Muqeeth commited on Nov 28, 2025

Commit

c8b87cc

verified ·

1 Parent(s): cce9419

Add files using upload-large-folder tool

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

.hydra/hydra.yaml +154 -0
.hydra/overrides.yaml +1 -0
run.log +0 -0
seed_9999/Qwen/Qwen2.5-7B-Instruct/adapters/README.md +207 -0
seed_9999/Qwen/Qwen2.5-7B-Instruct/adapters/agent_adapter/adapter_config.json +42 -0
seed_9999/Qwen/Qwen2.5-7B-Instruct/adapters/critic_adapter/adapter_config.json +42 -0
src_code_for_reproducibility/__init__.py +0 -0
src_code_for_reproducibility/__pycache__/__init__.cpython-312.pyc +0 -0
src_code_for_reproducibility/chat_utils/__pycache__/chat_turn.cpython-312.pyc +0 -0
src_code_for_reproducibility/docs/Makefile +19 -0
src_code_for_reproducibility/docs/make.bat +35 -0
src_code_for_reproducibility/docs/source/environments/diplomacy.rst +459 -0
src_code_for_reproducibility/docs/source/installation.rst +10 -0
src_code_for_reproducibility/docs/source/media/runbatch.png +0 -0
src_code_for_reproducibility/docs/source/src.environments.dond.dond_agent.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.dond.dond_game.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.dond.dond_log_funcs.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.dond.dond_return_funcs.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.dond.dond_statistics_funcs.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.environment_imports.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.ipd.ipd_game.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.ipd.ipd_log_funcs.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.ipd.ipd_statistics_funcs.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.rst +25 -0
src_code_for_reproducibility/docs/source/src.experiments.dond_run_train.rst +7 -0
src_code_for_reproducibility/docs/source/src.experiments.generate_and_train.rst +7 -0
src_code_for_reproducibility/docs/source/src.experiments.last_completion.rst +7 -0
src_code_for_reproducibility/docs/source/src.generation.rst +15 -0
src_code_for_reproducibility/docs/source/src.models.dummy_hf_agent.rst +7 -0
src_code_for_reproducibility/docs/source/src.models.rst +20 -0
src_code_for_reproducibility/docs/source/src.models.updatable_worker.rst +7 -0
src_code_for_reproducibility/docs/source/src.models.vllm_worker_wrap.rst +7 -0
src_code_for_reproducibility/docs/source/src.training.ppo_train.rst +7 -0
src_code_for_reproducibility/docs/source/src.training.reinforce_training.rst +7 -0
src_code_for_reproducibility/docs/source/src.training.rl_convs_processing.rst +7 -0
src_code_for_reproducibility/docs/source/src.training.rst +19 -0
src_code_for_reproducibility/docs/source/src.training.train_main.rst +7 -0
src_code_for_reproducibility/docs/source/src.utils.common_imports.rst +7 -0
src_code_for_reproducibility/docs/source/src.utils.log_statistics.rst +7 -0
src_code_for_reproducibility/docs/source/src.utils.parallel_shuffle.rst +7 -0
src_code_for_reproducibility/docs/source/src.utils.rst +24 -0
src_code_for_reproducibility/markov_games/__init__.py +0 -0
src_code_for_reproducibility/markov_games/agent.py +76 -0
src_code_for_reproducibility/markov_games/alternative_actions_runner.py +138 -0
src_code_for_reproducibility/markov_games/group_timesteps.py +150 -0
src_code_for_reproducibility/markov_games/linear_runner.py +30 -0
src_code_for_reproducibility/markov_games/markov_game.py +208 -0
src_code_for_reproducibility/markov_games/mg_utils.py +89 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/nego_hard_coded_policies.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/rollout_tree.py +86 -0

.hydra/hydra.yaml ADDED Viewed

	@@ -0,0 +1,154 @@

+hydra:
+  run:
+    dir: ${oc.env:SCRATCH}/llm_negotiation/${now:%Y_%m}/${experiment.name}
+  sweep:
+    dir: multirun/${now:%Y-%m-%d}/${now:%H-%M-%S}
+    subdir: ${hydra.job.num}
+  launcher:
+    _target_: hydra._internal.core_plugins.basic_launcher.BasicLauncher
+  sweeper:
+    _target_: hydra._internal.core_plugins.basic_sweeper.BasicSweeper
+    max_batch_size: null
+    params: null
+  help:
+    app_name: ${hydra.job.name}
+    header: '${hydra.help.app_name} is powered by Hydra.
+      '
+    footer: 'Powered by Hydra (https://hydra.cc)
+      Use --hydra-help to view Hydra specific help
+      '
+    template: '${hydra.help.header}
+      == Configuration groups ==
+      Compose your configuration from those groups (group=option)
+      $APP_CONFIG_GROUPS
+      == Config ==
+      Override anything in the config (foo.bar=value)
+      $CONFIG
+      ${hydra.help.footer}
+      '
+  hydra_help:
+    template: 'Hydra (${hydra.runtime.version})
+      See https://hydra.cc for more info.
+      == Flags ==
+      $FLAGS_HELP
+      == Configuration groups ==
+      Compose your configuration from those groups (For example, append hydra/job_logging=disabled
+      to command line)
+      $HYDRA_CONFIG_GROUPS
+      Use ''--cfg hydra'' to Show the Hydra config.
+      '
+    hydra_help: ???
+  hydra_logging:
+    version: 1
+    formatters:
+      simple:
+        format: '[%(asctime)s][HYDRA] %(message)s'
+    handlers:
+      console:
+        class: logging.StreamHandler
+        formatter: simple
+        stream: ext://sys.stdout
+    root:
+      level: INFO
+      handlers:
+      - console
+    loggers:
+      logging_example:
+        level: DEBUG
+    disable_existing_loggers: false
+  job_logging:
+    version: 1
+    formatters:
+      simple:
+        format: '[%(asctime)s][%(name)s][%(levelname)s] - %(message)s'
+    handlers:
+      console:
+        class: logging.StreamHandler
+        formatter: simple
+        stream: ext://sys.stdout
+      file:
+        class: logging.FileHandler
+        formatter: simple
+        filename: ${hydra.runtime.output_dir}/${hydra.job.name}.log
+    root:
+      level: INFO
+      handlers:
+      - console
+      - file
+    disable_existing_loggers: false
+  env: {}
+  mode: RUN
+  searchpath: []
+  callbacks: {}
+  output_subdir: .hydra
+  overrides:
+    hydra:
+    - hydra.mode=RUN
+    task: []
+  job:
+    name: run
+    chdir: false
+    override_dirname: ''
+    id: ???
+    num: ???
+    config_name: no_press_10_1_ties_ad_align_nocurrtimestep_seed9999.yaml
+    env_set: {}
+    env_copy: []
+    config:
+      override_dirname:
+        kv_sep: '='
+        item_sep: ','
+        exclude_keys: []
+  runtime:
+    version: 1.3.2
+    version_base: '1.1'
+    cwd: /scratch/m/muqeeth/llm_negotiation
+    config_sources:
+    - path: hydra.conf
+      schema: pkg
+      provider: hydra
+    - path: /scratch/m/muqeeth/llm_negotiation/configs
+      schema: file
+      provider: main
+    - path: ''
+      schema: structured
+      provider: schema
+    output_dir: /scratch/m/muqeeth/llm_negotiation/2025_11/no_press_10_1_ties_ad_align_nocurrtimestep_seed9999
+    choices:
+      hydra/env: default
+      hydra/callbacks: null
+      hydra/job_logging: default
+      hydra/hydra_logging: default
+      hydra/hydra_help: default
+      hydra/help: default
+      hydra/sweeper: basic
+      hydra/launcher: basic
+      hydra/output: default
+  verbose: false

.hydra/overrides.yaml ADDED Viewed

	@@ -0,0 +1 @@


1	+ []

run.log ADDED Viewed

The diff for this file is too large to render. See raw diff

seed_9999/Qwen/Qwen2.5-7B-Instruct/adapters/README.md ADDED Viewed

	@@ -0,0 +1,207 @@

+---
+base_model: Qwen/Qwen2.5-7B-Instruct
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-7B-Instruct
+- lora
+- transformers
+---
+# Model Card for Model ID
+<!-- Provide a quick summary of what the model is/does. -->
+## Model Details
+### Model Description
+<!-- Provide a longer summary of what this model is. -->
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+### Model Sources [optional]
+<!-- Provide the basic links for the model. -->
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+## Uses
+<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
+### Direct Use
+<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
+[More Information Needed]
+### Downstream Use [optional]
+<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
+[More Information Needed]
+### Out-of-Scope Use
+<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
+[More Information Needed]
+## Bias, Risks, and Limitations
+<!-- This section is meant to convey both technical and sociotechnical limitations. -->
+[More Information Needed]
+### Recommendations
+<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+## How to Get Started with the Model
+Use the code below to get started with the model.
+[More Information Needed]
+## Training Details
+### Training Data
+<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
+[More Information Needed]
+### Training Procedure
+<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
+#### Preprocessing [optional]
+[More Information Needed]
+#### Training Hyperparameters
+- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
+#### Speeds, Sizes, Times [optional]
+<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
+[More Information Needed]
+## Evaluation
+<!-- This section describes the evaluation protocols and provides the results. -->
+### Testing Data, Factors & Metrics
+#### Testing Data
+<!-- This should link to a Dataset Card if possible. -->
+[More Information Needed]
+#### Factors
+<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
+[More Information Needed]
+#### Metrics
+<!-- These are the evaluation metrics being used, ideally with a description of why. -->
+[More Information Needed]
+### Results
+[More Information Needed]
+#### Summary
+## Model Examination [optional]
+<!-- Relevant interpretability work for the model goes here -->
+[More Information Needed]
+## Environmental Impact
+<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+## Technical Specifications [optional]
+### Model Architecture and Objective
+[More Information Needed]
+### Compute Infrastructure
+[More Information Needed]
+#### Hardware
+[More Information Needed]
+#### Software
+[More Information Needed]
+## Citation [optional]
+<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
+**BibTeX:**
+[More Information Needed]
+**APA:**
+[More Information Needed]
+## Glossary [optional]
+<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
+[More Information Needed]
+## More Information [optional]
+[More Information Needed]
+## Model Card Authors [optional]
+[More Information Needed]
+## Model Card Contact
+[More Information Needed]
+### Framework versions
+- PEFT 0.17.1

seed_9999/Qwen/Qwen2.5-7B-Instruct/adapters/agent_adapter/adapter_config.json ADDED Viewed

	@@ -0,0 +1,42 @@

+{
+  "alpha_pattern": {},
+  "auto_mapping": null,
+  "base_model_name_or_path": "Qwen/Qwen2.5-7B-Instruct",
+  "bias": "none",
+  "corda_config": null,
+  "eva_config": null,
+  "exclude_modules": null,
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": true,
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 64,
+  "lora_bias": false,
+  "lora_dropout": 0.0,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "qalora_group_size": 16,
+  "r": 32,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": [
+    "v_proj",
+    "k_proj",
+    "gate_proj",
+    "q_proj",
+    "up_proj",
+    "down_proj",
+    "o_proj"
+  ],
+  "target_parameters": null,
+  "task_type": "CAUSAL_LM",
+  "trainable_token_indices": null,
+  "use_dora": false,
+  "use_qalora": false,
+  "use_rslora": false
+}

seed_9999/Qwen/Qwen2.5-7B-Instruct/adapters/critic_adapter/adapter_config.json ADDED Viewed

	@@ -0,0 +1,42 @@

+{
+  "alpha_pattern": {},
+  "auto_mapping": null,
+  "base_model_name_or_path": "Qwen/Qwen2.5-7B-Instruct",
+  "bias": "none",
+  "corda_config": null,
+  "eva_config": null,
+  "exclude_modules": null,
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": true,
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 64,
+  "lora_bias": false,
+  "lora_dropout": 0.0,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "qalora_group_size": 16,
+  "r": 32,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": [
+    "v_proj",
+    "k_proj",
+    "gate_proj",
+    "q_proj",
+    "up_proj",
+    "down_proj",
+    "o_proj"
+  ],
+  "target_parameters": null,
+  "task_type": "CAUSAL_LM",
+  "trainable_token_indices": null,
+  "use_dora": false,
+  "use_qalora": false,
+  "use_rslora": false
+}

src_code_for_reproducibility/__init__.py ADDED Viewed

File without changes

src_code_for_reproducibility/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (148 Bytes). View file

src_code_for_reproducibility/chat_utils/__pycache__/chat_turn.cpython-312.pyc ADDED Viewed

Binary file (1.32 kB). View file

src_code_for_reproducibility/docs/Makefile ADDED Viewed

	@@ -0,0 +1,19 @@

+# Minimal makefile for Sphinx documentation
+# You can set these variables from the command line, and also
+# from the environment for the first two.
+SPHINXOPTS    ?=
+SPHINXBUILD   ?= sphinx-build
+SOURCEDIR     = source
+BUILDDIR      = build
+# Put it first so that "make" without argument is like "make help".
+help:
+	@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(SPHINXFLAGS)
+.PHONY: help Makefile
+# Catch-all target: route all unknown targets to Sphinx using the new
+# "make mode" option.  $(O) is meant as a shortcut for $(SPHINXOPTS).
+%: Makefile
+	@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(SPHINXFLAGS)

src_code_for_reproducibility/docs/make.bat ADDED Viewed

	@@ -0,0 +1,35 @@

+@ECHO OFF
+pushd %~dp0
+REM Command file for Sphinx documentation
+if "%SPHINXBUILD%" == "" (
+	set SPHINXBUILD=sphinx-build
+)
+set SOURCEDIR=source
+set BUILDDIR=build
+%SPHINXBUILD% >NUL 2>NUL
+if errorlevel 9009 (
+	echo.
+	echo.The 'sphinx-build' command was not found. Make sure you have Sphinx
+	echo.installed, then set the SPHINXBUILD environment variable to point
+	echo.to the full path of the 'sphinx-build' executable. Alternatively you
+	echo.may add the Sphinx directory to PATH.
+	echo.
+	echo.If you don't have Sphinx installed, grab it from
+	echo.https://www.sphinx-doc.org/
+	exit /b 1
+)
+if "%1" == "" goto help
+%SPHINXBUILD% -M %1 %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
+goto end
+:help
+%SPHINXBUILD% -M help %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
+:end
+popd

src_code_for_reproducibility/docs/source/environments/diplomacy.rst ADDED Viewed

	@@ -0,0 +1,459 @@

+=================
+Diplomacy
+=================
+The Diplomacy environment provides a multi-agent negotiation interface for the classic board game Diplomacy,
+based on DeepMind's implementation. This document describes the API for interacting with the Diplomacy environment
+and its associated agent handler.
+Overview
+--------
+Diplomacy is a strategic board game set in Europe before World War I, where players control one of seven European powers
+and negotiate with each other to gain control of supply centers. The game is played in turns, with each turn consisting
+of movement phases, retreat phases, and build phases.
+Our implementation adapts DeepMind's Diplomacy code to the Multi-Agent Negotiation Environment standard, allowing it
+to be used with LLM agents through a text-based interface.
+Game Rules
+----------
+### Game Board and Powers
+Diplomacy is played on a map of Europe divided into provinces. The game features seven Great Powers that players can control:
+- England (blue)
+- France (light blue)
+- Germany (black)
+- Italy (green)
+- Austria-Hungary (red)
+- Russia (white)
+- Turkey (yellow)
+Each power begins with three supply centers (except Russia, which starts with four) and an equal number of units.
+### Units and Movement
+There are two types of units in Diplomacy:
+- **Armies (A)**: Can move to adjacent land provinces or be convoyed across water by fleets
+- **Fleets (F)**: Can move to adjacent coastal provinces and sea regions
+During movement phases, each unit can execute one of these orders:
+- **Hold**: The unit remains in its current province (e.g., "A PAR H")
+  - Format: [Unit Type] [Province] H
+  - Example: "A PAR H" means "Army in Paris holds its position"
+- **Move**: The unit attempts to move to an adjacent province (e.g., "A PAR - BUR")
+  - Format: [Unit Type] [Current Province] - [Destination Province]
+  - Example: "A PAR - BUR" means "Army in Paris moves to Burgundy"
+  - Example: "F BRE - ENG" means "Fleet in Brest moves to the English Channel"
+- **Support**: The unit supports another unit's move or hold (e.g., "A PAR S A MAR - BUR")
+  - Format for supporting a move: [Unit Type] [Province] S [Unit Type] [Province] - [Destination]
+  - Format for supporting a hold: [Unit Type] [Province] S [Unit Type] [Province]
+  - Example: "A PAR S A MAR - BUR" means "Army in Paris supports the Army in Marseille's move to Burgundy"
+  - Example: "F LON S F NTH" means "Fleet in London supports the Fleet in North Sea holding its position"
+- **Convoy**: A fleet can convoy an army across water (e.g., "F ENG C A LON - BRE")
+  - Format: [Fleet] [Sea Province] C [Army] [Coastal Province] - [Coastal Province]
+  - Example: "F ENG C A LON - BRE" means "Fleet in English Channel convoys the Army in London to Brest"
+All orders are executed simultaneously, and conflicts are resolved based on strength (number of supporting units).
+### Common Province Abbreviations
+Diplomacy uses three-letter abbreviations for provinces. Some common ones include:
+- **PAR**: Paris
+- **LON**: London
+- **BER**: Berlin
+- **MUN**: Munich
+- **BUR**: Burgundy
+- **MAR**: Marseilles
+- **BRE**: Brest
+- **ENG**: English Channel
+- **NTH**: North Sea
+- **VIE**: Vienna
+- **ROM**: Rome
+- **VEN**: Venice
+- **MOW**: Moscow
+- **CON**: Constantinople
+### Example: Movement and Conflicts
+For example, if France orders "A PAR - BUR" and Germany orders "A MUN - BUR", neither move succeeds as they have equal strength. However, if France also orders "A MAR S A PAR - BUR", then the French army from Paris would successfully move to Burgundy with strength of 2 against Germany's strength of 1.
+### Turn Structure
+A game year consists of five phases:
+1. **Spring Movement**: All powers submit orders for their units
+2. **Spring Retreat**: Units dislodged in the movement phase must retreat or be disbanded
+3. **Fall Movement**: Another round of movement orders
+4. **Fall Retreat**: Retreat orders for dislodged units
+5. **Winter Adjustment**: Powers gain or lose units based on the number of supply centers they control
+### Supply Centers and Building
+Supply centers (marked on the map) are key to victory. When a power occupies a supply center during a Fall turn, they gain control of it. During the Winter Adjustment phase:
+- If you control more supply centers than you have units, you can build new units in your home supply centers
+- If you control fewer supply centers than you have units, you must remove excess units
+### Example: Building and Removing Units
+If France controls 5 supply centers but only has 4 units, during the Winter phase they can build one new unit in an unoccupied home supply center (Paris, Marseilles, or Brest). Conversely, if France controls only 3 supply centers but has 4 units, they must remove one unit of their choice.
+### Negotiation
+A critical component of Diplomacy is the negotiation between players. Before submitting orders, players can communicate freely to form alliances, coordinate attacks, or mislead opponents. These negotiations are not binding, and betrayal is a common strategy.
+### Example: Alliance and Betrayal
+England and France might agree to an alliance against Germany, with England promising to support France's move into Belgium. However, England could secretly order their fleet to move into Belgium themselves or support a German move instead.
+### Victory Conditions
+The game ends when one power controls 18 or more supply centers (majority of the 34 total centers), or when players agree to a draw. In tournament settings, games may also end after a predetermined number of game years.
+DiplomacyEnv
+------------
+The ``DiplomacyEnv`` class provides an interface to the Diplomacy game environment that follows the Multi-Agent
+Negotiation Environment standard.
+.. code-block:: python
+    class DiplomacyEnv:
+        """
+        Multi-Agent Negotiation Environment for Diplomacy, adapting Deepmind's implementation
+        to the MarlEnvironment standard.
+        """
+        def __init__(self,
+                    initial_state: Optional[DiplomacyState] = None,
+                    max_turns: int = 100,
+                    points_per_supply_centre: bool = True,
+                    forced_draw_probability: float = 0.0,
+                    min_years_forced_draw: int = 35):
+            """Initialize the Diplomacy environment.
+            Args:
+                initial_state: Initial DiplomacyState (optional)
+                max_turns: Maximum number of turns in the game
+                points_per_supply_centre: Whether to award points per supply center in case of a draw
+                forced_draw_probability: Probability of forcing a draw after min_years_forced_draw
+                min_years_forced_draw: Minimum years before considering a forced draw
+            """
+            # ...
+        def reset(self):
+            """Reset the environment to an initial state and return the initial observation.
+            Returns:
+                observation (dict): A dictionary where keys are agent identifiers and values are observations.
+                Each observation contains:
+                - board_state: Current state of the board
+                - current_season: Current season in the game
+                - player_index: Index of the player's power
+                - possible_actions: List of possible actions in DeepMind's format
+                - human_readable_actions: List of human-readable action descriptions
+                - supply_centers: List of supply centers owned by the player
+                - units: List of units owned by the player
+                - year: Current year in the game
+            """
+            # ...
+        def step(self, actions):
+            """Take a step in the environment using the provided actions.
+            Args:
+                actions (dict): A dictionary where keys are agent identifiers and values are actions.
+                    Actions can be:
+                    - List of integer actions in DeepMind's format
+                    - List of string actions in text format (e.g., "A MUN - BER")
+            Returns:
+                observations (dict): A dictionary where keys are agent identifiers and values are observations.
+                    Each observation has the same structure as in reset().
+                done (bool): Whether the episode has ended.
+                info (dict): Additional information about the environment, including:
+                    - turn: Current turn number
+                    - returns: Game returns if the game is done, otherwise None
+                    - waiting_for: List of agents that still need to provide actions (if not all actions are provided)
+            """
+            # ...
+        def get_log_info(self):
+            """Get additional information about the environment for logging.
+            Returns:
+                log_info (dict): Information about the environment required to log the game, including:
+                    - power_names: List of power names
+                    - game_history: History of the game
+                    - current_turn: Current turn number
+                    - current_season: Current season name
+                    - supply_centers: Dictionary mapping power names to supply center counts
+            """
+            # ...
+        def render(self):
+            """Render the current state of the environment.
+            Displays a visualization of the current game state.
+            """
+            # ...
+        def close(self):
+            """Perform any necessary cleanup."""
+            # ...
+Key Implementation Details
+~~~~~~~~~~~~~~~~~~~~~~~~~
+The ``DiplomacyEnv`` class implements several key features:
+1. **Multi-Agent Support**: The environment tracks multiple agents (powers) and manages their interactions.
+2. **Turn-Based Gameplay**: The environment enforces the turn structure of Diplomacy, including different phases.
+3. **Action Processing**: The environment can handle actions in both text format and DeepMind's integer format.
+4. **Observation Generation**: The environment generates detailed observations for each agent, including board state, supply centers, and possible actions.
+5. **Game Termination**: The environment tracks game termination conditions, including supply center victory and maximum turn limits.
+Observation Structure
+~~~~~~~~~~~~~~~~~~~~
+Each agent receives an observation dictionary with the following structure:
+.. code-block:: python
+    {
+        "board_state": np.ndarray,  # Board state representation
+        "current_season": int,      # Season index (0-4)
+        "player_index": int,        # Index of the player's power (0-6)
+        "possible_actions": [int],  # List of possible actions in DeepMind's format
+        "human_readable_actions": [str],  # List of human-readable action descriptions
+        "supply_centers": [str],    # List of supply centers owned by the player
+        "units": [dict],            # List of units owned by the player
+        "year": int                 # Current year in the game
+    }
+Action Structure
+~~~~~~~~~~~~~~~
+Actions can be provided in two formats:
+1. **Text Format**: String actions like ``"A MUN - BER"`` or ``"F NTH C A LON - BEL"``.
+2. **Integer Format**: Lists of integers corresponding to DeepMind's action representation.
+The environment will convert text actions to the internal format as needed.
+DiplomacyAgent
+--------------
+The ``DiplomacyAgent`` class implements the agent handler interface for Diplomacy, processing observations from the environment and generating actions through an LLM.
+.. code-block:: python
+    class DiplomacyAgent:
+        """
+        Agent handler for Diplomacy, implementing the AgentState interface
+        for the multi-agent negotiation standard.
+        """
+        def __init__(self,
+                    power_name: str,
+                    use_text_interface: bool = True,
+                    system_prompt: Optional[str] = None):
+            """Initialize the Diplomacy agent handler.
+            Args:
+                power_name: Name of the power this agent controls
+                use_text_interface: Whether to use text-based interface (vs. structured)
+                system_prompt: Optional system prompt to use for the LLM
+            """
+            # ...
+        def step(self, observation_from_env, policy_output=None):
+            """Update the agent state based on the observation and action.
+            Args:
+                observation_from_env: The observation from the environment, with structure:
+                    - board_state: Current state of the board
+                    - current_season: Current season in the game
+                    - player_index: Index of the player's power
+                    - possible_actions: List of possible actions
+                    - human_readable_actions: List of human-readable action descriptions
+                    - supply_centers: List of supply centers owned by the player
+                    - units: List of units owned by the player
+                    - year: Current year in the game
+                policy_output: The output of the policy (LLM response), or None for initial prompt
+            Returns:
+                policy_id (str): The policy identifier ("llm_policy")
+                policy_input (dict): The input to the policy, with structure:
+                    - messages: List of conversation messages in the format:
+                        [{"role": "system", "content": "..."},
+                         {"role": "user", "content": "..."}]
+                action: The official action to be sent to the environment, or None if not ready
+                done (bool): Whether the LLM action is ready to be sent to the environment
+                info (dict): Additional information about the agent:
+                    - valid_action: Whether the extracted action is valid
+            """
+            # ...
+        def get_log_info(self):
+            """Get information about the agent required to log a trajectory.
+            Returns:
+                log_info (dict): Information about the agent required to log a trajectory:
+                    - power_name: Name of the power this agent controls
+                    - conversation_history: List of conversation messages
+                    - current_action: The current action, if any
+            """
+            # ...
+        def render(self):
+            """Render the current state of the agent.
+            Displays the agent's current state, including conversation history.
+            """
+            # ...
+        def close(self):
+            """Perform any necessary cleanup."""
+            # ...
+Key Implementation Details
+~~~~~~~~~~~~~~~~~~~~~~~~~
+The ``DiplomacyAgent`` class implements several key features:
+1. **LLM Interaction**: The agent generates prompts for an LLM and processes the LLM's responses to extract actions.
+2. **Conversation Management**: The agent maintains a conversation history for coherent interactions with the LLM.
+3. **Action Validation**: The agent validates extracted actions against the set of possible actions provided by the environment.
+4. **Error Handling**: The agent generates clarification prompts when invalid actions are detected.
+5. **Text-Based Interface**: The agent formats game state information into human-readable text for the LLM.
+Prompt Structure
+~~~~~~~~~~~~~~~
+The agent generates prompts that include:
+1. **System Prompt**: Instructions and context for the LLM, explaining its role as a Diplomacy player.
+2. **Game State Description**: A text description of the current game state, including:
+   - Current year and season
+   - Supply centers owned
+   - Units controlled
+   - Possible actions
+3. **Action Request**: Instructions on how to format actions.
+Example system prompt:
+.. code-block:: text
+    You are playing the role of FRANCE in a game of Diplomacy.
+    Your goal is to control as many supply centers as possible.
+    You can negotiate with other players and form alliances, but remember that
+    these alliances are not binding. When you need to submit orders for your units,
+    write them in the correct format, with each order on a new line.
+Example game state description:
+.. code-block:: text
+    Year: 1901, Season: SPRING_MOVES
+    You are playing as FRANCE.
+    You currently control 3 supply centers: PAR, MAR, BRE.
+    Your units are: A PAR, A MAR, F BRE.
+    Please provide orders for your units. Here are your possible actions:
+    A PAR - BUR
+    A PAR - GAS
+    A PAR - PIC
+    A PAR H
+    ...
+    Submit your orders, one per line, in the format like: "A MUN - BER" or "F NTH C A LON - BEL"
+Running Diplomacy Games
+----------------------
+To run Diplomacy games with LLM agents, you can use the ``run_batched_matches`` function with the ``DiplomacyEnv`` and ``DiplomacyAgent`` classes:
+.. code-block:: python
+    from mllm.environments.diplomacy.diplomacy_env import DiplomacyEnv
+    from mllm.environments.diplomacy.diplomacy_agent import DiplomacyAgent
+    from mllm.run_matches import run_batched_matches
+    # Create environment and agent handlers
+    env = DiplomacyEnv(max_turns=30)
+    agent_handlers = {
+        "AUSTRIA": DiplomacyAgent(power_name="AUSTRIA"),
+        "ENGLAND": DiplomacyAgent(power_name="ENGLAND"),
+        "FRANCE": DiplomacyAgent(power_name="FRANCE"),
+        "GERMANY": DiplomacyAgent(power_name="GERMANY"),
+        "ITALY": DiplomacyAgent(power_name="ITALY"),
+        "RUSSIA": DiplomacyAgent(power_name="RUSSIA"),
+        "TURKEY": DiplomacyAgent(power_name="TURKEY")
+    }
+    # Define policy mapping (mapping from policy IDs to actual policy functions)
+    policy_mapping = {
+        "llm_policy": my_llm_policy_function
+    }
+    # Run the game
+    game_results = run_batched_matches(
+        envs=[env],
+        agent_handlers_per_env=[agent_handlers],
+        policy_mapping=policy_mapping,
+        max_parallel_matches=1
+    )
+    # Process results
+    for result in game_results:
+        print(f"Game finished. Winner: {result['winner']}")
+        print(f"Supply centers: {result['supply_centers']}")
+This setup allows you to run Diplomacy games with LLM agents using the Multi-Agent Negotiation Environment standard.
+Limitations and Considerations
+-----------------------------
+1. **Performance**: Processing observations and actions for seven powers using LLMs can be computationally intensive.
+2. **Action Parsing**: Extracting valid actions from LLM outputs may require sophisticated parsing and error handling.
+3. **Game Complexity**: Diplomacy is a complex game with many rules and edge cases, which may be challenging for LLMs to fully grasp.
+4. **Turn Duration**: Real Diplomacy games include negotiation phases of variable duration, which are not fully captured in this implementation.
+5. **Text Formatting**: The quality of LLM interactions depends heavily on the formatting and clarity of text prompts.
+Advanced Usage
+------------
+For advanced usage, you can customize:
+1. **System Prompts**: Modify agent behavior by providing custom system prompts.
+2. **Observation Processing**: Extend the observation processing to include additional information.
+3. **Action Parsing**: Implement more sophisticated action parsing for complex orders.
+4. **Visualization**: Add custom visualization methods to the environment's render function.
+5. **Logging**: Extend the logging capabilities to capture additional information about the game state.

src_code_for_reproducibility/docs/source/installation.rst ADDED Viewed

	@@ -0,0 +1,10 @@

+Installation
+===========
+To install the package, run:
+.. code-block:: bash
+   git clone https://github.com/yourusername/llm_negotiation.git
+   cd llm_negotiation
+   pip install -e .

src_code_for_reproducibility/docs/source/media/runbatch.png ADDED Viewed

src_code_for_reproducibility/docs/source/src.environments.dond.dond_agent.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.dond.dond\_agent module
+========================================
+.. automodule:: src.environments.dond.dond_agent
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.dond.dond_game.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.dond.dond\_game module
+=======================================
+.. automodule:: src.environments.dond.dond_game
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.dond.dond_log_funcs.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.dond.dond\_log\_funcs module
+=============================================
+.. automodule:: src.environments.dond.dond_log_funcs
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.dond.dond_return_funcs.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.dond.dond\_return\_funcs module
+================================================
+.. automodule:: src.environments.dond.dond_return_funcs
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.dond.dond_statistics_funcs.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.dond.dond\_statistics\_funcs module
+====================================================
+.. automodule:: src.environments.dond.dond_statistics_funcs
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.environment_imports.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.environment\_imports module
+============================================
+.. automodule:: src.environments.environment_imports
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.ipd.ipd_game.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.ipd.ipd\_game module
+=====================================
+.. automodule:: src.environments.ipd.ipd_game
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.ipd.ipd_log_funcs.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.ipd.ipd\_log\_funcs module
+===========================================
+.. automodule:: src.environments.ipd.ipd_log_funcs
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.ipd.ipd_statistics_funcs.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.ipd.ipd\_statistics\_funcs module
+==================================================
+.. automodule:: src.environments.ipd.ipd_statistics_funcs
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.rst ADDED Viewed

	@@ -0,0 +1,25 @@

+src.environments package
+========================
+.. automodule:: src.environments
+   :members:
+   :undoc-members:
+   :show-inheritance:
+Subpackages
+-----------
+.. toctree::
+   :maxdepth: 4
+   src.environments.dond
+   src.environments.ipd
+Submodules
+----------
+.. toctree::
+   :maxdepth: 4
+   src.environments.env_imports
+   src.environments.environment_imports

src_code_for_reproducibility/docs/source/src.experiments.dond_run_train.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.experiments.dond\_run\_train module
+=======================================
+.. automodule:: src.experiments.dond_run_train
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.experiments.generate_and_train.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.experiments.generate\_and\_train module
+===========================================
+.. automodule:: src.experiments.generate_and_train
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.experiments.last_completion.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.experiments.last\_completion module
+=======================================
+.. automodule:: src.experiments.last_completion
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.generation.rst ADDED Viewed

	@@ -0,0 +1,15 @@

+src.generation package
+======================
+.. automodule:: src.generation
+   :members:
+   :undoc-members:
+   :show-inheritance:
+Submodules
+----------
+.. toctree::
+   :maxdepth: 4
+   src.generation.run_games

src_code_for_reproducibility/docs/source/src.models.dummy_hf_agent.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.models.dummy\_hf\_agent module
+==================================
+.. automodule:: src.models.dummy_llm_agent
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.models.rst ADDED Viewed

	@@ -0,0 +1,20 @@

+src.models package
+==================
+.. automodule:: src.models
+   :members:
+   :undoc-members:
+   :show-inheritance:
+Submodules
+----------
+.. toctree::
+   :maxdepth: 4
+   src.models.dummy_local_llm
+   src.models.local_llm
+   src.models.new_local_llm
+   src.models.server_llm
+   src.models.updatable_worker
+   src.models.vllm_worker_wrap

src_code_for_reproducibility/docs/source/src.models.updatable_worker.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.models.updatable\_worker module
+===================================
+.. automodule:: src.models.updatable_worker
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.models.vllm_worker_wrap.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.models.vllm\_worker\_wrap module
+====================================
+.. automodule:: src.models.vllm_worker_wrap
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.training.ppo_train.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.training.ppo\_train module
+==============================
+.. automodule:: src.training.ppo_train
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.training.reinforce_training.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.training.reinforce\_training module
+=======================================
+.. automodule:: src.training.reinforce_training
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.training.rl_convs_processing.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.training.rl\_convs\_processing module
+=========================================
+.. automodule:: src.training.rl_convs_processing
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.training.rst ADDED Viewed

	@@ -0,0 +1,19 @@

+src.training package
+====================
+.. automodule:: src.training
+   :members:
+   :undoc-members:
+   :show-inheritance:
+Submodules
+----------
+.. toctree::
+   :maxdepth: 4
+   src.training.ppo_train
+   src.training.ppo_train_value_head
+   src.training.reinforce_training
+   src.training.rl_convs_processing
+   src.training.train_main

src_code_for_reproducibility/docs/source/src.training.train_main.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.training.train\_main module
+===============================
+.. automodule:: src.training.train_main
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.utils.common_imports.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.utils.common\_imports module
+================================
+.. automodule:: src.utils.common_imports
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.utils.log_statistics.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.utils.log\_statistics module
+================================
+.. automodule:: src.utils.log_statistics
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.utils.parallel_shuffle.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.utils.parallel\_shuffle module
+==================================
+.. automodule:: src.utils.parallel_shuffle
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.utils.rst ADDED Viewed

	@@ -0,0 +1,24 @@

+src.utils package
+=================
+.. automodule:: src.utils
+   :members:
+   :undoc-members:
+   :show-inheritance:
+Submodules
+----------
+.. toctree::
+   :maxdepth: 4
+   src.utils.common_imports
+   src.utils.export_ppo_training_set
+   src.utils.extra_stats
+   src.utils.inherit_args
+   src.utils.log_gpu_usage
+   src.utils.log_statistics
+   src.utils.model_to_cpu
+   src.utils.parallel_shuffle
+   src.utils.quick_stats
+   src.utils.update_start_epoch

src_code_for_reproducibility/markov_games/__init__.py ADDED Viewed

File without changes

src_code_for_reproducibility/markov_games/agent.py ADDED Viewed

	@@ -0,0 +1,76 @@

+"""
+In simple RL paradise, where the action dimensions are constant and well defined,
+Agent classes are not necessary. But in MARL, with LLM's, there isn't always
+a direct path from policy to action. For instance, from the observation of the environment,
+a prompt must be created. Then, the outputs of the policy might be incorrect, so a second
+request to the LLM must be sent before the action is well defined. This is why this Agent class exists.
+It acts as a mini environment, bridging the gap between the core simulation and
+the LLM policies.
+"""
+from abc import ABC, abstractmethod
+from collections.abc import Callable
+from typing import Any, Tuple
+from numpy.random import default_rng
+from mllm.markov_games.rollout_tree import AgentActLog
+class Agent(ABC):
+    @abstractmethod
+    def __init__(
+        self,
+        seed: int,
+        agent_id: str,
+        agent_name: str,
+        agent_policy: Callable[[list[dict]], str],
+        *args,
+        **kwargs,
+    ):
+        """
+        Initialize the agent state.
+        """
+        self.seed = seed
+        self.agent_id = agent_id
+        self.agent_name = agent_name
+        self.policy = policy
+        self.rng = default_rng(self.seed)
+        raise NotImplementedError
+    async def act(self, observation) -> Tuple[Any, AgentActLog]:
+        """
+        Query (possibly multiple times) a policy (or possibly a pool of policies) to
+        obtain the action of the agent.
+        Example:
+        action = None
+        prompt = self.observation_to_prompt(observation)
+        while not self.valid(action):
+            output = await self.policy.generate(prompt)
+            action = self.policy_output_to_action(output)
+        return action
+        Returns:
+            action
+            step_info
+        """
+        raise NotImplementedError
+    def get_safe_copy(self):
+        """
+        Return copy of the agent object that is decorrelated from the original object.
+        """
+        raise NotImplementedError
+    def reset(self):
+        raise NotImplementedError
+    def render(self):
+        raise NotImplementedError
+    def close(self):
+        raise NotImplementedError
+    def get_agent_info(self):
+        raise NotImplementedError

src_code_for_reproducibility/markov_games/alternative_actions_runner.py ADDED Viewed

	@@ -0,0 +1,138 @@

+import asyncio
+import copy
+import json
+import os.path
+from typing import Any, Tuple
+from mllm.markov_games.markov_game import AgentAndActionSafeCopy, MarkovGame
+from mllm.markov_games.rollout_tree import (
+    AgentActLog,
+    RolloutTreeBranchNode,
+    RolloutTreeNode,
+    RolloutTreeRootNode,
+    StepLog,
+)
+AgentId = str
+async def run_with_unilateral_alt_action(
+    markov_game: MarkovGame,
+    agent_id: AgentId,
+    time_step: int,
+    branch_node: RolloutTreeBranchNode,
+    max_depth: int,
+):
+    """
+    This function is used to generate a new branch for a given agent.
+    """
+    # Generate alternative action and take a step
+    await markov_game.set_action_of_agent(agent_id)
+    terminated: bool = markov_game.take_simulation_step()
+    step_log = markov_game.get_step_log()
+    first_alternative_node = RolloutTreeNode(
+        step_log=step_log,
+        time_step=time_step,
+    )
+    # Generate rest of trajectory up to max depth
+    time_step += 1
+    counter = 1
+    previous_node = first_alternative_node
+    while not terminated and counter <= max_depth:
+        terminated, step_log = await markov_game.step()
+        current_node = RolloutTreeNode(step_log=step_log, time_step=time_step)
+        previous_node.child = current_node
+        previous_node = current_node
+        counter += 1
+        time_step += 1
+    if branch_node.branches == None:
+        branch_node.branches = {agent_id: [first_alternative_node]}
+    else:
+        agent_branches = branch_node.branches.get(agent_id, [])
+        agent_branches.append(first_alternative_node)
+        branch_node.branches[agent_id] = agent_branches
+async def AlternativeActionsRunner(
+    markov_game: MarkovGame,
+    output_folder: str,
+    nb_alternative_actions: int,
+    max_depth: int,
+    branch_only_on_new_round: bool = False,
+):
+    """
+    This method generates a trajectory with partially completed branches,
+    where the branching comes from taking unilateraly different actions.
+    The resulting data is used to estimate the updated advantage alignment policy gradient terms.
+    Let k := nb_sub_steps. Then the number of steps generated is O(Tk), where T is
+    the maximum trajectory length.
+    """
+    tasks = []
+    time_step = 0
+    terminated = False
+    root = RolloutTreeRootNode(
+        id=markov_game.get_id(),
+        crn_id=markov_game.get_crn_id()
+    )
+    previous_node = root
+    while not terminated:
+        mg_before_action = markov_game.get_safe_copy()
+        # Get safe copies for main branch
+        agent_action_safe_copies: dict[
+            AgentId, AgentAndActionSafeCopy
+        ] = await markov_game.get_actions_of_agents_without_side_effects()
+        markov_game.set_actions_of_agents_manually(agent_action_safe_copies)
+        terminated = markov_game.take_simulation_step()
+        main_node = RolloutTreeNode(
+            step_log=markov_game.get_step_log(), time_step=time_step
+        )
+        branch_node = RolloutTreeBranchNode(main_child=main_node)
+        previous_node.child = branch_node
+        previous_node = main_node
+        # Get alternative branches by generating new unilateral actions
+        for agent_id in markov_game.agent_ids:
+            for _ in range(nb_alternative_actions):
+                # Get safe copies for branches
+                branch_agent_action_safe_copies: dict[
+                    AgentId, AgentAndActionSafeCopy
+                ] = {
+                    agent_id: AgentAndActionSafeCopy(
+                        action=copy.deepcopy(agent_action_safe_copy.action),
+                        action_info=copy.deepcopy(agent_action_safe_copy.action_info),
+                        agent_after_action=agent_action_safe_copy.agent_after_action.get_safe_copy(),
+                    )
+                    for agent_id, agent_action_safe_copy in agent_action_safe_copies.items()
+                }
+                mg_branch: MarkovGame = mg_before_action.get_safe_copy()
+                other_agent_id = [id for id in mg_branch.agent_ids if id != agent_id][0]
+                mg_branch.set_action_and_agent_after_action_manually(
+                    agent_id=other_agent_id,
+                    agent_action_safe_copy=branch_agent_action_safe_copies[
+                        other_agent_id
+                    ],
+                )
+                task = asyncio.create_task(
+                    run_with_unilateral_alt_action(
+                        markov_game=mg_branch,
+                        time_step=time_step,
+                        agent_id=agent_id,
+                        branch_node=branch_node,
+                        max_depth=max_depth,
+                    )
+                )
+                tasks.append(task)
+        time_step += 1
+    # wait for all branches to complete
+    await asyncio.gather(*tasks)
+    return root

src_code_for_reproducibility/markov_games/group_timesteps.py ADDED Viewed

	@@ -0,0 +1,150 @@

+"""
+This module contains the logic for grouping time steps.
+"""
+import copy
+from typing import Callable
+from mllm.markov_games.markov_game import MarkovGame
+from mllm.markov_games.rollout_tree import (
+    AgentActLog,
+    RolloutTreeBranchNode,
+    RolloutTreeNode,
+    RolloutTreeRootNode,
+    StepLog,
+)
+from mllm.markov_games.simulation import SimulationStepLog
+AgentId = str
+def group_time_steps(
+    rollout_tree: RolloutTreeRootNode,
+    accumulation_stop_condition: Callable[[StepLog], bool],
+) -> RolloutTreeRootNode:
+    """
+    During generation, we create rollout trees according to the real time steps.
+    However, during training, we might want to treat groups of time steps as a single time step.
+    As a concrete example, take Trust-and-Split. At each round, say we have X time steps of communication and then one time step for the split.
+    Then the communication actions will not get any reward, and the split action will get the reward. During REINFORCE training, with discounting, this
+    can cause training instability. We could instead treat every action in the round as being part of a single action, and give it the reward of the split action.
+    This method helps to do this sort of grouping.
+    It accumulates actions until the accumulation_stop_condition is met, and then creates a new node with the accumulated actions.
+    It then recursively calls itself on the child node.
+    Details:
+    - The reward for the group is the reward of the last time step in the group.
+    - The simulation log for the group is the simulation log of the last time step in the group.
+    - The state end for the group becomes the first state end in the group.
+    - The agent info for the group is the agent info of the last time step in the group.
+    """
+    def group_step_logs(step_logs: list[StepLog]) -> StepLog:
+        """
+        Concatenate per-agent chat turns across steps; keep only the first is_state_end.
+        """
+        last_sim_log = step_logs[-1].simulation_step_log
+        agent_ids = {aid for s in step_logs for aid in s.action_logs.keys()}
+        grouped_logs: dict[AgentId, AgentActLog] = {}
+        for aid in agent_ids:
+            turns = []
+            for s in step_logs:
+                act = s.action_logs.get(aid)
+                if act and act.chat_turns:
+                    turns.extend(copy.deepcopy(act.chat_turns))
+            disable_is_state_end = False
+            # Only the first state_end should be True, the rest should be False
+            for t in turns:
+                if t.is_state_end:
+                    if disable_is_state_end:
+                        t.is_state_end = False
+                    else:
+                        disable_is_state_end = True
+                    continue
+            grouped_logs[aid] = AgentActLog(
+                chat_turns=turns, info=step_logs[-1].action_logs[aid].info
+            )
+        return StepLog(action_logs=grouped_logs, simulation_step_log=last_sim_log)
+    def group_time_steps_rec(
+        current_node: RolloutTreeNode | RolloutTreeBranchNode,
+        group_time_step: int,
+        accumulation_step_logs: list[StepLog],
+    ) -> RolloutTreeNode | RolloutTreeBranchNode:
+        """
+        Groups time steps. Recursion is used to handle branches.
+        """
+        assert isinstance(current_node, RolloutTreeNode) or isinstance(
+            current_node, RolloutTreeBranchNode
+        ), "Current node must be a tree node or a branch node. Is of type: " + str(
+            type(current_node)
+        )
+        first_group_node = None
+        current_group_node = None
+        while current_node is not None:
+            if isinstance(current_node, RolloutTreeBranchNode):
+                raise Exception(
+                    "Grouping timesteps by round is not supported for branching trajectories yet."
+                )
+            # Special recursive case for branches
+            # if isinstance(current_node, RolloutTreeBranchNode):
+            #     branches = {}
+            #     for agent_id, branch_nodes in current_node.branches.items():
+            #         branch_group_nodes = []
+            #         for branch_node in branch_nodes:
+            #             branch_group_node = group_time_steps_rec(
+            #                 current_node=branch_node,
+            #                 group_time_step=group_time_step,
+            #                 accumulation_step_logs=copy.deepcopy(accumulation_step_logs))
+            #             branch_group_nodes.append(branch_group_node)
+            #         branches[agent_id] = branch_group_nodes
+            #     main_child_group_node = group_time_steps_rec(
+            #         current_node=current_node.main_child,
+            #         group_time_step=group_time_step,
+            #         accumulation_step_logs=copy.deepcopy(accumulation_step_logs))
+            #     return RolloutTreeBranchNode(main_child=main_child_group_node, branches=branches)
+            # Accumulate
+            accumulation_step_logs.append(current_node.step_log)
+            if accumulation_stop_condition(current_node.step_log):
+                grouped_step_logs = group_step_logs(accumulation_step_logs)
+                accumulation_step_logs = []
+                new_group_node = RolloutTreeNode(
+                    step_log=grouped_step_logs, time_step=group_time_step, child=None
+                )
+                if first_group_node == None:
+                    first_group_node = new_group_node
+                group_time_step += 1
+                if current_group_node is not None:
+                    current_group_node.child = new_group_node
+                current_group_node = new_group_node
+            current_node = current_node.child
+        return first_group_node
+    node = group_time_steps_rec(
+        current_node=rollout_tree.child, group_time_step=0, accumulation_step_logs=[]
+    )
+    return RolloutTreeRootNode(
+        id=rollout_tree.id,
+        crn_id=rollout_tree.crn_id,
+        child=node,
+        agent_ids=rollout_tree.agent_ids,
+    )
+def stop_when_round_ends(step_log: StepLog) -> bool:
+    """
+    Simplest stop condition. Will return True if step log is the last time step of a round.
+    This will throw an error if this information is not available in the simulation info.
+    """
+    assert (
+        "is_last_timestep_in_round" in step_log.simulation_step_log.info.keys()
+    ), "To group by round, is_last_timestep_in_round must be set in the info of your simulation step log at each time step."
+    return step_log.simulation_step_log.info["is_last_timestep_in_round"]
+def group_by_round(rollout_tree: RolloutTreeRootNode) -> RolloutTreeRootNode:
+    """
+    Groups time steps by round.
+    """
+    return group_time_steps(rollout_tree, stop_when_round_ends)

src_code_for_reproducibility/markov_games/linear_runner.py ADDED Viewed

	@@ -0,0 +1,30 @@

+import asyncio
+import json
+import os.path
+from mllm.markov_games.markov_game import MarkovGame
+from mllm.markov_games.rollout_tree import RolloutTreeNode, RolloutTreeRootNode
+async def LinearRunner(
+    markov_game: MarkovGame, output_folder: str
+) -> RolloutTreeRootNode:
+    """
+    This method generates a trajectory without branching.
+    """
+    time_step = 0
+    terminated = False
+    root = RolloutTreeRootNode(
+        id=markov_game.get_id(),
+        crn_id=markov_game.get_crn_id(),
+        agent_ids=markov_game.get_agent_ids(),
+    )
+    previous_node = root
+    while not terminated:
+        terminated, step_log = await markov_game.step()
+        current_node = RolloutTreeNode(step_log=step_log, time_step=time_step)
+        previous_node.child = current_node
+        previous_node = current_node
+        time_step += 1
+    return root

src_code_for_reproducibility/markov_games/markov_game.py ADDED Viewed

	@@ -0,0 +1,208 @@

+"""
+This class unifies a simulation, and the agents acting in it (see `simulation.py` & `agent.py`).
+In a MarkovGame step,
+    1) each agent takes an action,
+    2) the state transitions with respect to these actions,
+    3) all relevant data of the step is appended to the historical data list
+In order to perform 3), the agents and the simulation are expected, at each time step,
+to return a log of the state transition (from their perspective).
+For instance, the Simulation might send rewards and the agents might send prompting contexts to be used later to generate the training data.
+A different approach would be to simply have the agents keep their data private and log it upon completion of a trajectory.
+The approach we use here centralizes the data gathering aspect,
+making it easy to create sub-trajectories (in the `runners` defined in `runners.py`) descriptions that
+only log information for step transitions occuring after the branching out.
+"""
+import asyncio
+import copy
+import json
+import os
+from dataclasses import dataclass
+from typing import Any, List, Literal, Optional, Tuple
+from transformers.models.idefics2 import Idefics2Config
+from mllm.markov_games.agent import Agent
+from mllm.markov_games.rollout_tree import AgentActLog, StepLog
+from mllm.markov_games.simulation import Simulation
+AgentId = str
+@dataclass
+class AgentAndActionSafeCopy:
+    action: Any
+    action_info: AgentActLog
+    agent_after_action: type[Agent]
+class MarkovGame(object):
+    def __init__(
+        self,
+        id: int,
+        agents: dict[AgentId, type[Agent]],
+        simulation: type[Simulation],
+        crn_id: int,
+    ):
+        """
+        Args:
+            agents:
+            output_path:
+                Path where the step infos are saved.
+            simulation:
+                Simulation object. Example: IPDSimulation
+        """
+        self.agents = agents
+        self.agent_ids = self.agents.keys()
+        self.simulation = simulation
+        self.simulation_step_log = None
+        self.agent_step_logs = {agent_id: None for agent_id in self.agent_ids}
+        self.actions = {}
+        self.id = id
+        self.crn_id = crn_id
+    def get_id(self) -> str:
+        return self.id
+    def get_crn_id(self) -> int:
+        return self.crn_id
+    def get_agent_ids(self) -> List[AgentId]:
+        return list(self.agent_ids)
+    async def get_action_of_agent_without_side_effects(
+        self, agent_id: AgentId
+    ) -> Tuple[Any, AgentActLog]:
+        """
+        Safe function to get an action of an agent without modifying the agent or the simulation.
+        """
+        agent = self.agents[agent_id]
+        agent_before_action = agent.get_safe_copy()
+        obs = self.simulation.get_obs_agent(agent_id)
+        action, action_info = await agent.act(observation=obs)
+        self.agents[agent_id] = agent_before_action
+        agent_after_action = agent.get_safe_copy()
+        return AgentAndActionSafeCopy(action, action_info, agent_after_action)
+    async def get_actions_of_agents_without_side_effects(
+        self,
+    ) -> dict[AgentId, AgentAndActionSafeCopy]:
+        """
+        Safe function to get an action of an agent without modifying the agent or the simulation.
+        """
+        tasks = []
+        for agent_id in self.agent_ids:
+            task = asyncio.create_task(
+                self.get_action_of_agent_without_side_effects(agent_id)
+            )
+            tasks.append(task)
+        agent_and_action_safe_copies: list[
+            AgentAndActionSafeCopy
+        ] = await asyncio.gather(*tasks)
+        return {
+            agent_id: agent_and_action_safe_copy
+            for agent_id, agent_and_action_safe_copy in zip(
+                self.agent_ids, agent_and_action_safe_copies
+            )
+        }
+    def set_action_and_agent_after_action_manually(
+        self,
+        agent_id: AgentId,
+        agent_action_safe_copy: AgentAndActionSafeCopy,
+    ):
+        """
+        Set the action and the agent after action manually.
+        """
+        self.actions[agent_id] = agent_action_safe_copy.action
+        self.agent_step_logs[agent_id] = agent_action_safe_copy.action_info
+        self.agents[agent_id] = agent_action_safe_copy.agent_after_action
+    def set_actions_of_agents_manually(
+        self, actions: dict[AgentId, AgentAndActionSafeCopy]
+    ):
+        """
+        Set the actions of agents manually.
+        """
+        for agent_id, agent_action_safe_copy in actions.items():
+            self.set_action_and_agent_after_action_manually(
+                agent_id, agent_action_safe_copy
+            )
+    async def set_action_of_agent(self, agent_id: AgentId):
+        """
+        TOWRITE
+        """
+        agent = self.agents[agent_id]
+        obs = self.simulation.get_obs_agent(agent_id)
+        action, action_info = await agent.act(observation=obs)
+        self.actions[agent_id] = action
+        self.agent_step_logs[agent_id] = action_info
+    async def set_actions(self):
+        """
+        TOWRITE
+        """
+        # background_tasks = set()
+        tasks = []
+        for agent_id in self.agent_ids:
+            task = asyncio.create_task(self.set_action_of_agent(agent_id))
+            tasks.append(task)
+        await asyncio.gather(*tasks)
+    def take_simulation_step(self):
+        """
+        TOWRITE
+        """
+        terminated, self.simulation_step_log = self.simulation.step(self.actions)
+        return terminated
+    def get_step_log(self) -> StepLog:
+        """
+        TOWRITE
+        TODO: assert actions and simulation have taken step
+        """
+        step_log = StepLog(
+            simulation_step_log=self.simulation_step_log,
+            action_logs=self.agent_step_logs,
+        )
+        return step_log
+    async def step(self) -> Tuple[bool, StepLog]:
+        """
+        TOWRITE
+        """
+        await self.set_actions()
+        terminated = self.take_simulation_step()
+        step_log = self.get_step_log()
+        return terminated, step_log
+    def get_safe_copy(self):
+        """
+        TOWRITE
+        """
+        new_markov_game = copy.copy(self)
+        new_simulation = self.simulation.get_safe_copy()
+        new_agents = {
+            agent_id: agent.get_safe_copy() for agent_id, agent in self.agents.items()
+        }
+        # Reassign copied components
+        new_markov_game.simulation = new_simulation
+        new_markov_game.agents = new_agents
+        # IMPORTANT: ensure agent_ids references the new agents dict, not the original
+        new_markov_game.agent_ids = new_markov_game.agents.keys()
+        # Deep-copy step data to avoid correlation
+        new_markov_game.simulation_step_log = copy.deepcopy(self.simulation_step_log)
+        new_markov_game.actions = copy.deepcopy(self.actions)
+        # Rebuild logs to align exactly with new agent ids
+        old_agent_step_logs = copy.deepcopy(self.agent_step_logs)
+        new_markov_game.agent_step_logs = {
+            agent_id: old_agent_step_logs.get(agent_id)
+            for agent_id in new_markov_game.agent_ids
+        }
+        return new_markov_game

src_code_for_reproducibility/markov_games/mg_utils.py ADDED Viewed

	@@ -0,0 +1,89 @@

+import asyncio
+import copy
+from collections.abc import Callable
+from dataclasses import dataclass
+from mllm.markov_games.ipd.ipd_agent import IPDAgent
+from mllm.markov_games.ipd.ipd_simulation import IPD
+from mllm.markov_games.markov_game import MarkovGame
+from mllm.markov_games.negotiation.dond_agent import DealNoDealAgent
+from mllm.markov_games.negotiation.dond_simulation import DealNoDealSimulation
+from mllm.markov_games.negotiation.nego_hard_coded_policies import (
+    HardCodedNegoGreedyPolicy,
+    HardCodedNegoWelfareMaximizingPolicy,
+)
+from mllm.markov_games.ipd.Ipd_hard_coded_agents import AlwaysCooperateIPDAgent, AlwaysDefectIPDAgent
+from mllm.markov_games.negotiation.no_press_nego_agent import NoPressAgent
+from mllm.markov_games.negotiation.no_press_nego_simulation import NoPressSimulation
+from mllm.markov_games.negotiation.tas_agent import TrustAndSplitAgent
+from mllm.markov_games.negotiation.tas_rps_agent import TrustAndSplitRPSAgent
+from mllm.markov_games.negotiation.tas_rps_simulation import TrustAndSplitRPSSimulation
+from mllm.markov_games.negotiation.tas_simple_agent import TrustAndSplitSimpleAgent
+from mllm.markov_games.negotiation.tas_simple_simulation import (
+    TrustAndSplitSimpleSimulation,
+)
+from mllm.markov_games.negotiation.tas_simulation import TrustAndSplitSimulation
+from mllm.markov_games.rollout_tree import (
+    AgentActLog,
+    RolloutTreeBranchNode,
+    RolloutTreeNode,
+    RolloutTreeRootNode,
+    StepLog,
+)
+from mllm.markov_games.simulation import SimulationStepLog
+AgentId = str
+@dataclass
+class AgentConfig:
+    agent_id: str
+    agent_name: str
+    agent_class_name: str
+    policy_id: str
+    init_kwargs: dict
+@dataclass
+class MarkovGameConfig:
+    id: int
+    seed: int
+    simulation_class_name: str
+    simulation_init_args: dict
+    agent_configs: list[AgentConfig]
+def init_markov_game_components(
+    config: MarkovGameConfig, policies: dict[str, Callable[[list[dict]], str]]
+):
+    """
+    TOWRITE
+    """
+    agents = {}
+    agent_names = []
+    for agent_config in config.agent_configs:
+        agent_id = agent_config.agent_id
+        agent_name = agent_config.agent_name
+        agent_class = eval(agent_config.agent_class_name)
+        agent = agent_class(
+            seed=config.seed,
+            agent_id=agent_id,
+            agent_name=agent_name,
+            policy=policies[agent_config.policy_id],
+            **agent_config.init_kwargs,
+        )
+        agents[agent_id] = agent
+        agent_names.append(agent_name)
+    simulation = eval(config.simulation_class_name)(
+        seed=config.seed,
+        agent_ids=list(agents.keys()),
+        agent_names=agent_names,
+        **config.simulation_init_args,
+    )
+    markov_game = MarkovGame(
+        id=config.id,
+        crn_id=config.seed,
+        agents=agents,
+        simulation=simulation,
+    )
+    return markov_game

src_code_for_reproducibility/markov_games/negotiation/__pycache__/nego_hard_coded_policies.cpython-312.pyc ADDED Viewed

Binary file (3.23 kB). View file

src_code_for_reproducibility/markov_games/rollout_tree.py ADDED Viewed

	@@ -0,0 +1,86 @@

+"""
+TODO: add parent to nodes so that some verification can be done. For instance, to ensure that node reward keys match the parent node.
+"""
+from __future__ import annotations
+import json
+from dataclasses import dataclass
+from pathlib import Path
+from typing import Any, List, Literal, Optional, Tuple
+import jsonschema
+from pydantic import BaseModel, Field, model_validator
+from mllm.chat_utils.chat_turn import ChatTurn
+AgentId = str
+class SimulationStepLog(BaseModel):
+    rewards: dict[AgentId, float]
+    info: Any = None
+class AgentActLog(BaseModel):
+    chat_turns: list[ChatTurn] | None
+    info: Any = None
+    @model_validator(mode="after")
+    def _exactly_one_state_end(self):
+        """
+        This method is used to enforce that for each AgentActLog, there is exactly one ChatTurn which is a state end.
+        """
+        if self.chat_turns != []:
+            n = sum(1 for t in self.chat_turns if t.is_state_end)
+            if n != 1:
+                raise ValueError(
+                    f"AgentActLog must have exactly one ChatTurn with is_state_end=True; got {self.chat_turns}."
+                )
+            return self
+        else:
+            return self
+class StepLog(BaseModel):
+    action_logs: dict[AgentId, AgentActLog]
+    simulation_step_log: SimulationStepLog
+# BranchType = Literal["unilateral_deviation", "common_deviation"] # might not be necessary
+# class BranchNodeInfo(BaseModel):
+#     branch_id: str
+#     branch_for: AgentId
+#     branch_type: BranchType
+class RolloutTreeNode(BaseModel):
+    step_log: StepLog
+    time_step: int
+    child: RolloutTreeNode | RolloutTreeBranchNode | None = None
+class RolloutTreeBranchNode(BaseModel):
+    """
+    First item of the tuple indicates which agent "called" for an alternative branch.
+    """
+    main_child: RolloutTreeNode
+    branches: dict[AgentId, list[RolloutTreeNode]] | None = None
+class RolloutTreeRootNode(BaseModel):
+    id: int
+    crn_id: int  # ID of the rng used to generate this rollout tree
+    child: RolloutTreeNode | RolloutTreeBranchNode | None = None
+    agent_ids: List[AgentId] = Field(min_length=1)
+# class RolloutTreeLeafNode(BaseModel):
+#     step_log: StepLog
+#     time_step: int
+# Necessary for self-referential stuff in pydantic
+RolloutTreeBranchNode.model_rebuild()
+RolloutTreeNode.model_rebuild()