dereckpichemila commited on Sep 10, 2025

Commit

40c85cf

verified ·

1 Parent(s): fa30e5a

Add files using upload-large-folder tool

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

.hydra/config.yaml +159 -0
.hydra/overrides.yaml +1 -0
run.log +0 -0
src_code_for_reproducibility/__init__.py +0 -0
src_code_for_reproducibility/docs/Makefile +19 -0
src_code_for_reproducibility/docs/generate_docs.py +249 -0
src_code_for_reproducibility/docs/make.bat +35 -0
src_code_for_reproducibility/docs/source/index.rst +22 -0
src_code_for_reproducibility/docs/source/installation.rst +10 -0
src_code_for_reproducibility/docs/source/launch.rst +0 -0
src_code_for_reproducibility/docs/source/marl_standard.rst +141 -0
src_code_for_reproducibility/docs/source/src.environments.dond.dond_return_funcs.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.dond.dond_training_data_funcs.rst +7 -0
src_code_for_reproducibility/docs/source/src.environments.dond.rst +19 -0
src_code_for_reproducibility/docs/source/src.experiments.arithmetic_test.rst +7 -0
src_code_for_reproducibility/docs/source/src.experiments.generate_and_train.rst +7 -0
src_code_for_reproducibility/docs/source/src.experiments.last_completion.rst +7 -0
src_code_for_reproducibility/docs/source/src.generation.rst +15 -0
src_code_for_reproducibility/docs/source/src.models.local_llm.rst +7 -0
src_code_for_reproducibility/docs/source/src.models.new_local_llm.rst +7 -0
src_code_for_reproducibility/docs/source/src.models.oai_agent.rst +7 -0
src_code_for_reproducibility/docs/source/src.rst +28 -0
src_code_for_reproducibility/docs/source/src.training.rst +19 -0
src_code_for_reproducibility/docs/source/src.training.train_main.rst +7 -0
src_code_for_reproducibility/markov_games/__pycache__/export_utils.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/__init__.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_agent.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_agent.cpython-311.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_game.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_log_funcs.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_log_match.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_player.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_simulation.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_simulation.cpython-311.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_statistics.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_statistics.cpython-311.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_statistics_funcs.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_training_data.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_training_data_funcs.cpython-310.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/ipd_agent.py +122 -0
src_code_for_reproducibility/markov_games/ipd/ipd_simulation.py +162 -0
src_code_for_reproducibility/markov_games/runners/__pycache__/alternative_actions_runner.cpython-311.pyc +0 -0
src_code_for_reproducibility/markov_games/runners/__pycache__/linear_runner.cpython-311.pyc +0 -0
src_code_for_reproducibility/utils/__init__.py +0 -0
src_code_for_reproducibility/utils/__pycache__/__init__.cpython-310.pyc +0 -0
src_code_for_reproducibility/utils/__pycache__/__init__.cpython-311.pyc +0 -0
src_code_for_reproducibility/utils/__pycache__/common_imports.cpython-310.pyc +0 -0
src_code_for_reproducibility/utils/__pycache__/dict_get_path.cpython-310.pyc +0 -0
src_code_for_reproducibility/utils/__pycache__/extra_stats.cpython-310.pyc +0 -0
src_code_for_reproducibility/utils/__pycache__/get_coagent_id.cpython-310.pyc +0 -0

.hydra/config.yaml ADDED Viewed

	@@ -0,0 +1,159 @@

+experiment:
+  nb_epochs: 1000
+  nb_matches_per_iteration: 64
+  reinit_matches_each_it: true
+  checkpoint_every_n_iterations: 10
+  start_epoch: 0
+  resume_experiment: true
+  base_seed: 0
+  seed_group_size: 1
+  train: true
+  name: tas_rps_no_regex_prev_ad_align_buffer_gae
+  agent_buffer: true
+  keep_agent_buffer_count: ${lora_count}
+  agent_buffer_recent_k: -1
+temperature: 1.0
+markov_games:
+  runner_method_name: LinearRunner
+  runner_kwargs: {}
+  group_by_round: true
+  simulation_class_name: TrustAndSplitRPSSimulation
+  simulation_init_args:
+    nb_of_rounds: 10
+    quota_messages_per_agent_per_round: 1
+  agents:
+    0:
+      agent_id: ${agent_0_id}
+      agent_name: Alice
+      agent_class_name: TrustAndSplitRPSAgent
+      policy_id: base_llm/agent_adapter
+      init_kwargs:
+        goal: Maximize your total points over the whole game.
+        num_message_chars: 500
+    1:
+      agent_id: ${agent_1_id}
+      agent_name: Bob
+      agent_class_name: TrustAndSplitRPSAgent
+      policy_id: base_llm/agent_adapter
+      init_kwargs:
+        goal: Maximize your total points over the whole game.
+        num_message_chars: 500
+models:
+  base_llm:
+    class: LeanLocalLLM
+    init_args:
+      llm_id: base_llm
+      model_name: Qwen/Qwen3-4B-Instruct-2507
+      inference_backend: vllm
+      hf_kwargs:
+        device_map: auto
+        torch_dtype: bfloat16
+        max_memory:
+          0: 20GiB
+        attn_implementation: flash_attention_2
+      inference_backend_init_kwargs:
+        seed: ${experiment.base_seed}
+        enable_prefix_caching: true
+        max_model_len: 10000.0
+        gpu_memory_utilization: 0.5
+        dtype: bfloat16
+        trust_remote_code: true
+        max_lora_rank: 32
+        enforce_eager: false
+        max_loras: ${lora_count}
+        max_cpu_loras: ${lora_count}
+        enable_sleep_mode: true
+        enable_lora: true
+      inference_backend_sampling_params:
+        temperature: ${temperature}
+        top_p: 1.0
+        max_tokens: 400
+        top_k: -1
+      adapter_configs:
+        agent_adapter:
+          task_type: CAUSAL_LM
+          r: 32
+          lora_alpha: 64
+          lora_dropout: 0.0
+          target_modules: all-linear
+        critic_adapter:
+          task_type: CAUSAL_LM
+          r: 32
+          lora_alpha: 64
+          lora_dropout: 0.0
+          target_modules: all-linear
+      enable_thinking: false
+      regex_max_attempts: 3
+critics:
+  agent_critic:
+    module_pointer:
+    - base_llm
+    - critic_adapter
+optimizers:
+  agent_optimizer:
+    module_pointer:
+    - base_llm
+    - agent_adapter
+    optimizer_class_name: torch.optim.Adam
+    init_args:
+      lr: 3.0e-06
+      weight_decay: 0.0
+  critic_optimizer:
+    module_pointer: agent_critic
+    optimizer_class_name: torch.optim.Adam
+    init_args:
+      lr: 3.0e-06
+      weight_decay: 0.0
+trainers:
+  agent_trainer:
+    class: TrainerAdAlign
+    module_pointers:
+      policy:
+      - base_llm
+      - agent_adapter
+      policy_optimizer: agent_optimizer
+      critic: agent_critic
+      critic_optimizer: critic_optimizer
+    kwargs:
+      entropy_coeff: 0.0
+      kl_coeff: 0.0
+      gradient_clipping: 1.0
+      restrict_tokens: null
+      mini_batch_size: 1
+      use_gradient_checkpointing: false
+      temperature: ${temperature}
+      device: cuda:0
+      use_gae: true
+      whiten_advantages: false
+      whiten_advantages_time_step_wise: false
+      skip_discounted_state_visitation: true
+      use_gae_lambda_annealing: false
+      gae_lambda_annealing_method: None
+      gae_lambda_annealing_method_params: None
+      gae_lambda_annealing_limit: 0.96
+      discount_factor: 0.98
+      use_rloo: false
+      enable_tokenwise_logging: false
+      pg_loss_normalization: batch
+      ad_align_force_coop_first_step: false
+      ad_align_clipping: null
+      ad_align_gamma: 0.98
+      ad_align_exclude_k_equals_t: false
+      ad_align_use_sign: false
+      ad_align_beta: 1.0
+      use_old_ad_align: true
+      use_time_regularization: false
+      rloo_branch: false
+      reuse_baseline: false
+      reward_normalizing_constant: 100.0
+train_on_which_data:
+  agent_trainer: ${agent_ids}
+lora_count: 10
+common_agent_kwargs:
+  goal: Maximize your total points over the whole game.
+  num_message_chars: 500
+agent_0_id: Alice
+agent_1_id: Bob
+agent_ids:
+- Alice
+- Bob

.hydra/overrides.yaml ADDED Viewed

	@@ -0,0 +1 @@


1	+ - experiment.name=tas_rps_no_regex_prev_ad_align_buffer_gae

run.log ADDED Viewed

The diff for this file is too large to render. See raw diff

src_code_for_reproducibility/__init__.py ADDED Viewed

File without changes

src_code_for_reproducibility/docs/Makefile ADDED Viewed

	@@ -0,0 +1,19 @@

+# Minimal makefile for Sphinx documentation
+# You can set these variables from the command line, and also
+# from the environment for the first two.
+SPHINXOPTS    ?=
+SPHINXBUILD   ?= sphinx-build
+SOURCEDIR     = source
+BUILDDIR      = build
+# Put it first so that "make" without argument is like "make help".
+help:
+	@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(SPHINXFLAGS)
+.PHONY: help Makefile
+# Catch-all target: route all unknown targets to Sphinx using the new
+# "make mode" option.  $(O) is meant as a shortcut for $(SPHINXOPTS).
+%: Makefile
+	@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(SPHINXFLAGS)

src_code_for_reproducibility/docs/generate_docs.py ADDED Viewed

	@@ -0,0 +1,249 @@

+#!/usr/bin/env python3
+"""
+Script to automatically generate Sphinx documentation for all modules and build the HTML website.
+"""
+import importlib.util
+import os
+import subprocess
+import sys
+def check_and_install_dependencies():
+    """Check for required dependencies and install them if missing."""
+    required_packages = [
+        "sphinx",
+        "sphinx-rtd-theme",
+        "sphinxcontrib-napoleon",
+        "sphinxcontrib-mermaid",
+        "sphinx-autodoc-typehints",
+    ]
+    missing_packages = []
+    for package in required_packages:
+        # Convert package name to module name (replace - with _)
+        module_name = package.replace("-", "_")
+        # Check if the package is installed
+        if importlib.util.find_spec(module_name) is None:
+            missing_packages.append(package)
+    # Install missing packages
+    if missing_packages:
+        print(f"Installing missing dependencies: {', '.join(missing_packages)}")
+        subprocess.check_call(
+            [sys.executable, "-m", "pip", "install"] + missing_packages
+        )
+        print("Dependencies installed successfully")
+    else:
+        print("All required dependencies are already installed")
+def create_makefile(docs_dir):
+    """Create a Makefile for Sphinx documentation if it doesn't exist."""
+    makefile_path = os.path.join(docs_dir, "Makefile")
+    if os.path.exists(makefile_path):
+        print(f"Makefile already exists at {makefile_path}")
+        return
+    print(f"Creating Makefile at {makefile_path}")
+    makefile_content = """# Minimal makefile for Sphinx documentation
+# You can set these variables from the command line, and also
+# from the environment for the first two.
+SPHINXOPTS    ?=
+SPHINXBUILD   ?= sphinx-build
+SOURCEDIR     = source
+BUILDDIR      = build
+# Put it first so that "make" without argument is like "make help".
+help:
+	@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(SPHINXFLAGS)
+.PHONY: help Makefile
+# Catch-all target: route all unknown targets to Sphinx using the new
+# "make mode" option.  $(O) is meant as a shortcut for $(SPHINXOPTS).
+%: Makefile
+	@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(SPHINXFLAGS)
+"""
+    with open(makefile_path, "w") as f:
+        f.write(makefile_content)
+    print("Makefile created successfully")
+def create_make_bat(docs_dir):
+    """Create a make.bat file for Windows if it doesn't exist."""
+    make_bat_path = os.path.join(docs_dir, "make.bat")
+    if os.path.exists(make_bat_path):
+        print(f"make.bat already exists at {make_bat_path}")
+        return
+    print(f"Creating make.bat at {make_bat_path}")
+    make_bat_content = """@ECHO OFF
+pushd %~dp0
+REM Command file for Sphinx documentation
+if "%SPHINXBUILD%" == "" (
+	set SPHINXBUILD=sphinx-build
+)
+set SOURCEDIR=source
+set BUILDDIR=build
+%SPHINXBUILD% >NUL 2>NUL
+if errorlevel 9009 (
+	echo.
+	echo.The 'sphinx-build' command was not found. Make sure you have Sphinx
+	echo.installed, then set the SPHINXBUILD environment variable to point
+	echo.to the full path of the 'sphinx-build' executable. Alternatively you
+	echo.may add the Sphinx directory to PATH.
+	echo.
+	echo.If you don't have Sphinx installed, grab it from
+	echo.https://www.sphinx-doc.org/
+	exit /b 1
+)
+if "%1" == "" goto help
+%SPHINXBUILD% -M %1 %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
+goto end
+:help
+%SPHINXBUILD% -M help %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
+:end
+popd
+"""
+    with open(make_bat_path, "w") as f:
+        f.write(make_bat_content)
+    print("make.bat created successfully")
+def main():
+    # Check and install required dependencies
+    print("=== Checking dependencies ===")
+    check_and_install_dependencies()
+    # Get the directory of this script
+    script_dir = os.path.dirname(os.path.abspath(__file__))
+    # Path to the project root
+    project_root = os.path.dirname(script_dir)
+    # Path to the source directory
+    source_dir = os.path.join(project_root, "src")
+    # Path to the docs source directory
+    docs_source_dir = os.path.join(script_dir, "source")
+    # Print paths for debugging
+    print(f"Script directory: {script_dir}")
+    print(f"Project root: {project_root}")
+    print(f"Source directory: {source_dir}")
+    print(f"Docs source directory: {docs_source_dir}")
+    # Make sure the source directory exists
+    if not os.path.exists(source_dir):
+        print(f"Error: Source directory {source_dir} does not exist!")
+        sys.exit(1)
+    # Make sure the docs source directory exists
+    if not os.path.exists(docs_source_dir):
+        print(f"Creating docs source directory: {docs_source_dir}")
+        os.makedirs(docs_source_dir)
+    # Step 1: Run sphinx-apidoc to generate .rst files for all modules
+    print("\n=== Generating API documentation ===")
+    cmd = [
+        "sphinx-apidoc",
+        "-f",  # Force overwriting of existing files
+        "-e",  # Put module documentation before submodule documentation
+        "-M",  # Put module documentation before subpackage documentation
+        "-o",
+        docs_source_dir,  # Output directory
+        source_dir,  # Source code directory
+    ]
+    print(f"Running command: {' '.join(cmd)}")
+    result = subprocess.run(cmd, capture_output=True, text=True)
+    # Print the output of the command
+    print("STDOUT:")
+    print(result.stdout)
+    print("STDERR:")
+    print(result.stderr)
+    if result.returncode != 0:
+        print(f"Error: sphinx-apidoc failed with return code {result.returncode}")
+        sys.exit(1)
+    # List the files in the docs source directory
+    print("\nFiles in docs/source directory:")
+    for file in sorted(os.listdir(docs_source_dir)):
+        print(f"  {file}")
+    print("\nDocumentation source files generated successfully!")
+    # Step 2: Create Makefile and make.bat if they don't exist
+    create_makefile(script_dir)
+    create_make_bat(script_dir)
+    # Step 3: Build the HTML documentation
+    print("\n=== Building HTML documentation ===")
+    # Determine the build command based on the platform
+    if os.name == "nt":  # Windows
+        build_cmd = ["make.bat", "html"]
+    else:  # Unix/Linux/Mac
+        build_cmd = ["make", "html"]
+    # Change to the docs directory to run the build command
+    os.chdir(script_dir)
+    print(f"Running command: {' '.join(build_cmd)}")
+    build_result = subprocess.run(build_cmd, capture_output=True, text=True)
+    # Print the output of the build command
+    print("STDOUT:")
+    print(build_result.stdout)
+    print("STDERR:")
+    print(build_result.stderr)
+    if build_result.returncode != 0:
+        print(f"Error: HTML build failed with return code {build_result.returncode}")
+        sys.exit(1)
+    # Get the path to the built HTML documentation
+    html_dir = os.path.join(script_dir, "build", "html")
+    index_path = os.path.join(html_dir, "index.html")
+    if os.path.exists(index_path):
+        print(f"\nHTML documentation built successfully!")
+        print(f"You can view it by opening: {index_path}")
+        # Try to open the documentation in a browser
+        try:
+            import webbrowser
+            print("\nAttempting to open documentation in your default browser...")
+            webbrowser.open(f"file://{index_path}")
+        except Exception as e:
+            print(f"Could not open browser automatically: {e}")
+    else:
+        print(f"\nWarning: HTML index file not found at {index_path}")
+if __name__ == "__main__":
+    main()

src_code_for_reproducibility/docs/make.bat ADDED Viewed

	@@ -0,0 +1,35 @@

+@ECHO OFF
+pushd %~dp0
+REM Command file for Sphinx documentation
+if "%SPHINXBUILD%" == "" (
+	set SPHINXBUILD=sphinx-build
+)
+set SOURCEDIR=source
+set BUILDDIR=build
+%SPHINXBUILD% >NUL 2>NUL
+if errorlevel 9009 (
+	echo.
+	echo.The 'sphinx-build' command was not found. Make sure you have Sphinx
+	echo.installed, then set the SPHINXBUILD environment variable to point
+	echo.to the full path of the 'sphinx-build' executable. Alternatively you
+	echo.may add the Sphinx directory to PATH.
+	echo.
+	echo.If you don't have Sphinx installed, grab it from
+	echo.https://www.sphinx-doc.org/
+	exit /b 1
+)
+if "%1" == "" goto help
+%SPHINXBUILD% -M %1 %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
+goto end
+:help
+%SPHINXBUILD% -M help %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O%
+:end
+popd

src_code_for_reproducibility/docs/source/index.rst ADDED Viewed

	@@ -0,0 +1,22 @@

+Welcome to LLM Negotiation's documentation!
+===========================================
+This library is a collection of tools for training and evaluating LLM-based agents in multi-agent environments. It is designed to be easy to use and extend.
+.. toctree::
+   :maxdepth: 3
+   :caption: Contents:
+   installation
+   marl_standard
+   environments
+   launch
+   usage
+   modules
+   contributing
+Indices and tables
+==================
+* :ref:`genindex`
+* :ref:`modindex`
+* :ref:`search`

src_code_for_reproducibility/docs/source/installation.rst ADDED Viewed

	@@ -0,0 +1,10 @@

+Installation
+===========
+To install the package, run:
+.. code-block:: bash
+   git clone https://github.com/yourusername/llm_negotiation.git
+   cd llm_negotiation
+   pip install -e .

src_code_for_reproducibility/docs/source/launch.rst ADDED Viewed

File without changes

src_code_for_reproducibility/docs/source/marl_standard.rst ADDED Viewed

	@@ -0,0 +1,141 @@

+=================
+Abstract Standard for Multi-Agent Negotiation Environments
+=================
+Multi-Agent Negotiation Environments require more features than gymnasium environments in order to be used as interfaces in general game running code.
+The two fundamental differences between gymnasium environments and Multi-Agent Negotiation Environments are:
+1. Response from the LLM is a text action, not a discrete action. Therefore, appropriate parsing of the text is required. The model may need to be run multiple times to get the full action.
+    This is why we introduce the `AgentHandler` class, which is responsible for parsing the LLM's response.
+2. The environment needs to be able to handle multi-agent interactions.
+    This is why we introduce the `NegotiationEnvironment` class, which is responsible for handling the multi-agent interactions.
+3. MARL environments are complex to describe. In different contexts, the same environment may be described differently. Therefore, both the environement and the agent handlers are
+    responsible for describing a particular trajectory. This information is given by the `get_log_info` method.
+4. There might be a lot of overlap between the neural networks used by each agent. For instance, the same model may be used for all agents. This motivates a requirement for a
+    policy identifier for each agent.
+Taking inspiration from the `gymnasium <https://gymnasium.farama.org/>`_ library, we introduce a new standard for Multi-Agent Negotiation Environments.
+Our standard is based on the following features:
+Environments are of the form:
+.. code-block:: python
+    class MarlEnvironment():
+        def __init__(self):
+            """Initialize the environment."""
+            pass
+        def reset(self):
+            """Reset the environment to an initial state and return the initial observation.
+            Returns:
+                observation (dict): A dictionary where keys are agent identifiers and values are observations.
+            """
+            # (...)
+            return observation
+        def step(self, actions):
+            """Take a step in the environment using the provided actions.
+            Args:
+                actions (dict): A dictionary where keys are agent identifiers and values are actions.
+            Returns:
+                observations (dict): A dictionary where keys are agent identifiers and values are observations.
+                reward (dict): A dictionary where keys are agent identifiers and values are rewards.
+                done (bool): Whether the episode has ended.
+                info (dict): Additional information about the environment.
+            """
+            # (...)
+            return observations, done, info
+        def get_log_info(self):
+            """Get additional information about the environment. This information is used to log the game.
+            Returns:
+                log_info (dict): Information about the environment required to log the game.
+            """
+            # (...)
+            return log_info
+        def render(self):
+            """Render the current state of the environment."""
+            pass
+        def close(self):
+            """Perform any necessary cleanup."""
+            pass
+    class AgentState():
+        def __init__(self):
+            """Initialize the agent state."""
+            pass
+        def step(self, observation_from_env, policy_output=None):
+            """Update the agent state based on the observation and action.
+            The action is the output of the LLM.
+            """
+            Args:
+                observation_from_env (dict): The observation of the environment.
+                policy_output : The output of the policy.
+            Returns:
+                policy_id (str): The policy identifier.
+                policy_input (dict): The input to the policy.
+                action : The official action to be sent to the environment.
+                done (bool): Whether the LLM action is ready to be sent to the environment.
+                info (dict): Additional information about the agent.
+            """
+            # (...)
+            return policy_id, policy_input, action, done, info
+        def get_log_info(self):
+            """Get information about the agent required to log a trajectory.
+            Returns:
+                log_info (dict): Information about the agent required to log a trajectory.
+            """
+            # (...)
+            return log_info
+        def render(self):
+            """Render the current state of the environment."""
+            pass
+        def close(self):
+            """Perform any necessary cleanup."""
+            pass
+Implicitely, the keys of the `observations` in the `step` method of the `MarlEnvironment` interface represent the set of agents from which an action is expected at the current step. The next step should only expect actions from the agents in the `observations` dictionary.
+As you can see, both classes have a `get_log_info` method. This method is used to log the game. It returns a dictionary with keys being the agent identifiers and values being the information to log. The reason we need this is because the environment and the agent handler may need to log different information. It makes it easier to log from the perspective of each agent. The core environment class should not need to know about the details of the agent handler.
+Running Environments in Parallel
+--------------------------------
+This standard allows the use of the `run_batched_matches` function (TODO: link) to run environments in an efficient way. The core idea is to batch the policy calls for all agents in the environment.
+.. note::
+   The ``run_batched_matches`` function allows you to run multiple negotiation games, or "matches," in parallel.
+   After each environment is initialized, the function continuously loops over all active matches and checks which agents
+   are still pending actions. Each agent's logic can require multiple calls to the policy (e.g., an LLM) before an action
+   becomes "ready" to be sent to the environment. (For instance, an agent might need multiple policy calls before having a string which can be parsed into a valid action.) While an agent is waiting for a policy output, these calls for all agents across all matches are grouped together by unique policy identifier and processed in batch for efficiency. This is the core functionality of the ``run_batched_matches`` function.
+   Only once all actions from the required agents at a given step for an environment are ready does the function make a single ``env.step(...)`` call; this ensures
+   every match moves forward in lockstep for all its active agents. As soon as an environment signals it is done, the function
+   retrieves logged information from both the environment and the agent states before removing this match from the active set.
+   If there are more matches waiting to be processed, they are then started one by one to maintain the specified degree of parallelism.
+   This batching approach provides an efficient mechanism to handle multi-agent or multi-policy environments, ensuring minimal
+   overhead and a clear, unified flow for stepping through matches.
+Here is a diagram that shows how the `run_batched_matches` function works at a high level:
+.. image:: media/runbatch.png
+   :alt: Alternate text for the image
+   :width: 1000px

src_code_for_reproducibility/docs/source/src.environments.dond.dond_return_funcs.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.dond.dond\_return\_funcs module
+================================================
+.. automodule:: src.environments.dond.dond_return_funcs
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.dond.dond_training_data_funcs.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.environments.dond.dond\_training\_data\_funcs module
+========================================================
+.. automodule:: src.environments.dond.dond_training_data_funcs
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.environments.dond.rst ADDED Viewed

	@@ -0,0 +1,19 @@

+src.environments.dond package
+=============================
+.. automodule:: src.environments.dond
+   :members:
+   :undoc-members:
+   :show-inheritance:
+Submodules
+----------
+.. toctree::
+   :maxdepth: 4
+   src.environments.dond.dond_agent
+   src.environments.dond.dond_game
+   src.environments.dond.dond_log_funcs
+   src.environments.dond.dond_statistics_funcs
+   src.environments.dond.dond_training_data_funcs

src_code_for_reproducibility/docs/source/src.experiments.arithmetic_test.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.experiments.arithmetic\_test module
+=======================================
+.. automodule:: src.experiments.arithmetic_test
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.experiments.generate_and_train.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.experiments.generate\_and\_train module
+===========================================
+.. automodule:: src.experiments.generate_and_train
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.experiments.last_completion.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.experiments.last\_completion module
+=======================================
+.. automodule:: src.experiments.last_completion
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.generation.rst ADDED Viewed

	@@ -0,0 +1,15 @@

+src.generation package
+======================
+.. automodule:: src.generation
+   :members:
+   :undoc-members:
+   :show-inheritance:
+Submodules
+----------
+.. toctree::
+   :maxdepth: 4
+   src.generation.run_games

src_code_for_reproducibility/docs/source/src.models.local_llm.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.models.local\_llm module
+============================
+.. automodule:: src.models.local_llm
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.models.new_local_llm.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.models.new\_local\_llm module
+=================================
+.. automodule:: src.models.new_local_llm
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.models.oai_agent.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.models.oai\_agent module
+============================
+.. automodule:: src.models.oai_agent
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/docs/source/src.rst ADDED Viewed

	@@ -0,0 +1,28 @@

+src package
+===========
+.. automodule:: src
+   :members:
+   :undoc-members:
+   :show-inheritance:
+Subpackages
+-----------
+.. toctree::
+   :maxdepth: 4
+   src.environments
+   src.experiments
+   src.generation
+   src.models
+   src.training
+   src.utils
+Submodules
+----------
+.. toctree::
+   :maxdepth: 4
+   src.run

src_code_for_reproducibility/docs/source/src.training.rst ADDED Viewed

	@@ -0,0 +1,19 @@

+src.training package
+====================
+.. automodule:: src.training
+   :members:
+   :undoc-members:
+   :show-inheritance:
+Submodules
+----------
+.. toctree::
+   :maxdepth: 4
+   src.training.ppo_train
+   src.training.ppo_train_value_head
+   src.training.reinforce_training
+   src.training.rl_convs_processing
+   src.training.train_main

src_code_for_reproducibility/docs/source/src.training.train_main.rst ADDED Viewed

	@@ -0,0 +1,7 @@

+src.training.train\_main module
+===============================
+.. automodule:: src.training.train_main
+   :members:
+   :undoc-members:
+   :show-inheritance:

src_code_for_reproducibility/markov_games/__pycache__/export_utils.cpython-310.pyc ADDED Viewed

Binary file (7.17 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/__init__.cpython-310.pyc ADDED Viewed

Binary file (168 Bytes). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_agent.cpython-310.pyc ADDED Viewed

Binary file (3.37 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_agent.cpython-311.pyc ADDED Viewed

Binary file (5.31 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_game.cpython-310.pyc ADDED Viewed

Binary file (5.39 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_log_funcs.cpython-310.pyc ADDED Viewed

Binary file (1.36 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_log_match.cpython-310.pyc ADDED Viewed

Binary file (1.75 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_player.cpython-310.pyc ADDED Viewed

Binary file (8.22 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_simulation.cpython-310.pyc ADDED Viewed

Binary file (6.16 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_simulation.cpython-311.pyc ADDED Viewed

Binary file (7.06 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_statistics.cpython-310.pyc ADDED Viewed

Binary file (11.7 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_statistics.cpython-311.pyc ADDED Viewed

Binary file (915 Bytes). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_statistics_funcs.cpython-310.pyc ADDED Viewed

Binary file (1.63 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_training_data.cpython-310.pyc ADDED Viewed

Binary file (5.59 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_training_data_funcs.cpython-310.pyc ADDED Viewed

Binary file (6.11 kB). View file

src_code_for_reproducibility/markov_games/ipd/ipd_agent.py ADDED Viewed

	@@ -0,0 +1,122 @@

+import copy
+import json
+import random
+import re
+from collections.abc import Callable
+from copy import deepcopy
+from dataclasses import dataclass, field
+from typing import Any, Dict, List, Optional, Tuple, Union
+from mllm.markov_games.agent import Agent
+from mllm.markov_games.rollout_tree import AgentActLog, ChatTurn
+@dataclass
+class IPDAgentState:
+    """
+    TOWRITE
+    """
+    nb_retries: int
+    round_nb: int
+    chat_counter: int
+    chat_history: List[ChatTurn]
+@dataclass
+class IPDAgent(Agent):
+    seed: int
+    agent_id: str
+    agent_name: str
+    policy: Callable[[List[Dict]], str]
+    intro_prompt: str  # Introduction prompt explaining the game rules
+    goal_prompt: str  # Prompt explaining the agent's goal
+    strategy_prompt: str  # Prompt suggesting a strategy to the agent
+    max_errors: int  # Maximum number of errors allowed before default action
+    allow_reasoning: bool  # Whether to allow reasoning in the response
+    max_reasoning_chars: int  # Maximum number of characters for reasoning
+    cooperate_string: str  # string parsed as playing cooperate by simulation
+    defect_string: str  # string parsed as playing defect by simulation
+    def __post_init__(self):
+        self.state = IPDAgentState(
+            nb_retries=0, round_nb=0, chat_counter=0, chat_history=[]
+        )
+    async def act(self, observation) -> Tuple[Any, AgentActLog]:
+        """
+        TOWRITE
+        """
+        action = None
+        action_is_ready = False
+        round_nb = observation.round_nb
+        # If it's the first round, we need to send the intro prompt
+        if round_nb == 0 and self.state.chat_counter == 0:
+            self.state.chat_history.append(
+                ChatTurn(
+                    agent_id=self.agent_id,
+                    role="user",
+                    content=self.intro_prompt,
+                    is_state_end=True,
+                )
+            )
+        # If new round
+        if round_nb > self.state.round_nb:
+            coagent_action = observation.last_coagent_move
+            user_message = f"Last round, the other agent played {coagent_action}."
+            self.state.chat_history.append(
+                ChatTurn(
+                    agent_id=self.agent_id,
+                    role="user",
+                    content=user_message,
+                    is_state_end=True,
+                )
+            )
+        # If not new round, try to get valid action from policy
+        prompt = [chat_item.dict() for chat_item in self.state.chat_history]
+        policy_output = await self.policy(
+            prompt=prompt, regex=f"({self.cooperate_string}|{self.defect_string})"
+        )
+        self.state.chat_history.append(
+            ChatTurn(
+                agent_id=self.agent_id,
+                role="assistant",
+                content=policy_output,
+                is_state_end=False,
+            )
+        )
+        action = policy_output
+        agent_step_log = AgentActLog(
+            chat_turns=self.state.chat_history[self.state.chat_counter :], info=None
+        )
+        self.state.chat_counter = len(self.state.chat_history)
+        self.state.round_nb = round_nb
+        return action, agent_step_log
+    def get_safe_copy(self):
+        """
+        Return a safe copy of the agent.
+        """
+        agent_copy = copy.copy(self)
+        agent_copy.state = copy.deepcopy(self.state)
+        return agent_copy
+    def reset(self):
+        self.state = IPDAgentState()
+        raise NotImplementedError
+    def render(self):
+        pass
+    def close(self):
+        pass
+    def get_agent_info(self):
+        pass

src_code_for_reproducibility/markov_games/ipd/ipd_simulation.py ADDED Viewed

	@@ -0,0 +1,162 @@

+import copy
+import random
+from dataclasses import dataclass
+from typing import Any, Dict, List, Optional, Tuple
+import numpy as np
+from mllm.markov_games.markov_game import Simulation
+from mllm.markov_games.rollout_tree import SimulationStepLog
+from mllm.utils.get_coagent_id import get_coagent_id
+@dataclass
+class IPDState:
+    """
+    State of the Iterated Prisoner's Dilemma game.
+    """
+    round_nb: int = 0
+    done: bool = False
+    last_moves: Dict[str, str] | None = None
+@dataclass
+class IPDObs:
+    """
+    Observation in Iterated Prisoner's Dilemma game.
+    """
+    round_nb: int
+    last_coagent_move: str | None
+class IPD(Simulation):
+    """
+    Iterated Prisoner's Dilemma simulation following the standard.
+    In each round of the game, two agents simultaneously choose to either cooperate (C) or defect (D).
+    The payoffs are as follows:
+    - If both cooperate: Both receive the "reward" (usually 3 points)
+    - If both defect: Both receive the "punishment" (usually 1 point)
+    - If one cooperates and one defects: The defector receives the "temptation" (usually 5 points)
+      and the cooperator receives the "sucker" payoff (usually 0 points)
+    The game is played for a specified number of rounds.
+    """
+    def __init__(
+        self,
+        agent_ids: List[str],
+        agent_names: List[str],
+        seed: int,
+        rounds_per_game: int,
+        reward: float,  # Both cooperate
+        punishment: float,  # Both defect
+        temptation: float,  # Defector's reward when other cooperates
+        sucker: float,  # Cooperator's reward when other defects
+        cooperate_actions: List[str],
+        defect_actions: List[str],
+    ):
+        self.agent_ids = agent_ids
+        self.agent_names = agent_names
+        self.seed = seed
+        self.rounds_per_game = rounds_per_game
+        self.reward = reward
+        self.punishment = punishment
+        self.temptation = temptation
+        self.sucker = sucker
+        self.cooperate_actions = cooperate_actions
+        self.defect_actions = defect_actions
+        self.state = IPDState()
+    def step(self, actions: Dict[str, str]) -> Tuple[bool, SimulationStepLog]:
+        """
+        Take a step in the environment using the provided actions.
+        Here, the observations are just the states of the game.
+        Args:
+            actions (dict): A dictionary where keys are agent identifiers and values are actions ('C' or 'D').
+        Returns:
+            observations (dict): A dictionary where keys are agent identifiers and values are observations.
+            done (bool): Whether the episode has ended.
+            info (dict): Additional information about the environment.
+        """
+        # Calculate rewards using payoff matrix
+        agent0_action = actions[self.agent_ids[0]]
+        agent1_action = actions[self.agent_ids[1]]
+        # Normalize actions to standard cooperate/defect/gibberish format
+        def normalize_action(action):
+            if action in self.cooperate_actions:
+                return "C"
+            elif action in self.defect_actions:
+                return "D"
+            else:
+                return "D"
+        norm_action0 = normalize_action(agent0_action)
+        norm_action1 = normalize_action(agent1_action)
+        payoffs = {
+            ("C", "C"): [self.reward, self.reward],
+            ("C", "D"): [self.sucker, self.temptation],
+            ("D", "C"): [self.temptation, self.sucker],
+            ("D", "D"): [self.punishment, self.punishment],
+        }
+        round_rewards = {
+            self.agent_ids[0]: payoffs[(norm_action0, norm_action1)][0],
+            self.agent_ids[1]: payoffs[(norm_action0, norm_action1)][1],
+        }
+        # Update game state
+        self.state.round_nb += 1
+        self.state.last_moves = copy.deepcopy(actions)
+        done = self.state.round_nb >= self.rounds_per_game
+        step_log = SimulationStepLog(
+            rewards=round_rewards,
+            info={
+                "actions": {
+                    self.agent_ids[0]: norm_action0,
+                    self.agent_ids[1]: norm_action1,
+                }
+            },
+        )
+        return done, step_log
+    def get_obs(self):
+        """Returns all agent observations in dict
+        Returns:
+            observations
+        """
+        observations = {}
+        for agent_id in self.agent_ids:
+            observations[agent_id] = self.get_obs_agent(agent_id)
+        return observations
+    def get_obs_agent(self, agent_id):
+        """Returns observation for agent_id"""
+        if self.state.last_moves != None:
+            other_id = get_coagent_id(self.agent_ids, agent_id)
+            last_coagent_move = self.state.last_moves[other_id]
+        else:
+            last_coagent_move = None
+        obs = IPDObs(round_nb=self.state.round_nb, last_coagent_move=last_coagent_move)
+        return obs
+    def reset(self):
+        """Returns initial observations and states"""
+        self.state = IPDState()
+        return self.get_obs()
+    def get_safe_copy(self):
+        """
+        Return a safe copy of the simulation.
+        """
+        simulation_copy = copy.copy(self)
+        simulation_copy.state = copy.deepcopy(self.state)
+        return simulation_copy

src_code_for_reproducibility/markov_games/runners/__pycache__/alternative_actions_runner.cpython-311.pyc ADDED Viewed

Binary file (5.8 kB). View file

src_code_for_reproducibility/markov_games/runners/__pycache__/linear_runner.cpython-311.pyc ADDED Viewed

Binary file (2.08 kB). View file

src_code_for_reproducibility/utils/__init__.py ADDED Viewed

File without changes

src_code_for_reproducibility/utils/__pycache__/__init__.cpython-310.pyc ADDED Viewed

Binary file (157 Bytes). View file

src_code_for_reproducibility/utils/__pycache__/__init__.cpython-311.pyc ADDED Viewed

Binary file (173 Bytes). View file

src_code_for_reproducibility/utils/__pycache__/common_imports.cpython-310.pyc ADDED Viewed

Binary file (524 Bytes). View file

src_code_for_reproducibility/utils/__pycache__/dict_get_path.cpython-310.pyc ADDED Viewed

Binary file (418 Bytes). View file

src_code_for_reproducibility/utils/__pycache__/extra_stats.cpython-310.pyc ADDED Viewed

Binary file (190 Bytes). View file

src_code_for_reproducibility/utils/__pycache__/get_coagent_id.cpython-310.pyc ADDED Viewed

Binary file (364 Bytes). View file