Muqeeth commited on Mar 31

Commit

6a6ad9b

verified ·

1 Parent(s): d2db893

Add files using upload-large-folder tool

Browse files

Files changed (50) hide show

.hydra/config.yaml +178 -0
.hydra/hydra.yaml +154 -0
.hydra/overrides.yaml +1 -0
seed_42/Qwen/Qwen2.5-7B-Instruct/adapters/README.md +207 -0
seed_42/Qwen/Qwen2.5-7B-Instruct/adapters/agent_adapter/adapter_config.json +46 -0
seed_42/Qwen/Qwen2.5-7B-Instruct/adapters/critic_adapter/adapter_config.json +46 -0
src_code_for_reproducibility/__init__.py +4 -0
src_code_for_reproducibility/chat_utils/__pycache__/apply_template.cpython-312.pyc +0 -0
src_code_for_reproducibility/chat_utils/__pycache__/chat_turn.cpython-312.pyc +0 -0
src_code_for_reproducibility/chat_utils/__pycache__/template_specific.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/__pycache__/__init__.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/__pycache__/agent.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/__pycache__/alternative_actions_runner.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/__pycache__/group_timesteps.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/__pycache__/markov_game.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/__pycache__/mg_utils.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/__pycache__/rollout_tree.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/__pycache__/run_markov_games.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/__pycache__/simulation.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/Ipd_hard_coded_agents.py +76 -0
src_code_for_reproducibility/markov_games/ipd/__init__.py +11 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/Ipd_hard_coded_agents.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/__init__.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_agent.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_simulation.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_statistics.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/ipd/ipd_agent.py +120 -0
src_code_for_reproducibility/markov_games/ipd/ipd_simulation.py +167 -0
src_code_for_reproducibility/markov_games/ipd/ipd_statistics.py +24 -0
src_code_for_reproducibility/markov_games/negotiation/README.md +27 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/dond_agent.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/dond_simulation.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/nego_agent.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/nego_hard_coded_policies.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/nego_simulation.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/negotiation_statistics.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/no_press_nego_agent.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/no_press_nego_simulation.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/tas_rps_agent.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/tas_rps_simulation.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/nego_hard_coded_policies.py +70 -0
src_code_for_reproducibility/markov_games/negotiation/tas_agent.py +118 -0
src_code_for_reproducibility/training/trainer_common.py +1032 -0
src_code_for_reproducibility/utils/get_coagent_id.py +10 -0
src_code_for_reproducibility/utils/get_stochastic_game_lengths.py +33 -0
src_code_for_reproducibility/utils/resource_context.py +83 -0
src_code_for_reproducibility/utils/rollout_tree_chat_htmls.py +1597 -0
src_code_for_reproducibility/utils/stat_pack.py +117 -0
src_code_for_reproducibility/utils/update_start_epoch.py +17 -0
src_code_for_reproducibility/utils/wandb_utils.py +170 -0

.hydra/config.yaml ADDED Viewed

	@@ -0,0 +1,178 @@

+experiment:
+  wandb_enabled: true
+  nb_epochs: 3000
+  nb_matches_per_iteration: 64
+  reinit_matches_each_it: true
+  checkpoint_every_n_iterations: 50
+  start_epoch: 0
+  resume_experiment: true
+  base_seed: 42
+  seed_group_size: 8
+  train: true
+  stat_methods_for_live_wandb: mllm.markov_games.negotiation.negotiation_statistics
+  name: split_no_comm_vanilla_ad_align_no_agent_buffer_seed42
+  agent_buffer: false
+  keep_agent_buffer_count: ${lora_count}
+  agent_buffer_recent_k: -1
+logging:
+  wandb:
+    enabled: false
+    project: llm-negotiation
+    entity: null
+    mode: online
+    name: null
+    group: null
+    tags: []
+    notes: null
+temperature: 1.0
+markov_games:
+  runner_method_name: LinearRunner
+  runner_kwargs: {}
+  group_by_round: true
+  simulation_class_name: NoPressSimulation
+  simulation_init_args:
+    nb_of_rounds: 10
+    quota_messages_per_agent_per_round: 0
+    game_type: 10-1-ties
+    atleast_one_conflict: true
+    item_types:
+    - hats
+    - books
+    - balls
+  agents:
+    0:
+      agent_id: ${agent_0_id}
+      agent_name: Alice
+      agent_class_name: NoPressAgent
+      policy_id: base_llm/agent_adapter
+      init_kwargs:
+        goal: Maximize your total points over the whole game.
+    1:
+      agent_id: ${agent_1_id}
+      agent_name: Bob
+      agent_class_name: NoPressAgent
+      policy_id: base_llm/agent_adapter
+      init_kwargs:
+        goal: Maximize your total points over the whole game.
+models:
+  base_llm:
+    class: LeanLocalLLM
+    init_args:
+      llm_id: base_llm
+      model_name: Qwen/Qwen2.5-7B-Instruct
+      inference_backend: vllm
+      hf_kwargs:
+        device_map: auto
+        torch_dtype: bfloat16
+        max_memory:
+          0: 20GiB
+        attn_implementation: flash_attention_2
+      inference_backend_init_kwargs:
+        enable_lora: true
+        seed: ${experiment.base_seed}
+        enable_prefix_caching: true
+        max_model_len: 10000.0
+        gpu_memory_utilization: 0.5
+        dtype: bfloat16
+        trust_remote_code: true
+        max_lora_rank: 32
+        enforce_eager: false
+        max_loras: ${lora_count}
+        max_cpu_loras: ${lora_count}
+        enable_sleep_mode: true
+      inference_backend_sampling_params:
+        temperature: ${temperature}
+        top_p: 1.0
+        max_tokens: 400
+        top_k: -1
+        logprobs: 0
+      adapter_configs:
+        agent_adapter:
+          task_type: CAUSAL_LM
+          r: 32
+          lora_alpha: 64
+          lora_dropout: 0.0
+          target_modules: all-linear
+        critic_adapter:
+          task_type: CAUSAL_LM
+          r: 32
+          lora_alpha: 64
+          lora_dropout: 0.0
+          target_modules: all-linear
+      enable_thinking: null
+      regex_max_attempts: 3
+critics:
+  agent_critic:
+    module_pointer:
+    - base_llm
+    - critic_adapter
+optimizers:
+  agent_optimizer:
+    module_pointer:
+    - base_llm
+    - agent_adapter
+    optimizer_class_name: torch.optim.Adam
+    init_args:
+      lr: 3.0e-06
+      weight_decay: 0.0
+  critic_optimizer:
+    module_pointer: agent_critic
+    optimizer_class_name: torch.optim.Adam
+    init_args:
+      lr: 3.0e-06
+      weight_decay: 0.0
+trainers:
+  agent_trainer:
+    class: TrainerAdAlign
+    module_pointers:
+      policy:
+      - base_llm
+      - agent_adapter
+      policy_optimizer: agent_optimizer
+      critic: agent_critic
+      critic_optimizer: critic_optimizer
+    kwargs:
+      entropy_coeff: 0.0
+      entropy_topk: null
+      entropy_mask_regex: null
+      kl_coeff: 0.001
+      gradient_clipping: 1.0
+      restrict_tokens: null
+      mini_batch_size: 1
+      use_gradient_checkpointing: false
+      temperature: ${temperature}
+      device: cuda:0
+      use_gae: false
+      whiten_advantages: false
+      whiten_advantages_time_step_wise: false
+      skip_discounted_state_visitation: true
+      use_gae_lambda_annealing: false
+      gae_lambda_annealing_method: None
+      gae_lambda_annealing_method_params: None
+      gae_lambda_annealing_limit: 0.95
+      discount_factor: 0.9
+      use_rloo: true
+      enable_tokenwise_logging: false
+      pg_loss_normalization: nb_tokens
+      truncated_importance_sampling_ratio_cap: 2.0
+      reward_normalizing_constant: 100.0
+      ad_align_force_coop_first_step: false
+      ad_align_clipping: null
+      ad_align_gamma: 0.9
+      ad_align_exclude_k_equals_t: true
+      ad_align_use_sign: false
+      ad_align_beta: 1.0
+      use_old_ad_align: true
+      use_time_regularization: false
+      rloo_branch: false
+      reuse_baseline: false
+train_on_which_data:
+  agent_trainer: ${agent_ids}
+lora_count: 30
+common_agent_kwargs:
+  goal: Maximize your total points over the whole game.
+agent_0_id: Alice
+agent_1_id: Bob
+agent_ids:
+- Alice
+- Bob

.hydra/hydra.yaml ADDED Viewed

	@@ -0,0 +1,154 @@

+hydra:
+  run:
+    dir: ${oc.env:SCRATCH}/llm_negotiation/${now:%Y_%m}/${experiment.name}
+  sweep:
+    dir: multirun/${now:%Y-%m-%d}/${now:%H-%M-%S}
+    subdir: ${hydra.job.num}
+  launcher:
+    _target_: hydra._internal.core_plugins.basic_launcher.BasicLauncher
+  sweeper:
+    _target_: hydra._internal.core_plugins.basic_sweeper.BasicSweeper
+    max_batch_size: null
+    params: null
+  help:
+    app_name: ${hydra.job.name}
+    header: '${hydra.help.app_name} is powered by Hydra.
+      '
+    footer: 'Powered by Hydra (https://hydra.cc)
+      Use --hydra-help to view Hydra specific help
+      '
+    template: '${hydra.help.header}
+      == Configuration groups ==
+      Compose your configuration from those groups (group=option)
+      $APP_CONFIG_GROUPS
+      == Config ==
+      Override anything in the config (foo.bar=value)
+      $CONFIG
+      ${hydra.help.footer}
+      '
+  hydra_help:
+    template: 'Hydra (${hydra.runtime.version})
+      See https://hydra.cc for more info.
+      == Flags ==
+      $FLAGS_HELP
+      == Configuration groups ==
+      Compose your configuration from those groups (For example, append hydra/job_logging=disabled
+      to command line)
+      $HYDRA_CONFIG_GROUPS
+      Use ''--cfg hydra'' to Show the Hydra config.
+      '
+    hydra_help: ???
+  hydra_logging:
+    version: 1
+    formatters:
+      simple:
+        format: '[%(asctime)s][HYDRA] %(message)s'
+    handlers:
+      console:
+        class: logging.StreamHandler
+        formatter: simple
+        stream: ext://sys.stdout
+    root:
+      level: INFO
+      handlers:
+      - console
+    loggers:
+      logging_example:
+        level: DEBUG
+    disable_existing_loggers: false
+  job_logging:
+    version: 1
+    formatters:
+      simple:
+        format: '[%(asctime)s][%(name)s][%(levelname)s] - %(message)s'
+    handlers:
+      console:
+        class: logging.StreamHandler
+        formatter: simple
+        stream: ext://sys.stdout
+      file:
+        class: logging.FileHandler
+        formatter: simple
+        filename: ${hydra.runtime.output_dir}/${hydra.job.name}.log
+    root:
+      level: INFO
+      handlers:
+      - console
+      - file
+    disable_existing_loggers: false
+  env: {}
+  mode: RUN
+  searchpath: []
+  callbacks: {}
+  output_subdir: .hydra
+  overrides:
+    hydra:
+    - hydra.mode=RUN
+    task: []
+  job:
+    name: run
+    chdir: false
+    override_dirname: ''
+    id: ???
+    num: ???
+    config_name: split_no_comm_vanilla_ad_align_no_agent_buffer_seed42.yaml
+    env_set: {}
+    env_copy: []
+    config:
+      override_dirname:
+        kv_sep: '='
+        item_sep: ','
+        exclude_keys: []
+  runtime:
+    version: 1.3.2
+    version_base: '1.1'
+    cwd: /lustre10/scratch/muqeeth/AdAlignLLM
+    config_sources:
+    - path: hydra.conf
+      schema: pkg
+      provider: hydra
+    - path: /lustre10/scratch/muqeeth/AdAlignLLM/configs
+      schema: file
+      provider: main
+    - path: ''
+      schema: structured
+      provider: schema
+    output_dir: /scratch/muqeeth/llm_negotiation/2026_03/split_no_comm_vanilla_ad_align_no_agent_buffer_seed42
+    choices:
+      hydra/env: default
+      hydra/callbacks: null
+      hydra/job_logging: default
+      hydra/hydra_logging: default
+      hydra/hydra_help: default
+      hydra/help: default
+      hydra/sweeper: basic
+      hydra/launcher: basic
+      hydra/output: default
+  verbose: false

.hydra/overrides.yaml ADDED Viewed

	@@ -0,0 +1 @@


1	+ []

seed_42/Qwen/Qwen2.5-7B-Instruct/adapters/README.md ADDED Viewed

	@@ -0,0 +1,207 @@

+---
+base_model: Qwen/Qwen2.5-7B-Instruct
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-7B-Instruct
+- lora
+- transformers
+---
+# Model Card for Model ID
+<!-- Provide a quick summary of what the model is/does. -->
+## Model Details
+### Model Description
+<!-- Provide a longer summary of what this model is. -->
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+### Model Sources [optional]
+<!-- Provide the basic links for the model. -->
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+## Uses
+<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
+### Direct Use
+<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
+[More Information Needed]
+### Downstream Use [optional]
+<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
+[More Information Needed]
+### Out-of-Scope Use
+<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
+[More Information Needed]
+## Bias, Risks, and Limitations
+<!-- This section is meant to convey both technical and sociotechnical limitations. -->
+[More Information Needed]
+### Recommendations
+<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+## How to Get Started with the Model
+Use the code below to get started with the model.
+[More Information Needed]
+## Training Details
+### Training Data
+<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
+[More Information Needed]
+### Training Procedure
+<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
+#### Preprocessing [optional]
+[More Information Needed]
+#### Training Hyperparameters
+- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
+#### Speeds, Sizes, Times [optional]
+<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
+[More Information Needed]
+## Evaluation
+<!-- This section describes the evaluation protocols and provides the results. -->
+### Testing Data, Factors & Metrics
+#### Testing Data
+<!-- This should link to a Dataset Card if possible. -->
+[More Information Needed]
+#### Factors
+<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
+[More Information Needed]
+#### Metrics
+<!-- These are the evaluation metrics being used, ideally with a description of why. -->
+[More Information Needed]
+### Results
+[More Information Needed]
+#### Summary
+## Model Examination [optional]
+<!-- Relevant interpretability work for the model goes here -->
+[More Information Needed]
+## Environmental Impact
+<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+## Technical Specifications [optional]
+### Model Architecture and Objective
+[More Information Needed]
+### Compute Infrastructure
+[More Information Needed]
+#### Hardware
+[More Information Needed]
+#### Software
+[More Information Needed]
+## Citation [optional]
+<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
+**BibTeX:**
+[More Information Needed]
+**APA:**
+[More Information Needed]
+## Glossary [optional]
+<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
+[More Information Needed]
+## More Information [optional]
+[More Information Needed]
+## Model Card Authors [optional]
+[More Information Needed]
+## Model Card Contact
+[More Information Needed]
+### Framework versions
+- PEFT 0.18.1

seed_42/Qwen/Qwen2.5-7B-Instruct/adapters/agent_adapter/adapter_config.json ADDED Viewed

	@@ -0,0 +1,46 @@

+{
+  "alora_invocation_tokens": null,
+  "alpha_pattern": {},
+  "arrow_config": null,
+  "auto_mapping": null,
+  "base_model_name_or_path": "Qwen/Qwen2.5-7B-Instruct",
+  "bias": "none",
+  "corda_config": null,
+  "ensure_weight_tying": false,
+  "eva_config": null,
+  "exclude_modules": null,
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": true,
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 64,
+  "lora_bias": false,
+  "lora_dropout": 0.0,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "peft_version": "0.18.1",
+  "qalora_group_size": 16,
+  "r": 32,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": [
+    "up_proj",
+    "v_proj",
+    "o_proj",
+    "down_proj",
+    "q_proj",
+    "gate_proj",
+    "k_proj"
+  ],
+  "target_parameters": null,
+  "task_type": "CAUSAL_LM",
+  "trainable_token_indices": null,
+  "use_dora": false,
+  "use_qalora": false,
+  "use_rslora": false
+}

seed_42/Qwen/Qwen2.5-7B-Instruct/adapters/critic_adapter/adapter_config.json ADDED Viewed

	@@ -0,0 +1,46 @@

+{
+  "alora_invocation_tokens": null,
+  "alpha_pattern": {},
+  "arrow_config": null,
+  "auto_mapping": null,
+  "base_model_name_or_path": "Qwen/Qwen2.5-7B-Instruct",
+  "bias": "none",
+  "corda_config": null,
+  "ensure_weight_tying": false,
+  "eva_config": null,
+  "exclude_modules": null,
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": true,
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 64,
+  "lora_bias": false,
+  "lora_dropout": 0.0,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "peft_version": "0.18.1",
+  "qalora_group_size": 16,
+  "r": 32,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": [
+    "up_proj",
+    "v_proj",
+    "o_proj",
+    "down_proj",
+    "q_proj",
+    "gate_proj",
+    "k_proj"
+  ],
+  "target_parameters": null,
+  "task_type": "CAUSAL_LM",
+  "trainable_token_indices": null,
+  "use_dora": false,
+  "use_qalora": false,
+  "use_rslora": false
+}

src_code_for_reproducibility/__init__.py ADDED Viewed

	@@ -0,0 +1,4 @@

+"""
+File: mllm/__init__.py
+Summary: Initializes the multi-agent large language model package namespace.
+"""

src_code_for_reproducibility/chat_utils/__pycache__/apply_template.cpython-312.pyc ADDED Viewed

Binary file (4.13 kB). View file

src_code_for_reproducibility/chat_utils/__pycache__/chat_turn.cpython-312.pyc ADDED Viewed

Binary file (1.46 kB). View file

src_code_for_reproducibility/chat_utils/__pycache__/template_specific.cpython-312.pyc ADDED Viewed

Binary file (4.4 kB). View file

src_code_for_reproducibility/markov_games/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (297 Bytes). View file

src_code_for_reproducibility/markov_games/__pycache__/agent.cpython-312.pyc ADDED Viewed

Binary file (3.16 kB). View file

src_code_for_reproducibility/markov_games/__pycache__/alternative_actions_runner.cpython-312.pyc ADDED Viewed

Binary file (5.43 kB). View file

src_code_for_reproducibility/markov_games/__pycache__/group_timesteps.cpython-312.pyc ADDED Viewed

Binary file (6.23 kB). View file

src_code_for_reproducibility/markov_games/__pycache__/markov_game.cpython-312.pyc ADDED Viewed

Binary file (10.2 kB). View file

src_code_for_reproducibility/markov_games/__pycache__/mg_utils.cpython-312.pyc ADDED Viewed

Binary file (4.07 kB). View file

src_code_for_reproducibility/markov_games/__pycache__/rollout_tree.cpython-312.pyc ADDED Viewed

Binary file (3.97 kB). View file

src_code_for_reproducibility/markov_games/__pycache__/run_markov_games.cpython-312.pyc ADDED Viewed

Binary file (1.53 kB). View file

src_code_for_reproducibility/markov_games/__pycache__/simulation.cpython-312.pyc ADDED Viewed

Binary file (4.25 kB). View file

src_code_for_reproducibility/markov_games/ipd/Ipd_hard_coded_agents.py ADDED Viewed

	@@ -0,0 +1,76 @@

+"""
+File: mllm/markov_games/ipd/Ipd_hard_coded_agents.py
+Summary: Contains hand-crafted IPD policies used as deterministic baselines.
+"""
+from dataclasses import dataclass
+from typing import Any, Tuple
+from mllm.markov_games.ipd.ipd_agent import IPDAgent
+from mllm.markov_games.rollout_tree import AgentActLog, ChatTurn
+@dataclass
+class AlwaysCooperateIPDAgent(IPDAgent):
+    async def act(self, observation) -> Tuple[Any, AgentActLog]:
+        """
+        Always plays the cooperate action, ignoring observation.
+        Returns the configured cooperate_string so the simulation parses it as "C".
+        """
+        action = self.cooperate_string
+        # Log a minimal, structured chat turn for consistency with other agents
+        turn_text = f"Playing cooperate: {action}"
+        self.state.chat_history.append(
+            ChatTurn(
+                agent_id=self.agent_id,
+                role="assistant",
+                content=turn_text,
+                is_state_end=True,
+            )
+        )
+        act_log = AgentActLog(
+            chat_turns=[self.state.chat_history[-1]],
+            info=None,
+        )
+        # Advance internal counters similar to IPDAgent semantics
+        self.state.chat_counter = len(self.state.chat_history)
+        self.state.round_nb = observation.round_nb
+        return action, act_log
+@dataclass
+class AlwaysDefectIPDAgent(IPDAgent):
+    async def act(self, observation) -> Tuple[Any, AgentActLog]:
+        """
+        Always plays the defect action, ignoring observation.
+        Returns the configured defect_string so the simulation parses it as "D".
+        """
+        action = self.defect_string
+        # Log a minimal, structured chat turn for consistency with other agents
+        turn_text = f"Playing defect: {action}"
+        self.state.chat_history.append(
+            ChatTurn(
+                agent_id=self.agent_id,
+                role="assistant",
+                content=turn_text,
+                is_state_end=True,
+            )
+        )
+        act_log = AgentActLog(
+            chat_turns=[self.state.chat_history[-1]],
+            info=None,
+        )
+        # Advance internal counters similar to IPDAgent semantics
+        self.state.chat_counter = len(self.state.chat_history)
+        self.state.round_nb = observation.round_nb
+        return action, act_log

src_code_for_reproducibility/markov_games/ipd/__init__.py ADDED Viewed

	@@ -0,0 +1,11 @@

+"""
+File: mllm/markov_games/ipd/__init__.py
+Summary: Marks the Iterated Prisoner's Dilemma subpackage.
+"""
+from .Ipd_hard_coded_agents import AlwaysCooperateIPDAgent, AlwaysDefectIPDAgent
+__all__ = [
+    "AlwaysCooperateIPDAgent",
+    "AlwaysDefectIPDAgent",
+]

src_code_for_reproducibility/markov_games/ipd/__pycache__/Ipd_hard_coded_agents.cpython-312.pyc ADDED Viewed

Binary file (3.05 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (435 Bytes). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_agent.cpython-312.pyc ADDED Viewed

Binary file (4.97 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_simulation.cpython-312.pyc ADDED Viewed

Binary file (6.87 kB). View file

src_code_for_reproducibility/markov_games/ipd/__pycache__/ipd_statistics.cpython-312.pyc ADDED Viewed

Binary file (1.42 kB). View file

src_code_for_reproducibility/markov_games/ipd/ipd_agent.py ADDED Viewed

	@@ -0,0 +1,120 @@

+"""
+File: mllm/markov_games/ipd/ipd_agent.py
+Summary: Implements the IPD agent abstraction used during simulations.
+"""
+import copy
+import json
+import random
+import re
+from collections.abc import Callable
+from copy import deepcopy
+from dataclasses import dataclass, field
+from typing import Any, Dict, List, Optional, Tuple, Union
+from mllm.markov_games.agent import Agent
+from mllm.markov_games.rollout_tree import AgentActLog, ChatTurn
+@dataclass
+class IPDAgentState:
+    """
+    Tracks retry count, round index, and chat history for a single IPD agent.
+    """
+    nb_retries: int
+    round_nb: int
+    chat_counter: int
+    chat_history: List[ChatTurn]
+@dataclass
+class IPDAgent(Agent):
+    seed: int
+    agent_id: str
+    agent_name: str
+    policy: Callable[[List[Dict]], str]
+    intro_prompt: str  # Introduction prompt explaining the game rules
+    goal_prompt: str  # Prompt explaining the agent's goal
+    strategy_prompt: str  # Prompt suggesting a strategy to the agent
+    max_errors: int  # Maximum number of errors allowed before default action
+    allow_reasoning: bool  # Whether to allow reasoning in the response
+    max_reasoning_chars: int  # Maximum number of characters for reasoning
+    cooperate_string: str  # string parsed as playing cooperate by simulation
+    defect_string: str  # string parsed as playing defect by simulation
+    def __post_init__(self):
+        self.state = IPDAgentState(
+            nb_retries=0, round_nb=0, chat_counter=0, chat_history=[]
+        )
+    async def act(self, observation) -> Tuple[Any, AgentActLog]:
+        """
+        Run the LLM policy conversation until a valid cooperate/defect action is produced.
+        """
+        action = None
+        action_is_ready = False
+        round_nb = observation.round_nb
+        # If it's the first round, we need to send the intro prompt
+        if round_nb == 0 and self.state.chat_counter == 0:
+            self.state.chat_history.append(
+                ChatTurn(
+                    agent_id=self.agent_id,
+                    role="user",
+                    content=self.intro_prompt,
+                    is_state_end=True,
+                )
+            )
+        # If new round
+        if round_nb > self.state.round_nb:
+            coagent_action = observation.last_coagent_move
+            user_message = f"Last round, the other agent played {coagent_action}."
+            self.state.chat_history.append(
+                ChatTurn(
+                    agent_id=self.agent_id,
+                    role="user",
+                    content=user_message,
+                    is_state_end=True,
+                )
+            )
+        # If not new round, try to get valid action from policy
+        output_chat_turn: ChatTurn = await self.policy(
+            state=self.state.chat_history,
+            agent_id=self.agent_id,
+            regex=f"({self.cooperate_string}|{self.defect_string})",
+        )
+        self.state.chat_history.append(output_chat_turn)
+        action = output_chat_turn.content
+        agent_step_log = AgentActLog(
+            chat_turns=self.state.chat_history[self.state.chat_counter :], info=None
+        )
+        self.state.chat_counter = len(self.state.chat_history)
+        self.state.round_nb = round_nb
+        return action, agent_step_log
+    def get_safe_copy(self):
+        """
+        Return a safe copy of the agent.
+        """
+        agent_copy = copy.copy(self)
+        agent_copy.state = copy.deepcopy(self.state)
+        return agent_copy
+    def reset(self):
+        self.state = IPDAgentState()
+        raise NotImplementedError
+    def render(self):
+        pass
+    def close(self):
+        pass
+    def get_agent_info(self):
+        pass

src_code_for_reproducibility/markov_games/ipd/ipd_simulation.py ADDED Viewed

	@@ -0,0 +1,167 @@

+"""
+File: mllm/markov_games/ipd/ipd_simulation.py
+Summary: Runs Iterated Prisoner's Dilemma simulations under the Markov-game API.
+"""
+import copy
+import random
+from dataclasses import dataclass
+from typing import Any, Dict, List, Optional, Tuple
+import numpy as np
+from mllm.markov_games.markov_game import Simulation
+from mllm.markov_games.rollout_tree import SimulationStepLog
+from mllm.utils.get_coagent_id import get_coagent_id
+@dataclass
+class IPDState:
+    """
+    State of the Iterated Prisoner's Dilemma game.
+    """
+    round_nb: int = 0
+    done: bool = False
+    last_moves: Dict[str, str] | None = None
+@dataclass
+class IPDObs:
+    """
+    Observation in Iterated Prisoner's Dilemma game.
+    """
+    round_nb: int
+    last_coagent_move: str | None
+class IPD(Simulation):
+    """
+    Iterated Prisoner's Dilemma simulation following the standard.
+    In each round of the game, two agents simultaneously choose to either cooperate (C) or defect (D).
+    The payoffs are as follows:
+    - If both cooperate: Both receive the "reward" (usually 3 points)
+    - If both defect: Both receive the "punishment" (usually 1 point)
+    - If one cooperates and one defects: The defector receives the "temptation" (usually 5 points)
+      and the cooperator receives the "sucker" payoff (usually 0 points)
+    The game is played for a specified number of rounds.
+    """
+    def __init__(
+        self,
+        agent_ids: List[str],
+        agent_names: List[str],
+        seed: int,
+        rounds_per_game: int,
+        reward: float,  # Both cooperate
+        punishment: float,  # Both defect
+        temptation: float,  # Defector's reward when other cooperates
+        sucker: float,  # Cooperator's reward when other defects
+        cooperate_actions: List[str],
+        defect_actions: List[str],
+    ):
+        self.agent_ids = agent_ids
+        self.agent_names = agent_names
+        self.seed = seed
+        self.rounds_per_game = rounds_per_game
+        self.reward = reward
+        self.punishment = punishment
+        self.temptation = temptation
+        self.sucker = sucker
+        self.cooperate_actions = cooperate_actions
+        self.defect_actions = defect_actions
+        self.state = IPDState()
+    def step(self, actions: Dict[str, str]) -> Tuple[bool, SimulationStepLog]:
+        """
+        Take a step in the environment using the provided actions.
+        Here, the observations are just the states of the game.
+        Args:
+            actions (dict): A dictionary where keys are agent identifiers and values are actions ('C' or 'D').
+        Returns:
+            observations (dict): A dictionary where keys are agent identifiers and values are observations.
+            done (bool): Whether the episode has ended.
+            info (dict): Additional information about the environment.
+        """
+        # Calculate rewards using payoff matrix
+        agent0_action = actions[self.agent_ids[0]]
+        agent1_action = actions[self.agent_ids[1]]
+        # Normalize actions to standard cooperate/defect/gibberish format
+        def normalize_action(action):
+            if action in self.cooperate_actions:
+                return "C"
+            elif action in self.defect_actions:
+                return "D"
+            else:
+                return "D"
+        norm_action0 = normalize_action(agent0_action)
+        norm_action1 = normalize_action(agent1_action)
+        payoffs = {
+            ("C", "C"): [self.reward, self.reward],
+            ("C", "D"): [self.sucker, self.temptation],
+            ("D", "C"): [self.temptation, self.sucker],
+            ("D", "D"): [self.punishment, self.punishment],
+        }
+        round_rewards = {
+            self.agent_ids[0]: payoffs[(norm_action0, norm_action1)][0],
+            self.agent_ids[1]: payoffs[(norm_action0, norm_action1)][1],
+        }
+        # Update game state
+        self.state.round_nb += 1
+        self.state.last_moves = copy.deepcopy(actions)
+        done = self.state.round_nb >= self.rounds_per_game
+        step_log = SimulationStepLog(
+            rewards=round_rewards,
+            info={
+                "actions": {
+                    self.agent_ids[0]: norm_action0,
+                    self.agent_ids[1]: norm_action1,
+                }
+            },
+        )
+        return done, step_log
+    def get_obs(self):
+        """Returns all agent observations in dict
+        Returns:
+            observations
+        """
+        observations = {}
+        for agent_id in self.agent_ids:
+            observations[agent_id] = self.get_obs_agent(agent_id)
+        return observations
+    def get_obs_agent(self, agent_id):
+        """Returns observation for agent_id"""
+        if self.state.last_moves != None:
+            other_id = get_coagent_id(self.agent_ids, agent_id)
+            last_coagent_move = self.state.last_moves[other_id]
+        else:
+            last_coagent_move = None
+        obs = IPDObs(round_nb=self.state.round_nb, last_coagent_move=last_coagent_move)
+        return obs
+    def reset(self):
+        """Returns initial observations and states"""
+        self.state = IPDState()
+        return self.get_obs()
+    def get_safe_copy(self):
+        """
+        Return a safe copy of the simulation.
+        """
+        simulation_copy = copy.copy(self)
+        simulation_copy.state = copy.deepcopy(self.state)
+        return simulation_copy

src_code_for_reproducibility/markov_games/ipd/ipd_statistics.py ADDED Viewed

	@@ -0,0 +1,24 @@

+"""
+File: mllm/markov_games/ipd/ipd_statistics.py
+Summary: Computes statistics and summaries for IPD experiments.
+"""
+from __future__ import annotations
+from typing import Callable, Dict, List, Tuple
+from mllm.markov_games.rollout_tree import SimulationStepLog
+def avg_reward(sl: SimulationStepLog) -> List[Tuple[str, float]]:
+    for aid in sl.rewards.keys():
+        if "buffer" in str(aid) and "live" not in str(aid):
+            return None
+    # One value per agent at each step
+    rewards_dict = {f"reward-{aid}": float(v) for aid, v in (sl.rewards or {}).items()}
+    return [(key, value) for key, value in rewards_dict.items() if value is not None]
+stat_functs: list[Callable[[SimulationStepLog], List[Tuple[str, float]]]] = [
+    avg_reward,
+]

src_code_for_reproducibility/markov_games/negotiation/README.md ADDED Viewed

	@@ -0,0 +1,27 @@

+## Negotiation Games: core mechanics and variants
+This family of games feature two agents who, in each round, may briefly communicate and then simultaneously propose how to split a fixed resource (most commonly 10 coins). Rewards are the amount kept multiplied by an agent’s per-unit value. The starting speaker alternates deterministically across rounds.
+Communication is optional and variant-dependent: some settings encourage rich messaging to share private information, while others remove messaging entirely to focus on allocation behavior.
+Proportional splitting is used when the two proposals exceed the available total: allocations are scaled proportionally rather than discarded. This preserves a useful learning signal even when agents over-claim.
+### Variants (in increasing difficulty)
+- No‑Press Split
+  - Multiple item types (e.g., hats, balls, books)
+  - The item values for each agent are public.
+  - No communication; agents go straight to making split proposals.
+  - Motivation: mirrors no‑communication setups (e.g., Advantage Alignment) while keeping the split decision nontrivial.
+- Trust-and-Split RPS (TAS-RPS)
+  - Single item type (coins)
+  - Each round, a rock–paper–scissors hand draw creates a strong asymmetry: the winner’s per-coin value is 10, the loser’s is 1.
+  - Each agent initially sees only their own hand and must communicate to coordinate an optimal split.
+  - Motivation: enforce large value disparity so one’s own value reveals little about the other’s (avoiding ceiling effects) and incentivize meaningful communication.

src_code_for_reproducibility/markov_games/negotiation/__pycache__/dond_agent.cpython-312.pyc ADDED Viewed

Binary file (4.66 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/dond_simulation.cpython-312.pyc ADDED Viewed

Binary file (10.7 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/nego_agent.cpython-312.pyc ADDED Viewed

Binary file (11.7 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/nego_hard_coded_policies.cpython-312.pyc ADDED Viewed

Binary file (3.39 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/nego_simulation.cpython-312.pyc ADDED Viewed

Binary file (12.6 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/negotiation_statistics.cpython-312.pyc ADDED Viewed

Binary file (14.3 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/no_press_nego_agent.cpython-312.pyc ADDED Viewed

Binary file (6.11 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/no_press_nego_simulation.cpython-312.pyc ADDED Viewed

Binary file (9.72 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/tas_rps_agent.cpython-312.pyc ADDED Viewed

Binary file (6.05 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/tas_rps_simulation.cpython-312.pyc ADDED Viewed

Binary file (11.7 kB). View file

src_code_for_reproducibility/markov_games/negotiation/nego_hard_coded_policies.py ADDED Viewed

	@@ -0,0 +1,70 @@

+"""
+File: mllm/markov_games/negotiation/nego_hard_coded_policies.py
+Summary: Provides deterministic negotiation policies for testing and baselines.
+"""
+import asyncio
+from typing import Any, Optional, Tuple
+from mllm.markov_games.negotiation.nego_agent import NegotiationAgent
+from mllm.markov_games.negotiation.nego_simulation import Split
+from mllm.markov_games.negotiation.no_press_nego_agent import NoPressAgent
+from mllm.markov_games.negotiation.no_press_nego_simulation import NoPressObs
+from mllm.markov_games.rollout_tree import AgentActLog, ChatTurn
+class HardCodedNegoWelfareMaximizingPolicy(NoPressAgent):
+    async def act(self, observation: NoPressObs) -> Tuple[Any, AgentActLog]:
+        """
+        Policy that gives all of the items to the agent who values them more.
+        If the items are equally valued, give them to the agent who values them more.
+        """
+        quantities = observation.quantities
+        my_values = observation.value
+        other_values = observation.other_value
+        items_given_to_self = {}
+        for item, qty in quantities.items():
+            my_v = float(my_values.get(item, 0))
+            other_v = float(other_values.get(item, 0))
+            if my_v == other_v:
+                items_given_to_self[item] = int(qty) / 2
+            else:
+                items_given_to_self[item] = int(qty if my_v > other_v else 0)
+        action = Split(items_given_to_self=items_given_to_self)
+        act_log = AgentActLog(
+            chat_turns=[
+                ChatTurn(
+                    agent_id=self.agent_id,
+                    role="assistant",
+                    content="Using welfare-maximizing split (all to higher-value agent).",
+                    is_state_end=True,
+                )
+            ],
+            info=None,
+        )
+        return action, act_log
+class HardCodedNegoGreedyPolicy(NoPressAgent):
+    async def act(self, observation: NoPressObs) -> Tuple[Any, AgentActLog]:
+        """
+        Always gives itself all of the items.
+        """
+        quantities = observation.quantities
+        items_given_to_self = {item: int(qty) for item, qty in quantities.items()}
+        action = Split(items_given_to_self=items_given_to_self)
+        act_log = AgentActLog(
+            chat_turns=[
+                ChatTurn(
+                    agent_id=self.agent_id,
+                    role="assistant",
+                    content="Using greedy split (keep all items).",
+                    is_state_end=True,
+                )
+            ],
+            info=None,
+        )
+        return action, act_log

src_code_for_reproducibility/markov_games/negotiation/tas_agent.py ADDED Viewed

	@@ -0,0 +1,118 @@

+"""
+File: mllm/markov_games/negotiation/tas_agent.py
+Summary: Agent implementation for Take-and-Split negotiations.
+"""
+from mllm.markov_games.negotiation.nego_agent import NegotiationAgent
+from mllm.markov_games.negotiation.nego_simulation import Split
+from mllm.markov_games.negotiation.tas_simulation import TrustAndSplitObs
+class TrustAndSplitAgent(NegotiationAgent):
+    """Prompt/template wrapper for the classic multi-item Take-and-Split benchmark."""
+    def __init__(self, num_message_chars, *args, **kwargs):
+        self.num_message_chars = num_message_chars
+        super().__init__(*args, **kwargs)
+        self.intro_prompt = (
+            "Welcome to an iterated game. You are {agent}. The other agent is {other_agent}.\n"
+            "Setup:\n"
+            "1. The game has multiple independent rounds.\n"
+            "2. In each round, there are multiple items to split between the two agents.\n"
+            "3. Both agents are assigned a per-item value between 1 and 20 (inclusive) in each round.\n"
+            "4. You can only observe your own per-item values.\n"
+            "5. Because assignments are random, both agents are equally likely to have same expected per-item value.\n"
+            "\n"
+            "Protocol:\n"
+            "1. At the start of the round, one agent begins the conversation. The starting role alternates each round.\n"
+            "2. Agents exchange a short chat ({quota_messages_per_agent_per_round} messages per round per agent) to negotiate how to split the item.\n"
+            "   - Use this chat to communicate your private per-item value to make informed proposals.\n"
+            "3. After the chat, both agents simultaneously propose the amount of each item they will keep.\n"
+            "4. If the total sum of proposals is less than or equal to the item quantity, both agents receive their proposed amounts.\n"
+            "5. If the total sum of proposals exceeds the item quantity, they are allocated proportionally.\n"
+            "6. Your points for the round = (amount you receive per item) x (your per-item value for that round), added across all items.\n"
+            "7. Points are accumulated across rounds.\n"
+            "Your goal: {goal}\n"
+        )
+        self.new_round_prompt = (
+            "A New Round Begins\n"
+            "The items to split are {quantities}.\n"
+            "Your per-item values are {value}."
+        )
+        self.last_round_prompt = (
+            "Last Round Summary:\n"
+            "   - Items to split: {last_quantities}\n"
+            "   - Your per-item values: {last_value_agent}\n"
+            "   - {other_agent}'s per-item values: {last_value_coagent}\n"
+            "   - You proposed: {last_split_agent}\n"
+            "   - You earned: {last_points_agent} points\n"
+            "   - {other_agent} proposed: {last_split_coagent}\n"
+            "   - {other_agent} earned: {last_points_coagent} points\n"
+            "   - Round Complete.\n"
+        )
+        self.send_split_prompt = (
+            "Message quota is finished for this round.\n"
+            "{other_agent} has finalized their proposal.\n"
+            "Submit your finalization now\n"
+            "Respond with {proposal_style2}"
+        )
+        # self.wait_for_message_prompt = "Wait for {other_agent} to send a message..."
+        self.wait_for_message_prompt = ""
+        self.last_message_prompt = "{other_agent} said: {last_message}"
+        # self.send_message_prompt = (
+        #     f"Send your message now (max {self.num_message_chars} chars)."
+        # )
+        self.send_message_prompt = f"Send your message now in <message>...</message> (<={self.num_message_chars} chars)."
+    def get_message_regex(self, observation: TrustAndSplitObs) -> str:
+        """Constrain chat to bounded XML tags for stable parsing."""
+        return rf"<message>[\s\S]{{0,{self.num_message_chars}}}</message>"
+    # def get_message_regex(self, observation: TrustAndSplitObs) -> str:
+    #     return rf"(?s).{{0,{self.num_message_chars}}}"
+    def get_split_regex(self, observation: TrustAndSplitObs) -> str:
+        """Allow natural-language item names while still returning machine-parsable XML."""
+        items = list(observation.quantities.keys())
+        # Accept both singular and plural forms
+        item_pattern = "|".join(
+            [f"{item[:-1]}s?" if item.endswith("s") else f"{item}s?" for item in items]
+        )
+        regex = rf"(?i)<items_to_self> ?((?:\s*(?P<num>(10|[0-9]))\s*(?P<item>{item_pattern})\s*,?)+) ?</items_to_self>"
+        return regex
+    def get_split_action(
+        self, policy_output: str, observation: TrustAndSplitObs
+    ) -> Split:
+        """Convert human-readable allocation text back into canonical item IDs."""
+        items = list(observation.quantities.keys())
+        import re as _re
+        split_regex = self.get_split_regex(observation)
+        items_given_to_self = {item: 0 for item in items}
+        m = _re.match(split_regex, policy_output.strip())
+        if m:
+            # Find all (number, item) pairs
+            item_pattern = "|".join(
+                [
+                    f"{item[:-1]}s?" if item.endswith("s") else f"{item}s?"
+                    for item in items
+                ]
+            )
+            inner_regex = rf"(?i)(10|[0-9])\s*({item_pattern})"
+            def normalize_item_name(item_str):
+                for orig in items:
+                    if item_str.lower() == orig.lower():
+                        return orig
+                    if orig.endswith("s") and item_str.lower() == orig[:-1].lower():
+                        return orig
+                    if (
+                        not orig.endswith("s")
+                        and item_str.lower() == orig.lower() + "s"
+                    ):
+                        return orig
+            for num, item in _re.findall(inner_regex, m.group(1)):
+                items_given_to_self[normalize_item_name(item)] = int(num)
+        return Split(items_given_to_self=items_given_to_self)

src_code_for_reproducibility/training/trainer_common.py ADDED Viewed

	@@ -0,0 +1,1032 @@

+"""
+File: mllm/training/trainer_common.py
+Summary: Shared trainer utilities, base classes, and gradient helpers.
+"""
+import logging
+import os
+import pickle
+import sys
+from abc import ABC, abstractmethod
+from typing import Callable, Literal, Union
+import numpy as np
+import torch
+import torch.nn.functional as F
+from accelerate import Accelerator
+from pandas._libs.tslibs.offsets import CBMonthBegin
+from peft import LoraConfig
+from torch.nn.utils.rnn import pad_sequence
+from transformers import AutoModelForCausalLM, AutoTokenizer
+from mllm.markov_games.rollout_tree import *
+from mllm.markov_games.rollout_tree import RolloutTreeRootNode
+from mllm.training.annealing_methods import sigmoid_annealing
+from mllm.training.credit_methods import (
+    get_discounted_returns,
+    get_generalized_advantage_estimates,
+    get_rloo_credits,
+    whiten_advantages,
+    whiten_advantages_time_step_wise,
+)
+from mllm.training.tally_metrics import Tally
+from mllm.training.tally_rollout import RolloutTally, RolloutTallyItem
+from mllm.training.tally_tokenwise import ContextualizedTokenwiseTally
+from mllm.training.tokenize_chats import *
+from mllm.training.tokenize_chats import process_training_chat
+from mllm.training.training_data_utils import *
+from mllm.training.training_data_utils import (
+    TrainingBatch,
+    TrajectoryBatch,
+    get_tokenwise_credits,
+)
+from mllm.utils.resource_context import resource_logger_context
+logger = logging.getLogger(__name__)
+logger.addHandler(logging.StreamHandler(sys.stdout))
+@dataclass
+class TrainerAnnealingState:
+    annealing_step_counter: int = 0
+class BaseTrainer(ABC):
+    """
+    Shared scaffolding for policy-gradient trainers (optimizer wiring, logging, etc.).
+    Subclasses implement `set_agent_trajectory_data` / `share_advantage_data`
+    to plug in algorithm-specific behavior.
+    """
+    def __init__(
+        self,
+        policy: AutoModelForCausalLM,
+        policy_optimizer: torch.optim.Optimizer,
+        critic: Union[AutoModelForCausalLM, None],
+        critic_optimizer: Union[torch.optim.Optimizer, None],
+        tokenizer: AutoTokenizer,
+        lr_scheduler: torch.optim.lr_scheduler.LRScheduler,
+        critic_lr_scheduler: Union[torch.optim.lr_scheduler.LRScheduler, None],
+        ######################################################################
+        entropy_coeff: float,
+        entropy_topk: int,
+        entropy_mask_regex: Union[str, None],
+        kl_coeff: float,
+        gradient_clipping: Union[float, None],
+        restrict_tokens: Union[list[str], None],
+        mini_batch_size: int,
+        use_gradient_checkpointing: bool,
+        temperature: float,
+        device: str,
+        whiten_advantages: bool,
+        whiten_advantages_time_step_wise: bool,
+        use_gae: bool,
+        use_gae_lambda_annealing: bool,
+        gae_lambda_annealing_limit: float,
+        gae_lambda_annealing_method: Literal["sigmoid_annealing"],
+        gae_lambda_annealing_method_params: dict,
+        pg_loss_normalization: Literal["batch", "nb_tokens"],
+        use_rloo: bool,
+        skip_discounted_state_visitation: bool,
+        discount_factor: float,
+        enable_tokenwise_logging: bool,
+        save_path: str,
+        reward_normalizing_constant: float = 1.0,
+        critic_loss_type: Literal["mse", "huber"] = "huber",
+        exploration_prompts_to_remove: list[str] = [],
+        filter_higher_refprob_tokens_kl: bool = False,
+        truncated_importance_sampling_ratio_cap: float = 0.0,
+        importance_sampling_strategy: Literal[
+            "per_token", "per_sequence"
+        ] = "per_token",
+        no_rloo_grouping: bool = False,
+    ):
+        """
+        Initialize the REINFORCE trainer with reward shaping for multi-agent or single-agent training.
+        Args:
+            model (AutoModelForCausalLM): The main policy model.
+            tokenizer (AutoTokenizer): Tokenizer for the model.
+            optimizer (torch.optim.Optimizer): Optimizer for the policy model.
+            lr_scheduler (torch.optim.lr_scheduler.LRScheduler): Learning rate scheduler for the policy model.
+            critic (AutoModelForCausalLM or None): Critic model for value estimation (optional).
+            critic_optimizer (torch.optim.Optimizer or None): Optimizer for the critic model (optional).
+            critic_lr_scheduler (torch.optim.lr_scheduler.LRScheduler or None): LR scheduler for the critic (optional).
+            config (RtConfig): Configuration object for training.
+        """
+        self.tokenizer = tokenizer
+        # self.tokenizer.padding_side = "left"  # needed for flash attention
+        if self.tokenizer.pad_token_id is None:
+            self.tokenizer.pad_token_id = self.tokenizer.eos_token_id
+        self.lr_scheduler = lr_scheduler
+        self.accelerator = Accelerator()
+        (
+            self.policy,
+            self.policy_optimizer,
+            self.critic,
+            self.critic_optimizer,
+        ) = self.accelerator.prepare(policy, policy_optimizer, critic, critic_optimizer)
+        self.critic_lr_scheduler = critic_lr_scheduler
+        self.tally = Tally()
+        if use_gradient_checkpointing == True:
+            self.policy.gradient_checkpointing_enable(dict(use_reentrant=False))
+            if critic is not None:
+                self.critic.gradient_checkpointing_enable(dict(use_reentrant=False))
+        self.save_path = save_path
+        # Load trainer state if it exists
+        self.trainer_annealing_state_path = os.path.join(
+            self.save_path, "trainer_annealing_state.pkl"
+        )
+        if os.path.exists(self.trainer_annealing_state_path):
+            logger.info(
+                f"Loading trainer state from {self.trainer_annealing_state_path}"
+            )
+            self.trainer_annealing_state = pickle.load(
+                open(self.trainer_annealing_state_path, "rb")
+            )
+        else:
+            self.trainer_annealing_state = TrainerAnnealingState()
+        # Load policy optimizer state if it exists
+        self.policy_optimizer_path = os.path.join(
+            self.save_path, "policy_optimizer_state.pt"
+        )
+        if os.path.exists(self.policy_optimizer_path):
+            logger.info(
+                f"Loading policy optimizer state from {self.policy_optimizer_path}"
+            )
+            self.policy_optimizer.load_state_dict(
+                torch.load(self.policy_optimizer_path)
+            )
+        # Load critic optimizer state if it exists
+        self.critic_optimizer_path = os.path.join(
+            self.save_path, "critic_optimizer_state.pt"
+        )
+        if (
+            os.path.exists(self.critic_optimizer_path)
+            and self.critic_optimizer is not None
+        ):
+            logger.info(
+                f"Loading critic optimizer state from {self.critic_optimizer_path}"
+            )
+            self.critic_optimizer.load_state_dict(
+                torch.load(self.critic_optimizer_path)
+            )
+        self.device = self.accelerator.device
+        self.entropy_coeff = entropy_coeff
+        self.entropy_topk = entropy_topk
+        self.entropy_mask_regex = entropy_mask_regex
+        self.kl_coeff = kl_coeff
+        self.gradient_clipping = gradient_clipping
+        self.restrict_tokens = restrict_tokens
+        self.mini_batch_size = mini_batch_size
+        self.use_gradient_checkpointing = use_gradient_checkpointing
+        self.temperature = temperature
+        self.use_gae = use_gae
+        self.whiten_advantages = whiten_advantages
+        self.whiten_advantages_time_step_wise = whiten_advantages_time_step_wise
+        self.use_rloo = use_rloo
+        self.skip_discounted_state_visitation = skip_discounted_state_visitation
+        self.use_gae_lambda_annealing = use_gae_lambda_annealing
+        self.gae_lambda_annealing_limit = gae_lambda_annealing_limit
+        if use_gae_lambda_annealing:
+            self.gae_lambda_annealing_method: Callable[
+                [int], float
+            ] = lambda step: eval(gae_lambda_annealing_method)(
+                step=step, **gae_lambda_annealing_method_params
+            )
+        self.discount_factor = discount_factor
+        self.enable_tokenwise_logging = enable_tokenwise_logging
+        self.reward_normalizing_constant = reward_normalizing_constant
+        self.pg_loss_normalization = pg_loss_normalization
+        self.critic_loss_type = critic_loss_type
+        self.exploration_prompts_to_remove = exploration_prompts_to_remove
+        # Common containers used by all trainers
+        self.training_data: dict = {}
+        self.debug_path_list: list[str] = []
+        self.policy_gradient_data = None
+        self.tally = Tally()
+        self.rollout_tally = RolloutTally()
+        self.tokenwise_tally: Union[ContextualizedTokenwiseTally, None] = None
+        self.filter_higher_refprob_tokens_kl = filter_higher_refprob_tokens_kl
+        self.truncated_importance_sampling_ratio_cap = (
+            truncated_importance_sampling_ratio_cap
+        )
+        self.importance_sampling_strategy = importance_sampling_strategy
+        self.no_rloo_grouping = no_rloo_grouping
+    def mask_non_restricted_token_logits(self, logits: torch.Tensor) -> torch.Tensor:
+        """
+        Masks logits so that only allowed tokens (as specified in config.restrict_tokens)
+        and the EOS token are active.
+        All other logits are set to -inf, effectively removing them from the softmax.
+        Args:
+            logits (torch.Tensor): The logits tensor of shape (B, S, V).
+        Returns:
+            torch.Tensor: The masked logits tensor.
+        """
+        # Gradients flow only through the kept logits; masking is recomputed per batch for clarity.
+        if self.restrict_tokens is not None:
+            allowed_token_ids = []
+            for token in self.restrict_tokens:
+                token_ids = self.tokenizer(token, add_special_tokens=False)["input_ids"]
+                allowed_token_ids.append(token_ids[0])
+            allowed_token_ids.append(
+                self.tokenizer.eos_token_id
+            )  # This token should always be active
+            allowed_token_ids = torch.tensor(allowed_token_ids, device=logits.device)
+            # Mask log_probs and probs to only allowed tokens
+            mask = torch.zeros_like(logits).bool()  # (B, S, V)
+            mask[..., allowed_token_ids] = True
+            logits = torch.where(
+                mask,
+                logits,
+                torch.tensor(-float("inf"), device=logits.device),
+            )
+        return logits
+    def apply_reinforce_step(
+        self,
+        training_batch: TrainingBatch,
+    ) -> None:
+        """
+        Applies a single REINFORCE policy gradient step using the provided batch of rollouts.
+        Handles batching, loss computation (including entropy and KL regularization), gradient accumulation, and optimizer step.
+        Optionally logs various metrics and statistics.
+        Args:
+            paths (list[str]): List of game complete file paths for each rollout.
+            contexts (list[torch.Tensor]): List of context tensors for each rollout.
+            credits (list[torch.Tensor]): List of credit tensors (rewards/advantages) for each rollout.
+            action_masks (list[torch.Tensor]): List of action mask tensors for each rollout.
+        """
+        with resource_logger_context(logger, "Apply reinforce step"):
+            self.policy.train()
+            mb_size = self.mini_batch_size
+            nb_rollouts = len(training_batch)
+            # Initialize running mean logs
+            running_mean_logs = {
+                "rl_objective": 0.0,
+                "policy_gradient_loss": 0.0,
+                "policy_gradient_norm": 0.0,
+                "log_probs": 0.0,
+                "credits": 0.0,
+                "entropy": 0.0,
+                "engine_log_probs_diff_clampfrac": 0.0,
+                "tis_imp_ratio": 0.0,
+                "ref_log_probs_diff_clampfrac": 0.0,
+                "higher_refprob_frac": 0.0,
+                "tis_imp_ratio_clampfrac": 0.0,
+            }
+            if self.entropy_coeff != 0.0:
+                running_mean_logs["entropy"] = 0.0
+            if self.kl_coeff != 0.0:
+                running_mean_logs["kl_divergence"] = 0.0
+            # Get total number of tokens generated
+            total_tokens_generated = 0
+            for att_mask in training_batch.batch_action_mask:
+                total_tokens_generated += att_mask.sum()
+            # Obtain loss normalization
+            if self.pg_loss_normalization == "nb_tokens":
+                normalization_factor = total_tokens_generated
+            elif self.pg_loss_normalization == "batch":
+                normalization_factor = np.ceil(nb_rollouts / mb_size).astype(int)
+            else:
+                raise ValueError(
+                    f"Invalid pg_loss_normalization: {self.pg_loss_normalization}"
+                )
+            # Gradient accumulation for each mini-batch
+            for mb in range(0, nb_rollouts, mb_size):
+                logger.info(f"Processing mini-batch {mb} of {nb_rollouts}")
+                loss = 0.0
+                training_mb = training_batch[mb : mb + mb_size]
+                training_mb = training_mb.get_padded_tensors()
+                training_mb.to(self.device)
+                (
+                    tokens_mb,
+                    action_mask_mb,
+                    entropy_mask_mb,
+                    credits_mb,
+                    engine_log_probs_mb,
+                    timesteps_mb,
+                ) = (
+                    training_mb.batch_input_ids,
+                    training_mb.batch_action_mask,
+                    training_mb.batch_entropy_mask,
+                    training_mb.batch_credits,
+                    training_mb.batch_engine_log_probs,
+                    training_mb.batch_timesteps,
+                )
+                # Next token prediction
+                contexts_mb = tokens_mb[:, :-1]
+                shifted_contexts_mb = tokens_mb[:, 1:]
+                action_mask_mb = action_mask_mb[:, 1:]
+                entropy_mask_mb = entropy_mask_mb[:, 1:]
+                credits_mb = credits_mb[:, 1:]
+                engine_log_probs_mb = engine_log_probs_mb[:, 1:]
+                timesteps_mb = timesteps_mb[:, 1:]
+                if self.enable_tokenwise_logging:
+                    self.tokenwise_tally.set_action_mask(action_mask=action_mask_mb)
+                    self.tokenwise_tally.set_range(range=(mb, mb + mb_size))
+                    self.tokenwise_tally.add_contexts(contexts=contexts_mb)
+                    self.tokenwise_tally.add_data(
+                        metric_id="next_token",
+                        metrics=shifted_contexts_mb,
+                        to_tids=True,
+                    )
+                    self.tokenwise_tally.add_data(
+                        metric_id="entropy_mask",
+                        metrics=entropy_mask_mb,
+                    )
+                if self.enable_tokenwise_logging:
+                    self.tokenwise_tally.add_data(
+                        metric_id="next_token_credit", metrics=credits_mb
+                    )
+                # Forward pass + cast to FP-32 for higher prec. Causal LM attention masks are implicit;
+                # wire up a custom mask here only if the policy deviates from standard autoregressive behavior.
+                logits = self.policy(input_ids=contexts_mb)[0]  # (B, S, V)
+                # Mask non-restricted tokens
+                if self.restrict_tokens is not None:
+                    logits = self.mask_non_restricted_token_logits(logits)
+                logits /= self.temperature  # (B, S, V)
+                # Compute new log probabilities
+                log_probs = F.log_softmax(logits, dim=-1)  # (B, S, V)
+                # Get log probabilities of actions taken during rollouts
+                action_log_probs = log_probs.gather(
+                    dim=-1, index=shifted_contexts_mb.unsqueeze(-1)
+                ).squeeze(
+                    -1
+                )  # (B, S)
+                if self.pg_loss_normalization == "batch":
+                    den_running_mean = action_mask_mb.sum() * normalization_factor
+                else:
+                    den_running_mean = normalization_factor
+                running_mean_logs["log_probs"] += (
+                    action_log_probs * action_mask_mb
+                ).sum().item() / den_running_mean
+                running_mean_logs["credits"] += (
+                    credits_mb * action_mask_mb
+                ).sum().item() / den_running_mean
+                if self.enable_tokenwise_logging:
+                    self.tokenwise_tally.add_data(
+                        metric_id="next_token_log_prob",
+                        metrics=action_log_probs,
+                    )
+                    self.tokenwise_tally.add_data(
+                        metric_id="engine_next_token_log_prob",
+                        metrics=engine_log_probs_mb,
+                    )
+                    self.tokenwise_tally.add_data(
+                        metric_id="next_token_prob",
+                        metrics=torch.exp(action_log_probs),
+                    )
+                    top_k_indices = torch.topk(logits, k=5, dim=-1).indices
+                    self.tokenwise_tally.add_data(
+                        metric_id=f"top_{5}_tids",
+                        metrics=top_k_indices,
+                        to_tids=True,
+                    )
+                    self.tokenwise_tally.add_data(
+                        metric_id=f"top_{5}_probs",
+                        metrics=torch.exp(log_probs).gather(
+                            dim=-1, index=top_k_indices
+                        ),
+                    )
+                rewarded_action_log_probs = (
+                    action_mask_mb * credits_mb * action_log_probs
+                )
+                # (B, S)
+                INVALID_LOGPROB = 1.0
+                CLAMP_VALUE = 40.0
+                masked_action_log_probs = torch.masked_fill(
+                    action_log_probs, ~action_mask_mb, INVALID_LOGPROB
+                )
+                masked_engine_log_probs = torch.masked_fill(
+                    engine_log_probs_mb, ~action_mask_mb, INVALID_LOGPROB
+                )
+                with torch.no_grad():
+                    action_engine_log_probs_diff = (
+                        masked_action_log_probs - masked_engine_log_probs
+                    ).clamp(-CLAMP_VALUE, CLAMP_VALUE)
+                running_mean_logs["engine_log_probs_diff_clampfrac"] += (
+                    action_engine_log_probs_diff.abs()
+                    .eq(CLAMP_VALUE)
+                    .float()
+                    .sum()
+                    .item()
+                    / den_running_mean
+                )
+                if self.importance_sampling_strategy == "per_sequence":
+                    tis_imp_ratio = torch.zeros_like(action_engine_log_probs_diff)
+                    for mb_idx in range(action_engine_log_probs_diff.shape[0]):
+                        valid_token_mask = action_mask_mb[mb_idx]
+                        timestep_ids = timesteps_mb[mb_idx][valid_token_mask]
+                        timestep_logprob_diffs = action_engine_log_probs_diff[mb_idx][
+                            valid_token_mask
+                        ]
+                        max_timestep = int(timestep_ids.max().item()) + 1
+                        timestep_sums = torch.zeros(
+                            max_timestep,
+                            device=action_engine_log_probs_diff.device,
+                            dtype=action_engine_log_probs_diff.dtype,
+                        )
+                        timestep_sums.scatter_add_(
+                            0, timestep_ids, timestep_logprob_diffs
+                        )
+                        timestep_ratios = torch.exp(timestep_sums)
+                        tis_imp_ratio[
+                            mb_idx, valid_token_mask
+                        ] = timestep_ratios.gather(0, timestep_ids)
+                else:
+                    tis_imp_ratio = torch.exp(action_engine_log_probs_diff)
+                running_mean_logs["tis_imp_ratio"] += (
+                    tis_imp_ratio * action_mask_mb
+                ).sum().item() / den_running_mean
+                if self.truncated_importance_sampling_ratio_cap > 0.0:
+                    tis_imp_ratio = torch.clamp(
+                        tis_imp_ratio, max=self.truncated_importance_sampling_ratio_cap
+                    )
+                    running_mean_logs["tis_imp_ratio_clampfrac"] += (
+                        tis_imp_ratio.eq(self.truncated_importance_sampling_ratio_cap)
+                        .float()
+                        .sum()
+                        .item()
+                    ) / den_running_mean
+                    rewarded_action_log_probs = (
+                        rewarded_action_log_probs * tis_imp_ratio
+                    )
+                if self.enable_tokenwise_logging:
+                    self.tokenwise_tally.add_data(
+                        metric_id="next_token_clogπ",
+                        metrics=rewarded_action_log_probs,
+                    )
+                # Add value term to loss
+                if self.pg_loss_normalization == "batch":
+                    nb_act_tokens = action_mask_mb.sum()
+                    mb_value = -rewarded_action_log_probs.sum() / nb_act_tokens
+                else:
+                    mb_value = -rewarded_action_log_probs.sum()
+                loss += mb_value
+                running_mean_logs["rl_objective"] += mb_value.item() / den_running_mean
+                # -------------------------------------------------
+                # Entropy Regularization
+                # -------------------------------------------------
+                # Only apply entropy on distribution defined over most probable tokens
+                if self.entropy_topk is not None:
+                    top_k_indices = torch.topk(
+                        logits, k=self.entropy_topk, dim=-1
+                    ).indices
+                    entropy_logits = logits.gather(dim=-1, index=top_k_indices)
+                else:
+                    entropy_logits = logits
+                token_entropy_terms = -F.softmax(
+                    entropy_logits, dim=-1
+                ) * F.log_softmax(
+                    entropy_logits, dim=-1
+                )  # (B, S, T)
+                token_entropy_terms *= (
+                    action_mask_mb[:, :, None] * entropy_mask_mb[:, :, None]
+                )  # only get loss on specific action tokens
+                mb_entropy = token_entropy_terms.sum(dim=-1)
+                if self.enable_tokenwise_logging:
+                    self.tokenwise_tally.add_data(
+                        metric_id="entropy",
+                        metrics=mb_entropy,
+                    )
+                if self.pg_loss_normalization == "batch":
+                    nb_act_tokens = action_mask_mb.sum()
+                    mb_entropy = -mb_entropy.sum() / nb_act_tokens
+                else:
+                    mb_entropy = -mb_entropy.sum()
+                running_mean_logs["entropy"] += -mb_entropy.item() / den_running_mean
+                if self.entropy_coeff != 0.0:
+                    mb_entropy *= self.entropy_coeff
+                    loss += mb_entropy
+                # -------------------------------------------------
+                # KL-DIVERGENCE
+                # -------------------------------------------------
+                if self.kl_coeff != 0.0:
+                    ref_model_logits = self.policy.get_base_model_logits(contexts_mb)
+                    ref_model_logits = ref_model_logits / self.temperature
+                    # (B, S, V)
+                    ref_model_logits = self.mask_non_restricted_token_logits(
+                        logits=ref_model_logits
+                    )
+                    # (B, S, V)
+                    ref_model_log_probs = F.log_softmax(ref_model_logits, dim=-1)
+                    # (B, S, V)
+                    ref_model_action_log_probs = ref_model_log_probs.gather(
+                        dim=-1, index=shifted_contexts_mb.unsqueeze(-1)
+                    ).squeeze(
+                        -1
+                    )  # (B,S)
+                    # Approximating KL Divergence (see refs in docstring)
+                    # Ref 1: http://joschu.net/blog/kl-approx.html
+                    # Ref 2: https://github.dev/huggingface/trl/blob/main/trl/trainer/grpo_trainer.py#L1332
+                    masked_ref_model_action_log_probs = torch.masked_fill(
+                        ref_model_action_log_probs, ~action_mask_mb, INVALID_LOGPROB
+                    )
+                    action_log_probs_diff = (
+                        masked_ref_model_action_log_probs - masked_action_log_probs
+                    ).clamp(-CLAMP_VALUE, CLAMP_VALUE)
+                    running_mean_logs["ref_log_probs_diff_clampfrac"] += (
+                        action_log_probs_diff.abs().eq(CLAMP_VALUE).float().sum().item()
+                        / den_running_mean
+                    )
+                    if self.filter_higher_refprob_tokens_kl:
+                        higher_refprob_tokens_mask = action_log_probs_diff > 0.0
+                        running_mean_logs["higher_refprob_frac"] += (
+                            higher_refprob_tokens_mask.sum().item() / den_running_mean
+                        )
+                        action_log_probs_diff = action_log_probs_diff * (
+                            ~higher_refprob_tokens_mask
+                        )
+                    kl_div = torch.expm1(action_log_probs_diff) - action_log_probs_diff
+                    kl_div *= action_mask_mb  # We only care about KLD of action tokens
+                    if self.truncated_importance_sampling_ratio_cap > 0.0:
+                        kl_div = kl_div * tis_imp_ratio
+                    kl_div *= self.kl_coeff
+                    if self.enable_tokenwise_logging:
+                        self.tokenwise_tally.add_data(
+                            metric_id="ref_model_next_token_log_prob",
+                            metrics=ref_model_action_log_probs,
+                        )
+                        self.tokenwise_tally.add_data(
+                            metric_id="kl_divergence",
+                            metrics=kl_div,
+                        )
+                    if self.pg_loss_normalization == "batch":
+                        nb_act_tokens = action_mask_mb.sum()
+                        mb_kl = kl_div.sum() / nb_act_tokens
+                    else:
+                        mb_kl = kl_div.sum()
+                    running_mean_logs["kl_divergence"] += (
+                        mb_kl.item() / den_running_mean
+                    )
+                    loss += mb_kl
+                # Accumulate gradient
+                running_mean_logs["policy_gradient_loss"] += (
+                    loss.item() / den_running_mean
+                )
+                loss /= normalization_factor
+                self.accelerator.backward(loss)
+                # ensure gpu memory is freed
+                del training_mb
+                del log_probs
+                del logits
+                del loss
+                del action_log_probs
+                del rewarded_action_log_probs
+            logger.info(
+                f"Accumulated the policy gradient loss for {total_tokens_generated} tokens."
+            )
+            # Clip gradients and take step
+            if self.gradient_clipping is not None:
+                grad_norm = self.accelerator.clip_grad_norm_(
+                    self.policy.parameters(), self.gradient_clipping
+                )
+                running_mean_logs["policy_gradient_norm"] += grad_norm.item()
+            # Take step
+            self.policy_optimizer.step()
+            self.policy_optimizer.zero_grad()
+            # Store logs
+            for key, value in running_mean_logs.items():
+                self.tally.add_metric(path=key, metric=value)
+            # Clear accelerator state so we do not accumulate references between optimizer steps.
+            self.accelerator.clear(self.policy, self.policy_optimizer)
+            import gc
+            gc.collect()
+            torch.cuda.empty_cache()
+            return running_mean_logs
+    def get_advantages_with_critic_gradient_accumulation(
+        self, trajectories: TrajectoryBatch, critic_loss_scaling_factor: float = 2.0
+    ) -> torch.FloatTensor:
+        """
+        Compute (and optionally whiten) advantages while training the critic in mini-batches.
+        Uses GAE if enabled, otherwise uses Monte Carlo returns.
+        Optionally trains the critic if GAE is used.
+        Returns:
+            advantages: NestedFloatTensors
+        """
+        mb_size = self.mini_batch_size
+        batch_size = trajectories.rollout_ids.shape[0]
+        agent_id = trajectories.agent_ids[0]
+        batch_rewards = trajectories.batch_rewards
+        ######################################
+        # use critic for advantage estimation
+        ######################################
+        if self.use_gae:
+            if "buffer" in agent_id:
+                self.critic.eval()
+                training = False
+            else:
+                self.critic.train()
+                training = True
+            advantages = []
+            # critic_loss_scaling_factor comes learning single critic for two agents
+            normalization_factor = (
+                np.ceil(batch_size / mb_size).astype(int) * critic_loss_scaling_factor
+            )
+            # For each minibatch
+            for mb in range(0, batch_size, mb_size):
+                trajectory_mb = trajectories[mb : mb + mb_size]
+                trajectory_mb.to(self.device)
+                rewards_mb = trajectory_mb.batch_rewards
+                (
+                    tokens_mb,
+                    state_ends_mask_mb,
+                    timestep_counts,
+                ) = trajectory_mb.get_padded_tensors_for_critic()
+                # critic causal attention up to end flags
+                if training:
+                    vals_estimate_full = self.critic(tokens_mb)
+                else:
+                    with torch.no_grad():
+                        vals_estimate_full = self.critic(tokens_mb)
+                # if vals_estimate_full.dim() == 3:
+                #     vals_estimate_full = vals_estimate_full.squeeze(-1)
+                # Select only positions where states end, per sample → list of (jT,)
+                B = tokens_mb.shape[0]
+                vals_list = [
+                    vals_estimate_full[b][state_ends_mask_mb[b]] for b in range(B)
+                ]
+                # Pad to (B, max_jT) = (B, S)
+                vals_estimate_mb = pad_sequence(
+                    vals_list, batch_first=True, padding_value=0.0
+                )
+                dtype = vals_estimate_mb.dtype
+                rewards_mb = pad_sequence(
+                    rewards_mb, batch_first=True, padding_value=0.0
+                ).to(
+                    dtype=dtype
+                )  # (B, S)
+                self.rollout_tally.add_metric(
+                    path=["batch_rewards"],
+                    rollout_tally_item=RolloutTallyItem(
+                        crn_ids=trajectory_mb.crn_ids,
+                        rollout_ids=trajectory_mb.rollout_ids,
+                        agent_ids=trajectory_mb.agent_ids,
+                        metric_matrix=rewards_mb,
+                    ),
+                )
+                if self.reward_normalizing_constant != 1.0:
+                    rewards_mb /= self.reward_normalizing_constant
+                det_vals_estimate_mb = vals_estimate_mb.detach()  # (B, max_jT)
+                self.rollout_tally.add_metric(
+                    path=["mb_value_estimates_critic"],
+                    rollout_tally_item=RolloutTallyItem(
+                        crn_ids=trajectory_mb.crn_ids,
+                        rollout_ids=trajectory_mb.rollout_ids,
+                        agent_ids=trajectory_mb.agent_ids,
+                        metric_matrix=det_vals_estimate_mb,
+                    ),
+                )
+                # Append a 0 value to the end of the value estimates
+                if det_vals_estimate_mb.shape[1] == rewards_mb.shape[1]:
+                    Bsize = det_vals_estimate_mb.shape[0]
+                    device = det_vals_estimate_mb.device
+                    dtype = det_vals_estimate_mb.dtype
+                    det_vals_estimate_mb = torch.cat(
+                        [
+                            det_vals_estimate_mb,
+                            torch.zeros((Bsize, 1), device=device, dtype=dtype),
+                        ],
+                        dim=1,
+                    )  # (B, max_jT+1)
+                else:
+                    raise ValueError(
+                        "Incompatible shapes for value estimates and rewards."
+                    )
+                # Get annealed lambda
+                if self.use_gae_lambda_annealing:
+                    annealing_constant = self.gae_lambda_annealing_method(
+                        step=self.trainer_annealing_state.annealing_step_counter
+                    )
+                    annealed_lambda = (
+                        self.gae_lambda_annealing_limit * annealing_constant
+                    )
+                    self.tally.add_metric(
+                        path="annealed_lambda", metric=annealed_lambda
+                    )
+                else:
+                    annealed_lambda = self.gae_lambda_annealing_limit
+                # Get GAE advantages
+                gae_advantages = get_generalized_advantage_estimates(
+                    rewards=rewards_mb,
+                    value_estimates=det_vals_estimate_mb,
+                    discount_factor=self.discount_factor,
+                    lambda_coef=annealed_lambda,
+                )  # (B, max_jT)
+                self.rollout_tally.add_metric(
+                    path=["mb_gae_advantages"],
+                    rollout_tally_item=RolloutTallyItem(
+                        crn_ids=trajectory_mb.crn_ids,
+                        rollout_ids=trajectory_mb.rollout_ids,
+                        agent_ids=trajectory_mb.agent_ids,
+                        metric_matrix=gae_advantages,
+                    ),
+                )
+                if training:
+                    targets = (
+                        gae_advantages.to(dtype=dtype) + det_vals_estimate_mb[:, :-1]
+                    )  # (B, max_jT) # A(s, a, b) + V(s) = Q(s, a, b)
+                    self.rollout_tally.add_metric(
+                        path=["mb_targets_critic"],
+                        rollout_tally_item=RolloutTallyItem(
+                            crn_ids=trajectory_mb.crn_ids,
+                            rollout_ids=trajectory_mb.rollout_ids,
+                            agent_ids=trajectory_mb.agent_ids,
+                            metric_matrix=targets,
+                        ),
+                    )
+                    if self.critic_loss_type == "mse":
+                        loss = F.mse_loss(
+                            input=vals_estimate_mb,
+                            target=targets,
+                        )
+                    elif self.critic_loss_type == "huber":
+                        loss = F.huber_loss(
+                            input=vals_estimate_mb,
+                            target=targets,
+                        )
+                    self.tally.add_metric(path=["mb_critic_loss"], metric=loss.item())
+                    # Accumulate gradient
+                    loss /= normalization_factor
+                    self.accelerator.backward(loss)
+                    del loss
+                    del targets
+                    del vals_estimate_mb
+                del trajectory_mb
+                del vals_estimate_full
+                # Get jagged back using timestep_counts
+                advantages.extend(
+                    [gae_advantages[i, : timestep_counts[i]] for i in range(B)]
+                )
+        ######################################
+        # use exclusively Monte Carlo returns & rloo for advantage estimation
+        ######################################
+        else:
+            lengths = [len(c) for c in batch_rewards]
+            padded_rewards = pad_sequence(
+                batch_rewards, batch_first=True, padding_value=0.0
+            )
+            self.rollout_tally.add_metric(
+                path=["mb_rewards"],
+                rollout_tally_item=RolloutTallyItem(
+                    crn_ids=trajectories.crn_ids,
+                    rollout_ids=trajectories.rollout_ids,
+                    agent_ids=trajectories.agent_ids,
+                    metric_matrix=padded_rewards,
+                ),
+            )
+            if self.reward_normalizing_constant != 1.0:
+                padded_rewards /= self.reward_normalizing_constant
+            padded_advantages = get_discounted_returns(
+                rewards=padded_rewards,
+                discount_factor=self.discount_factor,
+            )  # no baseline for now
+            if self.use_rloo:
+                is_grouped_by_rng = (
+                    trajectories.crn_ids.unique().shape[0]
+                    != trajectories.crn_ids.shape[0]
+                )
+                if is_grouped_by_rng and not self.no_rloo_grouping:
+                    for crn_id in trajectories.crn_ids.unique():
+                        rng_mask = trajectories.crn_ids == crn_id
+                        rng_advantages = padded_advantages[rng_mask]
+                        rng_advantages, _ = get_rloo_credits(credits=rng_advantages)
+                        padded_advantages[rng_mask] = rng_advantages
+                else:
+                    padded_advantages, _ = get_rloo_credits(credits=padded_advantages)
+                self.rollout_tally.add_metric(
+                    path=["mb_rloo_advantages"],
+                    rollout_tally_item=RolloutTallyItem(
+                        crn_ids=trajectories.crn_ids,
+                        rollout_ids=trajectories.rollout_ids,
+                        agent_ids=trajectories.agent_ids,
+                        metric_matrix=padded_advantages,
+                    ),
+                )
+            advantages = [
+                padded_advantages[i, : lengths[i]]
+                for i in range(padded_advantages.shape[0])
+            ]
+        if self.whiten_advantages_time_step_wise or self.whiten_advantages:
+            lengths = [len(c) for c in advantages]
+            padded_advantages = pad_sequence(
+                advantages, batch_first=True, padding_value=0.0
+            )
+            if self.whiten_advantages_time_step_wise:
+                whitened_padded_advantages = whiten_advantages_time_step_wise(
+                    padded_advantages
+                )
+                path = ["mb_whitened_advantages_time_step_wise"]
+            elif self.whiten_advantages:
+                whitened_padded_advantages = whiten_advantages(padded_advantages)
+                path = ["mb_whitened_advantages"]
+            self.rollout_tally.add_metric(
+                path=path,
+                rollout_tally_item=RolloutTallyItem(
+                    crn_ids=trajectories.crn_ids,
+                    rollout_ids=trajectories.rollout_ids,
+                    agent_ids=trajectories.agent_ids,
+                    metric_matrix=whitened_padded_advantages,
+                ),
+            )
+            advantages = [
+                whitened_padded_advantages[i, : lengths[i]]
+                for i in range(whitened_padded_advantages.shape[0])
+            ]
+        self.trainer_annealing_state.annealing_step_counter += 1
+        return advantages
+    @abstractmethod
+    def set_agent_trajectory_data(
+        self, agent_id: str, roots: list[RolloutTreeRootNode]
+    ) -> None:
+        """
+        Populate self.training_data for a single agent using the provided rollout trees.
+        """
+        pass
+    def set_trajectory_data(
+        self, roots: list[RolloutTreeRootNode], agent_ids: list[str]
+    ) -> None:
+        """
+        Convenience wrapper to ingest trajectory data for every training agent.
+        """
+        for agent_id in agent_ids:
+            self.set_agent_trajectory_data(agent_id, roots)
+    @abstractmethod
+    def share_advantage_data(self) -> list[AdvantagePacket]:
+        pass
+    @abstractmethod
+    def receive_advantage_data(self, advantage_packets: list[AdvantagePacket]) -> None:
+        pass
+    def set_policy_gradient_data(self, agent_ids: list[str]) -> None:
+        """
+        Reset and rebuild the policy-gradient minibatches before iterating through agents.
+        """
+        self.policy_gradient_data = None
+        for agent_id in agent_ids:
+            assert "buffer" not in agent_id, "Buffer agents do not train policy"
+            trajectory_batch = self.training_data[agent_id]
+            tokenwise_batch_credits = get_tokenwise_credits(
+                batch_timesteps=trajectory_batch.batch_timesteps,
+                batch_credits=trajectory_batch.batch_credits,
+            )
+            policy_gradient_data = TrainingBatch(
+                rollout_ids=trajectory_batch.rollout_ids,
+                batch_input_ids=trajectory_batch.batch_input_ids,
+                batch_action_mask=trajectory_batch.batch_action_mask,
+                batch_entropy_mask=trajectory_batch.batch_entropy_mask,
+                batch_credits=tokenwise_batch_credits,
+                batch_engine_log_probs=trajectory_batch.batch_engine_log_probs,
+                batch_timesteps=trajectory_batch.batch_timesteps,
+            )
+            if self.policy_gradient_data is None:
+                self.policy_gradient_data = policy_gradient_data
+            else:
+                self.policy_gradient_data.append(policy_gradient_data)
+        self.training_data = {}
+        self.tokenwise_tally = ContextualizedTokenwiseTally(
+            tokenizer=self.tokenizer,
+            paths=self.debug_path_list,
+        )
+    def train(self) -> None:
+        """
+        Entry point for policy updates: prepare batches, compute gradients, and update parameters.
+        """
+        assert self.policy_gradient_data is not None, "Policy gradient data is not set"
+        if self.critic_optimizer is not None:
+            if self.gradient_clipping is not None:
+                grad_norm = self.accelerator.clip_grad_norm_(
+                    self.critic.parameters(), self.gradient_clipping
+                )
+                self.tally.add_metric(
+                    path="gradient_norm_critic", metric=grad_norm.item()
+                )
+            # Take step
+            self.critic_optimizer.step()
+            self.critic_optimizer.zero_grad()
+            self.accelerator.clear(self.critic, self.critic_optimizer)
+            import gc
+            gc.collect()
+            torch.cuda.empty_cache()
+        running_mean_logs = self.apply_reinforce_step(
+            training_batch=self.policy_gradient_data
+        )
+        return running_mean_logs
+    def export_training_tally(self, identifier: str, folder: str) -> None:
+        """
+        Saves and resets the collected training metrics using the tally object.
+        """
+        os.makedirs(folder, exist_ok=True)
+        self.tally.save(identifier=identifier, folder=folder)
+        self.tokenwise_tally.save(
+            path=os.path.join(folder, f"{identifier}_tokenwise.csv")
+        )
+        self.rollout_tally.save(identifier=identifier, folder=folder)
+        self.tally.reset()
+        self.tokenwise_tally = None
+        self.rollout_tally.reset()
+        self.debug_path_list = []
+    def export_optimizer_states(self) -> None:
+        """
+        Saves the optimizer states for both the main model and critic (if it exists).
+        """
+        try:
+            os.makedirs(self.save_path, exist_ok=True)
+            torch.save(self.policy_optimizer.state_dict(), self.policy_optimizer_path)
+            logger.info(f"Saved main optimizer state to {self.policy_optimizer_path}")
+            if self.critic_optimizer is not None:
+                torch.save(
+                    self.critic_optimizer.state_dict(), self.critic_optimizer_path
+                )
+                logger.info(
+                    f"Saved critic optimizer state to {self.critic_optimizer_path}"
+                )
+        except Exception as e:
+            logger.error(f"Error saving optimizer states: {str(e)}")
+            raise
+    def export_trainer_annealing_state(self) -> None:
+        """
+        Saves the trainer state.
+        """
+        with open(self.trainer_annealing_state_path, "wb") as f:
+            pickle.dump(self.trainer_annealing_state, f)
+        logger.info(f"Saved trainer state to {self.trainer_annealing_state_path}")
+    def export_trainer_states(self) -> None:
+        """
+        Saves the trainer states.
+        """
+        self.export_optimizer_states()
+        self.export_trainer_annealing_state()

src_code_for_reproducibility/utils/get_coagent_id.py ADDED Viewed

	@@ -0,0 +1,10 @@

+"""
+File: mllm/utils/get_coagent_id.py
+Summary: Helper for deriving co-agent identifiers from rollout metadata.
+"""
+def get_coagent_id(ids: list[str], agent_id: str) -> str | None:
+    for id in ids:
+        if id != agent_id:
+            return id

src_code_for_reproducibility/utils/get_stochastic_game_lengths.py ADDED Viewed

	@@ -0,0 +1,33 @@

+"""
+File: mllm/utils/get_stochastic_game_lengths.py
+Summary: Computes distributions over stochastic game lengths.
+"""
+import numpy as np
+def get_stochastic_game_lengths(
+    max_length, nb_games, continuation_prob, same_length_batch=False
+):
+    """
+    Generates stochastic game lengths based on a geometric distribution.
+    Args:
+        max_length (int): The maximum length a game can have.
+        nb_games (int): The number of games to generate lengths for.
+        continuation_prob (float): The probability of the game continuing after each round.
+        same_length_batch (bool): If True, all games will have the same length.
+    Returns:
+        Array: An array of game lengths.
+    """
+    if continuation_prob == 1:
+        return [max_length] * nb_games
+    if same_length_batch:
+        length = np.random.geometric(1 - continuation_prob, 1)
+        game_lengths = np.repeat(length, nb_games)
+    else:
+        game_lengths = np.random.geometric(1 - continuation_prob, nb_games)
+    game_lengths = np.where(game_lengths > max_length, max_length, game_lengths)
+    return game_lengths.tolist()

src_code_for_reproducibility/utils/resource_context.py ADDED Viewed

	@@ -0,0 +1,83 @@

+"""
+File: mllm/utils/resource_context.py
+Summary: Tracks system resource usage via a context manager.
+"""
+import logging
+import time
+from contextlib import contextmanager
+import torch
+def vram_usage():
+    output = ""
+    for i in range(torch.cuda.device_count()):
+        gpu_memory_allocated = torch.cuda.memory_allocated(i) / (
+            1024**3
+        )  # Convert bytes to GB
+        gpu_memory_reserved = torch.cuda.memory_reserved(i) / (
+            1024**3
+        )  # Convert bytes to GB
+        output += f"GPU {i}: Memory Allocated: {gpu_memory_allocated:.2f} GB, Memory Reserved: {gpu_memory_reserved:.2f} GB"
+    return output
+def ram_usage():
+    import psutil
+    process = psutil.Process()
+    memory_info = process.memory_info()
+    ram_used = memory_info.rss / (1024**3)  # Convert bytes to GB
+    return f"RAM Usage: {ram_used:.2f} GB"
+@contextmanager
+def resource_logger_context(logger: logging.Logger, task_description: str):
+    """
+    Context manager to log the resource usage of the current task.
+    Args:
+        logger: The logger to use to log the resource usage.
+        task_description: The description of the task to log.
+    Returns:
+        None
+    """
+    try:
+        initial_time = time.time()
+        # Assume CUDA is available and use device 0 only
+        total_mem_bytes = torch.cuda.get_device_properties(0).total_memory
+        initial_total_bytes = torch.cuda.memory_allocated(
+            0
+        ) + torch.cuda.memory_reserved(0)
+        torch.cuda.reset_peak_memory_stats(0)
+        yield None
+    finally:
+        final_time = time.time()
+        # Ensure kernels within the block are accounted for
+        torch.cuda.synchronize()
+        # Compute metrics
+        final_allocated_bytes = torch.cuda.memory_allocated(0)
+        final_reserved_bytes = torch.cuda.memory_reserved(0)
+        final_total_bytes = final_allocated_bytes + final_reserved_bytes
+        delta_vram_percent_total = (
+            100 * (final_total_bytes - initial_total_bytes) / total_mem_bytes
+            if total_mem_bytes
+            else 0.0
+        )
+        current_percent_vram_taken = (
+            100 * final_total_bytes / total_mem_bytes if total_mem_bytes else 0.0
+        )
+        block_peak_percent = (
+            100 * torch.cuda.max_memory_allocated(0) / total_mem_bytes
+            if total_mem_bytes
+            else 0.0
+        )
+        delta_time_str = time.strftime(
+            "%H:%M:%S", time.gmtime(final_time - initial_time)
+        )
+        logger.info(
+            f"For task: {task_description}, ΔVRAM % (total): {delta_vram_percent_total:.2f}%, Current % of VRAM taken: {current_percent_vram_taken:.2f}%, Block Peak % of device VRAM: {block_peak_percent:.2f}%, ΔTime: {delta_time_str}"
+        )

src_code_for_reproducibility/utils/rollout_tree_chat_htmls.py ADDED Viewed

	@@ -0,0 +1,1597 @@

+"""
+File: mllm/utils/rollout_tree_chat_htmls.py
+Summary: Renders rollout tree chat transcripts into HTML artifacts.
+"""
+from pathlib import Path
+from typing import List
+from mllm.utils.rollout_tree_gather_utils import *
+def html_from_chat_turns(chat_turns: List[ChatTurnLog]) -> str:
+    """
+    Render chat turns as a single, wrapping sequence of messages in time order.
+    Keep badge and message bubble styles, include time on every badge and
+    include rewards on assistant badges. Each message is individually
+    hide/show by click; when hidden, only the badge remains and "(...)" is
+    shown inline (not inside a bubble).
+    """
+    import html
+    import re as _re
+    # Prepare ordering: sort by (time_step, original_index) to keep stable order within same step
+    indexed_turns = list(enumerate(chat_turns))
+    indexed_turns.sort(key=lambda t: (t[1].time_step, t[0]))
+    # Get unique agent IDs and sort alphabetically for consistent assignment
+    # Agent with alphabetically lower name gets agent-0 (left, green)
+    # Agent with alphabetically higher name gets agent-1 (right, orange)
+    unique_agent_ids = sorted(
+        set(turn.agent_id for turn in chat_turns if turn.role == "assistant")
+    )
+    agent_id_to_index = {aid: idx for idx, aid in enumerate(unique_agent_ids)}
+    # CSS styles (simplified layout; no time-step or agent-column backgrounds)
+    css = """
+    <style>
+        :root {
+            --font-family: 'Segoe UI', Tahoma, Geneva, Verdana, sans-serif;
+            --bg: #ffffff;
+            --text: #1c0b00;
+            --muted-text: #2C3E50;
+            --accent-muted: #BDC3C7;
+            --accent-muted-2: #D0D7DE;
+            --panel-bg: #F8FAFC;
+            --reward-color: #3a2e00; /* dark text for reward pill */
+            --font-size: 14px;
+            --border-width: 2px;
+            --corner-radius: 6px;
+            --pill-radius-left: 999px 0 0 999px;
+            --pill-radius-right: 0 999px 999px 0;
+            --inset-shadow: 0 1px 0 rgba(0,0,0,0.03) inset;
+            /* Chat View Colors */
+            --agent-0-bg: #dcf8c6;
+            --agent-0-border: #0eb224;
+            --agent-1-bg: #ffe4cc;
+            --agent-1-border: #ef8323;
+            --user-bg: #f5f5f5;
+            --chat-bg: #ffffff;
+        }
+        body {
+            font-family: var(--font-family);
+            margin: 12px;
+            background-color: var(--bg);
+            color: var(--text);
+            font-size: var(--font-size);
+            line-height: 1.5;
+        }
+        /* Chat View Styles */
+        #flow-chat {
+            max-width: 900px;
+            margin: 0 auto;
+            background: var(--chat-bg);
+            padding: 12px 16px 12px 8px;
+            border-radius: 8px;
+        }
+        .simultaneous-messages {
+            display: flex !important;
+            flex-direction: row !important;
+            flex-wrap: nowrap !important;
+            gap: 8px;
+            margin-bottom: 4px;
+            align-items: flex-start;
+            width: 100%;
+            overflow: hidden;
+            box-sizing: border-box;
+        }
+        .simultaneous-messages .chat-message {
+            flex: 1 1 0 !important;
+            margin-bottom: 0 !important;
+            display: flex !important;
+            flex-direction: row !important;
+            align-items: flex-start !important;
+            margin-left: 0 !important;
+            min-width: 0 !important;
+            max-width: 50% !important;
+            gap: 0 !important;
+            overflow: hidden !important;
+        }
+        .simultaneous-messages .chat-message-content {
+            max-width: 100% !important;
+            width: 100%;
+            align-items: flex-start !important;
+            margin-left: 0 !important;
+            overflow: hidden !important;
+        }
+        .simultaneous-messages .chat-message.agent-0 {
+            justify-content: flex-start !important;
+        }
+        .simultaneous-messages .chat-message.agent-1 {
+            justify-content: flex-end !important;
+        }
+        .simultaneous-messages .chat-message.agent-0 .chat-message-content {
+            margin-left: 0 !important;
+            align-items: flex-start !important;
+        }
+        .simultaneous-messages .chat-message.agent-1 .chat-message-content {
+            margin-left: auto !important;
+            margin-right: 0 !important;
+            align-items: flex-end !important;
+        }
+        .simultaneous-messages .chat-bubble {
+            max-width: 100%;
+            word-break: break-word;
+            overflow-wrap: break-word;
+            box-sizing: border-box;
+        }
+        .simultaneous-messages .chat-message.agent-0 .chat-bubble {
+            border-radius: 10px;
+        }
+        .simultaneous-messages .chat-message.agent-1 .chat-bubble {
+            border-radius: 10px;
+        }
+        .simultaneous-messages .chat-message.agent-0 .chat-header {
+            justify-content: flex-start;
+            flex-shrink: 0;
+        }
+        .simultaneous-messages .chat-message.agent-1 .chat-header {
+            justify-content: flex-end;
+            flex-shrink: 0;
+        }
+        .simultaneous-messages .chat-reasoning {
+            max-width: 100%;
+            overflow-wrap: break-word;
+        }
+        /* Styling for user prompts in simultaneous-messages */
+        .simultaneous-messages .chat-message.role-user {
+            flex: 1 1 0 !important;
+            margin-bottom: 0 !important;
+            display: flex !important;
+            opacity: 0.7;
+            cursor: pointer;
+        }
+        .simultaneous-messages .chat-message.role-user:hover {
+            opacity: 1;
+        }
+        .simultaneous-messages .chat-message.role-user.collapsed .chat-bubble {
+            display: none;
+        }
+        .simultaneous-messages .chat-message.role-user.collapsed .chat-header::after {
+            content: ' (collapsed)';
+            font-weight: normal;
+            font-style: italic;
+            color: #999;
+            font-size: 0.9em;
+        }
+        .simultaneous-messages .chat-message.role-user.agent-0 {
+            justify-content: flex-start !important;
+        }
+        .simultaneous-messages .chat-message.role-user.agent-1 {
+            justify-content: flex-end !important;
+        }
+        .simultaneous-messages .chat-message.role-user.agent-0 .chat-message-content {
+            margin-left: 0 !important;
+            align-items: flex-start !important;
+        }
+        .simultaneous-messages .chat-message.role-user.agent-1 .chat-message-content {
+            margin-left: auto !important;
+            margin-right: 0 !important;
+            align-items: flex-end !important;
+        }
+        /* Styling for split-agent-context when wrapped */
+        .simultaneous-messages .split-agent-context {
+            width: 100%;
+            display: flex !important;
+        }
+        .chat-message {
+            display: flex;
+            margin-bottom: 2px;
+            align-items: flex-end;
+            gap: 6px;
+            position: relative;
+            margin-left: 36px;
+        }
+        .chat-message.agent-0 {
+            margin-left: 0;
+        }
+        .chat-message.agent-1 {
+            margin-left: 0;
+        }
+        .chat-message.agent-0::before {
+            left: 0;
+        }
+        .chat-message.agent-1::before {
+            left: 0;
+        }
+        .chat-message.role-user {
+            opacity: 0.7;
+            cursor: pointer;
+        }
+        .chat-message.role-user.collapsed .chat-bubble {
+            display: none;
+        }
+        .chat-message.role-user.collapsed .chat-header::after {
+            content: ' (collapsed)';
+            font-weight: normal;
+            font-style: italic;
+            color: #999;
+            font-size: 0.9em;
+        }
+        .chat-message.role-user:hover {
+            opacity: 1;
+        }
+        .chat-message::before {
+            content: '';
+            position: absolute;
+            left: -36px;
+            top: 0;
+            bottom: 0;
+            width: 36px;
+            pointer-events: auto;
+        }
+        .merge-btn {
+            position: absolute;
+            left: -30px;
+            top: 50%;
+            transform: translateY(-50%);
+            width: 26px;
+            height: 26px;
+            border-radius: 4px;
+            border: 1.5px solid var(--accent-muted);
+            background: white;
+            cursor: pointer;
+            font-size: var(--font-size);
+            opacity: 0;
+            display: flex;
+            align-items: center;
+            justify-content: center;
+            transition: opacity 0.2s ease, transform 0.1s ease;
+            padding: 0;
+            line-height: 1;
+            z-index: 10;
+        }
+        .chat-message:hover .merge-btn,
+        .merge-btn:hover {
+            opacity: 1;
+        }
+        .merge-btn:hover {
+            background: var(--panel-bg);
+            border-color: var(--accent-muted-2);
+            transform: translateY(-50%) scale(1.15);
+            box-shadow: 0 2px 4px rgba(0, 0, 0, 0.15);
+        }
+        .merge-btn:active {
+            transform: translateY(-50%) scale(0.95);
+        }
+        .chat-message.agent-0 .merge-btn {
+            left: -30px;
+        }
+        .chat-message.agent-1 .merge-btn {
+            left: -30px;
+        }
+        .chat-message.role-user .merge-btn {
+            display: none !important;
+        }
+        .simultaneous-messages .merge-btn {
+            opacity: 0 !important;
+            pointer-events: none;
+        }
+        .simultaneous-messages {
+            padding: 6px 0 6px 0 !important;
+            margin-left: 0 !important;
+            margin-right: 0 !important;
+            position: relative !important;
+            background: transparent !important;
+            border-radius: 0 !important;
+            box-sizing: border-box !important;
+            overflow: visible !important;
+            max-width: 100% !important;
+            border: none !important;
+            transition: padding 0.2s ease !important;
+        }
+        .simultaneous-messages:hover {
+            padding-top: 40px !important;
+        }
+        .simultaneous-messages::before {
+            content: '⇅ Merged';
+            position: absolute;
+            left: 0 !important;
+            top: 8px !important;
+            font-size: var(--font-size);
+            font-weight: 500;
+            color: #888;
+            pointer-events: none;
+            opacity: 0;
+            transition: opacity 0.2s ease;
+        }
+        .simultaneous-messages:hover::before {
+            opacity: 1;
+        }
+        .unmerge-btn {
+            position: absolute !important;
+            right: 0 !important;
+            top: 6px !important;
+            width: 36px !important;
+            height: 28px !important;
+            border-radius: 5px !important;
+            border: 2px solid #d63031 !important;
+            background: white !important;
+            cursor: pointer !important;
+            font-size: var(--font-size) !important;
+            font-weight: bold !important;
+            color: #d63031 !important;
+            display: flex !important;
+            align-items: center !important;
+            justify-content: center !important;
+            transition: all 0.2s ease !important;
+            padding: 0 !important;
+            line-height: 1 !important;
+            z-index: 1000 !important;
+            flex: none !important;
+            pointer-events: auto !important;
+            box-shadow: 0 2px 6px rgba(214, 48, 49, 0.3) !important;
+            opacity: 0 !important;
+        }
+        .simultaneous-messages:hover .unmerge-btn {
+            opacity: 1 !important;
+        }
+        .unmerge-btn:hover {
+            background: #ffe5e5 !important;
+            border-color: #b71c1c !important;
+            transform: scale(1.1) !important;
+            box-shadow: 0 3px 8px rgba(214, 48, 49, 0.4) !important;
+        }
+        .unmerge-btn:active {
+            transform: scale(0.95) !important;
+            background: #ffcccc !important;
+        }
+        .chat-message-content {
+            max-width: 72%;
+            display: flex;
+            flex-direction: column;
+            gap: 2px;
+        }
+        .chat-message.agent-0 .chat-message-content {
+            align-items: flex-start;
+        }
+        .chat-message.agent-1 .chat-message-content {
+            align-items: flex-end;
+            margin-left: auto;
+        }
+        .chat-bubble {
+            padding: 6px 10px;
+            border-radius: 10px;
+            word-wrap: break-word;
+            position: relative;
+            box-shadow: 0 1px 3px rgba(0, 0, 0, 0.1);
+            line-height: 1.4;
+        }
+        .chat-message.agent-0 .chat-bubble {
+            background: var(--agent-0-bg);
+            border: 2px solid var(--agent-0-border);
+            border-radius: 10px 10px 10px 2px;
+        }
+        .chat-message.agent-1 .chat-bubble {
+            background: var(--agent-1-bg);
+            border: 2px solid var(--agent-1-border);
+            border-radius: 10px 10px 2px 10px;
+        }
+        .chat-message.role-user .chat-bubble {
+            background: var(--user-bg);
+            border: 2px solid #d0d0d0;
+        }
+        .chat-header {
+            display: flex;
+            align-items: center;
+            gap: 4px;
+            margin-bottom: 2px;
+            font-size: var(--font-size);
+            font-weight: 600;
+            line-height: 1.2;
+        }
+        .chat-message.agent-0 .chat-header {
+            color: var(--agent-0-border);
+        }
+        .chat-message.agent-1 .chat-header {
+            color: var(--agent-1-border);
+        }
+        .chat-timestamp {
+            font-size: var(--font-size);
+            color: var(--muted-text);
+            margin-top: 1px;
+            opacity: 0.75;
+        }
+        .chat-reward {
+            display: inline-flex;
+            align-items: center;
+            background: linear-gradient(90deg, #fffdf2 0%, #ffffff 75%);
+            color: #000000;
+            font-weight: 600;
+            font-size: var(--font-size);
+            padding: 1px 5px;
+            border-radius: 3px;
+            border: 1px solid #f4e6a8;
+            margin-left: 4px;
+            line-height: 1.3;
+        }
+        .chat-reasoning {
+            font-size: var(--font-size);
+            font-style: italic;
+            color: #555;
+            margin-bottom: 2px;
+            padding: 4px 8px;
+            background: rgba(0, 0, 0, 0.03);
+            border-radius: 5px;
+            cursor: pointer;
+            line-height: 1.3;
+        }
+        .chat-reasoning.collapsed .reasoning-text {
+            display: none;
+        }
+        .chat-reasoning.collapsed::after {
+            content: ' (click to expand)';
+            color: #777;
+        }
+        .chat-group-divider {
+            display: flex;
+            align-items: center;
+            gap: 8px;
+            width: 100%;
+            margin: 8px 0 4px 0;
+            position: relative;
+            cursor: pointer;
+            user-select: none;
+        }
+        .chat-group-divider::before,
+        .chat-group-divider::after {
+            content: "";
+            flex: 1 1 auto;
+            height: 2px;
+            background: linear-gradient(90deg, rgba(224,230,235,0), var(--accent-muted-2) 30%, var(--accent-muted-2) 70%, rgba(224,230,235,0));
+        }
+        .chat-group-label {
+            display: inline-block;
+            background: white;
+            padding: 2px 12px;
+            border-radius: 999px;
+            font-size: var(--font-size);
+            font-weight: 700;
+            color: var(--muted-text);
+            border: 1.5px solid var(--accent-muted);
+            box-shadow: 0 1px 3px rgba(0, 0, 0, 0.08);
+            line-height: 1.4;
+            position: relative;
+            transition: background 0.2s ease;
+        }
+        .chat-group-divider:hover .chat-group-label {
+            background: var(--panel-bg);
+        }
+        .chat-group-label::before {
+            content: '▼ ';
+            font-size: 0.8em;
+            display: inline-block;
+            transition: transform 0.2s ease;
+            opacity: 0;
+        }
+        .chat-group-divider:hover .chat-group-label::before {
+            opacity: 1;
+        }
+        .chat-group-divider.collapsed .chat-group-label::before {
+            content: '▶ ';
+            opacity: 1;
+        }
+        .chat-group-divider.collapsed + * {
+            display: none !important;
+        }
+        /* Hide collapsed rounds in strong hide mode */
+        .strong-hide .chat-group-divider.collapsed {
+            display: none !important;
+        }
+        /* Chat view width control */
+        #flow-chat {
+            --chat-width: 900px;
+            max-width: var(--chat-width);
+            margin: 0 auto;
+        }
+        /* Hide user messages when toggle is on */
+        #flow-chat.hide-user-messages .chat-message.role-user {
+            display: none;
+        }
+        /* Hide rewards when hiding user messages */
+        #flow-chat.hide-user-messages .chat-reward {
+            display: none;
+        }
+        /* Round context annotations */
+        .round-context {
+            text-align: center;
+            margin: 4px auto;
+            max-width: 100%;
+        }
+        .round-context-edit {
+            min-height: 20px;
+            padding: 5px 10px;
+            border: 1.5px dashed var(--accent-muted);
+            border-radius: 6px;
+            background: #fafafa;
+            cursor: text;
+            transition: all 0.2s ease;
+            outline: none;
+            font-size: var(--font-size);
+            line-height: 1.3;
+            user-select: text;
+            -webkit-user-select: text;
+            -moz-user-select: text;
+            -ms-user-select: text;
+        }
+        .round-context-edit:focus {
+            border-style: solid;
+            border-color: var(--accent-muted-2);
+            background: #ffffff;
+            box-shadow: 0 2px 8px rgba(0, 0, 0, 0.1);
+        }
+        .round-context-edit:empty:before {
+            content: attr(data-placeholder);
+            color: #999;
+            font-style: italic;
+        }
+        .round-context-controls {
+            display: none;
+            justify-content: center;
+            gap: 4px;
+            margin-top: 4px;
+            flex-wrap: wrap;
+        }
+        .round-context-edit:focus + .round-context-controls,
+        .round-context-controls:hover,
+        .round-context:focus-within .round-context-controls {
+            display: flex;
+        }
+        .context-color-btn {
+            width: 22px;
+            height: 22px;
+            border-radius: 50%;
+            border: 1.5px solid #fff;
+            box-shadow: 0 1px 2px rgba(0, 0, 0, 0.15);
+            cursor: pointer;
+            transition: transform 0.1s ease;
+        }
+        .context-color-btn:hover {
+            transform: scale(1.15);
+        }
+        .context-color-btn:active {
+            transform: scale(0.95);
+        }
+        /* Split agent context boxes */
+        .split-agent-context {
+            display: flex;
+            gap: 6px;
+            margin: 4px auto;
+            max-width: 100%;
+            align-items: flex-start;
+        }
+        .agent-context-box {
+            flex: 1;
+            min-width: 0;
+            position: relative;
+        }
+        .agent-context-box .round-context-edit {
+            margin: 0;
+            border-radius: 6px;
+            padding: 4px 8px;
+            min-height: 18px;
+        }
+        .agent-context-box.agent-0 .round-context-edit {
+            border-color: var(--agent-0-border);
+            background: rgba(14, 178, 36, 0.03);
+        }
+        .agent-context-box.agent-1 .round-context-edit {
+            border-color: var(--agent-1-border);
+            background: rgba(239, 131, 35, 0.03);
+        }
+        .agent-context-box.agent-0 .round-context-edit:focus {
+            border-color: var(--agent-0-border);
+            box-shadow: 0 2px 8px rgba(14, 178, 36, 0.2);
+            background: rgba(14, 178, 36, 0.05);
+        }
+        .agent-context-box.agent-1 .round-context-edit:focus {
+            border-color: var(--agent-1-border);
+            box-shadow: 0 2px 8px rgba(239, 131, 35, 0.2);
+            background: rgba(239, 131, 35, 0.05);
+        }
+        .agent-context-box .round-context-edit::before {
+            font-weight: 700;
+            font-size: var(--font-size);
+            margin-right: 5px;
+            letter-spacing: 0.2px;
+        }
+        .agent-context-box.agent-0 .round-context-edit::before {
+            content: 'Agent 0 Prompt Summary:';
+            color: var(--agent-0-border);
+        }
+        .agent-context-box.agent-1 .round-context-edit::before {
+            content: 'Agent 1 Prompt Summary:';
+            color: var(--agent-1-border);
+        }
+        /* Empty context boxes will be hidden by JavaScript when strong hide is enabled */
+        .toolbar {
+            display: flex;
+            align-items: center;
+            gap: 8px;
+            margin-bottom: 0;
+            font-size: var(--font-size);
+            max-height: 0;
+            overflow: hidden;
+            opacity: 0;
+            pointer-events: none;
+            transition: max-height 0.2s ease, opacity 0.2s ease;
+            flex-wrap: wrap;
+        }
+        .toolbar-wrap { position: sticky; top: 0; z-index: 10; background: var(--bg); }
+        .toolbar-hotzone { height: 6px; }
+        .toolbar-wrap:hover .toolbar { max-height: 500px; opacity: 1; pointer-events: auto; margin-bottom: 12px; }
+        .toolbar * { pointer-events: auto !important; }
+        .toolbar input,
+        .toolbar select { z-index: 100 !important; position: relative; }
+        .toolbar input[type="number"],
+        .toolbar input[type="text"],
+        .toolbar select {
+            width: 72px;
+            padding: 2px 6px;
+            border: 1px solid var(--accent-muted);
+            border-radius: var(--corner-radius);
+            background: var(--bg);
+            user-select: text !important;
+            -webkit-user-select: text !important;
+            -moz-user-select: text !important;
+            -ms-user-select: text !important;
+            pointer-events: auto !important;
+            cursor: pointer !important;
+        }
+        .toolbar input[type="text"] {
+            cursor: text !important;
+        }
+        .toolbar input[type="text"]:focus,
+        .toolbar input[type="number"]:focus,
+        .toolbar select:focus {
+            outline: 2px solid #0066cc;
+            outline-offset: 1px;
+        }
+        .toolbar button {
+            padding: 4px 8px;
+            border: 1px solid var(--accent-muted);
+            background: var(--panel-bg);
+            border-radius: var(--corner-radius);
+            cursor: pointer;
+        }
+        .emoji-bw { filter: grayscale(100%); opacity: 0.95; font-size: var(--font-size); vertical-align: baseline; margin: 0; position: relative; top: -1px; line-height: 1; display: inline-block; }
+    </style>
+    """
+    # HTML structure
+    html_parts = [
+        "<!DOCTYPE html>",
+        "<html>",
+        "<head>",
+        "<meta charset='UTF-8'>",
+        "<title>Chat Turns</title>",
+        css,
+        "<script>\n"
+        "document.addEventListener('DOMContentLoaded', function() {\n"
+        "  const chatFlow = document.getElementById('flow-chat');\n"
+        "  let strongHideOn = false;\n"
+        "  let hideUserMessages = false;\n"
+        "  const hideUserBtn = document.getElementById('toggle-hide-user-messages');\n"
+        "  const hideUserStateEl = document.getElementById('hide-user-state');\n"
+        "  const widthControl = document.getElementById('chat-width-control');\n"
+        "  const widthSlider = document.getElementById('chat-width-slider');\n"
+        "  const widthValue = document.getElementById('chat-width-value');\n"
+        "  const strongHideBtn = document.getElementById('toggle-strong-hide');\n"
+        "  const strongHideStateEl = document.getElementById('strong-hide-state');\n"
+        "  if (strongHideBtn) {\n"
+        "    const setLabel = () => { if (strongHideStateEl) { strongHideStateEl.textContent = strongHideOn ? 'On' : 'Off'; } };\n"
+        "    strongHideBtn.addEventListener('click', () => { strongHideOn = !strongHideOn; chatFlow.classList.toggle('strong-hide', strongHideOn); setLabel(); applyStrongHideToChat(); });\n"
+        "    setLabel();\n"
+        "  }\n"
+        "  if (hideUserBtn && hideUserStateEl && chatFlow) {\n"
+        "    const updateHideUser = () => { hideUserStateEl.textContent = hideUserMessages ? 'On' : 'Off'; };\n"
+        "    hideUserBtn.addEventListener('click', () => {\n"
+        "      hideUserMessages = !hideUserMessages;\n"
+        "      chatFlow.classList.toggle('hide-user-messages', hideUserMessages);\n"
+        "      updateHideUser();\n"
+        "    });\n"
+        "    updateHideUser();\n"
+        "  }\n"
+        "  if (widthSlider && widthValue && chatFlow) {\n"
+        "    const savedWidth = localStorage.getItem('chat-view-width');\n"
+        "    if (savedWidth) {\n"
+        "      widthSlider.value = savedWidth;\n"
+        "      chatFlow.style.setProperty('--chat-width', savedWidth + 'px');\n"
+        "      widthValue.textContent = savedWidth + 'px';\n"
+        "    }\n"
+        "    widthSlider.addEventListener('input', (e) => {\n"
+        "      const width = e.target.value;\n"
+        "      chatFlow.style.setProperty('--chat-width', width + 'px');\n"
+        "      widthValue.textContent = width + 'px';\n"
+        "      localStorage.setItem('chat-view-width', width);\n"
+        "    });\n"
+        "  }\n"
+        "  const fontFamilySelect = document.getElementById('font-family-select');\n"
+        "  const fontSizeInput = document.getElementById('font-size-input');\n"
+        "  if (fontFamilySelect) {\n"
+        "    const savedFont = localStorage.getItem('render-font-family');\n"
+        "    if (savedFont) {\n"
+        "      fontFamilySelect.value = savedFont;\n"
+        "      document.body.style.setProperty('--font-family', savedFont);\n"
+        "    }\n"
+        "    fontFamilySelect.addEventListener('change', (e) => {\n"
+        "      const font = e.target.value;\n"
+        "      document.body.style.setProperty('--font-family', font);\n"
+        "      localStorage.setItem('render-font-family', font);\n"
+        "    });\n"
+        "  }\n"
+        "  if (fontSizeInput) {\n"
+        "    const savedSize = localStorage.getItem('render-font-size');\n"
+        "    if (savedSize) {\n"
+        "      fontSizeInput.value = savedSize;\n"
+        "      document.body.style.setProperty('--font-size', savedSize + 'px');\n"
+        "    }\n"
+        "    fontSizeInput.addEventListener('input', (e) => {\n"
+        "      const size = e.target.value;\n"
+        "      document.body.style.setProperty('--font-size', size + 'px');\n"
+        "      localStorage.setItem('render-font-size', size);\n"
+        "    });\n"
+        "  }\n"
+        "  const agent0EmojiInput = document.getElementById('agent0-emoji-input');\n"
+        "  const agent0NameInput = document.getElementById('agent0-name-input');\n"
+        "  const agent1EmojiInput = document.getElementById('agent1-emoji-input');\n"
+        "  const agent1NameInput = document.getElementById('agent1-name-input');\n"
+        "  const applyAgentNamesBtn = document.getElementById('apply-agent-names');\n"
+        "  function loadAgentNames() {\n"
+        "    if (agent0EmojiInput && agent0NameInput && agent1EmojiInput && agent1NameInput) {\n"
+        "      const savedAgent0Emoji = localStorage.getItem('agent0-emoji') || '🤖';\n"
+        "      const savedAgent0Name = localStorage.getItem('agent0-name') || document.getElementById('agent0-name-input').placeholder;\n"
+        "      const savedAgent1Emoji = localStorage.getItem('agent1-emoji') || '🤖';\n"
+        "      const savedAgent1Name = localStorage.getItem('agent1-name') || document.getElementById('agent1-name-input').placeholder;\n"
+        "      agent0EmojiInput.value = savedAgent0Emoji;\n"
+        "      agent0NameInput.value = savedAgent0Name;\n"
+        "      agent1EmojiInput.value = savedAgent1Emoji;\n"
+        "      agent1NameInput.value = savedAgent1Name;\n"
+        "      applyAgentNamesToDOM(savedAgent0Emoji, savedAgent0Name, savedAgent1Emoji, savedAgent1Name);\n"
+        "    }\n"
+        "  }\n"
+        "  function applyAgentNamesToDOM(agent0Emoji, agent0Name, agent1Emoji, agent1Name) {\n"
+        "    const agentMap = { '0': { name: agent0Name, emoji: agent0Emoji }, '1': { name: agent1Name, emoji: agent1Emoji } };\n"
+        "    document.querySelectorAll('[data-agent-index]').forEach(el => {\n"
+        "      const agentIndex = el.getAttribute('data-agent-index');\n"
+        "      if (!agentMap[agentIndex]) return;\n"
+        "      if (el.classList.contains('agent-name')) {\n"
+        "        el.textContent = agentMap[agentIndex].name;\n"
+        "      } else if (el.classList.contains('emoji-bw')) {\n"
+        "        const currentEmoji = el.textContent.trim();\n"
+        "        if (currentEmoji === '🤖' || currentEmoji === '👤') {\n"
+        "          el.textContent = agentMap[agentIndex].emoji;\n"
+        "        }\n"
+        "      }\n"
+        "    });\n"
+        "    const style = document.createElement('style');\n"
+        "    style.id = 'dynamic-agent-names-style';\n"
+        "    const existingStyle = document.getElementById('dynamic-agent-names-style');\n"
+        "    if (existingStyle) existingStyle.remove();\n"
+        "    style.textContent = `\n"
+        "      .agent-context-box.agent-0 .round-context-edit::before {\n"
+        "        content: '${agent0Name} Prompt Summary:';\n"
+        "      }\n"
+        "      .agent-context-box.agent-1 .round-context-edit::before {\n"
+        "        content: '${agent1Name} Prompt Summary:';\n"
+        "      }\n"
+        "    `;\n"
+        "    document.head.appendChild(style);\n"
+        "  }\n"
+        "  if (applyAgentNamesBtn && agent0EmojiInput && agent0NameInput && agent1EmojiInput && agent1NameInput) {\n"
+        "    [agent0EmojiInput, agent0NameInput, agent1EmojiInput, agent1NameInput].forEach(input => {\n"
+        "      input.style.pointerEvents = 'auto';\n"
+        "      if (input.tagName === 'INPUT') {\n"
+        "        input.style.userSelect = 'text';\n"
+        "        input.style.webkitUserSelect = 'text';\n"
+        "        input.readOnly = false;\n"
+        "      }\n"
+        "      input.disabled = false;\n"
+        "      const stopAll = (e) => { e.stopPropagation(); e.stopImmediatePropagation(); };\n"
+        "      input.addEventListener('mousedown', stopAll, true);\n"
+        "      input.addEventListener('mouseup', stopAll, true);\n"
+        "      input.addEventListener('click', stopAll, true);\n"
+        "      input.addEventListener('dblclick', stopAll, true);\n"
+        "      input.addEventListener('focus', stopAll, true);\n"
+        "      input.addEventListener('blur', stopAll, true);\n"
+        "      input.addEventListener('paste', stopAll, true);\n"
+        "      input.addEventListener('cut', stopAll, true);\n"
+        "      input.addEventListener('copy', stopAll, true);\n"
+        "      input.addEventListener('select', stopAll, true);\n"
+        "      input.addEventListener('selectstart', stopAll, true);\n"
+        "      input.addEventListener('keydown', stopAll, true);\n"
+        "      input.addEventListener('keyup', stopAll, true);\n"
+        "      input.addEventListener('keypress', stopAll, true);\n"
+        "      input.addEventListener('input', stopAll, true);\n"
+        "      input.addEventListener('change', stopAll, true);\n"
+        "      input.addEventListener('contextmenu', stopAll, true);\n"
+        "    });\n"
+        "    const applyNames = () => {\n"
+        "      const agent0Emoji = agent0EmojiInput.value || '🤖';\n"
+        "      const agent0Name = agent0NameInput.value.trim() || agent0NameInput.placeholder;\n"
+        "      const agent1Emoji = agent1EmojiInput.value || '🤖';\n"
+        "      const agent1Name = agent1NameInput.value.trim() || agent1NameInput.placeholder;\n"
+        "      localStorage.setItem('agent0-emoji', agent0Emoji);\n"
+        "      localStorage.setItem('agent0-name', agent0Name);\n"
+        "      localStorage.setItem('agent1-emoji', agent1Emoji);\n"
+        "      localStorage.setItem('agent1-name', agent1Name);\n"
+        "      applyAgentNamesToDOM(agent0Emoji, agent0Name, agent1Emoji, agent1Name);\n"
+        "    };\n"
+        "    applyAgentNamesBtn.addEventListener('click', applyNames);\n"
+        "    [agent0NameInput, agent1NameInput].forEach(input => {\n"
+        "      input.addEventListener('keydown', (e) => {\n"
+        "        if (e.key === 'Enter') {\n"
+        "          e.preventDefault();\n"
+        "          e.stopPropagation();\n"
+        "          e.stopImmediatePropagation();\n"
+        "          applyNames();\n"
+        "        }\n"
+        "      }, true);\n"
+        "    });\n"
+        "    [agent0EmojiInput, agent1EmojiInput].forEach(select => {\n"
+        "      select.addEventListener('change', applyNames);\n"
+        "    });\n"
+        "  }\n"
+        "  loadAgentNames();\n"
+        "  function setupRoundCollapse() {\n"
+        "    document.addEventListener('click', function(e) {\n"
+        "      if (e.target.closest('input, textarea, select, button, .round-context-edit, .toolbar')) { return; }\n"
+        "      const divider = e.target.closest('.chat-group-divider, .group-divider');\n"
+        "      if (!divider) return;\n"
+        "      divider.classList.toggle('collapsed');\n"
+        "      const isCollapsed = divider.classList.contains('collapsed');\n"
+        "      let nextElement = divider.nextElementSibling;\n"
+        "      while (nextElement) {\n"
+        "        if (nextElement.classList.contains('chat-group-divider') || nextElement.classList.contains('group-divider')) {\n"
+        "          break;\n"
+        "        }\n"
+        "        if (isCollapsed) {\n"
+        "          if (!nextElement.dataset.originalDisplay) {\n"
+        "            nextElement.dataset.originalDisplay = nextElement.style.display || getComputedStyle(nextElement).display;\n"
+        "          }\n"
+        "          nextElement.style.display = 'none';\n"
+        "        } else {\n"
+        "          if (nextElement.dataset.originalDisplay) {\n"
+        "            const originalDisplay = nextElement.dataset.originalDisplay;\n"
+        "            nextElement.style.display = originalDisplay === 'none' ? '' : originalDisplay;\n"
+        "            if (nextElement.style.display === originalDisplay && originalDisplay !== 'none') {\n"
+        "              nextElement.style.display = '';\n"
+        "            }\n"
+        "            delete nextElement.dataset.originalDisplay;\n"
+        "          } else {\n"
+        "            nextElement.style.display = '';\n"
+        "          }\n"
+        "        }\n"
+        "        nextElement = nextElement.nextElementSibling;\n"
+        "      }\n"
+        "      e.stopPropagation();\n"
+        "    });\n"
+        "  }\n"
+        "  setupRoundCollapse();\n"
+        "  const strongHideBtnChat = document.getElementById('toggle-strong-hide');\n"
+        "  function applyStrongHideToChat() {\n"
+        "    if (!chatFlow) return;\n"
+        "    chatFlow.classList.toggle('strong-hide', strongHideOn);\n"
+        "    const contextEdits = chatFlow.querySelectorAll('.round-context-edit');\n"
+        "    contextEdits.forEach(edit => {\n"
+        "      const parent = edit.closest('.round-context, .agent-context-box, .split-agent-context');\n"
+        "      if (parent) {\n"
+        "        if (strongHideOn && edit.textContent.trim() === '') {\n"
+        "          parent.style.display = 'none';\n"
+        "        } else {\n"
+        "          parent.style.display = '';\n"
+        "        }\n"
+        "      }\n"
+        "    });\n"
+        "    const splitContexts = chatFlow.querySelectorAll('.split-agent-context');\n"
+        "    splitContexts.forEach(split => {\n"
+        "      if (strongHideOn) {\n"
+        "        const boxes = split.querySelectorAll('.agent-context-box');\n"
+        "        const allEmpty = Array.from(boxes).every(box => {\n"
+        "          const edit = box.querySelector('.round-context-edit');\n"
+        "          return edit && edit.textContent.trim() === '';\n"
+        "        });\n"
+        "        if (allEmpty) split.style.display = 'none';\n"
+        "      }\n"
+        "    });\n"
+        "  }\n"
+        "  if (strongHideBtnChat && chatFlow) {\n"
+        "    strongHideBtnChat.addEventListener('click', () => {\n"
+        "      setTimeout(() => applyStrongHideToChat(), 0);\n"
+        "    });\n"
+        "  }\n"
+        "  document.addEventListener('click', function(e) {\n"
+        "    if (e.target.closest('input, textarea, select, .round-context-edit, .toolbar')) { return; }\n"
+        "    const chatReasoning = e.target.closest('.chat-reasoning');\n"
+        "    if (chatReasoning) {\n"
+        "      chatReasoning.classList.toggle('collapsed');\n"
+        "      return;\n"
+        "    }\n"
+        "    const userMessage = e.target.closest('.chat-message.role-user');\n"
+        "    if (userMessage && !e.target.closest('.merge-btn, .unmerge-btn')) {\n"
+        "      userMessage.classList.toggle('collapsed');\n"
+        "    }\n"
+        "  });\n"
+        "  function applyColorToSelection(color, element) {\n"
+        "    const selection = window.getSelection();\n"
+        "    if (!selection.rangeCount) return false;\n"
+        "    const range = selection.getRangeAt(0);\n"
+        "    if (!element.contains(range.commonAncestorContainer)) return false;\n"
+        "    const selectedText = range.toString();\n"
+        "    if (!selectedText) return false;\n"
+        "    if (color === 'default') {\n"
+        "      // Remove styling - just extract the text content\n"
+        "      const textNode = document.createTextNode(selectedText);\n"
+        "      range.deleteContents();\n"
+        "      range.insertNode(textNode);\n"
+        "    } else {\n"
+        "      const span = document.createElement('span');\n"
+        "      span.style.color = color;\n"
+        "      span.style.fontWeight = '600';\n"
+        "      try {\n"
+        "        range.surroundContents(span);\n"
+        "      } catch (e) {\n"
+        "        const contents = range.extractContents();\n"
+        "        span.appendChild(contents);\n"
+        "        range.insertNode(span);\n"
+        "      }\n"
+        "    }\n"
+        "    return true;\n"
+        "  }\n"
+        "  let lastFocusedContextEdit = null;\n"
+        "  document.addEventListener('focusin', function(e) {\n"
+        "    if (e.target.classList.contains('round-context-edit')) {\n"
+        "      lastFocusedContextEdit = e.target;\n"
+        "    }\n"
+        "  });\n"
+        "  document.addEventListener('mousedown', function(e) {\n"
+        "    if (e.target.classList.contains('context-color-btn')) {\n"
+        "      e.preventDefault();\n"
+        "    }\n"
+        "  });\n"
+        "  document.addEventListener('click', function(e) {\n"
+        "    if (e.target.closest('input:not(.round-context-edit), textarea, select') && !e.target.classList.contains('context-color-btn')) { return; }\n"
+        "    if (e.target.classList.contains('context-color-btn')) {\n"
+        "      e.preventDefault();\n"
+        "      const color = e.target.dataset.color;\n"
+        "      const controls = e.target.closest('.round-context-controls');\n"
+        "      const contextEdit = controls ? controls.previousElementSibling : null;\n"
+        "      if (contextEdit && contextEdit.classList.contains('round-context-edit')) {\n"
+        "        contextEdit.focus();\n"
+        "        const selection = window.getSelection();\n"
+        "        if (selection.rangeCount > 0 && selection.toString().length > 0 && contextEdit.contains(selection.anchorNode)) {\n"
+        "          if (applyColorToSelection(color, contextEdit)) {\n"
+        "            const key = contextEdit.dataset.contextKey;\n"
+        "            localStorage.setItem(key, contextEdit.innerHTML);\n"
+        "          }\n"
+        "        } else {\n"
+        "          try {\n"
+        "            if (color !== 'default') {\n"
+        "              document.execCommand('styleWithCSS', false, true);\n"
+        "              document.execCommand('foreColor', false, color);\n"
+        "            }\n"
+        "            const key = contextEdit.dataset.contextKey;\n"
+        "            setTimeout(() => localStorage.setItem(key, contextEdit.innerHTML), 10);\n"
+        "          } catch (e) {\n"
+        "            console.log('Color command failed:', e);\n"
+        "          }\n"
+        "        }\n"
+        "      }\n"
+        "    }\n"
+        "  });\n"
+        "  const contextEdits = document.querySelectorAll('.round-context-edit');\n"
+        "  contextEdits.forEach(edit => {\n"
+        "    edit.addEventListener('input', function() {\n"
+        "      const key = this.dataset.contextKey;\n"
+        "      localStorage.setItem(key, this.innerHTML);\n"
+        "    });\n"
+        "    const key = edit.dataset.contextKey;\n"
+        "    const saved = localStorage.getItem(key);\n"
+        "    if (saved) {\n"
+        "      edit.innerHTML = saved;\n"
+        "    }\n"
+        "  });\n"
+        "  document.addEventListener('click', function(e) {\n"
+        "    if (e.target.closest('input, textarea, select, .round-context-edit') && !e.target.classList.contains('merge-btn') && !e.target.classList.contains('unmerge-btn')) { return; }\n"
+        "    if (e.target.classList.contains('merge-btn')) {\n"
+        "      e.preventDefault();\n"
+        "      e.stopPropagation();\n"
+        "      const msgId = e.target.dataset.msgId;\n"
+        "      const currentMsg = e.target.closest('.chat-message');\n"
+        "      if (!currentMsg) return;\n"
+        "      if (currentMsg.classList.contains('role-user')) {\n"
+        "        alert('Cannot merge user messages');\n"
+        "        return;\n"
+        "      }\n"
+        "      let nextMsg = currentMsg.nextElementSibling;\n"
+        "      while (nextMsg && !nextMsg.classList.contains('chat-message')) {\n"
+        "        nextMsg = nextMsg.nextElementSibling;\n"
+        "      }\n"
+        "      while (nextMsg && nextMsg.classList.contains('role-user')) {\n"
+        "        nextMsg = nextMsg.nextElementSibling;\n"
+        "        while (nextMsg && !nextMsg.classList.contains('chat-message')) {\n"
+        "          nextMsg = nextMsg.nextElementSibling;\n"
+        "        }\n"
+        "      }\n"
+        "      if (!nextMsg || nextMsg.classList.contains('chat-message') === false) {\n"
+        "        alert('No next assistant message to merge with');\n"
+        "        return;\n"
+        "      }\n"
+        "      if (nextMsg.classList.contains('role-user')) {\n"
+        "        alert('Cannot merge with user messages');\n"
+        "        return;\n"
+        "      }\n"
+        "      \n"
+        "      // Find the user prompts that precede each assistant message\n"
+        "      let currentPrompt = currentMsg.previousElementSibling;\n"
+        "      while (currentPrompt && !currentPrompt.classList.contains('chat-message')) {\n"
+        "        currentPrompt = currentPrompt.previousElementSibling;\n"
+        "      }\n"
+        "      if (currentPrompt && !currentPrompt.classList.contains('role-user')) {\n"
+        "        currentPrompt = null;\n"
+        "      }\n"
+        "      \n"
+        "      let nextPrompt = nextMsg.previousElementSibling;\n"
+        "      while (nextPrompt && !nextPrompt.classList.contains('chat-message')) {\n"
+        "        nextPrompt = nextPrompt.previousElementSibling;\n"
+        "      }\n"
+        "      if (nextPrompt && !nextPrompt.classList.contains('role-user')) {\n"
+        "        nextPrompt = null;\n"
+        "      }\n"
+        "      \n"
+        "      // Find the split-agent-context that precedes the first prompt or assistant message\n"
+        "      let splitContext = null;\n"
+        "      let searchStart = currentPrompt || currentMsg;\n"
+        "      let elem = searchStart.previousElementSibling;\n"
+        "      while (elem) {\n"
+        "        if (elem.classList.contains('split-agent-context')) {\n"
+        "          splitContext = elem;\n"
+        "          break;\n"
+        "        }\n"
+        "        if (elem.classList.contains('chat-message') || elem.classList.contains('chat-group-divider')) {\n"
+        "          break;\n"
+        "        }\n"
+        "        elem = elem.previousElementSibling;\n"
+        "      }\n"
+        "      \n"
+        "      const parent = currentMsg.parentElement;\n"
+        "      if (parent.classList.contains('simultaneous-messages')) {\n"
+        "        const wrapper = parent;\n"
+        "        currentMsg.style.display = '';\n"
+        "        currentMsg.classList.remove('merged');\n"
+        "        const refNode = wrapper.nextElementSibling;\n"
+        "        parent.parentElement.insertBefore(currentMsg, refNode);\n"
+        "        if (nextMsg.parentElement === wrapper) {\n"
+        "          parent.parentElement.insertBefore(nextMsg, refNode);\n"
+        "        }\n"
+        "        if (wrapper.children.length === 0) {\n"
+        "          wrapper.remove();\n"
+        "        }\n"
+        "      } else {\n"
+        "        // If split-agent-context exists, wrap it\n"
+        "        if (splitContext && !splitContext.classList.contains('merged')) {\n"
+        "          const splitWrapper = document.createElement('div');\n"
+        "          splitWrapper.className = 'simultaneous-messages';\n"
+        "          const splitUnmergeBtn = document.createElement('button');\n"
+        "          splitUnmergeBtn.className = 'unmerge-btn';\n"
+        "          splitUnmergeBtn.innerHTML = '✕';\n"
+        "          splitUnmergeBtn.title = 'Click to unmerge messages';\n"
+        "          splitWrapper.appendChild(splitUnmergeBtn);\n"
+        "          splitWrapper.dataset.isSplitContext = 'true';\n"
+        "          parent.insertBefore(splitWrapper, splitContext);\n"
+        "          splitWrapper.appendChild(splitContext);\n"
+        "          splitContext.classList.add('merged');\n"
+        "        }\n"
+        "        \n"
+        "        // Create wrapper for prompts if both exist\n"
+        "        if (currentPrompt && nextPrompt) {\n"
+        "          const promptWrapper = document.createElement('div');\n"
+        "          promptWrapper.className = 'simultaneous-messages';\n"
+        "          const promptUnmergeBtn = document.createElement('button');\n"
+        "          promptUnmergeBtn.className = 'unmerge-btn';\n"
+        "          promptUnmergeBtn.innerHTML = '✕';\n"
+        "          promptUnmergeBtn.title = 'Click to unmerge messages';\n"
+        "          promptWrapper.appendChild(promptUnmergeBtn);\n"
+        "          promptWrapper.dataset.firstMsgId = currentPrompt.dataset.msgId;\n"
+        "          promptWrapper.dataset.secondMsgId = nextPrompt.dataset.msgId;\n"
+        "          \n"
+        "          // Determine order: agent-0 first, agent-1 second\n"
+        "          const firstPrompt = currentPrompt.classList.contains('agent-0') ? currentPrompt : nextPrompt;\n"
+        "          const secondPrompt = currentPrompt.classList.contains('agent-0') ? nextPrompt : currentPrompt;\n"
+        "          \n"
+        "          parent.insertBefore(promptWrapper, currentPrompt);\n"
+        "          promptWrapper.appendChild(firstPrompt);\n"
+        "          promptWrapper.appendChild(secondPrompt);\n"
+        "          currentPrompt.classList.add('merged');\n"
+        "          nextPrompt.classList.add('merged');\n"
+        "        }\n"
+        "        \n"
+        "        // Create wrapper for assistant messages\n"
+        "        const wrapper = document.createElement('div');\n"
+        "        wrapper.className = 'simultaneous-messages';\n"
+        "        const unmergeBtn = document.createElement('button');\n"
+        "        unmergeBtn.className = 'unmerge-btn';\n"
+        "        unmergeBtn.innerHTML = '✕';\n"
+        "        unmergeBtn.title = 'Click to unmerge messages';\n"
+        "        wrapper.appendChild(unmergeBtn);\n"
+        "        wrapper.dataset.firstMsgId = currentMsg.dataset.msgId;\n"
+        "        wrapper.dataset.secondMsgId = nextMsg.dataset.msgId;\n"
+        "        \n"
+        "        // Determine order: agent-0 first, agent-1 second\n"
+        "        const firstAssistant = currentMsg.classList.contains('agent-0') ? currentMsg : nextMsg;\n"
+        "        const secondAssistant = currentMsg.classList.contains('agent-0') ? nextMsg : currentMsg;\n"
+        "        \n"
+        "        parent.insertBefore(wrapper, currentMsg);\n"
+        "        wrapper.appendChild(firstAssistant);\n"
+        "        wrapper.appendChild(secondAssistant);\n"
+        "        currentMsg.classList.add('merged');\n"
+        "        nextMsg.classList.add('merged');\n"
+        "      }\n"
+        "    }\n"
+        "    if (e.target.classList.contains('unmerge-btn')) {\n"
+        "      const wrapper = e.target.closest('.simultaneous-messages');\n"
+        "      if (!wrapper) return;\n"
+        "      const parent = wrapper.parentElement;\n"
+        "      \n"
+        "      // Check if this is a split-context wrapper\n"
+        "      if (wrapper.dataset.isSplitContext === 'true') {\n"
+        "        const splitContext = wrapper.querySelector('.split-agent-context');\n"
+        "        if (splitContext) {\n"
+        "          splitContext.classList.remove('merged');\n"
+        "          parent.insertBefore(splitContext, wrapper.nextElementSibling);\n"
+        "        }\n"
+        "        wrapper.remove();\n"
+        "        return;\n"
+        "      }\n"
+        "      \n"
+        "      const firstMsgId = wrapper.dataset.firstMsgId;\n"
+        "      const secondMsgId = wrapper.dataset.secondMsgId;\n"
+        "      const messages = Array.from(wrapper.querySelectorAll('.chat-message'));\n"
+        "      const refNode = wrapper.nextElementSibling;\n"
+        "      const firstMsg = messages.find(m => m.dataset.msgId === firstMsgId);\n"
+        "      const secondMsg = messages.find(m => m.dataset.msgId === secondMsgId);\n"
+        "      \n"
+        "      // Check for preceding wrappers to also unmerge (prompts and split-context)\n"
+        "      let currentElem = wrapper.previousElementSibling;\n"
+        "      const wrappersToUnmerge = [];\n"
+        "      \n"
+        "      while (currentElem) {\n"
+        "        if (currentElem.classList.contains('simultaneous-messages')) {\n"
+        "          wrappersToUnmerge.push(currentElem);\n"
+        "        } else if (currentElem.classList.contains('chat-message') || currentElem.classList.contains('chat-group-divider')) {\n"
+        "          break;\n"
+        "        }\n"
+        "        currentElem = currentElem.previousElementSibling;\n"
+        "      }\n"
+        "      \n"
+        "      // Unmerge preceding wrappers\n"
+        "      for (const prevWrapper of wrappersToUnmerge) {\n"
+        "        if (prevWrapper.dataset.isSplitContext === 'true') {\n"
+        "          const splitContext = prevWrapper.querySelector('.split-agent-context');\n"
+        "          if (splitContext) {\n"
+        "            splitContext.classList.remove('merged');\n"
+        "            parent.insertBefore(splitContext, prevWrapper.nextElementSibling);\n"
+        "          }\n"
+        "          prevWrapper.remove();\n"
+        "        } else {\n"
+        "          const prevMessages = Array.from(prevWrapper.querySelectorAll('.chat-message'));\n"
+        "          const prevFirstMsgId = prevWrapper.dataset.firstMsgId;\n"
+        "          const prevSecondMsgId = prevWrapper.dataset.secondMsgId;\n"
+        "          const prevFirstMsg = prevMessages.find(m => m.dataset.msgId === prevFirstMsgId);\n"
+        "          const prevSecondMsg = prevMessages.find(m => m.dataset.msgId === prevSecondMsgId);\n"
+        "          const prevRefNode = prevWrapper.nextElementSibling;\n"
+        "          \n"
+        "          if (prevFirstMsg) {\n"
+        "            prevFirstMsg.classList.remove('merged');\n"
+        "            prevFirstMsg.style.display = '';\n"
+        "            parent.insertBefore(prevFirstMsg, prevRefNode);\n"
+        "          }\n"
+        "          if (prevSecondMsg) {\n"
+        "            prevSecondMsg.classList.remove('merged');\n"
+        "            prevSecondMsg.style.display = '';\n"
+        "            parent.insertBefore(prevSecondMsg, prevRefNode);\n"
+        "          }\n"
+        "          prevWrapper.remove();\n"
+        "        }\n"
+        "      }\n"
+        "      \n"
+        "      // Unmerge the main assistant messages\n"
+        "      if (firstMsg) {\n"
+        "        firstMsg.classList.remove('merged');\n"
+        "        firstMsg.style.display = '';\n"
+        "        parent.insertBefore(firstMsg, refNode);\n"
+        "      }\n"
+        "      if (secondMsg) {\n"
+        "        secondMsg.classList.remove('merged');\n"
+        "        secondMsg.style.display = '';\n"
+        "        parent.insertBefore(secondMsg, refNode);\n"
+        "      }\n"
+        "      wrapper.remove();\n"
+        "    }\n"
+        "  });\n"
+        "});\n"
+        "</script>",
+        "</head>",
+        "<body>",
+        '<div class="toolbar-wrap">',
+        '<div class="toolbar-hotzone"></div>',
+        '<div class="toolbar">',
+        '<button id="toggle-strong-hide"><span class="emoji-bw">🗜️</span> Strong Hide: <span id="strong-hide-state">Off</span></button>',
+        '<button id="toggle-hide-user-messages"><span class="emoji-bw">👁️</span> Hide Prompts: <span id="hide-user-state">Off</span></button>',
+        '<span id="chat-width-control" style="margin-left:8px;">',
+        '<label for="chat-width-slider"><span class="emoji-bw">↔️</span> Width:</label>',
+        '<input id="chat-width-slider" type="range" min="600" max="1600" step="50" value="900" style="width:120px; vertical-align:middle;" />',
+        '<span id="chat-width-value" style="margin-left:4px;">900px</span>',
+        "</span>",
+        '<span style="margin-left:12px;">',
+        '<label for="font-family-select"><span class="emoji-bw">🔤</span> Font:</label>',
+        '<select id="font-family-select" style="padding:2px 6px; border:1px solid var(--accent-muted); border-radius:var(--corner-radius); background:var(--bg);">',
+        "<option value=\"'Segoe UI', Tahoma, Geneva, Verdana, sans-serif\">Segoe UI</option>",
+        '<option value="Arial, sans-serif">Arial</option>',
+        "<option value=\"'Helvetica Neue', Helvetica, sans-serif\">Helvetica</option>",
+        "<option value=\"'Times New Roman', Times, serif\">Times New Roman</option>",
+        '<option value="Georgia, serif">Georgia</option>',
+        "<option value=\"'Courier New', Courier, monospace\">Courier New</option>",
+        "<option value=\"'Comic Sans MS', cursive\">Comic Sans</option>",
+        "<option value=\"'Trebuchet MS', sans-serif\">Trebuchet MS</option>",
+        '<option value="Verdana, sans-serif">Verdana</option>',
+        "<option value=\"'Palatino Linotype', 'Book Antiqua', Palatino, serif\">Palatino</option>",
+        "<option value=\"'Lucida Console', Monaco, monospace\">Lucida Console</option>",
+        "</select>",
+        "</span>",
+        '<span style="margin-left:8px;">',
+        '<label for="font-size-input"><span class="emoji-bw">📏</span> Size:</label>',
+        '<input id="font-size-input" type="number" min="8" max="24" step="1" value="14" style="width:50px;" />',
+        "<span>px</span>",
+        "</span>",
+        '<span style="margin-left:12px; display:flex; align-items:center; gap:8px;">',
+        '<label style="font-weight:600;">Agent Names:</label>',
+        f'<select id="agent0-emoji-input" style="width:65px; padding:2px 6px; border:1px solid var(--accent-muted); border-radius:var(--corner-radius); background:var(--bg);">',
+        '<option value="🤖">🤖 Robot</option>',
+        '<option value="👤">👤 Human</option>',
+        "</select>",
+        f'<input id="agent0-name-input" type="text" placeholder="{html.escape(unique_agent_ids[0]) if len(unique_agent_ids) > 0 else "Agent 0"}" style="width:80px; padding:2px 6px; border:1px solid var(--accent-muted); border-radius:var(--corner-radius); background:var(--bg);" />',
+        '<span style="margin:0 4px;">|</span>',
+        f'<select id="agent1-emoji-input" style="width:65px; padding:2px 6px; border:1px solid var(--accent-muted); border-radius:var(--corner-radius); background:var(--bg);">',
+        '<option value="🤖">🤖 Robot</option>',
+        '<option value="👤">👤 Human</option>',
+        "</select>",
+        f'<input id="agent1-name-input" type="text" placeholder="{html.escape(unique_agent_ids[1]) if len(unique_agent_ids) > 1 else "Agent 1"}" style="width:80px; padding:2px 6px; border:1px solid var(--accent-muted); border-radius:var(--corner-radius); background:var(--bg);" />',
+        '<button id="apply-agent-names" style="padding:4px 8px; border:1px solid var(--accent-muted); background:var(--panel-bg); border-radius:var(--corner-radius); cursor:pointer;">Apply</button>',
+        "</span>",
+        "</div>",
+        "</div>",
+    ]
+    # Add Chat View
+    import html as _html_mod
+    html_parts.append('<div id="flow-chat" class="messages-flow">')
+    # Helper function to add context annotation areas
+    def add_context_area(position: str, time_step: int):
+        context_key = f"round-context-{position}-{time_step}"
+        placeholder = f"Add context {position} round {time_step}..."
+        color_buttons = ""
+        # Add default/reset color button first
+        color_buttons += (
+            f'<div class="context-color-btn" data-color="default" '
+            f'style="background: linear-gradient(135deg, #000 25%, transparent 25%, transparent 75%, #000 75%), '
+            f"linear-gradient(135deg, #000 25%, transparent 25%, transparent 75%, #000 75%); "
+            f"background-size: 4px 4px; background-position: 0 0, 2px 2px; "
+            f'background-color: #fff;" title="Default color"></div>'
+        )
+        for color_name, color_value in [
+            ("red", "#d32f2f"),
+            ("orange", "#f57c00"),
+            ("yellow", "#f9a825"),
+            ("green", "#388e3c"),
+            ("blue", "#1976d2"),
+            ("purple", "#7b1fa2"),
+            ("gray", "#666666"),
+        ]:
+            color_buttons += (
+                f'<div class="context-color-btn" data-color="{color_value}" '
+                f'style="background-color: {color_value};" title="{color_name}"></div>'
+            )
+        html_parts.append(
+            f'<div class="round-context">'
+            f'<div class="round-context-edit" contenteditable="true" spellcheck="true" '
+            f'data-context-key="{context_key}" '
+            f'data-placeholder="{placeholder}"></div>'
+            f'<div class="round-context-controls">{color_buttons}</div>'
+            f"</div>"
+        )
+    # Helper function to add split agent context boxes
+    def add_split_agent_contexts(position: str, time_step: int):
+        color_buttons = ""
+        # Add default/reset color button first
+        color_buttons += (
+            f'<div class="context-color-btn" data-color="default" '
+            f'style="background: linear-gradient(135deg, #000 25%, transparent 25%, transparent 75%, #000 75%), '
+            f"linear-gradient(135deg, #000 25%, transparent 25%, transparent 75%, #000 75%); "
+            f"background-size: 4px 4px; background-position: 0 0, 2px 2px; "
+            f'background-color: #fff;" title="Default color"></div>'
+        )
+        for color_name, color_value in [
+            ("red", "#d32f2f"),
+            ("orange", "#f57c00"),
+            ("yellow", "#f9a825"),
+            ("green", "#388e3c"),
+            ("blue", "#1976d2"),
+            ("purple", "#7b1fa2"),
+            ("gray", "#666666"),
+        ]:
+            color_buttons += (
+                f'<div class="context-color-btn" data-color="{color_value}" '
+                f'style="background-color: {color_value};" title="{color_name}"></div>'
+            )
+        html_parts.append('<div class="split-agent-context">')
+        # Agent 0 box
+        agent0_key = f"agent-context-0-{position}-{time_step}"
+        agent0_placeholder = f"..."
+        html_parts.append(
+            f'<div class="agent-context-box agent-0">'
+            f'<div class="round-context-edit" contenteditable="true" spellcheck="true" '
+            f'data-context-key="{agent0_key}" '
+            f'data-placeholder="{agent0_placeholder}"></div>'
+            f'<div class="round-context-controls">{color_buttons}</div>'
+            f"</div>"
+        )
+        # Agent 1 box
+        agent1_key = f"agent-context-1-{position}-{time_step}"
+        agent1_placeholder = f"..."
+        html_parts.append(
+            f'<div class="agent-context-box agent-1">'
+            f'<div class="round-context-edit" contenteditable="true" spellcheck="true" '
+            f'data-context-key="{agent1_key}" '
+            f'data-placeholder="{agent1_placeholder}"></div>'
+            f'<div class="round-context-controls">{color_buttons}</div>'
+            f"</div>"
+        )
+        html_parts.append("</div>")  # split-agent-context
+    last_time_step_chat = None
+    for original_index, turn in indexed_turns:
+        # Use agent index for CSS class (agent-0 or agent-1) instead of agent ID
+        agent_index = agent_id_to_index.get(turn.agent_id, 0)
+        agent_class = f"agent-{agent_index}"
+        role_class = f"role-{turn.role}"
+        # Add time step divider and beginning context
+        if last_time_step_chat is None or turn.time_step != last_time_step_chat:
+            # Add end contexts for previous round (only regular context, not prompt summary)
+            if last_time_step_chat is not None:
+                add_context_area("end", last_time_step_chat)
+            html_parts.append(
+                f'<div class="chat-group-divider">'
+                f'<span class="chat-group-label">⏱ Round {turn.time_step + 1}</span>'
+                f"</div>"
+            )
+            # Add beginning contexts for new round (both context and prompt summary)
+            add_context_area("beginning", turn.time_step)
+            add_split_agent_contexts("beginning", turn.time_step)
+            last_time_step_chat = turn.time_step
+        # Build chat message with merge controls
+        html_parts.append(
+            f'<div class="chat-message {agent_class} {role_class}" data-msg-id="{original_index}">'
+        )
+        # Add merge control button
+        html_parts.append(
+            f'<button class="merge-btn" title="Merge with next message" data-msg-id="{original_index}">⇄</button>'
+        )
+        html_parts.append('<div class="chat-message-content">')
+        # Header with agent name and reward (always show reward)
+        if turn.role == "assistant":
+            name = _html_mod.escape(turn.agent_id)
+            raw_val = turn.reward
+            if isinstance(raw_val, (int, float)):
+                reward_val = f"{raw_val:.4f}".rstrip("0").rstrip(".")
+                if len(reward_val) > 8:
+                    reward_val = reward_val[:8] + "…"
+            else:
+                reward_val = str(raw_val)
+            header_html = (
+                f'<div class="chat-header">'
+                f'<span class="emoji-bw" data-agent-index="{agent_index}">🤖</span> <span class="agent-name" data-agent-index="{agent_index}">{name}</span>'
+                f'<span class="chat-reward">⚑ {reward_val}</span>'
+                f"</div>"
+            )
+        else:
+            name = _html_mod.escape(turn.agent_id)
+            header_html = f'<div class="chat-header">Prompt of <span class="agent-name" data-agent-index="{agent_index}">{name}</span></div>'
+        html_parts.append(header_html)
+        # Reasoning content if present
+        if turn.reasoning_content:
+            _raw_reasoning = turn.reasoning_content.replace("\r\n", "\n")
+            _raw_reasoning = _re.sub(r"^\s*\n+", "", _raw_reasoning)
+            esc_reasoning = _html_mod.escape(_raw_reasoning)
+            html_parts.append(
+                f'<div class="chat-reasoning collapsed">'
+                f'<span class="reasoning-icon">💭</span> '
+                f'<span class="reasoning-text">{esc_reasoning}</span>'
+                f"</div>"
+            )
+        # Message bubble
+        esc_content = _html_mod.escape(turn.content)
+        html_parts.append(f'<div class="chat-bubble">{esc_content}</div>')
+        html_parts.append("</div>")  # chat-message-content
+        html_parts.append("</div>")  # chat-message
+    # Add end contexts for the last round (only regular context, not prompt summary)
+    if last_time_step_chat is not None:
+        add_context_area("end", last_time_step_chat)
+    html_parts.append("</div>")  # flow-chat
+    html_parts.extend(["</body>", "</html>"])
+    return "\n".join(html_parts)
+def export_html_from_rollout_tree(path: Path, outdir: Path, main_only: bool = False):
+    """Process a rollout tree file and generate HTML files for each path.
+    Creates separate HTML files for the main path and each branch path.
+    The main path is saved in the root output directory, while branch paths
+    are saved in a 'branches' subdirectory.
+    Args:
+        path: Path to the rollout tree JSON file
+        outdir: Output directory for HTML files
+        main_only: If True, only export the main trajectory (default: False)
+    """
+    root = load_rollout_tree(path)
+    mgid = root.id
+    main_path, branch_paths = get_rollout_tree_paths(root)
+    outdir.mkdir(parents=True, exist_ok=True)
+    # Create branches subdirectory if we have branch paths
+    if not main_only and branch_paths:
+        branches_dir = outdir / f"mgid:{mgid}_branches_html_renders"
+        branches_dir.mkdir(parents=True, exist_ok=True)
+    # Generate HTML for the main path
+    chat_turns = gather_all_chat_turns_for_path(main_path)
+    html_content = html_from_chat_turns(chat_turns)
+    output_file = outdir / f"mgid:{mgid}_main_html_render.render.html"
+    with open(output_file, "w", encoding="utf-8") as f:
+        f.write(html_content)
+    # Generate HTML for each branch path
+    for path_obj in branch_paths:
+        chat_turns = gather_all_chat_turns_for_path(path_obj)
+        html_content = html_from_chat_turns(chat_turns)
+        path_id: str = path_obj.id
+        output_filename = f"{path_id}_html_render.render.html"
+        output_file = branches_dir / output_filename
+        with open(output_file, "w", encoding="utf-8") as f:
+            f.write(html_content)

src_code_for_reproducibility/utils/stat_pack.py ADDED Viewed

	@@ -0,0 +1,117 @@

+"""
+File: mllm/utils/stat_pack.py
+Summary: Implements the StatPack container for incremental statistics.
+"""
+import csv
+import json
+import os
+import pickle
+from collections import Counter
+from copy import deepcopy
+from locale import strcoll
+from statistics import mean
+from typing import Any, Dict, Iterator, List, Optional, Tuple, TypedDict
+import matplotlib.pyplot as plt
+import numpy as np
+style_path = os.environ.get("ADALIGN_MPLSTYLE")
+if style_path:
+    plt.style.use(style_path)
+import wandb
+from . import wandb_utils
+class StatPack:
+    def __init__(self):
+        self.data = {}
+    def add_stat(self, key: str, value: float | int | None):
+        assert (
+            isinstance(value, float) or isinstance(value, int) or value is None
+        ), f"Value {value} is not a valid type"
+        if key not in self.data:
+            self.data[key] = []
+        self.data[key].append(value)
+    def add_stats(self, other: "StatPack"):
+        for key in other.keys():
+            self.add_stat(key, other[key])
+    def __getitem__(self, key: str):
+        return self.data[key]
+    def __setitem__(self, key: str, value: Any):
+        self.data[key] = value
+    def __contains__(self, key: str):
+        return key in self.data
+    def __len__(self):
+        return len(self.data)
+    def __iter__(self):
+        return iter(self.data)
+    def keys(self):
+        return self.data.keys()
+    def values(self):
+        return self.data.values()
+    def items(self):
+        return self.data.items()
+    def mean(self):
+        mean_st = StatPack()
+        for key in self.keys():
+            if isinstance(self[key], list):
+                # Ignore None entries so missing measurements do not bias the mean.
+                non_none_values = [v for v in self[key] if v is not None]
+                if non_none_values:
+                    mean_st[key] = np.mean(np.array(non_none_values))
+                else:
+                    mean_st[key] = None
+        return mean_st
+    def store_plots(self, folder: str):
+        os.makedirs(folder, exist_ok=True)
+        for key in self.keys():
+            plt.figure(figsize=(10, 5))
+            plt.plot(self[key])
+            plt.title(key)
+            plt.savefig(os.path.join(folder, f"{key}.pdf"))
+            plt.close()
+    def store_numpy(self, folder: str):
+        os.makedirs(folder, exist_ok=True)
+        for key in self.keys():
+            # Sanitize filename components (avoid slashes, spaces, etc.)
+            safe_key = str(key).replace(os.sep, "_").replace("/", "_").replace(" ", "_")
+            values = self[key]
+            # Convert None to NaN for numpy compatibility
+            arr = np.array(
+                [(np.nan if (v is None) else v) for v in values], dtype=float
+            )
+            np.save(os.path.join(folder, f"{safe_key}.npy"), arr)
+    def store_json(self, folder: str, filename: str = "stats.json"):
+        os.makedirs(folder, exist_ok=True)
+        with open(os.path.join(folder, filename), "w") as f:
+            json.dump(self.data, f, indent=4)
+    def store_csv(self, folder: str):
+        os.makedirs(folder, exist_ok=True)
+        for key in self.keys():
+            with open(os.path.join(folder, f"stats.csv"), "w") as f:
+                writer = csv.writer(f)
+                writer.writerow([key] + self[key])
+    def store_pickle(self, folder: str):
+        os.makedirs(folder, exist_ok=True)
+        for key in self.keys():
+            with open(os.path.join(folder, f"stats.pkl"), "wb") as f:
+                pickle.dump(self[key], f)

src_code_for_reproducibility/utils/update_start_epoch.py ADDED Viewed

	@@ -0,0 +1,17 @@

+"""
+File: mllm/utils/update_start_epoch.py
+Summary: Updates persisted start-epoch metadata when resuming runs.
+"""
+import os
+# During run, set hydra.run.dir=./outputs/{folder}
+def update_start_epoch(cfg, output_directory):
+    if cfg["experiment"]["resume_experiment"]:
+        folders = [
+            f for f in os.listdir(output_directory) if f.startswith("iteration_")
+        ]
+        iterations = [int(f.split("_")[1]) for f in folders] if folders else [0]
+        cfg["experiment"]["start_epoch"] = max(iterations)
+    return None

src_code_for_reproducibility/utils/wandb_utils.py ADDED Viewed

	@@ -0,0 +1,170 @@

+"""
+File: mllm/utils/wandb_utils.py
+Summary: Shared Weights & Biases helper functions.
+"""
+import os
+from typing import Any, Dict, Optional
+_WANDB_AVAILABLE = False
+_WANDB_RUN = None
+def _try_import_wandb():
+    global _WANDB_AVAILABLE
+    if _WANDB_AVAILABLE:
+        return True
+    try:
+        import wandb  # type: ignore
+        _WANDB_AVAILABLE = True
+        return True
+    except Exception:
+        _WANDB_AVAILABLE = False
+        return False
+def _safe_get(cfg: Dict[str, Any], path: list[str], default: Any = None) -> Any:
+    cur: Any = cfg
+    for key in path:
+        if not isinstance(cur, dict) or key not in cur:
+            return default
+        cur = cur[key]
+    return cur
+def is_enabled(cfg: Dict[str, Any]) -> bool:
+    return bool(_safe_get(cfg, ["logging", "wandb", "enabled"], False))
+def init(cfg: Dict[str, Any], run_dir: str, run_name: Optional[str] = None) -> None:
+    """
+    Initialize Weights & Biases if enabled in config. No-op if disabled or wandb not installed.
+    """
+    global _WANDB_RUN
+    if not is_enabled(cfg):
+        return
+    if not _try_import_wandb():
+        return
+    import wandb  # type: ignore
+    project = _safe_get(cfg, ["logging", "wandb", "project"], "llm-negotiation")
+    entity = _safe_get(cfg, ["logging", "wandb", "entity"], None)
+    mode = _safe_get(cfg, ["logging", "wandb", "mode"], "online")
+    tags = _safe_get(cfg, ["logging", "wandb", "tags"], []) or []
+    notes = _safe_get(cfg, ["logging", "wandb", "notes"], None)
+    group = _safe_get(cfg, ["logging", "wandb", "group"], None)
+    name = _safe_get(cfg, ["logging", "wandb", "name"], run_name)
+    # Ensure files are written into the hydra run directory
+    os.makedirs(run_dir, exist_ok=True)
+    os.environ.setdefault("WANDB_DIR", run_dir)
+    # Convert cfg to plain types for W&B config; fallback to minimal dictionary
+    try:
+        from omegaconf import OmegaConf  # type: ignore
+        cfg_container = OmegaConf.to_container(cfg, resolve=True)  # type: ignore
+    except Exception:
+        cfg_container = cfg
+    _WANDB_RUN = wandb.init(
+        project=project,
+        entity=entity,
+        mode=mode,
+        name=name,
+        group=group,
+        tags=tags,
+        notes=notes,
+        config=cfg_container,
+        dir=run_dir,
+        reinit=True,
+    )
+def log(metrics: Dict[str, Any], step: Optional[int] = None) -> None:
+    """Log a flat dictionary of metrics to W&B if active."""
+    if not _WANDB_AVAILABLE or _WANDB_RUN is None:
+        return
+    try:
+        import wandb  # type: ignore
+        wandb.log(metrics if step is None else dict(metrics, step=step))
+    except Exception:
+        pass
+def _flatten(prefix: str, data: Dict[str, Any], out: Dict[str, Any]) -> None:
+    for k, v in data.items():
+        key = f"{prefix}.{k}" if prefix else k
+        if isinstance(v, dict):
+            _flatten(key, v, out)
+        else:
+            out[key] = v
+def _summarize_value(value: Any) -> Dict[str, Any]:
+    import numpy as np  # local import to avoid hard dependency during disabled mode
+    if value is None:
+        return {"none": 1}
+    # Scalars
+    if isinstance(value, (int, float)):
+        return {"value": float(value)}
+    # Lists or arrays
+    try:
+        arr = np.asarray(value)
+        if arr.size == 0:
+            return {"size": 0}
+        return {
+            "mean": float(np.nanmean(arr)),
+            "min": float(np.nanmin(arr)),
+            "max": float(np.nanmax(arr)),
+            "last": float(arr.reshape(-1)[-1]),
+            "size": int(arr.size),
+        }
+    except Exception:
+        # Fallback: string repr
+        return {"text": str(value)}
+def log_tally(
+    array_tally: Dict[str, Any], prefix: str = "", step: Optional[int] = None
+) -> None:
+    """
+    Flatten and summarize Tally.array_tally and log to WandB.
+    Each leaf list/array is summarized with mean/min/max/last/size.
+    """
+    if not _WANDB_AVAILABLE or _WANDB_RUN is None:
+        return
+    summarized: Dict[str, Any] = {}
+    def walk(node: Any, path: list[str]):
+        if isinstance(node, dict):
+            for k, v in node.items():
+                walk(v, path + [k])
+            return
+        # node is a list of values accumulated over time
+        key = ".".join([p for p in ([prefix] if prefix else []) + path])
+        try:
+            summary = _summarize_value(node)
+            for sk, sv in summary.items():
+                summarized[f"{key}.{sk}"] = sv
+        except Exception:
+            summarized[f"{key}.error"] = 1
+    walk(array_tally, [])
+    if summarized:
+        log(summarized, step=step)
+def log_flat_stats(
+    stats: Dict[str, Any], prefix: str = "", step: Optional[int] = None
+) -> None:
+    if not _WANDB_AVAILABLE or _WANDB_RUN is None:
+        return
+    flat: Dict[str, Any] = {}
+    _flatten(prefix, stats, flat)
+    if flat:
+        log(flat, step=step)