Add files using upload-large-folder tool

Browse files

Files changed (17) hide show

.gitattributes +1 -0
run.log +3 -0
seed_1/Qwen/Qwen2.5-7B-Instruct/adapters/agent_adapter/adapter_model.safetensors +3 -0
seed_1/Qwen/Qwen2.5-7B-Instruct/adapters/critic_adapter/adapter_model.safetensors +3 -0
seed_1/Qwen/Qwen2.5-7B-Instruct/adapters/fixed_ad_align_adapter/adapter_model.safetensors +3 -0
seed_1/agent_trainer/policy_optimizer_state.pt +3 -0
seed_1/agent_trainer/trainer_annealing_state.pkl +3 -0
seed_1/random_state.pkl +3 -0
src_code_for_reproducibility/markov_games/__pycache__/alternative_actions_runner.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/__pycache__/negotiation_statistics.cpython-312.pyc +0 -0
src_code_for_reproducibility/markov_games/negotiation/dond_simulation.py +176 -0
src_code_for_reproducibility/markov_games/negotiation/negotiation_statistics.py +249 -0
src_code_for_reproducibility/markov_games/negotiation/tas_rps_agent.py +128 -0
src_code_for_reproducibility/models/__pycache__/__init__.cpython-312.pyc +0 -0
src_code_for_reproducibility/models/__pycache__/inference_backend.cpython-312.pyc +0 -0
src_code_for_reproducibility/training/__pycache__/tokenize_chats.cpython-312.pyc +0 -0
src_code_for_reproducibility/training/__pycache__/trainer_sum_rewards.cpython-312.pyc +0 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+run.log filter=lfs diff=lfs merge=lfs -text

run.log ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3c50803dd26fc6b9db435c1427981e4017e7e156753c5cf5e60710cf5afdd64d
+size 10900718

seed_1/Qwen/Qwen2.5-7B-Instruct/adapters/agent_adapter/adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6ab958fd2facd005cae5571b0df06ed3be786697f4c31d6435ffcbc655d2920b
+size 323014168

seed_1/Qwen/Qwen2.5-7B-Instruct/adapters/critic_adapter/adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4c44c3464099d92dfebb2b132524339800fbf19760b378a02c3c527ac3380b88
+size 323014168

seed_1/Qwen/Qwen2.5-7B-Instruct/adapters/fixed_ad_align_adapter/adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:cf6bb8f8d702f23ed3c0797660ebbc16bdee9cbac5c984ffbad4a1dc3ba2215c
+size 323014168

seed_1/agent_trainer/policy_optimizer_state.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1dca3a51476df532d5a63aa1f269f3467830fee1573883a3cb10d0857ddd4111
+size 646269121

seed_1/agent_trainer/trainer_annealing_state.pkl ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:17f3ead2dac3c925aeb1b3176d071b434c765c0606d4e707e423de4498633e52
+size 104

seed_1/random_state.pkl ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:06d527c81a0ed8e596458de353799680ab01076dfc3d43cd3b1a2ebea4439ac5
+size 12218

src_code_for_reproducibility/markov_games/__pycache__/alternative_actions_runner.cpython-312.pyc ADDED Viewed

Binary file (5.42 kB). View file

src_code_for_reproducibility/markov_games/negotiation/__pycache__/negotiation_statistics.cpython-312.pyc ADDED Viewed

Binary file (14.2 kB). View file

src_code_for_reproducibility/markov_games/negotiation/dond_simulation.py ADDED Viewed

	@@ -0,0 +1,176 @@

+"""
+File: mllm/markov_games/negotiation/dond_simulation.py
+Summary: Simulates Deal-or-No-Deal negotiation games and logs rollouts.
+"""
+import copy
+from dataclasses import dataclass
+from typing import Any, Dict, List, Tuple
+from numpy.random import default_rng
+from mllm.markov_games.negotiation.nego_simulation import (
+    NegotiationObs,
+    NegotiationSimulation,
+    NegotiationState,
+    Split,
+)
+from mllm.markov_games.rollout_tree import SimulationStepLog
+from mllm.utils.get_coagent_id import get_coagent_id
+AgentId = str
+@dataclass
+class DealNoDealState(NegotiationState):
+    """NegotiationState with per-agent value tables and item taxonomy."""
+    item_types: List[str]
+    values: Dict[AgentId, Dict[str, int]]
+@dataclass
+class DealNoDealObs(NegotiationObs):
+    """Observation that reveals own values and (lagged) opponent values."""
+    my_values: Dict[str, int]
+    item_types: List[str]
+    previous_values_coagent: Dict[str, int] | None
+def random_partition_integer(rng, total: int, parts: int) -> List[int]:
+    """Sample non-negative integers summing to ``total`` across ``parts`` buckets."""
+    if parts <= 0:
+        return []
+    if total <= 0:
+        return [0 for _ in range(parts)]
+    cuts = sorted(rng.integers(0, total + 1, size=parts - 1).tolist())
+    vals = []
+    prev = 0
+    for c in cuts + [total]:
+        vals.append(c - prev)
+        prev = c
+    return vals
+class DealNoDealSimulation(NegotiationSimulation):
+    """NegotiationSimulation variant implementing the Rubinstein-style Deal-or-No-Deal."""
+    def __init__(
+        self,
+        item_types: List[str] = ["books", "hats", "balls"],
+        *args,
+        **kwargs,
+    ):
+        super().__init__(item_types=item_types, *args, **kwargs)
+        self.reset()
+    def _other(self, agent_id: AgentId) -> AgentId:
+        return get_coagent_id(self.agent_ids, agent_id)
+    def _sample_stock(self) -> Dict[str, int]:
+        # total items between 5 and 7
+        total_items = int(self.rng.integers(5, 8))
+        # nonnegative per-type counts summing to total_items
+        parts = random_partition_integer(self.rng, total_items, len(self.item_types))
+        # allow zeros per type
+        return {t: int(c) for t, c in zip(self.item_types, parts)}
+    def _sample_values_pair(self) -> Dict[AgentId, Dict[str, int]]:
+        # Each agent has integer non-negative values that sum to 10
+        # Each item type valued by at least one agent
+        # Some item type valued by both agents
+        while True:
+            vals_a = random_partition_integer(self.rng, 10, len(self.item_types))
+            vals_b = random_partition_integer(self.rng, 10, len(self.item_types))
+            a = {t: int(v) for t, v in zip(self.item_types, vals_a)}
+            b = {t: int(v) for t, v in zip(self.item_types, vals_b)}
+            # each item valued by at least one
+            ok1 = all((a[t] > 0) or (b[t] > 0) for t in self.item_types)
+            # some item valued by both
+            ok2 = any((a[t] > 0) and (b[t] > 0) for t in self.item_types)
+            if ok1 and ok2:
+                return {self.agent_ids[0]: a, self.agent_ids[1]: b}
+    def _is_valid_allocation(
+        self, allocation: Dict[str, int], stock: Dict[str, int]
+    ) -> bool:
+        for t in self.item_types:
+            v = allocation.get(t)
+            if v is None:
+                return False
+            if not isinstance(v, int):
+                return False
+            if v < 0 or v > int(stock.get(t, 0)):
+                return False
+        return True
+    def set_new_round_of_variant(self):
+        # Keep same values, resample stock
+        self.state.quantities = self._sample_stock()
+    def get_info_of_variant(
+        self, state: NegotiationState, actions: Dict[AgentId, Any]
+    ) -> Dict[str, Any]:
+        return {
+            "quantities": copy.deepcopy(state.quantities),
+            "values": copy.deepcopy(state.values),
+            "splits": copy.deepcopy(state.splits),
+        }
+    def get_rewards(self, splits: Dict[AgentId, Split]) -> Dict[AgentId, float]:
+        """
+        Returns the rewards for each agent.
+        """
+        split_a = splits[self.agent_ids[0]].items_given_to_self
+        split_b = splits[self.agent_ids[1]].items_given_to_self
+        rewards = {self.agent_ids[0]: 0, self.agent_ids[1]: 0}
+        for t in self.item_types:
+            # If not complementary, return 0!
+            if not split_a[t] + split_b[t] == self.state.quantities[t]:
+                return {self.agent_ids[0]: 0, self.agent_ids[1]: 0}
+            rewards[self.agent_ids[0]] += (
+                split_a[t] * self.state.values[self.agent_ids[0]][t]
+            )
+            rewards[self.agent_ids[1]] += (
+                split_b[t] * self.state.values[self.agent_ids[1]][t]
+            )
+        return rewards
+    def get_obs(self):
+        return {agent_id: self.get_obs_agent(agent_id) for agent_id in self.agent_ids}
+    def get_obs_agent(self, agent_id):
+        other_id = self._other(agent_id)
+        obs = DealNoDealObs(
+            round_nb=self.state.round_nb,
+            last_message=self.state.last_message,
+            current_agent=self.state.current_agent,
+            quantities=copy.deepcopy(self.state.quantities),
+            value=0.0,  # unused in DOND
+            other_agent_split=None,  # not meaningful until split
+            split_phase=self.state.split_phase,
+            quota_messages_per_agent_per_round=self.quota_messages_per_agent_per_round,
+            my_values=copy.deepcopy(self.state.values[agent_id]),
+            item_types=list(self.item_types),
+            previous_values_coagent=copy.deepcopy(self.state.values.get(other_id, {})),
+        )
+        return obs
+    def reset(self):
+        start_agent = self.agent_ids[self._starting_agent_index]
+        stock = self._sample_stock()
+        values = self._sample_values_pair()
+        self.state = DealNoDealState(
+            round_nb=0,
+            last_message="",
+            current_agent=start_agent,
+            quantities=stock,
+            values=values,
+            previous_values=None,
+            splits={aid: None for aid in self.agent_ids},
+            nb_messages_sent={aid: 0 for aid in self.agent_ids},
+            split_phase=False,
+            item_types=list(self.item_types),
+        )
+        return self.get_obs()

src_code_for_reproducibility/markov_games/negotiation/negotiation_statistics.py ADDED Viewed

	@@ -0,0 +1,249 @@

+"""
+File: mllm/markov_games/negotiation/negotiation_statistics.py
+Summary: Aggregates and reports statistics for negotiation experiments.
+"""
+from __future__ import annotations
+from typing import Callable, Dict, List, Tuple
+from mllm.markov_games.negotiation.nego_simulation import Split
+from mllm.markov_games.rollout_tree import SimulationStepLog
+def avg_reward(sl: SimulationStepLog) -> List[Tuple[str, float]]:
+    """Average (per-step) reward for each agent and overall.
+    What it computes:
+            - Returns the raw reward for every (non-buffer) agent at the current
+                simulation step.
+            - Adds an aggregate key ``all_agents`` which is the simple arithmetic
+                mean across the agents present in ``sl.rewards``.
+    Rationale / motivation:
+            Monitoring the reward stream at each step helps:
+                * Diagnose reward shaping issues (e.g., unintended negative drift).
+                * Provide a fairness snapshot (are rewards systematically skewed?).
+                * Supply a ubiquitous baseline metric used by other higher‑level
+                    summaries (efficiency, surplus allocation, etc.).
+    Return shape:
+            { agent_id: float, ..., "all_agents": float }
+            If any agent id contains the substring "buffer" we treat this step as
+            an implementation artifact (e.g., rollout buffer) and return ``None``
+            to avoid polluting aggregates.
+    """
+    for aid in sl.rewards.keys():
+        if "buffer" in str(aid) and "live" not in str(aid):
+            return None
+    # One value per agent at each step
+    rewards_dict = {f"reward-{aid}": float(v) for aid, v in (sl.rewards or {}).items()}
+    return [(key, value) for key, value in rewards_dict.items() if value is not None]
+def split_efficiency(sl: SimulationStepLog) -> List[Tuple[str, float]] | None:
+    """Final‑round allocation efficiency relative to an upper bound.
+    What it computes (only on the last timestep of a negotiation round):
+            - Uses ``info['values']`` (per‑agent per‑item valuations) and
+                ``info['quantities']`` (available item counts) to form a greedy
+                *upper bound* on achievable total reward: allocate each unit of an
+                item to the single agent who values that item most.
+            - Compares the actually realized sum of rewards at that final
+                timestep to this constructed maximum.
+            - Emits a single scalar under key ``"all_agents"`` equal to
+                achieved / theoretical_max.
+    Motivation:
+            Efficiency (a core welfare notion) distinguishes between coordination
+            failures (low efficiency) versus strategic distributional disputes
+            (high efficiency but uneven splits). Tracking this per round helps
+            evaluate whether models learn to identify and realize joint surplus.
+    Notes / caveats:
+            - Only defined for 2+ non‑buffer agents; if a buffer agent is present
+                returns ``None`` to exclude spurious steps.
+            - Requires the environment to have populated ``values`` and
+                ``quantities``; otherwise returns ``None``.
+            - This is an optimistic bound (not necessarily reachable under
+                protocol constraints) but is simple, fast, and comparable across
+                runs.
+    """
+    info = sl.info or {}
+    if not info or not info.get("is_last_timestep_in_round"):
+        return None
+    quantities = info.get("quantities") or {}
+    values = info.get("values") or {}
+    if not values or not quantities:
+        return None
+    agent_ids = list(sl.rewards.keys())
+    if type(values[agent_ids[0]]) is dict:
+        item_keys = list(values.values())[0].keys()
+        max_vals, max_quantities = [], []
+        for item in item_keys:
+            max_val = max(float(agent_vals[item]) for agent_vals in values.values())
+            max_vals.append(max_val)
+            max_quantities.append(quantities[item])
+    else:
+        max_vals = [max(float(v) for v in values.values())]
+        max_quantities = [quantities[item] for item in quantities.keys()]
+    for aid in sl.rewards.keys():
+        if "buffer" in str(aid) and "live" not in str(aid):
+            return None
+    achieved = sum(float(v) for v in sl.rewards.values())
+    max_reward = sum(d * v for d, v in zip(max_quantities, max_vals))
+    # Efficiency is a global metric; emit same value for a special key "all"
+    return [("split_efficiency", achieved / max_reward)]
+def _extract_items_from_split(raw_split: Dict) -> Dict[str, float] | None:
+    """Return a mapping item->proposal amount from a split structure.
+    Supports both generic negotiation splits with nested structure
+    { 'items_given_to_self': {item: qty, ...}}
+    and TAS coin-only variants which may already be a flat mapping {'coins': qty}.
+    """
+    if raw_split is None:
+        return {}
+    elif isinstance(raw_split, Split):
+        return {k: float(v) for k, v in raw_split.items_given_to_self.items()}
+    elif isinstance(raw_split, dict):
+        if "items_given_to_self" in raw_split and isinstance(
+            raw_split["items_given_to_self"], dict
+        ):
+            return {k: float(v) for k, v in raw_split["items_given_to_self"].items()}
+        # Fallback: assume already flat mapping of items
+        elif hasattr(raw_split, "items_given_to_self"):
+            return {k: float(v) for k, v in raw_split["items_given_to_self"].items()}
+        return {
+            k: float(v) for k, v in raw_split.items() if isinstance(v, (int, float))
+        }
+    return {}
+def _average_proposal_relative_value(
+    sl: SimulationStepLog,
+    metric_name: str,
+    comparator: Callable[[float, float], bool],
+    opposite_comparator: Callable[[float, float], bool],
+) -> Dict[str, float | None] | None:
+    """Shared implementation for proposal size conditioned on relative value.
+    Parameters:
+            comparator: returns True when agent_0's value relation (e.g. < or >)
+                                    to agent_1 holds for an item and we should collect agent_0's
+                                    proposed quantity for that item.
+            opposite_comparator: inverse relation used to collect agent_1's items.
+    Behavior:
+            - Executes only on final timestep of a round (where the definitive
+                proposal / allocation is known via ``info['splits']``).
+            - For each item, classifies which agent's value satisfies the chosen
+                relation and records that agent's proposed quantity from the split.
+            - Averages (mean) across all qualifying items per agent; if no items
+                qualify for an agent returns ``None`` for that agent id.
+            - Adds ``all_agents`` mean across the numeric (non-None) agent values.
+    Why this matters:
+            Distinguishing how much an agent *asks for* when it subjectively
+            values items more (or less) than its counterpart reveals patterns of
+            opportunism vs. concession. This is especially useful when raw reward
+            differences are subtle but allocation *intent* differs.
+    """
+    info = sl.info or {}
+    if not info or not info.get("is_last_timestep_in_round"):
+        return None
+    quantities = info.get("quantities") or {}
+    splits = info.get("splits") or {}
+    values = info.get("values") or {}
+    agent_ids: List[str] = list(sl.rewards.keys())
+    if len(agent_ids) != 2:
+        return None  # Only defined for 2-agent case.
+    for aid in agent_ids:
+        if "buffer" in str(aid) and "live" not in str(aid):
+            return None
+    # Extract per-agent item proposals robustly
+    split_items = {aid: _extract_items_from_split(splits.get(aid)) for aid in agent_ids}
+    agent_0_vals: List[float] = []
+    agent_1_vals: List[float] = []
+    for item in quantities.keys():
+        # Values may be either a float (same for all items) or dict per item
+        v0_raw = values[agent_ids[0]]
+        v1_raw = values[agent_ids[1]]
+        v0 = float(v0_raw[item]) if isinstance(v0_raw, dict) else float(v0_raw)
+        v1 = float(v1_raw[item]) if isinstance(v1_raw, dict) else float(v1_raw)
+        if comparator(v0, v1):
+            agent_0_vals.append(split_items[agent_ids[0]].get(item, 0.0))
+        elif opposite_comparator(v0, v1):
+            agent_1_vals.append(split_items[agent_ids[1]].get(item, 0.0))
+    out: Dict[str, float | None] = {}
+    out[f"{metric_name}-{agent_ids[0]}"] = (
+        sum(agent_0_vals) / len(agent_0_vals) if agent_0_vals else None
+    )
+    out[f"{metric_name}-{agent_ids[1]}"] = (
+        sum(agent_1_vals) / len(agent_1_vals) if agent_1_vals else None
+    )
+    return [(key, value) for key, value in out.items() if value is not None]
+def average_proposal_when_agent_values_item_lower(
+    sl: SimulationStepLog,
+) -> List[Tuple[str, float | None]] | None:
+    """Mean quantity an agent proposes for items it values *less* than opponent.
+    Interpretation:
+        A higher value implies the agent still claims (or is allocated) a
+        notable share of items where it has a comparative *disadvantage* in
+        valuation, signaling either strategic over-claiming or protocol-driven
+        egalitarian splits. Conversely, very low numbers can indicate
+        efficient specialization or excessive concession.
+    Returns:
+        Mapping { agent_id: float | None, "all_agents": float | None } where
+        None indicates no qualifying items for that agent in the round.
+    """
+    return _average_proposal_relative_value(
+        sl,
+        "average_proposal_when_agent_values_item_lower",
+        lambda a, b: a < b,
+        lambda a, b: a > b,
+    )
+def average_proposal_when_agent_values_item_higher(
+    sl: SimulationStepLog,
+) -> List[Tuple[str, float | None]] | None:
+    """Mean quantity an agent proposes for items it values *more* than opponent.
+    Interpretation:
+        Captures how aggressively an agent claims items where it holds a
+        comparative *advantage*. Elevated values can reflect rational
+        specialization (efficient exploitation of comparative advantage) or
+        potentially unfair grabs if paired with low concession in the lower
+        valuation metric. Comparing this with the 'lower' counterpart helps
+        profile negotiation style (cooperative vs. exploitative).
+    Returns:
+        Mapping { agent_id: float | None, "all_agents": float | None } where
+        None indicates no qualifying items.
+    """
+    return _average_proposal_relative_value(
+        sl,
+        "average_proposal_when_agent_values_item_higher",
+        lambda a, b: a > b,
+        lambda a, b: a < b,
+    )
+# Explicit list of metric functions exported for rendering. Helper functions
+# starting with '_' are intentionally excluded. Update this list when adding
+# new public statistics so render.py can rely on it instead of introspecting
+# every callable in the module.
+stat_functs: list[Callable[[SimulationStepLog], List[Tuple[str, float]]]] = [
+    avg_reward,
+    average_proposal_when_agent_values_item_lower,
+    average_proposal_when_agent_values_item_higher,
+    split_efficiency,
+]

src_code_for_reproducibility/markov_games/negotiation/tas_rps_agent.py ADDED Viewed

	@@ -0,0 +1,128 @@

+"""
+File: mllm/markov_games/negotiation/tas_rps_agent.py
+Summary: Agent logic for TAS Rock-Paper-Scissors blended game.
+"""
+import copy
+from collections.abc import Callable
+from dataclasses import dataclass
+from typing import Any, Dict, List, Tuple
+from mllm.markov_games.agent import Agent
+from mllm.markov_games.negotiation.nego_agent import (
+    Message,
+    NegotiationAgent,
+    NegotiationAgentState,
+    Split,
+)
+from mllm.markov_games.negotiation.tas_rps_simulation import TrustAndSplitRPSObs
+from mllm.markov_games.rollout_tree import AgentActLog, ChatTurn
+class TrustAndSplitRPSAgent(NegotiationAgent):
+    """NegotiationAgent that reasons about hidden hands before submitting TAS splits."""
+    def __init__(
+        self,
+        num_message_chars: int,
+        message_start_end_format: bool = False,
+        proposal_start_end_format: bool = False,
+        *args,
+        **kwargs,
+    ):
+        self.num_message_chars = num_message_chars
+        self.message_start_end_format = message_start_end_format
+        self.proposal_start_end_format = proposal_start_end_format
+        super().__init__(*args, **kwargs)
+        self.intro_prompt = (
+            "Welcome to an iterated game. You are {agent}. The other agent is {other_agent}.\n"
+            "\n"
+            "Setup:\n"
+            "1. The game has multiple independent rounds.\n"
+            "2. In each round, there are 10 coins to split between the two agents.\n"
+            "3. Each agent's per-coin value for that round is determined as follows:\n"
+            "   - Both agents are randomly assigned a rock, paper or scissors hands\n"
+            "   - Rock has the upper hand over scissors, scissors has the upper hand over paper and paper has the upper hand over rock.\n"
+            "   - The agent with the upper hand has a per-coin value of 10.\n"
+            "   - The agent with the lower hand has a per-coin value of 1.\n"
+            "4. You only see your own hand, but you may communicate it in messages and infer your value based on the other agent's hand.\n"
+            "5. Over many rounds both agents are equally likely to have the upper and lower hand.\n"
+            "\n"
+            "Protocol:\n"
+            "1. At the start of the round, one agent begins the conversation. The starting role alternates each round.\n"
+            "2. Agents exchange a short chat ({quota_messages_per_agent_per_round} messages per round per agent) to negotiate how to split the 10 coins.\n"
+            "   - Use this chat to communicate your hand so that both agents can determine their per-coin values.\n"
+            "3. After the chat, both agents simultaneously propose how many coins they keep.\n"
+            "4. If the total sum of proposals is less than or equal to 10, both agents receive their proposals.\n"
+            "5. If the total sum of proposals exceeds 10, the coins are allocated proportionally.\n"
+            "6. Your points for the round = (coins you receive) x (your per-coin value for that round). \n"
+            "7. The points are accumulated across rounds.\n"
+            "Your goal: {goal}\n"
+        )
+        self.new_round_prompt = (
+            "A New Round Begins\n"
+            "Your hand is {hand}. You don't know {other_agent}'s hand yet.\n"
+        )
+        # self.last_round_prompt = (
+        #     "Last Round Summary:\n"
+        #     "   - Your hand: {last_hand_agent}\n"
+        #     "   - {other_agent}'s hand: {last_hand_coagent}\n"
+        #     "   - Your value per coin: {last_value_agent}\n"
+        #     "   - {other_agent}'s value per coin: {last_value_coagent}\n"
+        #     "   - You proposed: {last_split_agent} coins\n"
+        #     "   - You earned: {last_points_agent} points\n"
+        #     "   - {other_agent} proposed: {last_split_coagent} coins\n"
+        #     "   - {other_agent} earned: {last_points_coagent} points\n"
+        #     "   - Round Complete.\n"
+        # )
+        self.last_round_prompt = "In the previous round, {other_agent} had a {last_hand_value_coagent} hand and proposed {last_split_coagent} coins.\n"
+        if self.proposal_start_end_format:
+            self.send_split_prompt = (
+                "Submit your proposal\n"
+                "Respond with <<proposal_start>> x <<proposal_end>> where x is an integer in [0, 10]."
+            )
+        else:
+            self.send_split_prompt = (
+                "Submit your proposal\n"
+                "Respond with <coins_to_self> x </coins_to_self> where x is an integer in [0, 10]."
+            )
+        self.wait_for_message_prompt = "Wait for {other_agent} to send a message..."
+        # self.wait_for_message_prompt = ""
+        self.last_message_prompt = "{other_agent} said: {last_message}"
+        if self.message_start_end_format:
+            self.send_message_prompt = f"Send your message now in <<message_start>>...<<message_end>> (<={self.num_message_chars} chars)."
+        else:
+            self.send_message_prompt = f"Send your message now in <message>...</message> (<={self.num_message_chars} chars)."
+    def get_message_regex(self, observation: TrustAndSplitRPSObs) -> str:
+        """Switch between <message>...</message> and <<message_start>> formats on demand."""
+        if self.message_start_end_format:
+            return (
+                rf"<<message_start>>[\s\S]{{0,{self.num_message_chars}}}<<message_end>>"
+            )
+        else:
+            return rf"<message>[\s\S]{{0,{self.num_message_chars}}}</message>"
+    def get_split_regex(self, observation: TrustAndSplitRPSObs) -> str:
+        """Force single-number proposals inside whichever tag style the config selected."""
+        if self.proposal_start_end_format:
+            return r"<<proposal_start>> ?(10|[0-9]) ?<<proposal_end>>"
+        else:
+            return r"<coins_to_self> ?(10|[0-9]) ?</coins_to_self>"
+    def get_split_action(
+        self, policy_output: str, observation: TrustAndSplitRPSObs
+    ) -> Split:
+        """Parse the proposal tag (or raw integer fallback) into a Split."""
+        import re as _re
+        if self.proposal_start_end_format:
+            m = _re.search(
+                r"<<proposal_start>> ?(10|[0-9]) ?<<proposal_end>>", policy_output
+            )
+        else:
+            m = _re.search(
+                r"<coins_to_self> ?(10|[0-9]) ?</coins_to_self>", policy_output
+            )
+        coins_int = int(m.group(1)) if m else int(policy_output)
+        return Split(items_given_to_self={"coins": coins_int})

src_code_for_reproducibility/models/__pycache__/__init__.cpython-312.pyc ADDED Viewed

Binary file (260 Bytes). View file

src_code_for_reproducibility/models/__pycache__/inference_backend.cpython-312.pyc ADDED Viewed

Binary file (2.37 kB). View file

src_code_for_reproducibility/training/__pycache__/tokenize_chats.cpython-312.pyc ADDED Viewed

Binary file (5.97 kB). View file

src_code_for_reproducibility/training/__pycache__/trainer_sum_rewards.cpython-312.pyc ADDED Viewed

Binary file (6.02 kB). View file