Spaces:
Paused
Paused
Sync Space with HCM-21-Private main (72f5f30): reward recalibration, seeded RNG streams, MCP integration, poster artifacts
670ccf0 verified | """Reward and final score computation for the HR Productivity Environment. | |
| Per-quarter rewards are sparse (given at advance_quarter). | |
| Final episode score is computed at the end of Q6, normalized to [0, 1]. | |
| CALIBRATION (v2 β anti-gaming): | |
| The Fitz-enz efficiency ratios HCVA (per-FTE) and HCROI (per-employment-cost) | |
| can be trivially inflated by shedding headcount: a do-nothing agent that lets | |
| attrition collapse the workforce 300 -> 41 drives HCVA/HCROI *up* ~75% while | |
| QIPS and engagement crater. Under the original sigmoid-centred scoring this | |
| produced a flat ~0.66 floor where do-nothing, random, and heuristic policies | |
| were indistinguishable. | |
| v2 fixes this by (1) using one-sided ramps so "no improvement" scores 0 (not | |
| 0.5), (2) gating efficiency credit by workforce health so you cannot win by | |
| downsizing, and (3) promoting workforce sustainability (headcount + engagement | |
| retention) to a first-class scoring term. Measured calibration after the fix: | |
| do-nothing ~0.15, random ~0.20-0.30, heuristic ~0.50, leaving headroom for a | |
| trained strategic agent at ~0.65+. | |
| """ | |
| from __future__ import annotations | |
| import math | |
| from typing import Any, Dict, List | |
| # ββ Targets (full credit at these improvement levels) βββββββββββββββ | |
| HCVA_TARGET = 0.25 # +25% HCVA over the episode = full efficiency credit | |
| HCROI_TARGET = 0.25 # +25% HCROI | |
| QIPS_TARGET = 0.05 # +5% QIPS (operational quality is hard to move) | |
| EV_TARGET = 0.05 # +5% employee-value composite | |
| RETENTION_FULL = 0.60 # retain >=60% of starting headcount for full health gate | |
| def compute_quarterly_reward( | |
| current_metrics: Dict[str, Any], | |
| previous_metrics: Dict[str, Any], | |
| ) -> float: | |
| """Per-quarter reward in roughly [-1, 1], anti-gaming. | |
| Efficiency gains (HCVA/HCROI) are gated by workforce health this quarter so an | |
| agent cannot farm reward by letting people leave. Headcount and engagement | |
| changes are first-class signals (a quarter that bleeds staff is punished). | |
| Weights: | |
| - Gated efficiency (HCVA + HCROI improvement): 40% | |
| - Operational quality (QIPS improvement): 25% | |
| - Workforce (headcount + engagement change): 35% | |
| """ | |
| hcva_change = _pct_change(current_metrics.get("hcva", 0), previous_metrics.get("hcva", 0)) | |
| hcroi_change = _pct_change(current_metrics.get("hcroi", 0), previous_metrics.get("hcroi", 0)) | |
| qips_change = _pct_change( | |
| current_metrics.get("qips", {}).get("composite", 0), | |
| previous_metrics.get("qips", {}).get("composite", 0), | |
| ) | |
| cur_snap = current_metrics.get("snapshot", {}) | |
| prev_snap = previous_metrics.get("snapshot", {}) | |
| hc_change = _pct_change(cur_snap.get("headcount", 0), prev_snap.get("headcount", 0)) | |
| eng_change = _pct_change(cur_snap.get("avg_engagement", 0), prev_snap.get("avg_engagement", 0)) | |
| # Health gate: a quarter that loses >=20% of headcount zeroes efficiency credit; | |
| # growth is capped at full credit (1.0). | |
| health = min(1.0, max(0.0, 1.0 + min(0.0, hc_change) / 0.20)) | |
| efficiency = ( | |
| 0.5 * _sigmoid_reward(hcva_change, scale=5.0) | |
| + 0.5 * _sigmoid_reward(hcroi_change, scale=5.0) | |
| ) * health | |
| quality = _sigmoid_reward(qips_change, scale=5.0) | |
| workforce = ( | |
| 0.5 * _sigmoid_reward(hc_change, scale=5.0) | |
| + 0.5 * _sigmoid_reward(eng_change, scale=5.0) | |
| ) | |
| reward = 0.40 * efficiency + 0.25 * quality + 0.35 * workforce | |
| return round(reward, 4) | |
| def compute_final_score(metric_history: List[Dict[str, Any]]) -> float: | |
| """Final episode score at end of Q6, normalized to [0, 1]. | |
| Weights: | |
| - Gated efficiency (HCVA + HCROI trajectory): 20% | |
| - Operational quality (QIPS level + growth): 20% | |
| - Employee-value composite (level + growth): 10% | |
| - Workforce sustainability (retention): 30% | |
| - Financial solvency (gated by health): 20% | |
| """ | |
| if len(metric_history) < 2: | |
| return 0.0 | |
| first, last = metric_history[0], metric_history[-1] | |
| # --- Workforce health (the anti-gaming backbone) --- | |
| # When snapshot data is absent (synthetic inputs), assume sustained: no | |
| # headcount information must not be read as a workforce collapse. | |
| hc0 = first.get("snapshot", {}).get("headcount") | |
| hcL = last.get("snapshot", {}).get("headcount") | |
| if hc0 and hcL is not None: | |
| retention = min(1.0, max(0.0, hcL / hc0)) | |
| else: | |
| retention = 1.0 | |
| eng0 = first.get("snapshot", {}).get("avg_engagement") | |
| engL = last.get("snapshot", {}).get("avg_engagement") | |
| if eng0 and engL is not None: | |
| eng_retention = min(1.0, max(0.0, engL / eng0)) | |
| else: | |
| eng_retention = 1.0 | |
| # Full efficiency credit only if >=RETENTION_FULL of the workforce is kept. | |
| health_gate = min(1.0, retention / RETENTION_FULL) | |
| # --- Efficiency trajectory (gated) --- | |
| hcva_imp = _pct_change(last.get("hcva", 0), first.get("hcva", 0)) | |
| hcroi_imp = _pct_change(last.get("hcroi", 0), first.get("hcroi", 0)) | |
| efficiency = (0.5 * _ramp(hcva_imp, HCVA_TARGET) + 0.5 * _ramp(hcroi_imp, HCROI_TARGET)) * health_gate | |
| # --- Operational quality: both sustained level and growth --- | |
| q0 = first.get("qips", {}).get("composite", 0) | |
| qL = last.get("qips", {}).get("composite", 0) | |
| qips_imp = _pct_change(qL, q0) | |
| qips_score = 0.5 * _ramp(qips_imp, QIPS_TARGET) + 0.5 * min(1.0, max(0.0, qL)) | |
| # --- Workforce capability --- | |
| ev0 = first.get("employee_value", 0.5) | |
| evL = last.get("employee_value", 0.5) | |
| ev_score = 0.6 * _ramp(evL - ev0, EV_TARGET) + 0.4 * min(1.0, max(0.0, evL)) | |
| # --- Sustainability: headcount retention scaled by engagement retention --- | |
| sustainability = retention * (0.5 + 0.5 * eng_retention) | |
| # --- Financial solvency, gated by health (a profitable shell is not healthy) --- | |
| profits = [m.get("profit", 0) for m in metric_history if "profit" in m] | |
| if profits: | |
| solvency = sum(1 for p in profits if p > 0) / len(profits) | |
| else: | |
| solvency = 0.5 | |
| financial = solvency * health_gate | |
| score = ( | |
| 0.20 * efficiency | |
| + 0.20 * qips_score | |
| + 0.10 * ev_score | |
| + 0.30 * sustainability | |
| + 0.20 * financial | |
| ) | |
| return round(min(1.0, max(0.0, score)), 4) | |
| # ββ Helpers βββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def _pct_change(curr: float, prev: float) -> float: | |
| """Signed fractional change; 0 when previous is 0.""" | |
| if prev == 0: | |
| return 0.0 | |
| return (curr - prev) / abs(prev) | |
| def _ramp(x: float, target: float) -> float: | |
| """One-sided linear ramp: 0 at x<=0, 1 at x>=target, linear between. | |
| Unlike a sigmoid, "no improvement" maps to 0 rather than 0.5, so a do-nothing | |
| policy collects no free credit. | |
| """ | |
| if target <= 0: | |
| return 0.0 | |
| return max(0.0, min(1.0, x / target)) | |
| def _sigmoid_reward(x: float, scale: float = 1.0) -> float: | |
| """Map a change value to [-1, 1] using sigmoid (for signed per-quarter rewards).""" | |
| return 2.0 / (1.0 + math.exp(-scale * x)) - 1.0 | |
| def _bounded_score(x: float, scale: float = 1.0) -> float: | |
| """Map a value to [0, 1] using sigmoid. Retained for compatibility.""" | |
| return 1.0 / (1.0 + math.exp(-scale * x)) | |
| def _trend_slope(values: List[float]) -> float: | |
| """Linear regression slope for a series of values. Retained for compatibility.""" | |
| n = len(values) | |
| if n < 2: | |
| return 0.0 | |
| x_mean = (n - 1) / 2.0 | |
| y_mean = sum(values) / n | |
| numerator = sum((i - x_mean) * (v - y_mean) for i, v in enumerate(values)) | |
| denominator = sum((i - x_mean) ** 2 for i in range(n)) | |
| if denominator == 0: | |
| return 0.0 | |
| return numerator / denominator | |