hcm21 / hr_env /server /scoring.py
ParetoOptimal's picture
Sync Space with HCM-21-Private main (72f5f30): reward recalibration, seeded RNG streams, MCP integration, poster artifacts
670ccf0 verified
Raw
History Blame Contribute Delete
8.07 kB
"""Reward and final score computation for the HR Productivity Environment.
Per-quarter rewards are sparse (given at advance_quarter).
Final episode score is computed at the end of Q6, normalized to [0, 1].
CALIBRATION (v2 β€” anti-gaming):
The Fitz-enz efficiency ratios HCVA (per-FTE) and HCROI (per-employment-cost)
can be trivially inflated by shedding headcount: a do-nothing agent that lets
attrition collapse the workforce 300 -> 41 drives HCVA/HCROI *up* ~75% while
QIPS and engagement crater. Under the original sigmoid-centred scoring this
produced a flat ~0.66 floor where do-nothing, random, and heuristic policies
were indistinguishable.
v2 fixes this by (1) using one-sided ramps so "no improvement" scores 0 (not
0.5), (2) gating efficiency credit by workforce health so you cannot win by
downsizing, and (3) promoting workforce sustainability (headcount + engagement
retention) to a first-class scoring term. Measured calibration after the fix:
do-nothing ~0.15, random ~0.20-0.30, heuristic ~0.50, leaving headroom for a
trained strategic agent at ~0.65+.
"""
from __future__ import annotations
import math
from typing import Any, Dict, List
# ── Targets (full credit at these improvement levels) ───────────────
HCVA_TARGET = 0.25 # +25% HCVA over the episode = full efficiency credit
HCROI_TARGET = 0.25 # +25% HCROI
QIPS_TARGET = 0.05 # +5% QIPS (operational quality is hard to move)
EV_TARGET = 0.05 # +5% employee-value composite
RETENTION_FULL = 0.60 # retain >=60% of starting headcount for full health gate
def compute_quarterly_reward(
current_metrics: Dict[str, Any],
previous_metrics: Dict[str, Any],
) -> float:
"""Per-quarter reward in roughly [-1, 1], anti-gaming.
Efficiency gains (HCVA/HCROI) are gated by workforce health this quarter so an
agent cannot farm reward by letting people leave. Headcount and engagement
changes are first-class signals (a quarter that bleeds staff is punished).
Weights:
- Gated efficiency (HCVA + HCROI improvement): 40%
- Operational quality (QIPS improvement): 25%
- Workforce (headcount + engagement change): 35%
"""
hcva_change = _pct_change(current_metrics.get("hcva", 0), previous_metrics.get("hcva", 0))
hcroi_change = _pct_change(current_metrics.get("hcroi", 0), previous_metrics.get("hcroi", 0))
qips_change = _pct_change(
current_metrics.get("qips", {}).get("composite", 0),
previous_metrics.get("qips", {}).get("composite", 0),
)
cur_snap = current_metrics.get("snapshot", {})
prev_snap = previous_metrics.get("snapshot", {})
hc_change = _pct_change(cur_snap.get("headcount", 0), prev_snap.get("headcount", 0))
eng_change = _pct_change(cur_snap.get("avg_engagement", 0), prev_snap.get("avg_engagement", 0))
# Health gate: a quarter that loses >=20% of headcount zeroes efficiency credit;
# growth is capped at full credit (1.0).
health = min(1.0, max(0.0, 1.0 + min(0.0, hc_change) / 0.20))
efficiency = (
0.5 * _sigmoid_reward(hcva_change, scale=5.0)
+ 0.5 * _sigmoid_reward(hcroi_change, scale=5.0)
) * health
quality = _sigmoid_reward(qips_change, scale=5.0)
workforce = (
0.5 * _sigmoid_reward(hc_change, scale=5.0)
+ 0.5 * _sigmoid_reward(eng_change, scale=5.0)
)
reward = 0.40 * efficiency + 0.25 * quality + 0.35 * workforce
return round(reward, 4)
def compute_final_score(metric_history: List[Dict[str, Any]]) -> float:
"""Final episode score at end of Q6, normalized to [0, 1].
Weights:
- Gated efficiency (HCVA + HCROI trajectory): 20%
- Operational quality (QIPS level + growth): 20%
- Employee-value composite (level + growth): 10%
- Workforce sustainability (retention): 30%
- Financial solvency (gated by health): 20%
"""
if len(metric_history) < 2:
return 0.0
first, last = metric_history[0], metric_history[-1]
# --- Workforce health (the anti-gaming backbone) ---
# When snapshot data is absent (synthetic inputs), assume sustained: no
# headcount information must not be read as a workforce collapse.
hc0 = first.get("snapshot", {}).get("headcount")
hcL = last.get("snapshot", {}).get("headcount")
if hc0 and hcL is not None:
retention = min(1.0, max(0.0, hcL / hc0))
else:
retention = 1.0
eng0 = first.get("snapshot", {}).get("avg_engagement")
engL = last.get("snapshot", {}).get("avg_engagement")
if eng0 and engL is not None:
eng_retention = min(1.0, max(0.0, engL / eng0))
else:
eng_retention = 1.0
# Full efficiency credit only if >=RETENTION_FULL of the workforce is kept.
health_gate = min(1.0, retention / RETENTION_FULL)
# --- Efficiency trajectory (gated) ---
hcva_imp = _pct_change(last.get("hcva", 0), first.get("hcva", 0))
hcroi_imp = _pct_change(last.get("hcroi", 0), first.get("hcroi", 0))
efficiency = (0.5 * _ramp(hcva_imp, HCVA_TARGET) + 0.5 * _ramp(hcroi_imp, HCROI_TARGET)) * health_gate
# --- Operational quality: both sustained level and growth ---
q0 = first.get("qips", {}).get("composite", 0)
qL = last.get("qips", {}).get("composite", 0)
qips_imp = _pct_change(qL, q0)
qips_score = 0.5 * _ramp(qips_imp, QIPS_TARGET) + 0.5 * min(1.0, max(0.0, qL))
# --- Workforce capability ---
ev0 = first.get("employee_value", 0.5)
evL = last.get("employee_value", 0.5)
ev_score = 0.6 * _ramp(evL - ev0, EV_TARGET) + 0.4 * min(1.0, max(0.0, evL))
# --- Sustainability: headcount retention scaled by engagement retention ---
sustainability = retention * (0.5 + 0.5 * eng_retention)
# --- Financial solvency, gated by health (a profitable shell is not healthy) ---
profits = [m.get("profit", 0) for m in metric_history if "profit" in m]
if profits:
solvency = sum(1 for p in profits if p > 0) / len(profits)
else:
solvency = 0.5
financial = solvency * health_gate
score = (
0.20 * efficiency
+ 0.20 * qips_score
+ 0.10 * ev_score
+ 0.30 * sustainability
+ 0.20 * financial
)
return round(min(1.0, max(0.0, score)), 4)
# ── Helpers ─────────────────────────────────────────────────────────
def _pct_change(curr: float, prev: float) -> float:
"""Signed fractional change; 0 when previous is 0."""
if prev == 0:
return 0.0
return (curr - prev) / abs(prev)
def _ramp(x: float, target: float) -> float:
"""One-sided linear ramp: 0 at x<=0, 1 at x>=target, linear between.
Unlike a sigmoid, "no improvement" maps to 0 rather than 0.5, so a do-nothing
policy collects no free credit.
"""
if target <= 0:
return 0.0
return max(0.0, min(1.0, x / target))
def _sigmoid_reward(x: float, scale: float = 1.0) -> float:
"""Map a change value to [-1, 1] using sigmoid (for signed per-quarter rewards)."""
return 2.0 / (1.0 + math.exp(-scale * x)) - 1.0
def _bounded_score(x: float, scale: float = 1.0) -> float:
"""Map a value to [0, 1] using sigmoid. Retained for compatibility."""
return 1.0 / (1.0 + math.exp(-scale * x))
def _trend_slope(values: List[float]) -> float:
"""Linear regression slope for a series of values. Retained for compatibility."""
n = len(values)
if n < 2:
return 0.0
x_mean = (n - 1) / 2.0
y_mean = sum(values) / n
numerator = sum((i - x_mean) * (v - y_mean) for i, v in enumerate(values))
denominator = sum((i - x_mean) ** 2 for i in range(n))
if denominator == 0:
return 0.0
return numerator / denominator