poolcoach / examples /random_agent_position.py
masterdanh's picture
deploy: snapshot for HF Space
78738de
Raw
History Blame Contribute Delete
2.11 kB
#!/usr/bin/env python3
"""Sanity + baseline PositionPlayEnv (stage 2a): agent random đánh N cú.
Chạy từ gốc repo:
python examples/random_agent_position.py 300
Mong đợi (design doc §7): env chạy không lỗi; pot ~2-4%, contact ~17-18%,
scratch ~20-22% (tương tự stage 1 vì reward gốc giữ nguyên); Q|pot là baseline
mới cần ghi lại để so với PPO.
"""
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
import numpy as np
from poolcoach_rl.envs import PositionPlayEnv
def main(n_episodes: int = 300):
env = PositionPlayEnv(seed=42)
rewards, aims, qs_on_pot = [], [], []
potted = scratched = contacted = b2_potted = makeable = 0
for _ in range(n_episodes):
env.reset()
action = env.action_space.sample()
_, reward, terminated, truncated, info = env.step(action)
assert terminated and not truncated
rewards.append(reward)
potted += info["potted"]
scratched += info["scratch"]
contacted += info["contact"]
b2_potted += info["b2_potted"]
aims.append(info["aim_cos"])
if info["potted"] and not info["scratch"]:
qs_on_pot.append(info["pos_q"])
makeable += info["pos_q"] > 0.5
n_pot = len(qs_on_pot)
print(f"Episodes : {n_episodes}")
print(f"Pot rate : {potted / n_episodes:.1%}")
print(f"Contact rate : {contacted / n_episodes:.1%}")
print(f"Scratch rate : {scratched / n_episodes:.1%}")
print(f"B2 lucky pot : {b2_potted / n_episodes:.1%}")
print(f"Q | pot : {np.mean(qs_on_pot):.3f} (n={n_pot})"
if n_pot else "Q | pot : n/a (0 cú pot)")
print(f"Makeable Q>.5 : {makeable / n_pot:.1%}" if n_pot
else "Makeable Q>.5 : n/a")
print(f"Aim cos mean : {np.mean(aims):+.4f}")
print(f"Reward mean : {np.mean(rewards):+.4f}")
print(f"Reward min/max: {np.min(rewards):+.4f} / {np.max(rewards):+.4f}")
if __name__ == "__main__":
n = int(sys.argv[1]) if len(sys.argv) > 1 else 300
main(n)