File size: 6,042 Bytes
4be6a52 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 | """Use off-study toy episodes; export tests do not inspect held-out games."""
import copy
import hashlib
import importlib.util
import json
from pathlib import Path
import pytest
from stackcraft.players import HeuristicPlayer, RandomPlayer
from stackcraft.tournament import run_tournament
TOY_SEED = 991337
@pytest.fixture
def exporter(monkeypatch):
path = Path(__file__).parents[1] / "scripts/export_demo.py"
spec = importlib.util.spec_from_file_location("stackcraft_export_demo_test", path)
assert spec is not None and spec.loader is not None
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
# Production requires 30000; this seam keeps synthetic unit tests off test seeds.
monkeypatch.setattr(module, "DEMO_SEED", TOY_SEED)
return module
@pytest.fixture
def toy(tmp_path):
result = run_tournament({"random": RandomPlayer, "heuristic": HeuristicPlayer}, [TOY_SEED], 3)
path = tmp_path / "toy-tournament.json"
path.write_text(json.dumps(result))
return result, path
def test_export_preserves_real_replays_and_probabilities_with_source_hash(exporter, toy):
original, path = toy
result = exporter.export_manifest([path], ["heuristic", "random"], TOY_SEED)
assert result["selection"] == {"seed": TOY_SEED, "method": "fixed-before-results"}
assert result["sources"][0]["sha256"] == hashlib.sha256(path.read_bytes()).hexdigest()
assert [player["id"] for player in result["players"]] == ["heuristic", "random"]
for player in result["players"]:
episode = next(e for e in original["episodes"] if e["player_id"] == player["id"])
assert player["replay"] == episode["replay"]
assert player["outcome"] == episode["outcome"]
assert player["decisions"][0]["probabilities"] == episode["decisions"][0]["probabilities"]
assert "observation" not in player["decisions"][0]
assert "no live model inference" in result["disclaimer"]
def test_individual_episode_inputs_supported_and_duplicates_rejected(exporter, toy, tmp_path):
result, path = toy
paths = []
for index, episode in enumerate(result["episodes"]):
episode_path = tmp_path / f"episode-{index}.json"
episode_path.write_text(json.dumps(episode))
paths.append(episode_path)
assert len(exporter.export_manifest(paths, ["random", "heuristic"], TOY_SEED)["players"]) == 2
with pytest.raises(ValueError, match="duplicate"):
exporter.export_manifest([path, path], ["random", "heuristic"], TOY_SEED)
@pytest.mark.parametrize("corruption", ["outcome", "probability", "action", "observation", "cap"])
def test_export_rejects_inconsistent_evidence(exporter, toy, corruption):
result, path = toy
episode = result["episodes"][0]
if corruption == "outcome":
episode["outcome"]["lines"] += 1
elif corruption == "probability":
key = next(iter(episode["decisions"][0]["probabilities"]))
episode["decisions"][0]["probabilities"][key] = -0.5
elif corruption == "action":
episode["decisions"][0]["action_id"] = "wrong"
elif corruption == "observation":
episode["decisions"][0]["observation"]["current"] = "wrong"
elif corruption == "cap":
episode["max_pieces"] = 4
path.write_text(json.dumps(result))
with pytest.raises(ValueError):
exporter.export_manifest([path], ["random", "heuristic"], TOY_SEED)
def test_errors_are_preserved_instead_of_disguised_as_success(exporter, toy):
result, path = toy
original = result["episodes"][0]
episode = copy.deepcopy(original)
last = episode["decisions"].pop()
episode["replay"]["actions"].pop()
# Use the authoritative serializer after removing the final successful move.
from stackcraft.replay import make_replay
episode["replay"] = make_replay(TOY_SEED, episode["replay"]["actions"])
failure = {"turn": 2, "kind": "player_error", "message": "test inference failed"}
episode["errors"] = [failure]
last.pop("action_id")
last.pop("probabilities")
last["error"] = failure
episode["decisions"].append(last)
episode["outcome"] = {**episode["replay"]["final"], "cap_hit": False, "status": "error"}
result["episodes"][0] = episode
path.write_text(json.dumps(result))
exported = exporter.export_manifest([path], ["random", "heuristic"], TOY_SEED)
assert exported["players"][0]["errors"] == [failure]
assert exported["players"][0]["decisions"][-1]["error"] == failure
def test_fixed_seed_missing_player_and_existing_output_are_rejected(exporter, toy, tmp_path):
_, path = toy
with pytest.raises(ValueError, match="fixed"):
exporter.export_manifest([path], ["random", "heuristic"], TOY_SEED + 1)
with pytest.raises(ValueError, match="missing"):
exporter.export_manifest([path], ["random", "trained"], TOY_SEED)
destination = tmp_path / "existing.json"
destination.write_text("do not replace")
with pytest.raises(SystemExit):
exporter.main(["--input", str(path), "--output", str(destination)])
assert destination.read_text() == "do not replace"
@pytest.mark.parametrize(
"url",
[
"javascript:alert(1)",
"http://example.com/report",
"/report",
"https:///",
"https://user:pass@example.com/report",
"https://example.com/ bad",
],
)
def test_report_url_rejects_unsafe_or_non_https_destinations(exporter, toy, url):
_, path = toy
with pytest.raises(ValueError):
exporter.export_manifest([path], ["random", "heuristic"], TOY_SEED, report_url=url)
def test_optional_report_url_preserves_recorded_provenance(exporter, toy):
_, path = toy
original = exporter.export_manifest([path], ["random", "heuristic"], TOY_SEED)
assert "report_url" not in original
url = "https://example.com/study/results"
linked = exporter.export_manifest([path], ["random", "heuristic"], TOY_SEED, report_url=url)
assert linked.pop("report_url") == url
assert linked == original
|