Spaces:
Running
Running
| from __future__ import annotations | |
| import math | |
| import random | |
| from copy import deepcopy | |
| from statistics import mean, median | |
| from .models import SimulationConfig | |
| from .simulator import run_simulation | |
| STUDIES = { | |
| "prefix_cache": { | |
| "label": "Prefix reuse: off vs on", | |
| "baseline": "Prefix reuse off", | |
| "treatment": "Prefix reuse on", | |
| }, | |
| "pd_vs_colocated": { | |
| "label": "Topology: colocated vs P/D", | |
| "baseline": "Colocated", | |
| "treatment": "P/D disaggregated", | |
| }, | |
| "chunked_vs_fcfs": { | |
| "label": "Scheduling: FCFS vs chunked prefill", | |
| "baseline": "Continuous FCFS", | |
| "treatment": "Chunked prefill + SLO", | |
| }, | |
| "slo_vs_fcfs": { | |
| "label": "Scheduling: FCFS vs least-slack", | |
| "baseline": "Continuous FCFS", | |
| "treatment": "Continuous SLO", | |
| }, | |
| } | |
| METRICS = { | |
| "goodput_rps": {"direction": 1, "label": "Goodput", "unit": "req/s"}, | |
| "p95_ttft_ms": {"direction": -1, "label": "p95 TTFT", "unit": "ms"}, | |
| "p95_e2e_ms": {"direction": -1, "label": "p95 E2E", "unit": "ms"}, | |
| "slo_attainment": {"direction": 1, "label": "SLO attainment", "unit": "fraction"}, | |
| } | |
| def _study_configs(base: SimulationConfig, study: str) -> tuple[SimulationConfig, SimulationConfig]: | |
| if study not in STUDIES: | |
| raise ValueError(f"Unknown paired study: {study}") | |
| a = deepcopy(base) | |
| b = deepcopy(base) | |
| if study == "prefix_cache": | |
| a.prefix_cache_enabled = False | |
| b.prefix_cache_enabled = True | |
| elif study == "pd_vs_colocated": | |
| a.topology = "colocated" | |
| b.topology = "disaggregated_pd" | |
| if b.scheduler == "static_fcfs": | |
| b.scheduler = "continuous_fcfs" | |
| elif study == "chunked_vs_fcfs": | |
| a.topology = "colocated" | |
| b.topology = "colocated" | |
| a.scheduler = "continuous_fcfs" | |
| b.scheduler = "chunked_slo" | |
| elif study == "slo_vs_fcfs": | |
| a.topology = "colocated" | |
| b.topology = "colocated" | |
| a.scheduler = "continuous_fcfs" | |
| b.scheduler = "continuous_slo" | |
| return a, b | |
| def _extract(result: dict) -> dict[str, float]: | |
| return { | |
| "goodput_rps": float(result["summary"]["goodput_rps"]), | |
| "p95_ttft_ms": float(result["latency"]["ttft_ms"]["p95"]), | |
| "p95_e2e_ms": float(result["latency"]["e2e_ms"]["p95"]), | |
| "slo_attainment": float(result["summary"]["slo_attainment"]), | |
| } | |
| def _percentile(values: list[float], q: float) -> float: | |
| if not values: | |
| return 0.0 | |
| ordered = sorted(values) | |
| if len(ordered) == 1: | |
| return ordered[0] | |
| pos = min(max(q, 0.0), 1.0) * (len(ordered) - 1) | |
| lo = int(math.floor(pos)) | |
| hi = int(math.ceil(pos)) | |
| if lo == hi: | |
| return ordered[lo] | |
| frac = pos - lo | |
| return ordered[lo] * (1.0 - frac) + ordered[hi] * frac | |
| def _bootstrap_ci(deltas: list[float], samples: int, seed: int) -> tuple[float, float]: | |
| if not deltas: | |
| return (0.0, 0.0) | |
| rng = random.Random(seed) | |
| n = len(deltas) | |
| boot = [] | |
| for _ in range(max(samples, 50)): | |
| boot.append(mean(deltas[rng.randrange(n)] for _ in range(n))) | |
| return _percentile(boot, 0.025), _percentile(boot, 0.975) | |
| def paired_study(config: dict, study: str = "prefix_cache", repetitions: int = 12, bootstrap_samples: int = 500) -> dict: | |
| """Run a paired Monte Carlo A/B study using common random numbers. | |
| Baseline and treatment share the same seed on every repetition. This reduces | |
| workload-noise variance and makes the delta attributable to the controlled | |
| system change rather than to different synthetic request traces. | |
| """ | |
| base = SimulationConfig.from_dict(config) | |
| repetitions = max(2, min(int(repetitions), 64)) | |
| a_cfg, b_cfg = _study_configs(base, study) | |
| pairs: list[dict] = [] | |
| for rep in range(repetitions): | |
| seed = base.seed + rep * 1009 | |
| a_cfg.seed = seed | |
| b_cfg.seed = seed | |
| a_result = run_simulation(a_cfg.to_dict()) | |
| b_result = run_simulation(b_cfg.to_dict()) | |
| a_metrics = _extract(a_result) | |
| b_metrics = _extract(b_result) | |
| pairs.append({"rep": rep + 1, "seed": seed, "baseline": a_metrics, "treatment": b_metrics}) | |
| metrics = [] | |
| for key, meta in METRICS.items(): | |
| baseline = [row["baseline"][key] for row in pairs] | |
| treatment = [row["treatment"][key] for row in pairs] | |
| deltas = [b - a for a, b in zip(baseline, treatment, strict=True)] | |
| relative = [((b - a) / abs(a) * 100.0) if abs(a) > 1e-12 else 0.0 for a, b in zip(baseline, treatment, strict=True)] | |
| metric_seed = sum((idx + 1) * ord(ch) for idx, ch in enumerate(key)) | |
| ci_low, ci_high = _bootstrap_ci(deltas, bootstrap_samples, base.seed ^ metric_seed) | |
| direction = int(meta["direction"]) | |
| wins = sum(1 for delta in deltas if delta * direction > 0) | |
| ties = sum(1 for delta in deltas if abs(delta) <= 1e-12) | |
| metrics.append( | |
| { | |
| "metric": key, | |
| "label": meta["label"], | |
| "unit": meta["unit"], | |
| "baseline_mean": mean(baseline), | |
| "treatment_mean": mean(treatment), | |
| "delta_mean": mean(deltas), | |
| "delta_median": median(deltas), | |
| "delta_ci95_low": ci_low, | |
| "delta_ci95_high": ci_high, | |
| "relative_change_pct": mean(relative), | |
| "treatment_win_rate": wins / repetitions, | |
| "tie_rate": ties / repetitions, | |
| "preferred_direction": "higher" if direction > 0 else "lower", | |
| "ci_excludes_zero": ci_low > 0 or ci_high < 0, | |
| } | |
| ) | |
| return { | |
| "study": study, | |
| "label": STUDIES[study]["label"], | |
| "baseline_label": STUDIES[study]["baseline"], | |
| "treatment_label": STUDIES[study]["treatment"], | |
| "repetitions": repetitions, | |
| "bootstrap_samples": max(bootstrap_samples, 50), | |
| "protocol": "paired-common-random-numbers", | |
| "metrics": metrics, | |
| "pairs": pairs, | |
| } | |
| def robustness_study( | |
| config: dict, | |
| study: str = "pd_vs_colocated", | |
| samples: int = 32, | |
| uncertainty: float = 0.20, | |
| ) -> dict: | |
| """Stress-test an A/B conclusion under analytical latency uncertainty. | |
| Each sample draws shared prefill/decode/transfer scale factors and applies | |
| them to both alternatives. The goal is not a probability statement about | |
| real hardware; it is a sensitivity analysis showing whether a conclusion is | |
| fragile to plausible multiplicative error in the reference latency model. | |
| """ | |
| base = SimulationConfig.from_dict(config) | |
| samples = max(4, min(int(samples), 96)) | |
| uncertainty = min(max(float(uncertainty), 0.0), 0.75) | |
| rng = random.Random(base.seed ^ 0x51514A) | |
| rows = [] | |
| for idx in range(samples): | |
| prefill_scale = rng.uniform(1.0 - uncertainty, 1.0 + uncertainty) | |
| decode_scale = rng.uniform(1.0 - uncertainty, 1.0 + uncertainty) | |
| transfer_scale = rng.uniform(1.0 - uncertainty, 1.0 + uncertainty) | |
| a_cfg, b_cfg = _study_configs(base, study) | |
| seed = base.seed + idx * 1009 | |
| for cfg in (a_cfg, b_cfg): | |
| cfg.seed = seed | |
| cfg.prefill_time_scale = prefill_scale | |
| cfg.decode_time_scale = decode_scale | |
| cfg.transfer_time_scale = transfer_scale | |
| a = run_simulation(a_cfg.to_dict()) | |
| b = run_simulation(b_cfg.to_dict()) | |
| am = _extract(a) | |
| bm = _extract(b) | |
| rows.append( | |
| { | |
| "sample": idx + 1, | |
| "prefill_scale": prefill_scale, | |
| "decode_scale": decode_scale, | |
| "transfer_scale": transfer_scale, | |
| "baseline": am, | |
| "treatment": bm, | |
| "baseline_slo_pass": am["slo_attainment"] >= base.slo_attainment_target, | |
| "treatment_slo_pass": bm["slo_attainment"] >= base.slo_attainment_target, | |
| } | |
| ) | |
| def win_fraction(metric: str, direction: int) -> float: | |
| return mean( | |
| 1.0 if (row["treatment"][metric] - row["baseline"][metric]) * direction > 0 else 0.0 | |
| for row in rows | |
| ) | |
| goodput_deltas = [row["treatment"]["goodput_rps"] - row["baseline"]["goodput_rps"] for row in rows] | |
| ttft_deltas = [row["treatment"]["p95_ttft_ms"] - row["baseline"]["p95_ttft_ms"] for row in rows] | |
| e2e_deltas = [row["treatment"]["p95_e2e_ms"] - row["baseline"]["p95_e2e_ms"] for row in rows] | |
| return { | |
| "study": study, | |
| "label": STUDIES[study]["label"], | |
| "baseline_label": STUDIES[study]["baseline"], | |
| "treatment_label": STUDIES[study]["treatment"], | |
| "samples": samples, | |
| "uncertainty": uncertainty, | |
| "method": "shared-multiplicative-latency-perturbation", | |
| "summary": { | |
| "treatment_goodput_win_fraction": win_fraction("goodput_rps", 1), | |
| "treatment_ttft_win_fraction": win_fraction("p95_ttft_ms", -1), | |
| "treatment_e2e_win_fraction": win_fraction("p95_e2e_ms", -1), | |
| "baseline_slo_pass_fraction": mean(1.0 if row["baseline_slo_pass"] else 0.0 for row in rows), | |
| "treatment_slo_pass_fraction": mean(1.0 if row["treatment_slo_pass"] else 0.0 for row in rows), | |
| "median_goodput_delta": median(goodput_deltas), | |
| "median_ttft_delta_ms": median(ttft_deltas), | |
| "median_e2e_delta_ms": median(e2e_deltas), | |
| }, | |
| "rows": rows, | |
| } | |