Buckets:
| """Figures for the DANCE reproduction logbook (one per claim).""" | |
| import json, os | |
| import numpy as np | |
| import matplotlib | |
| matplotlib.use("Agg") | |
| import matplotlib.pyplot as plt | |
| os.makedirs("outputs", exist_ok=True) | |
| BLUE, RED, GRAY, GREEN = "#2D5F8B", "#C1666B", "#9AA0A6", "#4E8B6B" | |
| def fig_claim5(): | |
| d = json.load(open("outputs/claim5_complexity.json")) | |
| A = d["A_edges_O_Kq"]["vs_K"]; B = d["B_time_scaling"]["rows"]; C = d["C_encoding_cost"] | |
| fig, ax = plt.subplots(1, 3, figsize=(15, 4.2)) | |
| K = [r["K"] for r in A]; edges = [r["edges"] for r in A]; K2 = [r["K2"] for r in A] | |
| ax[0].plot(K, edges, "o-", color=BLUE, lw=2.5, label="DANCE edges (O(Kk))") | |
| ax[0].plot(K, K2, "s--", color=RED, label="dense O(K²)") | |
| ax[0].set_xlabel("core size K"); ax[0].set_ylabel("edge count"); ax[0].set_yscale("log") | |
| ax[0].set_title("Topology edges: O(Kk), not O(K²)"); ax[0].legend(); ax[0].grid(alpha=0.3) | |
| ns = [r["n"] for r in B]; ts = [r["time_s"] for r in B] | |
| ax[1].plot(ns, ts, "o-", color=BLUE, lw=2.5, label="measured") | |
| ref = np.array(ns, float); ref = ref**2 / ref[0]**2 * ts[0] | |
| ax[1].plot(ns, ref, "s--", color=RED, label="O(n²) ref") | |
| ax[1].set_xlabel("n (nodes)"); ax[1].set_ylabel("time / refresh (s)") | |
| ax[1].set_title(f"Pipeline time (exp≈{d['B_time_scaling']['empirical_time_exponent_p_in_n^p']})") | |
| ax[1].legend(); ax[1].grid(alpha=0.3) | |
| names = ["DANCE\n(encode once)", "LLM re-encode\n(condensed)", "full-graph\nLLM"] | |
| vals = [C["est_dance_encode_time_s"], C["est_llm_baseline_time_s"], C["est_fullgraph_llm_time_s"]] | |
| ax[2].bar(names, vals, color=[BLUE, RED, GRAY]) | |
| for i, v in enumerate(vals): ax[2].text(i, v, f"{v:.0f}s", ha="center", va="bottom") | |
| ax[2].set_ylabel("est. text-encoding time (s)") | |
| ax[2].set_title(f"Training-time: DANCE {C['dance_vs_llm_token_ratio']}×–{C['dance_vs_fullgraph_ratio']:.0f}× cheaper") | |
| plt.tight_layout(); plt.savefig("outputs/fig_claim5.png", dpi=130, bbox_inches="tight") | |
| print("wrote fig_claim5.png") | |
| def fig_claim4(): | |
| d = json.load(open("outputs/claim4_theorems.json")) | |
| t54, t56 = d["theorem_5_4"], d["theorem_5_6"] | |
| fig, ax = plt.subplots(1, 2, figsize=(11, 4)) | |
| ax[0].bar(["max", "p95", "mean"], | |
| [t54["max_ratio_lhs_over_rhs"], t54["p95_ratio"], t54["mean_ratio"]], | |
| color=[BLUE, BLUE, GRAY]) | |
| ax[0].axhline(1.0, color=RED, ls="--", label="bound (ratio=1)") | |
| ax[0].set_ylim(0, 1.15); ax[0].set_ylabel("‖t̃−t_full‖ / (2M·δ)") | |
| ax[0].set_title(f"Thm 5.4: 0/{t54['trials']} violations"); ax[0].legend() | |
| ax[1].bar(["inside ball\n(should stay)", "outside ball\n(may change)"], | |
| [t56["inside_ball_invariance_rate"] * 100, t56["outside_ball_change_rate"] * 100], | |
| color=[GREEN, RED]) | |
| ax[1].set_ylabel("%"); ax[1].set_ylim(0, 105) | |
| ax[1].set_title("Thm 5.6: top-B invariance vs drift") | |
| ax[1].text(0, t56["inside_ball_invariance_rate"] * 100, f"{t56['inside_ball_invariance_rate']*100:.0f}%\ninvariant", ha="center", va="top", color="white") | |
| plt.tight_layout(); plt.savefig("outputs/fig_claim4.png", dpi=130, bbox_inches="tight") | |
| print("wrote fig_claim4.png") | |
| def fig_claim3(): | |
| d = json.load(open("outputs/claim23_cora.json"))["claim3_evidence"] | |
| ks = d["ks"] | |
| fig, ax = plt.subplots(1, 2, figsize=(11, 4.2)) | |
| ax[0].plot(ks, d["insertion_selected_conf"], "o-", color=BLUE, lw=2.5, label="selected evidence") | |
| ax[0].plot(ks, d["insertion_random_conf"], "v--", color=GRAY, label="random") | |
| ax[0].set_xlabel("# evidence chunks kept (insertion)"); ax[0].set_ylabel("pred-class confidence") | |
| ax[0].set_title(f"Sufficiency (gap +{d['sufficiency_gap_conf']})"); ax[0].legend(); ax[0].grid(alpha=0.3) | |
| ax[1].plot(ks, d["deletion_selected_conf"], "o-", color=RED, lw=2.5, label="remove selected") | |
| ax[1].plot(ks, d["deletion_random_conf"], "v--", color=GRAY, label="remove random") | |
| ax[1].set_xlabel("# evidence chunks removed (deletion)"); ax[1].set_ylabel("pred-class confidence") | |
| ax[1].set_title(f"Necessity (gap +{d['necessity_gap_conf']})"); ax[1].legend(); ax[1].grid(alpha=0.3) | |
| plt.tight_layout(); plt.savefig("outputs/fig_claim3.png", dpi=130, bbox_inches="tight") | |
| print("wrote fig_claim3.png") | |
| def fig_claim1(): | |
| d = json.load(open("outputs/claim1_condensation.json")) | |
| acc = d["accuracy_pct"]; tok = d["tokens_per_node"] | |
| fig, ax = plt.subplots(1, 2, figsize=(11, 4.2)) | |
| names = ["random", "degree", "dance_label_aware"] | |
| means = [acc[n]["mean"] for n in names]; stds = [acc[n]["std"] for n in names] | |
| ax[0].bar(["random", "degree", "DANCE\n(label-aware)"], means, yerr=stds, | |
| color=[GRAY, GRAY, BLUE], capsize=4) | |
| for i, v in enumerate(means): ax[0].text(i, v, f"{v:.1f}", ha="center", va="bottom") | |
| ax[0].set_ylim(70, 85); ax[0].set_ylabel("test accuracy (%)") | |
| ax[0].set_title("8% core selection (Cora, node-selection proxy)") | |
| bars = ["DANCE\nB_tok", "full 0-hop\ntext", "full 1-hop\nmulti-hop"] | |
| vals = [tok["dance_B_tok"], tok["avg_present_words_0hop"], tok["avg_multihop_1hop_tokens"]] | |
| ax[1].bar(bars, vals, color=[BLUE, RED, GRAY]) | |
| for i, v in enumerate(vals): ax[1].text(i, v, f"{v:.0f}", ha="center", va="bottom") | |
| ax[1].set_ylabel("tokens per condensed node") | |
| ax[1].set_title(f"Token reduction: {tok['reduction_vs_0hop_pct']:.0f}%–{tok['reduction_vs_multihop_pct']:.0f}% (claim 33.42%)") | |
| plt.tight_layout(); plt.savefig("outputs/fig_claim1.png", dpi=130, bbox_inches="tight") | |
| print("wrote fig_claim1.png") | |
| if __name__ == "__main__": | |
| fig_claim4(); fig_claim5(); fig_claim3(); fig_claim1() | |
Xet Storage Details
- Size:
- 5.66 kB
- Xet hash:
- f63cb01a7583294e618808033903d3723d114861af1da3fea9731f01213cb4b1
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.