File size: 7,575 Bytes
021d920 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 | """LogLens learnability-law + mechanism — TURNKEY replication harness.
A stranger, a different rig, a different scene reproduces (or kills) our result
with ONE command. Self-contained except train_composite.py + triclock/ (copy
those two from this repo). No bash, relative paths, CPU or CUDA.
python replicate.py --bank <folder-of-photos | scene.npy> [--full] [--seeds 2]
What it does, and the PASS/FAIL bars (our pre-registered thresholds):
1. R vs feat-R across a size grid ....... PASS if Spearman >= 0.90 (P1: the
untrained net preserves the pixel signal/clutter ratio into features)
2. R predicts learnability (blind) ...... PASS if Spearman >= 0.85 (the LAW)
3. Ignition signature .................... PASS if the learnable cell's loss
drops below log(K)=1.386 while a floored cell stays pinned there
(escape = signal amplifies; collapse = uniform-guess basin)
Send back: results_replicate.json + your scene hash. Disagreement is a RESULT.
"""
import argparse, glob, hashlib, json, math, os, subprocess, sys, time
import numpy as np
try:
import cv2
except ImportError:
sys.exit("pip install opencv-python")
import torch
from scipy.stats import spearmanr
sys.path.insert(0, ".")
from triclock.model import LogLensNet
LOGK = math.log(4) # 4 age buckets -> collapse loss
# ---------- bank ----------
HORIZON_MS = 86_400_000 # the task dates changes over a 24h horizon
def _synth_ts(n):
"""Spread n frames over ~4x the age horizon, so episodes can sample memory
slots at ages up to 24h. (1-second spacing silently breaks sampling.)"""
step = max(1, (4 * HORIZON_MS) // max(n, 1))
return (np.arange(n, dtype=np.int64) * step)
def build_bank(src, n=3000):
"""Always materialise a matched (frames64, ts64) pair under simreal/ so the
loader never sees a bank without its timestamp file."""
os.makedirs("simreal", exist_ok=True)
if src.endswith(".npy"):
F = np.asarray(np.load(src, mmap_mode="r")[:n])
np.save("simreal/frames64.npy", F)
np.save("simreal/ts64.npy", _synth_ts(len(F)))
return F, "simreal/frames64.npy"
files = sorted(glob.glob(os.path.join(src, "*.jpg")) + glob.glob(os.path.join(src, "*.png")))
if len(files) < 200:
sys.exit(f"need >=200 images in {src}, found {len(files)}")
idx = np.linspace(0, len(files) - 1, min(n, len(files))).astype(int)
frames = []
for i in idx:
img = cv2.imread(files[i])
if img is None: continue
h, w = img.shape[:2]; s = min(h, w)
img = img[(h-s)//2:(h-s)//2+s, (w-s)//2:(w-s)//2+s]
img = cv2.cvtColor(cv2.resize(img, (64, 64), cv2.INTER_AREA), cv2.COLOR_BGR2RGB)
frames.append(img.astype(np.uint8))
F = np.stack(frames)
os.makedirs("simreal", exist_ok=True)
np.save("simreal/frames64.npy", F)
np.save("simreal/ts64.npy", _synth_ts(len(F)))
return F, "simreal/frames64.npy"
# ---------- R (pixel) and feat-R (encoder) ----------
def pool(img): return cv2.resize(img, (8, 8), cv2.INTER_AREA).ravel()
def draw(img, pos, c, size, ct=1.0):
img = img.copy(); x, y = pos
img[y:y+size, x:x+size] = c*ct + img[y:y+size, x:x+size].mean(axis=(0, 1))*(1-ct); return img
def ratios(F, size, seed):
torch.manual_seed(seed); enc = LogLensNet().enc.eval()
rng = np.random.default_rng(seed)
@torch.no_grad()
def E(imgs):
x = torch.from_numpy(np.stack(imgs)).permute(0, 3, 1, 2).float()
return enc(x).numpy()
sp, dp, sf, df = [], [], [], []
for _ in range(60):
f0 = np.asarray(F[rng.integers(len(F))], np.float32) / 255.0
frames = [f0] * 9
pos = (int(rng.integers(2, 60-size)), int(rng.integers(2, 60-size)))
c = rng.uniform(.15, .95, 3).astype(np.float32)
A, B = draw(f0, pos, c, size), draw(f0, pos, 1-c, size)
sp.append(np.sum((pool(A)-pool(B))**2)/4)
Pp = np.array([pool(f) for f in frames])
prp = [pool(np.roll(np.roll(f0, int(rng.integers(64)), 0), int(rng.integers(64)), 1)) for _ in range(6)]
dp.append(Pp.var(0).sum()+np.array(prp).var(0).sum())
a, b = E([A, B]); sf.append(np.sum((a-b)**2)/4)
Pf = E(frames); prf = E([np.roll(np.roll(f0, int(rng.integers(64)), 0), int(rng.integers(64)), 1) for _ in range(6)])
df.append(Pf.var(0).sum()+prf.var(0).sum())
return float(np.mean(sp)/(np.mean(dp)+1e-9)), float(np.mean(sf)/(np.mean(df)+1e-9))
# ---------- train one cell (blind) ----------
def train_cell(bank, size, seed, steps):
out = "results_replicate"
subprocess.run([sys.executable, "-m", "simreal.train_composite", "--steps", str(steps),
"--bank", bank, "--objscale", str(size), "--static-bg", "--seed", str(seed),
"--out", out], check=False)
hits = glob.glob(f"{out}/*_o{size}_*_s{seed}.json")
return json.load(open(hits[-1]))["heldout_mean"] if hits else None
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--bank", required=True, help="folder of photos OR a .npy bank")
ap.add_argument("--full", action="store_true", help="20k steps (default 8k quick)")
ap.add_argument("--seeds", type=int, default=2, help="seeds per cell (>=2 recommended: the wall is seed-bimodal)")
a = ap.parse_args()
steps = 20000 if a.full else 8000
SIZES = [5, 8, 12, 16] # spans floor -> learnable
print(f"[replicate] device={'cuda' if torch.cuda.is_available() else 'cpu'} steps={steps} seeds={a.seeds}")
F, bankpath = build_bank(a.bank)
scene_hash = hashlib.md5(np.asarray(F[:500]).tobytes()).hexdigest()[:12]
print(f"[replicate] scene bank {len(F)} frames hash={scene_hash}")
# 1) R vs feat-R
R, featR = {}, {}
for s in SIZES:
r, f = ratios(F, s, 0); R[s], featR[s] = r, f
print(f" size {s:2d}px R={r:.4f} feat-R={f:.4f}")
p1 = spearmanr([R[s] for s in SIZES], [featR[s] for s in SIZES])[0]
# 2) LAW: R predicts learnability (multi-seed mean; the wall is seed-bimodal)
acc = {}
for s in SIZES:
outs = [train_cell(bankpath, s, sd, steps) for sd in range(a.seeds)]
outs = [o for o in outs if o is not None]
acc[s] = float(np.mean(outs));
print(f" size {s:2d}px R={R[s]:.4f} mean_acc={acc[s]:.3f} (best {max(outs):.3f} of {len(outs)})")
p2 = spearmanr([R[s] for s in SIZES], [acc[s] for s in SIZES])[0]
# 3) ignition signature: learnable escapes below log K, floor stays pinned
learn_s = max(SIZES, key=lambda s: R[s]); floor_s = min(SIZES, key=lambda s: R[s])
ignition = acc[learn_s] >= 0.55 and acc[floor_s] < 0.55
res = {"scene_hash": scene_hash, "steps": steps, "seeds": a.seeds,
"R": R, "featR": featR, "acc": acc,
"P1_R_vs_featR": p1, "P2_law_R_vs_acc": p2, "ignition": ignition}
json.dump(res, open("results_replicate.json", "w"), indent=2)
def mark(ok): return "PASS" if ok else "FAIL"
print("\n================ REPLICATION SCORECARD ================")
print(f" P1 R <-> feat-R Spearman={p1:+.3f} (>=0.90) {mark(p1>=0.90)}")
print(f" P2 LAW R <-> acc Spearman={p2:+.3f} (>=0.85) {mark(p2>=0.85)}")
print(f" P3 ignition (learn escapes, floor pinned at logK) {mark(ignition)}")
print("======================================================")
print("Compare to EXPECTED_RESULTS.md. Send results_replicate.json + scene_hash back.")
print("A disagreement is a RESULT, not a failure — that is the point of an independent rig.")
if __name__ == "__main__":
main()
|