E-037: torchrun's first positional IS the script -- passing sys.executable made it compile the python binary and die with 'source code cannot contain null bytes' in 8.6s. Affects the Phase 4 launcher identically at session 1. Probe also split into smoke (20 steps) and soak (180) modes
Browse files- kernels/p4_stop_probe.py +27 -13
kernels/p4_stop_probe.py
CHANGED
|
@@ -50,11 +50,21 @@ from huggingface_hub import HfApi
|
|
| 50 |
MIX = "Cion-lab/ounce100m-mix-v1"
|
| 51 |
PROBE = "Cion-lab/ounce100m-ckpt-probe" # never the real run's repo, see the header
|
| 52 |
ROOT, RUN = "/kaggle/working/mixroot", "/kaggle/working/run"
|
| 53 |
-
#
|
| 54 |
-
#
|
| 55 |
-
#
|
| 56 |
-
|
| 57 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 58 |
T0 = time.time()
|
| 59 |
|
| 60 |
|
|
@@ -62,6 +72,7 @@ def run(argv, label, timeout):
|
|
| 62 |
print("=== " + label, flush=True)
|
| 63 |
t0 = time.time()
|
| 64 |
e = dict(os.environ); e["PYTHONPATH"] = "/kaggle/working"
|
|
|
|
| 65 |
p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
|
| 66 |
env=e, bufsize=1, start_new_session=True)
|
| 67 |
killed = []
|
|
@@ -105,22 +116,22 @@ rc, out = run([sys.executable, "-c",
|
|
| 105 |
"from huggingface_hub import snapshot_download\n"
|
| 106 |
"p = snapshot_download(repo_id='%s', repo_type='dataset',\n"
|
| 107 |
" local_dir='/kaggle/working/mixroot', max_workers=4)\n"
|
| 108 |
-
"print('mix at', p)\n" % MIX], "FETCH_MIX",
|
| 109 |
if rc != 0:
|
| 110 |
raise SystemExit("VERDICT P4PROBE_STOP could not fetch the published mix")
|
| 111 |
man = json.load(open(os.path.join(ROOT, "manifest.json")))
|
| 112 |
print("mix", man["n_shards"], "shards", format(int(man["total_tokens"]), ","), "tokens", flush=True)
|
| 113 |
|
| 114 |
-
common = [
|
| 115 |
"--hub-repo", PROBE, "--prune", "--seq-len", "1024", "--attn", "eager", "--grad-ckpt",
|
| 116 |
"--micro-batch", "4", "--accum", "32", "--no-grad-ckpt",
|
| 117 |
"--tokens", str(TOKENS), "--lr", "6e-4",
|
| 118 |
-
"--push-every-steps", str(PUSH_EVERY), "--val-tokens",
|
| 119 |
"--resume", "auto"]
|
| 120 |
|
| 121 |
shutil.rmtree(RUN, ignore_errors=True)
|
| 122 |
rc1, o1 = run(["torchrun", "--nproc_per_node=2"] + common + ["--stop-after-steps", str(STOP1)],
|
| 123 |
-
"LEG1_STOP_EARLY",
|
| 124 |
api = HfApi(token=os.environ["HF_TOKEN"])
|
| 125 |
try:
|
| 126 |
listed = sorted(hubckpt.hub_listing(PROBE, "dataset", token=os.environ["HF_TOKEN"]))
|
|
@@ -134,7 +145,7 @@ print("LEG1 pushed:", pushed, "pointer:", {k: ptr.get(k) for k in ("step", "path
|
|
| 134 |
|
| 135 |
shutil.rmtree(RUN, ignore_errors=True) # force the cold-resume path (E-029's shape)
|
| 136 |
rc2, o2 = run(["torchrun", "--nproc_per_node=2"] + common + ["--stop-after-steps", "0"],
|
| 137 |
-
"LEG2_TO_HORIZON",
|
| 138 |
try:
|
| 139 |
listed2 = sorted(hubckpt.hub_listing(PROBE, "dataset", token=os.environ["HF_TOKEN"]))
|
| 140 |
except Exception as e:
|
|
@@ -188,13 +199,16 @@ res["checks"] = {
|
|
| 188 |
"params_are_the_frozen_model": rj1.get("params") == 106194240,
|
| 189 |
"checkpointing_really_off": rj1.get("grad_ckpt") is False,
|
| 190 |
# 13.6 GB of ~14.56 usable leaves the run room for a fragmentation spike and for the two CUDA contexts;
|
| 191 |
-
# above that the answer to the user's question is "not on this hardware", not "keep going".
|
| 192 |
-
|
| 193 |
-
|
|
|
|
| 194 |
"no_nan_and_loss_moved": (rj1.get("final_loss") or 1e9) < 11.0
|
| 195 |
and (rj2.get("final_loss") or 1e9) < 11.0,
|
| 196 |
}
|
| 197 |
res["PROBE_PASSED"] = all(res["checks"].values()) and rc1 == 0 and rc2 == 0
|
|
|
|
|
|
|
| 198 |
print("PROBE_JSON_BEGIN")
|
| 199 |
print(json.dumps(res, indent=1, default=str))
|
| 200 |
print("PROBE_JSON_END")
|
|
|
|
| 50 |
MIX = "Cion-lab/ounce100m-mix-v1"
|
| 51 |
PROBE = "Cion-lab/ounce100m-ckpt-probe" # never the real run's repo, see the header
|
| 52 |
ROOT, RUN = "/kaggle/working/mixroot", "/kaggle/working/run"
|
| 53 |
+
# Two modes, one file, so the assertions are literally the same code in both. `smoke` is the user's
|
| 54 |
+
# suggestion and it is the right order: 20 steps costs ~12 minutes and answers "does the training code run
|
| 55 |
+
# at all, and does one checkpoint survive the push/verify/pointer/prune cycle" -- which is exactly what the
|
| 56 |
+
# first run of this probe failed at, in 8.6 seconds, on a malformed torchrun command line (E-037). `soak`
|
| 57 |
+
# is the 180-step memory question, and it is only worth 1.6 GPU-hours once the mechanics are known to work.
|
| 58 |
+
MODE = os.environ.get("P4_PROBE_MODE", "soak")
|
| 59 |
+
TPS = 262144 # the run's real step shape, in tokens
|
| 60 |
+
if MODE == "smoke":
|
| 61 |
+
STEPS, PUSH_EVERY, STOP1, VAL = 20, 10, 10, 200000
|
| 62 |
+
T_LEG1, T_LEG2, T_FRESH = 1500, 1200, 2400
|
| 63 |
+
else:
|
| 64 |
+
STEPS, PUSH_EVERY, STOP1, VAL = 180, 60, 120, 2000000
|
| 65 |
+
T_LEG1, T_LEG2, T_FRESH = 4200, 3000, 2400
|
| 66 |
+
TOKENS = STEPS * TPS
|
| 67 |
+
GATE_PEAK = (MODE != "smoke") # 20 steps says nothing about allocator drift
|
| 68 |
T0 = time.time()
|
| 69 |
|
| 70 |
|
|
|
|
| 72 |
print("=== " + label, flush=True)
|
| 73 |
t0 = time.time()
|
| 74 |
e = dict(os.environ); e["PYTHONPATH"] = "/kaggle/working"
|
| 75 |
+
e["PYTHONUNBUFFERED"] = "1" # the -u that torchrun cannot carry
|
| 76 |
p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
|
| 77 |
env=e, bufsize=1, start_new_session=True)
|
| 78 |
killed = []
|
|
|
|
| 116 |
"from huggingface_hub import snapshot_download\n"
|
| 117 |
"p = snapshot_download(repo_id='%s', repo_type='dataset',\n"
|
| 118 |
" local_dir='/kaggle/working/mixroot', max_workers=4)\n"
|
| 119 |
+
"print('mix at', p)\n" % MIX], "FETCH_MIX", T_FRESH)
|
| 120 |
if rc != 0:
|
| 121 |
raise SystemExit("VERDICT P4PROBE_STOP could not fetch the published mix")
|
| 122 |
man = json.load(open(os.path.join(ROOT, "manifest.json")))
|
| 123 |
print("mix", man["n_shards"], "shards", format(int(man["total_tokens"]), ","), "tokens", flush=True)
|
| 124 |
|
| 125 |
+
common = ["train_ounce100m.py", "--root", ROOT, "--out", RUN,
|
| 126 |
"--hub-repo", PROBE, "--prune", "--seq-len", "1024", "--attn", "eager", "--grad-ckpt",
|
| 127 |
"--micro-batch", "4", "--accum", "32", "--no-grad-ckpt",
|
| 128 |
"--tokens", str(TOKENS), "--lr", "6e-4",
|
| 129 |
+
"--push-every-steps", str(PUSH_EVERY), "--val-tokens", str(VAL), "--log-every", "5",
|
| 130 |
"--resume", "auto"]
|
| 131 |
|
| 132 |
shutil.rmtree(RUN, ignore_errors=True)
|
| 133 |
rc1, o1 = run(["torchrun", "--nproc_per_node=2"] + common + ["--stop-after-steps", str(STOP1)],
|
| 134 |
+
"LEG1_STOP_EARLY", T_LEG1)
|
| 135 |
api = HfApi(token=os.environ["HF_TOKEN"])
|
| 136 |
try:
|
| 137 |
listed = sorted(hubckpt.hub_listing(PROBE, "dataset", token=os.environ["HF_TOKEN"]))
|
|
|
|
| 145 |
|
| 146 |
shutil.rmtree(RUN, ignore_errors=True) # force the cold-resume path (E-029's shape)
|
| 147 |
rc2, o2 = run(["torchrun", "--nproc_per_node=2"] + common + ["--stop-after-steps", "0"],
|
| 148 |
+
"LEG2_TO_HORIZON", T_LEG2)
|
| 149 |
try:
|
| 150 |
listed2 = sorted(hubckpt.hub_listing(PROBE, "dataset", token=os.environ["HF_TOKEN"]))
|
| 151 |
except Exception as e:
|
|
|
|
| 199 |
"params_are_the_frozen_model": rj1.get("params") == 106194240,
|
| 200 |
"checkpointing_really_off": rj1.get("grad_ckpt") is False,
|
| 201 |
# 13.6 GB of ~14.56 usable leaves the run room for a fragmentation spike and for the two CUDA contexts;
|
| 202 |
+
# above that the answer to the user's question is "not on this hardware", not "keep going". Reported in
|
| 203 |
+
# both modes, gated only in the soak: a 20-step peak is not evidence about 3,814 steps.
|
| 204 |
+
"peak_memory_below_13_6_gb": (not GATE_PEAK) or ((rj1.get("peak_gpu_gb") or 99) <= 13.6
|
| 205 |
+
and (rj2.get("peak_gpu_gb") or 99) <= 13.6),
|
| 206 |
"no_nan_and_loss_moved": (rj1.get("final_loss") or 1e9) < 11.0
|
| 207 |
and (rj2.get("final_loss") or 1e9) < 11.0,
|
| 208 |
}
|
| 209 |
res["PROBE_PASSED"] = all(res["checks"].values()) and rc1 == 0 and rc2 == 0
|
| 210 |
+
print("PROBE_MODE", MODE, "steps", STEPS, "push_every", PUSH_EVERY,
|
| 211 |
+
"stop1", STOP1, "peak_gate", GATE_PEAK, flush=True)
|
| 212 |
print("PROBE_JSON_BEGIN")
|
| 213 |
print(json.dumps(res, indent=1, default=str))
|
| 214 |
print("PROBE_JSON_END")
|