Cion-lab commited on
Commit
1dd45f8
·
verified ·
1 Parent(s): a0cf9fa

E-037: torchrun's first positional IS the script -- passing sys.executable made it compile the python binary and die with 'source code cannot contain null bytes' in 8.6s. Affects the Phase 4 launcher identically at session 1. Probe also split into smoke (20 steps) and soak (180) modes

Browse files
Files changed (1) hide show
  1. kernels/p4_stop_probe.py +27 -13
kernels/p4_stop_probe.py CHANGED
@@ -50,11 +50,21 @@ from huggingface_hub import HfApi
50
  MIX = "Cion-lab/ounce100m-mix-v1"
51
  PROBE = "Cion-lab/ounce100m-ckpt-probe" # never the real run's repo, see the header
52
  ROOT, RUN = "/kaggle/working/mixroot", "/kaggle/working/run"
53
- # 180 steps at the run's real step shape (262,144 tokens/step) = 47,185,920 tokens of the published mix:
54
- # 139,842,880 at the mid-run stop, 61,118,720 after the cold resume. Pushes every 60 steps, which is also
55
- # what a no-checkpointing main run would need as its recovery interval.
56
- TOKENS, PUSH_EVERY, STOP1, STEPS = 47185920, 60, 120, 180
57
- TPS = 262144
 
 
 
 
 
 
 
 
 
 
58
  T0 = time.time()
59
 
60
 
@@ -62,6 +72,7 @@ def run(argv, label, timeout):
62
  print("=== " + label, flush=True)
63
  t0 = time.time()
64
  e = dict(os.environ); e["PYTHONPATH"] = "/kaggle/working"
 
65
  p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
66
  env=e, bufsize=1, start_new_session=True)
67
  killed = []
@@ -105,22 +116,22 @@ rc, out = run([sys.executable, "-c",
105
  "from huggingface_hub import snapshot_download\n"
106
  "p = snapshot_download(repo_id='%s', repo_type='dataset',\n"
107
  " local_dir='/kaggle/working/mixroot', max_workers=4)\n"
108
- "print('mix at', p)\n" % MIX], "FETCH_MIX", 2400)
109
  if rc != 0:
110
  raise SystemExit("VERDICT P4PROBE_STOP could not fetch the published mix")
111
  man = json.load(open(os.path.join(ROOT, "manifest.json")))
112
  print("mix", man["n_shards"], "shards", format(int(man["total_tokens"]), ","), "tokens", flush=True)
113
 
114
- common = [sys.executable, "-u", "train_ounce100m.py", "--root", ROOT, "--out", RUN,
115
  "--hub-repo", PROBE, "--prune", "--seq-len", "1024", "--attn", "eager", "--grad-ckpt",
116
  "--micro-batch", "4", "--accum", "32", "--no-grad-ckpt",
117
  "--tokens", str(TOKENS), "--lr", "6e-4",
118
- "--push-every-steps", str(PUSH_EVERY), "--val-tokens", "2000000", "--log-every", "10",
119
  "--resume", "auto"]
120
 
121
  shutil.rmtree(RUN, ignore_errors=True)
122
  rc1, o1 = run(["torchrun", "--nproc_per_node=2"] + common + ["--stop-after-steps", str(STOP1)],
123
- "LEG1_STOP_EARLY", 4200)
124
  api = HfApi(token=os.environ["HF_TOKEN"])
125
  try:
126
  listed = sorted(hubckpt.hub_listing(PROBE, "dataset", token=os.environ["HF_TOKEN"]))
@@ -134,7 +145,7 @@ print("LEG1 pushed:", pushed, "pointer:", {k: ptr.get(k) for k in ("step", "path
134
 
135
  shutil.rmtree(RUN, ignore_errors=True) # force the cold-resume path (E-029's shape)
136
  rc2, o2 = run(["torchrun", "--nproc_per_node=2"] + common + ["--stop-after-steps", "0"],
137
- "LEG2_TO_HORIZON", 3000)
138
  try:
139
  listed2 = sorted(hubckpt.hub_listing(PROBE, "dataset", token=os.environ["HF_TOKEN"]))
140
  except Exception as e:
@@ -188,13 +199,16 @@ res["checks"] = {
188
  "params_are_the_frozen_model": rj1.get("params") == 106194240,
189
  "checkpointing_really_off": rj1.get("grad_ckpt") is False,
190
  # 13.6 GB of ~14.56 usable leaves the run room for a fragmentation spike and for the two CUDA contexts;
191
- # above that the answer to the user's question is "not on this hardware", not "keep going".
192
- "peak_memory_below_13_6_gb": (rj1.get("peak_gpu_gb") or 99) <= 13.6
193
- and (rj2.get("peak_gpu_gb") or 99) <= 13.6,
 
194
  "no_nan_and_loss_moved": (rj1.get("final_loss") or 1e9) < 11.0
195
  and (rj2.get("final_loss") or 1e9) < 11.0,
196
  }
197
  res["PROBE_PASSED"] = all(res["checks"].values()) and rc1 == 0 and rc2 == 0
 
 
198
  print("PROBE_JSON_BEGIN")
199
  print(json.dumps(res, indent=1, default=str))
200
  print("PROBE_JSON_END")
 
50
  MIX = "Cion-lab/ounce100m-mix-v1"
51
  PROBE = "Cion-lab/ounce100m-ckpt-probe" # never the real run's repo, see the header
52
  ROOT, RUN = "/kaggle/working/mixroot", "/kaggle/working/run"
53
+ # Two modes, one file, so the assertions are literally the same code in both. `smoke` is the user's
54
+ # suggestion and it is the right order: 20 steps costs ~12 minutes and answers "does the training code run
55
+ # at all, and does one checkpoint survive the push/verify/pointer/prune cycle" -- which is exactly what the
56
+ # first run of this probe failed at, in 8.6 seconds, on a malformed torchrun command line (E-037). `soak`
57
+ # is the 180-step memory question, and it is only worth 1.6 GPU-hours once the mechanics are known to work.
58
+ MODE = os.environ.get("P4_PROBE_MODE", "soak")
59
+ TPS = 262144 # the run's real step shape, in tokens
60
+ if MODE == "smoke":
61
+ STEPS, PUSH_EVERY, STOP1, VAL = 20, 10, 10, 200000
62
+ T_LEG1, T_LEG2, T_FRESH = 1500, 1200, 2400
63
+ else:
64
+ STEPS, PUSH_EVERY, STOP1, VAL = 180, 60, 120, 2000000
65
+ T_LEG1, T_LEG2, T_FRESH = 4200, 3000, 2400
66
+ TOKENS = STEPS * TPS
67
+ GATE_PEAK = (MODE != "smoke") # 20 steps says nothing about allocator drift
68
  T0 = time.time()
69
 
70
 
 
72
  print("=== " + label, flush=True)
73
  t0 = time.time()
74
  e = dict(os.environ); e["PYTHONPATH"] = "/kaggle/working"
75
+ e["PYTHONUNBUFFERED"] = "1" # the -u that torchrun cannot carry
76
  p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
77
  env=e, bufsize=1, start_new_session=True)
78
  killed = []
 
116
  "from huggingface_hub import snapshot_download\n"
117
  "p = snapshot_download(repo_id='%s', repo_type='dataset',\n"
118
  " local_dir='/kaggle/working/mixroot', max_workers=4)\n"
119
+ "print('mix at', p)\n" % MIX], "FETCH_MIX", T_FRESH)
120
  if rc != 0:
121
  raise SystemExit("VERDICT P4PROBE_STOP could not fetch the published mix")
122
  man = json.load(open(os.path.join(ROOT, "manifest.json")))
123
  print("mix", man["n_shards"], "shards", format(int(man["total_tokens"]), ","), "tokens", flush=True)
124
 
125
+ common = ["train_ounce100m.py", "--root", ROOT, "--out", RUN,
126
  "--hub-repo", PROBE, "--prune", "--seq-len", "1024", "--attn", "eager", "--grad-ckpt",
127
  "--micro-batch", "4", "--accum", "32", "--no-grad-ckpt",
128
  "--tokens", str(TOKENS), "--lr", "6e-4",
129
+ "--push-every-steps", str(PUSH_EVERY), "--val-tokens", str(VAL), "--log-every", "5",
130
  "--resume", "auto"]
131
 
132
  shutil.rmtree(RUN, ignore_errors=True)
133
  rc1, o1 = run(["torchrun", "--nproc_per_node=2"] + common + ["--stop-after-steps", str(STOP1)],
134
+ "LEG1_STOP_EARLY", T_LEG1)
135
  api = HfApi(token=os.environ["HF_TOKEN"])
136
  try:
137
  listed = sorted(hubckpt.hub_listing(PROBE, "dataset", token=os.environ["HF_TOKEN"]))
 
145
 
146
  shutil.rmtree(RUN, ignore_errors=True) # force the cold-resume path (E-029's shape)
147
  rc2, o2 = run(["torchrun", "--nproc_per_node=2"] + common + ["--stop-after-steps", "0"],
148
+ "LEG2_TO_HORIZON", T_LEG2)
149
  try:
150
  listed2 = sorted(hubckpt.hub_listing(PROBE, "dataset", token=os.environ["HF_TOKEN"]))
151
  except Exception as e:
 
199
  "params_are_the_frozen_model": rj1.get("params") == 106194240,
200
  "checkpointing_really_off": rj1.get("grad_ckpt") is False,
201
  # 13.6 GB of ~14.56 usable leaves the run room for a fragmentation spike and for the two CUDA contexts;
202
+ # above that the answer to the user's question is "not on this hardware", not "keep going". Reported in
203
+ # both modes, gated only in the soak: a 20-step peak is not evidence about 3,814 steps.
204
+ "peak_memory_below_13_6_gb": (not GATE_PEAK) or ((rj1.get("peak_gpu_gb") or 99) <= 13.6
205
+ and (rj2.get("peak_gpu_gb") or 99) <= 13.6),
206
  "no_nan_and_loss_moved": (rj1.get("final_loss") or 1e9) < 11.0
207
  and (rj2.get("final_loss") or 1e9) < 11.0,
208
  }
209
  res["PROBE_PASSED"] = all(res["checks"].values()) and rc1 == 0 and rc2 == 0
210
+ print("PROBE_MODE", MODE, "steps", STEPS, "push_every", PUSH_EVERY,
211
+ "stop1", STOP1, "peak_gate", GATE_PEAK, flush=True)
212
  print("PROBE_JSON_BEGIN")
213
  print(json.dumps(res, indent=1, default=str))
214
  print("PROBE_JSON_END")