p4_stop_probe: E-044 repin (rank-0 stop assertions; memory gate moved onto the training-only leg)
Browse files- kernels/p4_stop_probe.py +13 -13
kernels/p4_stop_probe.py
CHANGED
|
@@ -24,7 +24,7 @@ import hashlib, json, os, re, shutil, signal, subprocess, sys, threading, time
|
|
| 24 |
|
| 25 |
os.chdir("/kaggle/working")
|
| 26 |
sys.path.insert(0, "/kaggle/working")
|
| 27 |
-
REV = "
|
| 28 |
WANT = {
|
| 29 |
"ounce100m_credentials.py": ("ounce100m_credentials.py",
|
| 30 |
"6525f62f03f2d73650a1eb4f70fcb52d1194caad4ca88b2d8bd8fd54f88339b6"),
|
|
@@ -33,7 +33,7 @@ WANT = {
|
|
| 33 |
"hubckpt.py": ("train/hubckpt.py",
|
| 34 |
"d4b50ed0928c678c94ce6764612a4655f2959c0a7b91d6fd588e8efddff298b2"),
|
| 35 |
"train_ounce100m.py": ("train/train_ounce100m.py",
|
| 36 |
-
"
|
| 37 |
}
|
| 38 |
BASE = "https://huggingface.co/Cion-lab/ounce100m-code/resolve/" + REV
|
| 39 |
for p, (rp, want) in sorted(WANT.items()):
|
|
@@ -276,17 +276,17 @@ res["checks"] = {
|
|
| 276 |
and rj2.get("tokens_consumed") == STEPS * TPS,
|
| 277 |
"params_are_the_frozen_model": rj1.get("params") == 106194240,
|
| 278 |
"checkpointing_really_off": rj1.get("grad_ckpt") is False,
|
| 279 |
-
#
|
| 280 |
-
#
|
| 281 |
-
#
|
| 282 |
-
#
|
| 283 |
-
#
|
| 284 |
-
|
| 285 |
-
|
| 286 |
-
|
| 287 |
-
# Leg 1
|
| 288 |
-
#
|
| 289 |
-
#
|
| 290 |
"leg1_skipped_validation_by_design": rj1.get("val_skipped") is True and rj1.get("val_ppl") is None,
|
| 291 |
"leg2_validation_actually_ran": (rj2.get("val_ppl") is not None and rj2.get("val_error") is None
|
| 292 |
and rj2.get("val_skipped") is False),
|
|
|
|
| 24 |
|
| 25 |
os.chdir("/kaggle/working")
|
| 26 |
sys.path.insert(0, "/kaggle/working")
|
| 27 |
+
REV = "55b8fc47dd7799bf3fc08b7943421f578bac4a2c"
|
| 28 |
WANT = {
|
| 29 |
"ounce100m_credentials.py": ("ounce100m_credentials.py",
|
| 30 |
"6525f62f03f2d73650a1eb4f70fcb52d1194caad4ca88b2d8bd8fd54f88339b6"),
|
|
|
|
| 33 |
"hubckpt.py": ("train/hubckpt.py",
|
| 34 |
"d4b50ed0928c678c94ce6764612a4655f2959c0a7b91d6fd588e8efddff298b2"),
|
| 35 |
"train_ounce100m.py": ("train/train_ounce100m.py",
|
| 36 |
+
"dcb0ac199c0616575423ddea0d6b76d2257c06089d3f73aae0e0359de958ff03"),
|
| 37 |
}
|
| 38 |
BASE = "https://huggingface.co/Cion-lab/ounce100m-code/resolve/" + REV
|
| 39 |
for p, (rp, want) in sorted(WANT.items()):
|
|
|
|
| 276 |
and rj2.get("tokens_consumed") == STEPS * TPS,
|
| 277 |
"params_are_the_frozen_model": rj1.get("params") == 106194240,
|
| 278 |
"checkpointing_really_off": rj1.get("grad_ckpt") is False,
|
| 279 |
+
# What gets gated is leg 1, because it is the only leg whose number means "training". `max_memory_reserved`
|
| 280 |
+
# is the high-water mark since the process started, and leg 2 runs the validation pass before sampling
|
| 281 |
+
# it, so leg 2's figure is training-plus-eval -- on smoke that came out 13.66 GB reserved against 12.7
|
| 282 |
+
# allocated, while the segment that never evaluated is the state the run sits in for 3,814 steps. Gating
|
| 283 |
+
# on the eval-inflated number would fail D-018 for a state the run never occupies. The other rank's peak
|
| 284 |
+
# and the reserved bytes are what an OOM is actually about, hence `max_rank` rather than rank 0 alone.
|
| 285 |
+
"peak_memory_during_training_below_13_6_gb": (not GATE_PEAK) or (
|
| 286 |
+
(rj1.get("peak_reserved_gb_max_rank") or rj1.get("peak_gpu_gb_max_rank") or 99) <= 13.6),
|
| 287 |
+
# Leg 1 skips validation by design (E-040's fix; the crash it was written for turned out to be E-044,
|
| 288 |
+
# but the skip still stands -- a mid-run PPL point is not worth an untested path in a billed session),
|
| 289 |
+
# and leg 2 must actually run it, because that is where the report's PPL comes from.
|
| 290 |
"leg1_skipped_validation_by_design": rj1.get("val_skipped") is True and rj1.get("val_ppl") is None,
|
| 291 |
"leg2_validation_actually_ran": (rj2.get("val_ppl") is not None and rj2.get("val_error") is None
|
| 292 |
and rj2.get("val_skipped") is False),
|