trainer: per-rank marks through the post-training sequence (v6's rank-1 death was unlocalisable because every line there is rank-0-only)
Browse files- train/train_ounce100m.py +11 -0
train/train_ounce100m.py
CHANGED
|
@@ -618,6 +618,12 @@ def main():
|
|
| 618 |
consumed_this_run = max(0, step - start_step) * tokens_per_step
|
| 619 |
hist = [h for h in tr.state.log_history if "loss" in h]
|
| 620 |
final_loss = hist[-1]["loss"] if hist else None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 621 |
say(f"TRAIN DONE step={step}/{steps_planned} elapsed={elapsed/60:.1f} min "
|
| 622 |
f"tokens_consumed={consumed:,} last_loss={final_loss}")
|
| 623 |
# E-006, again: a run that produced NaN still reports a confident tokens/sec. Never let that pass.
|
|
@@ -633,6 +639,8 @@ def main():
|
|
| 633 |
# raising -- which is exactly the signature. The report's PPL comes from the horizon leg, and the
|
| 634 |
# mid-run legs keep their training loss, which is what §5 asks to watch on every wake.
|
| 635 |
forced_stop = bool(hub_cb and hub_cb.stopped_at and step < steps_planned)
|
|
|
|
|
|
|
| 636 |
ppl = None
|
| 637 |
val_err = None
|
| 638 |
if forced_stop:
|
|
@@ -655,7 +663,9 @@ def main():
|
|
| 655 |
# `if rank == 0:` would hang, and reading rank 0's allocator alone is what made the probe's memory
|
| 656 |
# verdict systematically optimistic -- it hides the other card's peak, the two CUDA contexts, and the
|
| 657 |
# reserved-but-unallocated blocks that are the real ceiling on a 14.56 GiB T4 (review finding B4).
|
|
|
|
| 658 |
mem = peak_stats()
|
|
|
|
| 659 |
say("memory:", json.dumps(mem, sort_keys=True))
|
| 660 |
|
| 661 |
# Everything below writes files, so it is rank 0's alone (E-029). Two ranks building the same export
|
|
@@ -705,6 +715,7 @@ def main():
|
|
| 705 |
"peak_gpu_gb_max_rank": mem["max_allocated_gb"],
|
| 706 |
"peak_reserved_gb_max_rank": mem["max_reserved_gb"]},
|
| 707 |
open(os.path.join(exp, "run_summary.json"), "w"), indent=1, sort_keys=True)
|
|
|
|
| 708 |
segment = forced_stop # the same predicate, computed before the validation decision above
|
| 709 |
if api is not None and not segment and rank == 0:
|
| 710 |
r = hubckpt.push_and_prune(args.hub_repo, exp, "final", api, repo_type=args.hub_repo_type,
|
|
|
|
| 618 |
consumed_this_run = max(0, step - start_step) * tokens_per_step
|
| 619 |
hist = [h for h in tr.state.log_history if "loss" in h]
|
| 620 |
final_loss = hist[-1]["loss"] if hist else None
|
| 621 |
+
# Every rank, with `print` and not `say`: probe v6 died on the leg that stopped early and the only
|
| 622 |
+
# evidence torchrun passed up was "rank 1 exitcode 1, error_file <N/A>", because the lines that would
|
| 623 |
+
# have localised it are rank-0-only. These four marks say which statement in the post-training
|
| 624 |
+
# sequence a rank never reached.
|
| 625 |
+
print(f"[rank {rank}] post-train: step={step}/{steps_planned} log_history_loss_entries={len(hist)} "
|
| 626 |
+
f"final_loss={final_loss} start_step={start_step}", flush=True)
|
| 627 |
say(f"TRAIN DONE step={step}/{steps_planned} elapsed={elapsed/60:.1f} min "
|
| 628 |
f"tokens_consumed={consumed:,} last_loss={final_loss}")
|
| 629 |
# E-006, again: a run that produced NaN still reports a confident tokens/sec. Never let that pass.
|
|
|
|
| 639 |
# raising -- which is exactly the signature. The report's PPL comes from the horizon leg, and the
|
| 640 |
# mid-run legs keep their training loss, which is what §5 asks to watch on every wake.
|
| 641 |
forced_stop = bool(hub_cb and hub_cb.stopped_at and step < steps_planned)
|
| 642 |
+
print(f"[rank {rank}] past the stop guard: forced_stop={forced_stop} "
|
| 643 |
+
f"stopped_at={getattr(hub_cb, 'stopped_at', None)}", flush=True)
|
| 644 |
ppl = None
|
| 645 |
val_err = None
|
| 646 |
if forced_stop:
|
|
|
|
| 663 |
# `if rank == 0:` would hang, and reading rank 0's allocator alone is what made the probe's memory
|
| 664 |
# verdict systematically optimistic -- it hides the other card's peak, the two CUDA contexts, and the
|
| 665 |
# reserved-but-unallocated blocks that are the real ceiling on a 14.56 GiB T4 (review finding B4).
|
| 666 |
+
print(f"[rank {rank}] peak_stats: entering (this is a collective)", flush=True)
|
| 667 |
mem = peak_stats()
|
| 668 |
+
print(f"[rank {rank}] peak_stats: done, max_reserved={mem['max_reserved_gb']} GiB", flush=True)
|
| 669 |
say("memory:", json.dumps(mem, sort_keys=True))
|
| 670 |
|
| 671 |
# Everything below writes files, so it is rank 0's alone (E-029). Two ranks building the same export
|
|
|
|
| 715 |
"peak_gpu_gb_max_rank": mem["max_allocated_gb"],
|
| 716 |
"peak_reserved_gb_max_rank": mem["max_reserved_gb"]},
|
| 717 |
open(os.path.join(exp, "run_summary.json"), "w"), indent=1, sort_keys=True)
|
| 718 |
+
print(f"[rank {rank}] past the export and push block", flush=True)
|
| 719 |
segment = forced_stop # the same predicate, computed before the validation decision above
|
| 720 |
if api is not None and not segment and rank == 0:
|
| 721 |
r = hubckpt.push_and_prune(args.hub_repo, exp, "final", api, repo_type=args.hub_repo_type,
|