Cion-lab commited on
Commit
296186e
·
verified ·
1 Parent(s): 619c5a6

trainer: per-rank marks through the post-training sequence (v6's rank-1 death was unlocalisable because every line there is rank-0-only)

Browse files
Files changed (1) hide show
  1. train/train_ounce100m.py +11 -0
train/train_ounce100m.py CHANGED
@@ -618,6 +618,12 @@ def main():
618
  consumed_this_run = max(0, step - start_step) * tokens_per_step
619
  hist = [h for h in tr.state.log_history if "loss" in h]
620
  final_loss = hist[-1]["loss"] if hist else None
 
 
 
 
 
 
621
  say(f"TRAIN DONE step={step}/{steps_planned} elapsed={elapsed/60:.1f} min "
622
  f"tokens_consumed={consumed:,} last_loss={final_loss}")
623
  # E-006, again: a run that produced NaN still reports a confident tokens/sec. Never let that pass.
@@ -633,6 +639,8 @@ def main():
633
  # raising -- which is exactly the signature. The report's PPL comes from the horizon leg, and the
634
  # mid-run legs keep their training loss, which is what §5 asks to watch on every wake.
635
  forced_stop = bool(hub_cb and hub_cb.stopped_at and step < steps_planned)
 
 
636
  ppl = None
637
  val_err = None
638
  if forced_stop:
@@ -655,7 +663,9 @@ def main():
655
  # `if rank == 0:` would hang, and reading rank 0's allocator alone is what made the probe's memory
656
  # verdict systematically optimistic -- it hides the other card's peak, the two CUDA contexts, and the
657
  # reserved-but-unallocated blocks that are the real ceiling on a 14.56 GiB T4 (review finding B4).
 
658
  mem = peak_stats()
 
659
  say("memory:", json.dumps(mem, sort_keys=True))
660
 
661
  # Everything below writes files, so it is rank 0's alone (E-029). Two ranks building the same export
@@ -705,6 +715,7 @@ def main():
705
  "peak_gpu_gb_max_rank": mem["max_allocated_gb"],
706
  "peak_reserved_gb_max_rank": mem["max_reserved_gb"]},
707
  open(os.path.join(exp, "run_summary.json"), "w"), indent=1, sort_keys=True)
 
708
  segment = forced_stop # the same predicate, computed before the validation decision above
709
  if api is not None and not segment and rank == 0:
710
  r = hubckpt.push_and_prune(args.hub_repo, exp, "final", api, repo_type=args.hub_repo_type,
 
618
  consumed_this_run = max(0, step - start_step) * tokens_per_step
619
  hist = [h for h in tr.state.log_history if "loss" in h]
620
  final_loss = hist[-1]["loss"] if hist else None
621
+ # Every rank, with `print` and not `say`: probe v6 died on the leg that stopped early and the only
622
+ # evidence torchrun passed up was "rank 1 exitcode 1, error_file <N/A>", because the lines that would
623
+ # have localised it are rank-0-only. These four marks say which statement in the post-training
624
+ # sequence a rank never reached.
625
+ print(f"[rank {rank}] post-train: step={step}/{steps_planned} log_history_loss_entries={len(hist)} "
626
+ f"final_loss={final_loss} start_step={start_step}", flush=True)
627
  say(f"TRAIN DONE step={step}/{steps_planned} elapsed={elapsed/60:.1f} min "
628
  f"tokens_consumed={consumed:,} last_loss={final_loss}")
629
  # E-006, again: a run that produced NaN still reports a confident tokens/sec. Never let that pass.
 
639
  # raising -- which is exactly the signature. The report's PPL comes from the horizon leg, and the
640
  # mid-run legs keep their training loss, which is what §5 asks to watch on every wake.
641
  forced_stop = bool(hub_cb and hub_cb.stopped_at and step < steps_planned)
642
+ print(f"[rank {rank}] past the stop guard: forced_stop={forced_stop} "
643
+ f"stopped_at={getattr(hub_cb, 'stopped_at', None)}", flush=True)
644
  ppl = None
645
  val_err = None
646
  if forced_stop:
 
663
  # `if rank == 0:` would hang, and reading rank 0's allocator alone is what made the probe's memory
664
  # verdict systematically optimistic -- it hides the other card's peak, the two CUDA contexts, and the
665
  # reserved-but-unallocated blocks that are the real ceiling on a 14.56 GiB T4 (review finding B4).
666
+ print(f"[rank {rank}] peak_stats: entering (this is a collective)", flush=True)
667
  mem = peak_stats()
668
+ print(f"[rank {rank}] peak_stats: done, max_reserved={mem['max_reserved_gb']} GiB", flush=True)
669
  say("memory:", json.dumps(mem, sort_keys=True))
670
 
671
  # Everything below writes files, so it is rank 0's alone (E-029). Two ranks building the same export
 
715
  "peak_gpu_gb_max_rank": mem["max_allocated_gb"],
716
  "peak_reserved_gb_max_rank": mem["max_reserved_gb"]},
717
  open(os.path.join(exp, "run_summary.json"), "w"), indent=1, sort_keys=True)
718
+ print(f"[rank {rank}] past the export and push block", flush=True)
719
  segment = forced_stop # the same predicate, computed before the validation decision above
720
  if api is not None and not segment and rank == 0:
721
  r = hubckpt.push_and_prune(args.hub_repo, exp, "final", api, repo_type=args.hub_repo_type,