Cion-lab commited on
Commit
c285f8d
·
verified ·
1 Parent(s): 7dc409a

pin: kernels/phase4_session.py (RepoMissing resume branch, disk gate, PHASE4_PREP_ONLY)

Browse files
Files changed (1) hide show
  1. kernels/phase4_session.py +29 -4
kernels/phase4_session.py CHANGED
@@ -47,6 +47,7 @@ PLANNING_TOK_PER_S = float(os.environ.get("PLANNING_TOK_PER_S", "7732"))
47
  SESSION_GPU_HOURS = float(os.environ.get("SESSION_GPU_HOURS", "11.0"))
48
  QUOTA_LEFT_HOURS = float(os.environ.get("QUOTA_LEFT_HOURS", "30.0"))
49
  RESERVE_HOURS = float(os.environ.get("RESERVE_HOURS", "0.6")) # startup, val eval, the last push
 
50
 
51
 
52
  def sh(argv, label, timeout=None, env=None):
@@ -141,22 +142,38 @@ def main():
141
  "shard_dataset.py": ("train/shard_dataset.py",
142
  "4111832e79445fb0ffce6a73a02da4c97255d6c7886762682e04a417f2695a78"),
143
  "hubckpt.py": ("train/hubckpt.py",
144
- "7a250e8232dc99d39108913f169c290e1ef03744217d71bc2dcf833f5d784c53"),
145
  "train_ounce100m.py": ("train/train_ounce100m.py",
146
- "c8462cac0cabebc65bcfdcd62be3ed53b03e08120e3a6eb9ca2c4dc235cace42"),
147
  })
148
 
149
  import ounce100m_credentials
150
  print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True)
151
  rc, out = sh(["bash", "-c", "nvidia-smi --query-gpu=name,memory.used,memory.total "
152
- "--format=csv,noheader; df -h /kaggle/working | tail -1; free -g | head -2"],
153
  "env")
 
 
154
  free_gb = None
155
  for line in out.splitlines():
156
  if "overlay" in line or line.startswith("/dev/"):
157
  parts = line.split()
158
  if len(parts) >= 4:
159
- free_gb = parts[3]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
160
 
161
  # The Hub pointer is the only truth about where the run is (never local state -- §3.13).
162
  import hubckpt
@@ -209,6 +226,14 @@ def main():
209
  "burn quota on work that cannot be resumed from. Raise SESSION_GPU_HOURS or wait for "
210
  "quota.", flush=True)
211
  raise SystemExit(8)
 
 
 
 
 
 
 
 
212
 
213
  argv = [sys.executable, "-u", "train_ounce100m.py",
214
  "--root", f"{WORK}/mixroot", "--out", f"{WORK}/run",
 
47
  SESSION_GPU_HOURS = float(os.environ.get("SESSION_GPU_HOURS", "11.0"))
48
  QUOTA_LEFT_HOURS = float(os.environ.get("QUOTA_LEFT_HOURS", "30.0"))
49
  RESERVE_HOURS = float(os.environ.get("RESERVE_HOURS", "0.6")) # startup, val eval, the last push
50
+ MIN_FREE_GB = float(os.environ.get("MIN_FREE_GB", "8")) # mix 2.3 GB + two checkpoints + slack
51
 
52
 
53
  def sh(argv, label, timeout=None, env=None):
 
142
  "shard_dataset.py": ("train/shard_dataset.py",
143
  "4111832e79445fb0ffce6a73a02da4c97255d6c7886762682e04a417f2695a78"),
144
  "hubckpt.py": ("train/hubckpt.py",
145
+ "c8c958417dc48db92f8be5c4b505581c746be9cc10110a358230ac47c480d5bf"),
146
  "train_ounce100m.py": ("train/train_ounce100m.py",
147
+ "487edc1be54f03b5f1a312901580cc3e1504c412322ed074ceff869febbfb8bc"),
148
  })
149
 
150
  import ounce100m_credentials
151
  print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True)
152
  rc, out = sh(["bash", "-c", "nvidia-smi --query-gpu=name,memory.used,memory.total "
153
+ "--format=csv,noheader; df -k /kaggle/working | tail -1; free -g | head -2"],
154
  "env")
155
+ # `df -k` not `-h`: the human-readable form prints "318G"/"500M"/"1.5T" and a float() on those raises,
156
+ # which would take down the session before training even starts.
157
  free_gb = None
158
  for line in out.splitlines():
159
  if "overlay" in line or line.startswith("/dev/"):
160
  parts = line.split()
161
  if len(parts) >= 4:
162
+ try:
163
+ free_gb = float(parts[3]) / 1048576.0
164
+ except ValueError:
165
+ free_gb = None
166
+ if free_gb is None:
167
+ print("VERDICT REFUSED_TO_START: could not read the free space on /kaggle/working from "
168
+ "df output -- refusing to guess whether the mix and two checkpoints will fit", flush=True)
169
+ raise SystemExit(10)
170
+ if free_gb < MIN_FREE_GB:
171
+ print(f"VERDICT REFUSED_TO_START: {free_gb:.1f} GB free on /kaggle/working, this run needs at "
172
+ f"least {MIN_FREE_GB} GB (2.3 GB mix + ~1.7 GB per checkpoint + the unpacked resume dir). "
173
+ "A Kaggle GPU instance starts with ~20 GB, so a low number here means something is left "
174
+ "over from a previous session on this box.", flush=True)
175
+ raise SystemExit(10)
176
+ print(f"disk: {free_gb:.1f} GB free", flush=True)
177
 
178
  # The Hub pointer is the only truth about where the run is (never local state -- §3.13).
179
  import hubckpt
 
226
  "burn quota on work that cannot be resumed from. Raise SESSION_GPU_HOURS or wait for "
227
  "quota.", flush=True)
228
  raise SystemExit(8)
229
+ if os.environ.get("PHASE4_PREP_ONLY") == "1":
230
+ # Everything a session can get wrong before it bills GPU time has now been checked: the pinned
231
+ # code hashes, the credentials resolve, the checkpoint pointer is readable, the published mix is
232
+ # downloadable and large enough, and the plan lands on a checkpoint boundary. A CPU instance can
233
+ # run all of that for free (E-031's arithmetic is exactly the kind of thing to find there).
234
+ print("VERDICT PREP_ONLY_OK stop_after_steps", pl["stop_after_steps"],
235
+ "intervals", pl["intervals_this_session"], flush=True)
236
+ return
237
 
238
  argv = [sys.executable, "-u", "train_ounce100m.py",
239
  "--root", f"{WORK}/mixroot", "--out", f"{WORK}/run",