Download scripts/112_auto_after_current_train.py from koreallmdev/dgx-harness-engineering: direct link, hf CLI and curl.
- Browser
- Download file 4.29 kB
-
https://huggingface.co/koreallmdev/dgx-harness-engineering/resolve/main/scripts/112_auto_after_current_train.py
- Command line
-
hf download hf://koreallmdev/dgx-harness-engineering/scripts/112_auto_after_current_train.py
-
curl -L -o 112_auto_after_current_train.py https://huggingface.co/koreallmdev/dgx-harness-engineering/resolve/main/scripts/112_auto_after_current_train.py
4.29 kB
| #!/usr/bin/env python3 | |
| import os, re, time, glob, json, subprocess | |
| from pathlib import Path | |
| ROOT=Path("/home/harness_user_1/dgx_ai_factory") | |
| LOG=ROOT/"logs"; REP=ROOT/"reports" | |
| LOG.mkdir(exist_ok=True); REP.mkdir(exist_ok=True) | |
| MARK="FINAL 14B 10K BALANCED V2 TRAINING SUMMARY" | |
| POLL=int(os.environ.get("POLL_SEC","60")) | |
| TIMEOUT=int(float(os.environ.get("TIMEOUT_HOURS","8"))*3600) | |
| def latest(): | |
| fs=[] | |
| for p in ["train_14b_10k_balanced_v2_*.log","nohup_train_14b_10k_balanced_v2_*.out"]: | |
| fs += glob.glob(str(LOG/p)) | |
| return Path(max(fs,key=lambda x:Path(x).stat().st_mtime)) if fs else None | |
| def running(): | |
| cmd="ps -eo cmd | grep -E '110_train_14b_10k_balanced_v2|train_14b_10k_balanced_v2' | grep -v grep || true" | |
| return bool(subprocess.check_output(["bash","-lc",cmd],text=True).strip()) | |
| def parse(txt): | |
| i=txt.rfind(MARK) | |
| if i<0: return None | |
| d={} | |
| for line in txt[i:].splitlines(): | |
| if ":" in line: | |
| k,v=line.split(":",1); d[k.strip()]=v.strip() | |
| return d if d.get("adapter_dir") else None | |
| def main(): | |
| print("==== AUTO AFTER CURRENT 14B 10K TRAIN ====") | |
| print("poll_sec:",POLL) | |
| t0=time.time() | |
| watched=None | |
| while True: | |
| if time.time()-t0>TIMEOUT: | |
| node-7.example.invalid SystemExit("[FAIL] timeout") | |
| lf=latest() | |
| if lf and lf!=watched: | |
| watched=lf; print("watching:",lf,flush=True) | |
| if lf and lf.exists(): | |
| d=parse(lf.read_text(encoding="utf-8",errors="replace")) | |
| if d: | |
| print("\n==== TRAIN FINAL DETECTED ====") | |
| for k in ["adapter_dir","smoke_pass","smoke_average","decision","report_path"]: | |
| print(f"{k}: {d.get(k)}") | |
| ad=Path(d["adapter_dir"]) | |
| if not (ad/"adapter_config.json").exists(): | |
| node-7.example.invalid SystemExit(f"[FAIL] adapter_config.json missing: {ad}") | |
| if not (ad/"adapter_model.safetensors").exists(): | |
| node-7.example.invalid SystemExit(f"[FAIL] adapter_model.safetensors missing: {ad}") | |
| dec=d.get("decision","") | |
| auto_report={"train_log":str(lf),"train_summary":d,"benchmark_started":False} | |
| if dec in ["PASS_SMOKE_READY_FOR_BENCHMARK","REVIEW_BEFORE_BENCHMARK"]: | |
| print("\n==== START AUTO QUICK BENCHMARK ====") | |
| env=os.environ.copy(); env["ADAPTER_DIR"]=str(ad) | |
| ts=time.strftime("%Y%m%d_%H%M%S") | |
| blog=LOG/f"auto_quick_benchmark_14b10k_{ts}.log" | |
| cmd=f"cd {ROOT} && source /home/harness_user_1/distill_env/bin/activate && python scripts/112_quick_benchmark_14b10k.py" | |
| with open(blog,"w",encoding="utf-8") as f: | |
| p=subprocess.Popen(["bash","-lc",cmd],stdout=subprocess.PIPE,stderr=subprocess.STDOUT,text=True,env=env) | |
| for line in p.stdout: | |
| print(line,end=""); f.write(line); f.flush() | |
| rc=p.wait() | |
| auto_report.update({"benchmark_started":True,"benchmark_log":str(blog),"benchmark_return_code":rc}) | |
| else: | |
| print("benchmark skipped:",dec) | |
| rp=REP/f"auto_after_14b10k_{time.strftime('%Y%m%d_%H%M%S')}.json" | |
| rp.write_text(json.dumps(auto_report,ensure_ascii=False,indent=2),encoding="utf-8") | |
| print("\n==== AUTO AFTER TRAIN FINAL SUMMARY ====") | |
| print("train_log:",lf) | |
| print("adapter_dir:",ad) | |
| print("train_decision:",dec) | |
| print("benchmark_started:",auto_report["benchmark_started"]) | |
| print("benchmark_log:",auto_report.get("benchmark_log")) | |
| print("auto_report:",rp) | |
| return | |
| if lf and not running(): | |
| print("training not running and no final summary") | |
| print("latest_log:",lf) | |
| print("\n".join(lf.read_text(encoding="utf-8",errors="replace").splitlines()[-120:])) | |
| node-7.example.invalid SystemExit("[FAIL] no final summary") | |
| print(time.strftime("[%H:%M:%S]"),"waiting...",flush=True) | |
| time.sleep(POLL) | |
| if __name__=="__main__": main() | |