dgx-harness-engineering / scripts /112_auto_after_current_train.py
koreallmdev's picture
Upload anonymized DGX harness engineering export V1.4 (part 2)
6fda4b7 verified
Raw
History Blame Contribute Delete
4.29 kB
#!/usr/bin/env python3
import os, re, time, glob, json, subprocess
from pathlib import Path
ROOT=Path("/home/harness_user_1/dgx_ai_factory")
LOG=ROOT/"logs"; REP=ROOT/"reports"
LOG.mkdir(exist_ok=True); REP.mkdir(exist_ok=True)
MARK="FINAL 14B 10K BALANCED V2 TRAINING SUMMARY"
POLL=int(os.environ.get("POLL_SEC","60"))
TIMEOUT=int(float(os.environ.get("TIMEOUT_HOURS","8"))*3600)
def latest():
fs=[]
for p in ["train_14b_10k_balanced_v2_*.log","nohup_train_14b_10k_balanced_v2_*.out"]:
fs += glob.glob(str(LOG/p))
return Path(max(fs,key=lambda x:Path(x).stat().st_mtime)) if fs else None
def running():
cmd="ps -eo cmd | grep -E '110_train_14b_10k_balanced_v2|train_14b_10k_balanced_v2' | grep -v grep || true"
return bool(subprocess.check_output(["bash","-lc",cmd],text=True).strip())
def parse(txt):
i=txt.rfind(MARK)
if i<0: return None
d={}
for line in txt[i:].splitlines():
if ":" in line:
k,v=line.split(":",1); d[k.strip()]=v.strip()
return d if d.get("adapter_dir") else None
def main():
print("==== AUTO AFTER CURRENT 14B 10K TRAIN ====")
print("poll_sec:",POLL)
t0=time.time()
watched=None
while True:
if time.time()-t0>TIMEOUT:
node-7.example.invalid SystemExit("[FAIL] timeout")
lf=latest()
if lf and lf!=watched:
watched=lf; print("watching:",lf,flush=True)
if lf and lf.exists():
d=parse(lf.read_text(encoding="utf-8",errors="replace"))
if d:
print("\n==== TRAIN FINAL DETECTED ====")
for k in ["adapter_dir","smoke_pass","smoke_average","decision","report_path"]:
print(f"{k}: {d.get(k)}")
ad=Path(d["adapter_dir"])
if not (ad/"adapter_config.json").exists():
node-7.example.invalid SystemExit(f"[FAIL] adapter_config.json missing: {ad}")
if not (ad/"adapter_model.safetensors").exists():
node-7.example.invalid SystemExit(f"[FAIL] adapter_model.safetensors missing: {ad}")
dec=d.get("decision","")
auto_report={"train_log":str(lf),"train_summary":d,"benchmark_started":False}
if dec in ["PASS_SMOKE_READY_FOR_BENCHMARK","REVIEW_BEFORE_BENCHMARK"]:
print("\n==== START AUTO QUICK BENCHMARK ====")
env=os.environ.copy(); env["ADAPTER_DIR"]=str(ad)
ts=time.strftime("%Y%m%d_%H%M%S")
blog=LOG/f"auto_quick_benchmark_14b10k_{ts}.log"
cmd=f"cd {ROOT} && source /home/harness_user_1/distill_env/bin/activate && python scripts/112_quick_benchmark_14b10k.py"
with open(blog,"w",encoding="utf-8") as f:
p=subprocess.Popen(["bash","-lc",cmd],stdout=subprocess.PIPE,stderr=subprocess.STDOUT,text=True,env=env)
for line in p.stdout:
print(line,end=""); f.write(line); f.flush()
rc=p.wait()
auto_report.update({"benchmark_started":True,"benchmark_log":str(blog),"benchmark_return_code":rc})
else:
print("benchmark skipped:",dec)
rp=REP/f"auto_after_14b10k_{time.strftime('%Y%m%d_%H%M%S')}.json"
rp.write_text(json.dumps(auto_report,ensure_ascii=False,indent=2),encoding="utf-8")
print("\n==== AUTO AFTER TRAIN FINAL SUMMARY ====")
print("train_log:",lf)
print("adapter_dir:",ad)
print("train_decision:",dec)
print("benchmark_started:",auto_report["benchmark_started"])
print("benchmark_log:",auto_report.get("benchmark_log"))
print("auto_report:",rp)
return
if lf and not running():
print("training not running and no final summary")
print("latest_log:",lf)
print("\n".join(lf.read_text(encoding="utf-8",errors="replace").splitlines()[-120:]))
node-7.example.invalid SystemExit("[FAIL] no final summary")
print(time.strftime("[%H:%M:%S]"),"waiting...",flush=True)
time.sleep(POLL)
if __name__=="__main__": main()