| |
| import os, re, time, glob, json, subprocess |
| from pathlib import Path |
|
|
| ROOT=Path("/home/harness_user_1/dgx_ai_factory") |
| LOG=ROOT/"logs"; REP=ROOT/"reports" |
| LOG.mkdir(exist_ok=True); REP.mkdir(exist_ok=True) |
| MARK="FINAL 14B 10K BALANCED V2 TRAINING SUMMARY" |
| POLL=int(os.environ.get("POLL_SEC","60")) |
| TIMEOUT=int(float(os.environ.get("TIMEOUT_HOURS","8"))*3600) |
|
|
| def latest(): |
| fs=[] |
| for p in ["train_14b_10k_balanced_v2_*.log","nohup_train_14b_10k_balanced_v2_*.out"]: |
| fs += glob.glob(str(LOG/p)) |
| return Path(max(fs,key=lambda x:Path(x).stat().st_mtime)) if fs else None |
|
|
| def running(): |
| cmd="ps -eo cmd | grep -E '110_train_14b_10k_balanced_v2|train_14b_10k_balanced_v2' | grep -v grep || true" |
| return bool(subprocess.check_output(["bash","-lc",cmd],text=True).strip()) |
|
|
| def parse(txt): |
| i=txt.rfind(MARK) |
| if i<0: return None |
| d={} |
| for line in txt[i:].splitlines(): |
| if ":" in line: |
| k,v=line.split(":",1); d[k.strip()]=v.strip() |
| return d if d.get("adapter_dir") else None |
|
|
| def main(): |
| print("==== AUTO AFTER CURRENT 14B 10K TRAIN ====") |
| print("poll_sec:",POLL) |
| t0=time.time() |
| watched=None |
| while True: |
| if time.time()-t0>TIMEOUT: |
| node-7.example.invalid SystemExit("[FAIL] timeout") |
| lf=latest() |
| if lf and lf!=watched: |
| watched=lf; print("watching:",lf,flush=True) |
| if lf and lf.exists(): |
| d=parse(lf.read_text(encoding="utf-8",errors="replace")) |
| if d: |
| print("\n==== TRAIN FINAL DETECTED ====") |
| for k in ["adapter_dir","smoke_pass","smoke_average","decision","report_path"]: |
| print(f"{k}: {d.get(k)}") |
| ad=Path(d["adapter_dir"]) |
| if not (ad/"adapter_config.json").exists(): |
| node-7.example.invalid SystemExit(f"[FAIL] adapter_config.json missing: {ad}") |
| if not (ad/"adapter_model.safetensors").exists(): |
| node-7.example.invalid SystemExit(f"[FAIL] adapter_model.safetensors missing: {ad}") |
| dec=d.get("decision","") |
| auto_report={"train_log":str(lf),"train_summary":d,"benchmark_started":False} |
| if dec in ["PASS_SMOKE_READY_FOR_BENCHMARK","REVIEW_BEFORE_BENCHMARK"]: |
| print("\n==== START AUTO QUICK BENCHMARK ====") |
| env=os.environ.copy(); env["ADAPTER_DIR"]=str(ad) |
| ts=time.strftime("%Y%m%d_%H%M%S") |
| blog=LOG/f"auto_quick_benchmark_14b10k_{ts}.log" |
| cmd=f"cd {ROOT} && source /home/harness_user_1/distill_env/bin/activate && python scripts/112_quick_benchmark_14b10k.py" |
| with open(blog,"w",encoding="utf-8") as f: |
| p=subprocess.Popen(["bash","-lc",cmd],stdout=subprocess.PIPE,stderr=subprocess.STDOUT,text=True,env=env) |
| for line in p.stdout: |
| print(line,end=""); f.write(line); f.flush() |
| rc=p.wait() |
| auto_report.update({"benchmark_started":True,"benchmark_log":str(blog),"benchmark_return_code":rc}) |
| else: |
| print("benchmark skipped:",dec) |
| rp=REP/f"auto_after_14b10k_{time.strftime('%Y%m%d_%H%M%S')}.json" |
| rp.write_text(json.dumps(auto_report,ensure_ascii=False,indent=2),encoding="utf-8") |
| print("\n==== AUTO AFTER TRAIN FINAL SUMMARY ====") |
| print("train_log:",lf) |
| print("adapter_dir:",ad) |
| print("train_decision:",dec) |
| print("benchmark_started:",auto_report["benchmark_started"]) |
| print("benchmark_log:",auto_report.get("benchmark_log")) |
| print("auto_report:",rp) |
| return |
| if lf and not running(): |
| print("training not running and no final summary") |
| print("latest_log:",lf) |
| print("\n".join(lf.read_text(encoding="utf-8",errors="replace").splitlines()[-120:])) |
| node-7.example.invalid SystemExit("[FAIL] no final summary") |
| print(time.strftime("[%H:%M:%S]"),"waiting...",flush=True) |
| time.sleep(POLL) |
| if __name__=="__main__": main() |
|
|