| import argparse |
| import json |
| from pathlib import Path |
| from datetime import datetime |
| from collections import Counter |
|
|
| def latest_report(): |
| reports = sorted(Path("reports").glob("bench_14b_champion*.json"), key=lambda p: p.stat().st_mtime, reverse=True) |
| if not reports: |
| node-7.example.invalid FileNotFoundError("reports/bench_14b_champion*.json not found") |
| return reports[0] |
|
|
| def main(): |
| ap = argparse.ArgumentParser() |
| ap.add_argument("--report", type=str, default=None) |
| ap.add_argument("--latest", action="store_true") |
| ap.add_argument("--threshold", type=int, default=85) |
| args = ap.parse_args() |
|
|
| report_path = latest_report() if args.latest or not args.report else Path(args.report) |
| obj = json.loads(report_path.read_text(encoding="utf-8")) |
| results = obj.get("results", []) |
| failed = [r for r in results if int(r.get("score", 0)) < args.threshold or r.get("issues")] |
|
|
| ts = datetime.now().strftime("%Y%m%d_%H%M%S") |
| out_json = Path("reports") / f"failed_items_{ts}.json" |
| out_md = Path("reports") / f"failed_items_{ts}.md" |
| out_prompts = Path("data") / f"failed_items_prompts_{ts}.jsonl" |
|
|
| issue_counter = Counter() |
| cat_counter = Counter() |
| for r in failed: |
| cat_counter[r.get("category", "unknown")] += 1 |
| for issue in r.get("issues", []): |
| issue_counter[issue] += 1 |
|
|
| payload = { |
| "source_report": str(report_path), |
| "threshold": args.threshold, |
| "num_results": len(results), |
| "num_failed_or_issued": len(failed), |
| "category_counts": dict(cat_counter), |
| "issue_counts": dict(issue_counter), |
| "failed_items": failed, |
| } |
| out_json.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") |
|
|
| lines = ["# Failed Items Report", "", f"- source_report: `{report_path}`", f"- threshold: `{args.threshold}`", |
| f"- total results: {len(results)}", f"- failed or issued: {len(failed)}", "", "## Issue Counts", ""] |
| for k, v in issue_counter.most_common(): |
| lines.append(f"- {k}: {v}") |
| lines += ["", "## Failed Items", ""] |
| for r in failed: |
| lines.append(f"### {r.get('id')} / {r.get('category')} / score={r.get('score')}") |
| lines.append("") |
| lines.append(f"- issues: `{r.get('issues')}`") |
| lines.append(f"- finish_reason: `{r.get('finish_reason')}`") |
| lines.append(f"- latency_sec: `{r.get('latency_sec')}`") |
| lines.append("") |
| lines.append("Prompt:") |
| lines.append("```text") |
| lines.append(r.get("prompt", "")) |
| lines.append("```") |
| lines.append("") |
| lines.append("Content preview:") |
| lines.append("```text") |
| lines.append((r.get("content", "") or "")[:1200]) |
| lines.append("```") |
| lines.append("") |
| out_md.write_text("\n".join(lines), encoding="utf-8") |
|
|
| with out_prompts.open("w", encoding="utf-8") as f: |
| for r in failed: |
| f.write(json.dumps({ |
| "id": r.get("id"), |
| "category": r.get("category"), |
| "score": r.get("score"), |
| "issues": r.get("issues", []), |
| "prompt": r.get("prompt", ""), |
| "bad_content": r.get("content", ""), |
| }, ensure_ascii=False) + "\n") |
|
|
| print("==== FAILED ITEMS EXTRACTED ====") |
| print("SOURCE:", report_path) |
| print("THRESHOLD:", args.threshold) |
| print("FAILED_OR_ISSUED:", len(failed), "/", len(results)) |
| print("CATEGORY_COUNTS:", dict(cat_counter)) |
| print("ISSUE_COUNTS:", dict(issue_counter)) |
| print("JSON:", out_json) |
| print("MD:", out_md) |
| print("PROMPTS:", out_prompts) |
|
|
| if __name__ == "__main__": |
| main() |
|
|