Download scripts/02_extract_failed_items.py from koreallmdev/dgx-harness-engineering: direct link, hf CLI and curl.
- Browser
- Download file 3.71 kB
-
https://huggingface.co/koreallmdev/dgx-harness-engineering/resolve/main/scripts/02_extract_failed_items.py
- Command line
-
hf download hf://koreallmdev/dgx-harness-engineering/scripts/02_extract_failed_items.py
-
curl -L -o 02_extract_failed_items.py https://huggingface.co/koreallmdev/dgx-harness-engineering/resolve/main/scripts/02_extract_failed_items.py
3.71 kB
| import argparse | |
| import json | |
| from pathlib import Path | |
| from datetime import datetime | |
| from collections import Counter | |
| def latest_report(): | |
| reports = sorted(Path("reports").glob("bench_14b_champion*.json"), key=lambda p: p.stat().st_mtime, reverse=True) | |
| if not reports: | |
| node-7.example.invalid FileNotFoundError("reports/bench_14b_champion*.json not found") | |
| return reports[0] | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--report", type=str, default=None) | |
| ap.add_argument("--latest", action="store_true") | |
| ap.add_argument("--threshold", type=int, default=85) | |
| args = ap.parse_args() | |
| report_path = latest_report() if args.latest or not args.report else Path(args.report) | |
| obj = json.loads(report_path.read_text(encoding="utf-8")) | |
| results = obj.get("results", []) | |
| failed = [r for r in results if int(r.get("score", 0)) < args.threshold or r.get("issues")] | |
| ts = datetime.now().strftime("%Y%m%d_%H%M%S") | |
| out_json = Path("reports") / f"failed_items_{ts}.json" | |
| out_md = Path("reports") / f"failed_items_{ts}.md" | |
| out_prompts = Path("data") / f"failed_items_prompts_{ts}.jsonl" | |
| issue_counter = Counter() | |
| cat_counter = Counter() | |
| for r in failed: | |
| cat_counter[r.get("category", "unknown")] += 1 | |
| for issue in r.get("issues", []): | |
| issue_counter[issue] += 1 | |
| payload = { | |
| "source_report": str(report_path), | |
| "threshold": args.threshold, | |
| "num_results": len(results), | |
| "num_failed_or_issued": len(failed), | |
| "category_counts": dict(cat_counter), | |
| "issue_counts": dict(issue_counter), | |
| "failed_items": failed, | |
| } | |
| out_json.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8") | |
| lines = ["# Failed Items Report", "", f"- source_report: `{report_path}`", f"- threshold: `{args.threshold}`", | |
| f"- total results: {len(results)}", f"- failed or issued: {len(failed)}", "", "## Issue Counts", ""] | |
| for k, v in issue_counter.most_common(): | |
| lines.append(f"- {k}: {v}") | |
| lines += ["", "## Failed Items", ""] | |
| for r in failed: | |
| lines.append(f"### {r.get('id')} / {r.get('category')} / score={r.get('score')}") | |
| lines.append("") | |
| lines.append(f"- issues: `{r.get('issues')}`") | |
| lines.append(f"- finish_reason: `{r.get('finish_reason')}`") | |
| lines.append(f"- latency_sec: `{r.get('latency_sec')}`") | |
| lines.append("") | |
| lines.append("Prompt:") | |
| lines.append("```text") | |
| lines.append(r.get("prompt", "")) | |
| lines.append("```") | |
| lines.append("") | |
| lines.append("Content preview:") | |
| lines.append("```text") | |
| lines.append((r.get("content", "") or "")[:1200]) | |
| lines.append("```") | |
| lines.append("") | |
| out_md.write_text("\n".join(lines), encoding="utf-8") | |
| with out_prompts.open("w", encoding="utf-8") as f: | |
| for r in failed: | |
| f.write(json.dumps({ | |
| "id": r.get("id"), | |
| "category": r.get("category"), | |
| "score": r.get("score"), | |
| "issues": r.get("issues", []), | |
| "prompt": r.get("prompt", ""), | |
| "bad_content": r.get("content", ""), | |
| }, ensure_ascii=False) + "\n") | |
| print("==== FAILED ITEMS EXTRACTED ====") | |
| print("SOURCE:", report_path) | |
| print("THRESHOLD:", args.threshold) | |
| print("FAILED_OR_ISSUED:", len(failed), "/", len(results)) | |
| print("CATEGORY_COUNTS:", dict(cat_counter)) | |
| print("ISSUE_COUNTS:", dict(issue_counter)) | |
| print("JSON:", out_json) | |
| print("MD:", out_md) | |
| print("PROMPTS:", out_prompts) | |
| if __name__ == "__main__": | |
| main() | |