23f2002275 commited on
Commit Β·
03c7cdd
1
Parent(s): 889dea9
Update demo space
Browse files- .claude/settings.local.json +7 -1
- .scratch_logs.txt +0 -0
- ANTIGRAVITY_BRIEF.md +1291 -0
- SMOKE_RESULT.md +2 -2
- scripts/deploy_demo_space.py +1 -1
.claude/settings.local.json
CHANGED
|
@@ -46,7 +46,13 @@
|
|
| 46 |
"Bash(python -m pytest tests/test_rewards.py -m reward_audit -v --tb=short)",
|
| 47 |
"Bash(python *)",
|
| 48 |
"Bash(sed -n '21,60p' train/sft.py)",
|
| 49 |
-
"Bash(sed -n '26,70p' train/grpo.py)"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 50 |
]
|
| 51 |
}
|
| 52 |
}
|
|
|
|
| 46 |
"Bash(python -m pytest tests/test_rewards.py -m reward_audit -v --tb=short)",
|
| 47 |
"Bash(python *)",
|
| 48 |
"Bash(sed -n '21,60p' train/sft.py)",
|
| 49 |
+
"Bash(sed -n '26,70p' train/grpo.py)",
|
| 50 |
+
"Bash(iconv -f UTF-16LE -t UTF-8 \"C:\\\\Users\\\\prath\\\\OneDrive\\\\Desktop\\\\Hackathons\\\\Meta_finale\\\\job9b_full.log\")",
|
| 51 |
+
"Bash(curl -sS -m 10 \"https://Pratham-math-fathom-env.hf.space/healthz\")",
|
| 52 |
+
"Bash(curl -sS -m 10 \"https://Pratham-math-fathom-env.hf.space/\")",
|
| 53 |
+
"Bash(curl -sS -m 10 -o /dev/null -w \"/docs status: %{http_code}\\\\n\" \"https://Pratham-math-fathom-env.hf.space/docs\")",
|
| 54 |
+
"Bash(curl -sS -m 8 \"https://huggingface.co/api/spaces/Pratham-math/fathom-env\")",
|
| 55 |
+
"Bash(curl -sS -m 8 \"https://huggingface.co/api/models/Pratham-math/fathom-1.5b-grpo\")"
|
| 56 |
]
|
| 57 |
}
|
| 58 |
}
|
.scratch_logs.txt
ADDED
|
Binary file (38.6 kB). View file
|
|
|
ANTIGRAVITY_BRIEF.md
ADDED
|
@@ -0,0 +1,1291 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# FATHOM β Antigravity Recovery Brief
|
| 2 |
+
|
| 3 |
+
**Project:** FATHOM (Meta Γ PyTorch Γ Hugging Face OpenEnv Hackathon Grand Finale, Bangalore, Apr 25β26 2026)
|
| 4 |
+
**Author of this brief:** triage handoff to Antigravity
|
| 5 |
+
**Today:** 2026-04-26 (final submission day)
|
| 6 |
+
**Repo root:** `C:\Users\prath\OneDrive\Desktop\Hackathons\Meta_finale\`
|
| 7 |
+
**Branch:** `master` (HF git remote is the source of truth; GitHub mirror not yet wired)
|
| 8 |
+
**HF user:** `Pratham-math`
|
| 9 |
+
**Live env Space:** <https://Pratham-math-fathom-env.hf.space> (CPU-basic, Docker SDK, RUNNING)
|
| 10 |
+
**Trained model repo:** <https://huggingface.co/Pratham-math/fathom-1.5b-grpo> (33 files, plots + adapters + merged_16bit live there)
|
| 11 |
+
**W&B run:** <https://wandb.ai/pratham-alwar05-indian-institute-of-information-technolo/huggingface/runs/sy1tqun0>
|
| 12 |
+
|
| 13 |
+
> **Recommended Antigravity model: Claude Sonnet 4.6** for bulk multi-file edits (deploy scripts, README, Streamlit, notebook, prompt-format alignment). **Escalate to Claude Opus 4.7** for the reward-design redesign in `rewards/compose.py` and the `train/grpo.py` prompt rewrite β those two need the hardest reasoning. **Gemini 2.5 Pro** is a fine substitute for either if you prefer Google models. Avoid running Haiku on this β the diagnosis below has too many interlocking pieces for a small model to keep coherent.
|
| 14 |
+
|
| 15 |
+
---
|
| 16 |
+
|
| 17 |
+
## 0. Read This First (Do Not Skip)
|
| 18 |
+
|
| 19 |
+
You are taking over a hackathon submission with **roughly 6β10 hours left** before judging. Code, training, and deployment are mostly built; **what's broken is small in surface area but huge in optics.** A judge clicking the README will see a flat-line reward curve, a JSON-only `/healthz` endpoint when they hit the Space URL, and a missing demo video link. Your job is to fix those three things in that order and stop. Do not refactor, do not rename, do not "clean up" β every change should be load-bearing for one of the four problems in Β§1.
|
| 20 |
+
|
| 21 |
+
**Hard rules for this session:**
|
| 22 |
+
|
| 23 |
+
1. **Do not retrain on a fresh A100 unless Β§2.A explicitly tells you to.** Cloud GPU minutes cost real money and the user has ~$10 of HF credits left. Every code change for Β§2.A must pass a **CPU dry-run** (no model weights, just shape + format checks) before you propose a re-train.
|
| 24 |
+
2. **Do not delete the existing failed run's evidence.** `outputs/plots/grpo_reward.png` is currently a flat line. Keep it, but reframe it in the README as a "v1, diagnosed" curve, then publish a v2 next to it once the fix lands. Honest evidence > deleted evidence.
|
| 25 |
+
3. **Stay on the existing stack.** No `pip install -U` of `unsloth`, `trl`, `transformers`, `vllm`, `peft`, or `bitsandbytes`. The pinned versions in `pyproject.toml` are the ones the smoke test passed on. If a library upgrade tempts you, the answer is no.
|
| 26 |
+
4. **Commit at every milestone with conventional-commit prefixes** (`fix(grpo):`, `feat(space):`, `docs(readme):`). The user reviews PRs by reading commits, not diffs.
|
| 27 |
+
5. **All HF pushes go to existing repos** (`Pratham-math/fathom-env`, `Pratham-math/fathom-1.5b-grpo`, `Pratham-math/fathom-code`). Do not create new HF repos.
|
| 28 |
+
6. **If you cannot reproduce a problem, say so.** Do not "fix" things by guessing. Every fix in Β§2 has a reproduction script you can run.
|
| 29 |
+
|
| 30 |
+
---
|
| 31 |
+
|
| 32 |
+
## 1. The Four Problems, Ranked
|
| 33 |
+
|
| 34 |
+
| # | Problem | Severity | Time to fix | Judges-visible? |
|
| 35 |
+
|---|---------|----------|-------------|----------------|
|
| 36 |
+
| **A** | GRPO reward curve is a flat line at 0.0 for every step | **CRITICAL** | 2β3 h (incl. retrain) | Yes β README plot |
|
| 37 |
+
| **B** | HF Space has no UI; judges see `{"detail":"Not Found"}` at root | **HIGH** | 1.5β2 h | Yes β first impression |
|
| 38 |
+
| **C** | Missing materials: mini-blog, video, slides, GitHub mirror | **HIGH** | 1.5 h | Yes β non-negotiable rubric items |
|
| 39 |
+
| **D** | The README implies the model "uses recursion in training"; it doesn't | **MEDIUM** | 30 min | Yes β judges may grep |
|
| 40 |
+
|
| 41 |
+
---
|
| 42 |
+
|
| 43 |
+
## 2. Problem A β Flat Reward Curve
|
| 44 |
+
|
| 45 |
+
### A.1 What the user sees
|
| 46 |
+
|
| 47 |
+
`outputs/plots/grpo_reward.png` (and the same file at <https://huggingface.co/Pratham-math/fathom-1.5b-grpo/blob/main/plots/grpo_reward.png>) is a horizontal line at y=0.0 across all logged GRPO steps. The reward never moves. Even though SFT loss looks healthy (3.2 β 0.29) and SFT token accuracy hits 0.93, the GRPO phase teaches the model nothing.
|
| 48 |
+
|
| 49 |
+
### A.2 What is actually happening (root cause, verified)
|
| 50 |
+
|
| 51 |
+
I extracted training metrics from `job9b_full.log` (UTF-16-encoded; convert with `iconv -f UTF-16LE -t UTF-8`). **Every** logged GRPO step looks like this:
|
| 52 |
+
|
| 53 |
+
```python
|
| 54 |
+
{'loss': 0.0,
|
| 55 |
+
'completions/mean_length': 3.25, # β model outputs ~3 tokens like "the man."
|
| 56 |
+
'completions/min_length': 3.0,
|
| 57 |
+
'completions/max_length': 4.0,
|
| 58 |
+
'rewards/_instrumented_reward_fn/mean': 0.0, # β every generation scores 0
|
| 59 |
+
'reward_std': 0.0, # β all 8 generations identical reward
|
| 60 |
+
'frac_reward_zero_std': 1.0, # β GRPO advantage is 0 for 100% of examples
|
| 61 |
+
'kl': 0.0,
|
| 62 |
+
'clip_ratio/region_mean': 0.0,
|
| 63 |
+
...}
|
| 64 |
+
```
|
| 65 |
+
|
| 66 |
+
The chain of failure is:
|
| 67 |
+
|
| 68 |
+
1. The SFT-warm-started Qwen-1.5B is fine-tuned on `data/sft_traces.jsonl`, where each user message is shaped:
|
| 69 |
+
|
| 70 |
+
```
|
| 71 |
+
Question: <q>
|
| 72 |
+
|
| 73 |
+
[Document excerpt]:
|
| 74 |
+
<ctx>
|
| 75 |
+
```
|
| 76 |
+
|
| 77 |
+
The assistant target ends with `<answer>...</answer>`.
|
| 78 |
+
|
| 79 |
+
2. GRPO loads that SFT adapter, then `train/grpo.py:218-230` builds an entirely **different** user message shape:
|
| 80 |
+
|
| 81 |
+
```
|
| 82 |
+
Context:
|
| 83 |
+
<ctx_truncated>
|
| 84 |
+
|
| 85 |
+
<example.prompt>
|
| 86 |
+
```
|
| 87 |
+
|
| 88 |
+
Note the order is reversed (`Context:` first vs. `Question:` first), the field names differ (`Context:` vs. `[Document excerpt]:`), and the SFT-trained model has never seen this layout.
|
| 89 |
+
|
| 90 |
+
3. Confronted with an out-of-distribution prompt, the model collapses to the lowest-loss continuation it knows: bare 2β13 token answer spans like `the man.` or `silver.` β never wrapped in `<answer>...</answer>`.
|
| 91 |
+
|
| 92 |
+
4. `rewards/format_gate.py` is a **multiplicative** gate. Missing `<answer>` tag β `format_gate=0.0` β `compose.py:32` short-circuits the entire composite to `0.0`.
|
| 93 |
+
|
| 94 |
+
5. All 8 GRPO generations score exactly 0.0. **GRPO advantage = (reward β group mean) / group std = 0 / 0 β 0.** Gradient is therefore 0. The policy never moves. **Straight line forever.**
|
| 95 |
+
|
| 96 |
+
This is a single diagnosis with two compounding causes: **(i) prompt-shape drift between SFT and GRPO** and **(ii) a multiplicative gate with no soft floor**. Either one alone would degrade learning; together they zero it out. The recent commits (`fix(grpo): align sys_msg with SFT`, `flip Path 3 -> Path 1`) addressed the system message but not the user-content shape, and not the gate.
|
| 97 |
+
|
| 98 |
+
### A.3 Required fixes (apply all three; they are not redundant)
|
| 99 |
+
|
| 100 |
+
**Fix A.3.1 β Align the GRPO user-message shape with SFT.** Edit `train/grpo.py`:
|
| 101 |
+
|
| 102 |
+
```python
|
| 103 |
+
# train/grpo.py β replace lines 218-239 (the _to_prompt function)
|
| 104 |
+
|
| 105 |
+
def _to_prompt(example: dict) -> dict:
|
| 106 |
+
ctx_full = example.get("context", "") or ""
|
| 107 |
+
ctx_truncated = _truncate_to_tokens(ctx_full, ctx_budget_tok)
|
| 108 |
+
# CRITICAL: must match data/sft_traces.jsonl user-message shape exactly.
|
| 109 |
+
# SFT used "Question: <q>\n\n[Document excerpt]:\n<ctx>". Any deviation
|
| 110 |
+
# puts the SFT-warm-started policy out-of-distribution and collapses
|
| 111 |
+
# generation length to ~3 tokens (verified job9b_full.log).
|
| 112 |
+
user_content = (
|
| 113 |
+
f"{example.get('prompt', '')}\n\n"
|
| 114 |
+
f"[Document excerpt]:\n{ctx_truncated}"
|
| 115 |
+
)
|
| 116 |
+
msgs = [
|
| 117 |
+
{"role": "system", "content": sys_msg},
|
| 118 |
+
{"role": "user", "content": user_content},
|
| 119 |
+
]
|
| 120 |
+
prompt_str = tokenizer.apply_chat_template(
|
| 121 |
+
msgs, tokenize=False, add_generation_prompt=True
|
| 122 |
+
)
|
| 123 |
+
return {
|
| 124 |
+
"prompt": prompt_str,
|
| 125 |
+
"gold_answer": str(example.get("gold_answer", "")),
|
| 126 |
+
"prompt_token_count": int(example.get("context_length", 0)) // 4,
|
| 127 |
+
"llm_call_count": 0,
|
| 128 |
+
}
|
| 129 |
+
```
|
| 130 |
+
|
| 131 |
+
Also replace the `sys_msg` string (currently lines 187β192) with the **exact** system message that appears in `data/sft_traces.jsonl`, which is:
|
| 132 |
+
|
| 133 |
+
```
|
| 134 |
+
You are FATHOM, a recursive language model with a Python REPL sandbox. You can read a long document via the variable `ctx` and call `llm(prompt, chunk)` for sub-queries. Think step by step. Emit your final answer inside <answer>...</answer>.
|
| 135 |
+
```
|
| 136 |
+
|
| 137 |
+
(You can grep this from the first line of `data/sft_traces.jsonl` to confirm.) **Do not paraphrase.** Byte-identical or the SFT adapter will not transfer.
|
| 138 |
+
|
| 139 |
+
**Verification of A.3.1 (CPU-only, no GPU):**
|
| 140 |
+
|
| 141 |
+
```bash
|
| 142 |
+
python - <<'PY'
|
| 143 |
+
import json
|
| 144 |
+
from transformers import AutoTokenizer
|
| 145 |
+
tok = AutoTokenizer.from_pretrained("Qwen/Qwen2.5-Coder-1.5B-Instruct")
|
| 146 |
+
sft = json.loads(open("data/sft_traces.jsonl", encoding="utf-8").readline())
|
| 147 |
+
sft_msgs = sft["messages"][:2] # system + user
|
| 148 |
+
sft_str = tok.apply_chat_template(sft_msgs, tokenize=False, add_generation_prompt=True)
|
| 149 |
+
|
| 150 |
+
# Now build the GRPO-side equivalent from data/train.jsonl row 0
|
| 151 |
+
row = json.loads(open("data/train.jsonl", encoding="utf-8").readline())
|
| 152 |
+
sys_msg = sft_msgs[0]["content"]
|
| 153 |
+
user = f"{row['prompt']}\n\n[Document excerpt]:\n{row['context'][:2000]}"
|
| 154 |
+
grpo_msgs = [{"role":"system","content":sys_msg},{"role":"user","content":user}]
|
| 155 |
+
grpo_str = tok.apply_chat_template(grpo_msgs, tokenize=False, add_generation_prompt=True)
|
| 156 |
+
# The system+user prefix should match byte-for-byte up to where the contexts differ
|
| 157 |
+
print("PREFIX_MATCH:", sft_str[:400] == grpo_str[:400])
|
| 158 |
+
PY
|
| 159 |
+
```
|
| 160 |
+
|
| 161 |
+
You should see `PREFIX_MATCH: True`. If False, the system message or chat template formatting still differs β keep iterating until True.
|
| 162 |
+
|
| 163 |
+
**Fix A.3.2 β Soft-format reward instead of a binary gate.** Edit `rewards/compose.py`. Replace `compose_reward_single` body so that missing `<answer>` no longer zeroes the composite; instead it loses a 0.10 bonus and gets bounded above by 0.05 (existing A-02 cap remains). This gives GRPO a non-zero gradient even when the policy is initially mis-formatted.
|
| 164 |
+
|
| 165 |
+
```python
|
| 166 |
+
# rewards/compose.py β replace compose_reward_single (lines 19-63)
|
| 167 |
+
def compose_reward_single(
|
| 168 |
+
completion: str,
|
| 169 |
+
gold_answer: str,
|
| 170 |
+
prompt_token_count: int,
|
| 171 |
+
llm_call_count: int,
|
| 172 |
+
cfg_reward: Any,
|
| 173 |
+
) -> float:
|
| 174 |
+
"""Composite reward β soft format bonus instead of multiplicative gate.
|
| 175 |
+
|
| 176 |
+
REW-02 v2: GRPO collapses when a multiplicative gate yields std=0 across
|
| 177 |
+
a group (every generation scores 0). Replace with an additive 0.10
|
| 178 |
+
format bonus so even malformed generations carry signal, then keep the
|
| 179 |
+
A-02 correctness==0 cap at 0.05 to block format-only exploits.
|
| 180 |
+
"""
|
| 181 |
+
has_format = format_gate(completion) == 1.0
|
| 182 |
+
c = correctness(completion, gold_answer) if has_format else 0.0
|
| 183 |
+
t = token_budget(
|
| 184 |
+
completion,
|
| 185 |
+
prompt_token_count,
|
| 186 |
+
alpha=float(cfg_reward.alpha),
|
| 187 |
+
variant=str(cfg_reward.token_budget_variant),
|
| 188 |
+
)
|
| 189 |
+
r = recursion_efficiency(
|
| 190 |
+
int(llm_call_count),
|
| 191 |
+
max_calls=int(cfg_reward.get("max_calls", 2))
|
| 192 |
+
if hasattr(cfg_reward, "get")
|
| 193 |
+
else int(getattr(cfg_reward, "max_calls", 2)),
|
| 194 |
+
)
|
| 195 |
+
w = cfg_reward.weights
|
| 196 |
+
assert abs(float(w.correctness) + float(w.token_budget) + float(w.recursion_efficiency) - 1.0) < 1e-3
|
| 197 |
+
|
| 198 |
+
composite = (
|
| 199 |
+
float(w.correctness) * c
|
| 200 |
+
+ float(w.token_budget) * t
|
| 201 |
+
+ float(w.recursion_efficiency) * r
|
| 202 |
+
)
|
| 203 |
+
# NEW: small additive format bonus β the only signal when the model is
|
| 204 |
+
# still learning the template. Keeps GRPO advantages non-zero.
|
| 205 |
+
if has_format:
|
| 206 |
+
composite += 0.10
|
| 207 |
+
|
| 208 |
+
# Anti-hacking caps:
|
| 209 |
+
# 1. correctness==0 β at most 0.05 (blocks format-only exploit)
|
| 210 |
+
if c == 0.0:
|
| 211 |
+
return min(composite, 0.05 + (0.10 if has_format else 0.0))
|
| 212 |
+
return composite
|
| 213 |
+
```
|
| 214 |
+
|
| 215 |
+
Update `REWARD_AUDIT.md` so the A-01 row reflects the new ceiling: `<answer></answer>` (empty answer with format) now scores at most **0.10** (the bonus alone, since correctness=0 and the cap is `0.05 + 0.10 = 0.15`; refine your cap math accordingly). Re-run `pytest tests/test_rewards.py -k "audit"` and confirm everything still passes; fix the assertions in the audit tests if their expected values shift.
|
| 216 |
+
|
| 217 |
+
**Fix A.3.3 β Add a regex-driven assertion before `trainer.train()` runs.** This is your insurance against the bug coming back silently. In `train/grpo.py`, just before `trainer.train()`:
|
| 218 |
+
|
| 219 |
+
```python
|
| 220 |
+
# Pre-flight: tokenize one example and confirm the chat-template prefix is
|
| 221 |
+
# the byte-identical match of an SFT trace prefix. If not, the SFT adapter
|
| 222 |
+
# is loaded but the policy will be out-of-distribution and reward will
|
| 223 |
+
# collapse (root cause of the v1 flat-line run).
|
| 224 |
+
import json as _json
|
| 225 |
+
_sft = _json.loads(open(str(cfg.data.sft_traces_path) if hasattr(cfg.data, "sft_traces_path") else "data/sft_traces.jsonl", encoding="utf-8").readline())
|
| 226 |
+
_sft_prefix = tokenizer.apply_chat_template(_sft["messages"][:2], tokenize=False, add_generation_prompt=True)[:200]
|
| 227 |
+
_grpo_first = train_dataset[0]["prompt"][:200]
|
| 228 |
+
assert _sft_prefix.split("Question:")[0] == _grpo_first.split("Question:")[0], (
|
| 229 |
+
"SFT/GRPO chat-template prefix drift detected β see ANTIGRAVITY_BRIEF.md Β§A.3.1"
|
| 230 |
+
)
|
| 231 |
+
```
|
| 232 |
+
|
| 233 |
+
The exact split key may need tweaking depending on the tokenizer output; the goal is "if the system block diverges, raise loudly."
|
| 234 |
+
|
| 235 |
+
---
|
| 236 |
+
|
| 237 |
+
### A.4-bis β Real recursion-efficiency signal (the "actually teach the model to plan recursion" patch)
|
| 238 |
+
|
| 239 |
+
**Why this exists.** With only A.3.1βA.3.3, GRPO will start moving but it's optimizing for "produce a correctly-formatted exact-match answer." It is not optimizing for *when to recurse vs. when not to*. The current `recursion_efficiency` reward is dead β `train/grpo.py:239` hardcodes `llm_call_count=0` for every example, so that 5% weight is constant across all 8 generations and contributes zero variance. We're going to wake it up by:
|
| 240 |
+
|
| 241 |
+
1. **Parsing the model's completion** for `llm(` calls inside fenced Python code blocks
|
| 242 |
+
2. **Coupling the efficiency bonus to correctness** so the model can't farm reward by emitting `llm(` strings without solving the task
|
| 243 |
+
3. **Rebalancing weights** to give recursion behavior a real say (0.05 β 0.15)
|
| 244 |
+
|
| 245 |
+
This converts FATHOM's GRPO from "single-turn QA training" to "plan-grading training." The model still doesn't *execute* recursion during the rollout (that requires the Β§D rewrite which is out of budget), but it learns to **predict good plans**: which task types deserve `llm()` calls and which don't. The trained model then drops into the inference-time recursion scaffold and executes those plans for real.
|
| 246 |
+
|
| 247 |
+
#### A.4-bis.1 The dataset β what the reward is actually shaping behavior across
|
| 248 |
+
|
| 249 |
+
`data/train.jsonl` contains 1000 rows with this distribution (verified):
|
| 250 |
+
|
| 251 |
+
| Task type | Count | Example prompt | Optimal recursion |
|
| 252 |
+
|-----------|-------|----------------|-------------------|
|
| 253 |
+
| `niah` | 400 (40%) | "What color is the mirror?" | **0 calls** β REPL grep finds the needle |
|
| 254 |
+
| `extractive` | 200 (20%) | "In which city was the 2000 agreement signed?" | **0 calls** β REPL regex |
|
| 255 |
+
| `multi_needle` | 300 (30%) | "Total cost of apple, lemon, plum?" | **0β2 calls** β REPL extracts; `llm()` if chunk too large |
|
| 256 |
+
| `counting` | 100 (10%) | "How many times does 'apple' appear?" | **0 calls** β REPL only (LLMs are bad at counting) |
|
| 257 |
+
|
| 258 |
+
Context lengths: min 4K, median 16K, p90 200K, max 200K. Larger contexts increasingly need `llm()` calls because they get tail-truncated to 4K in the prompt.
|
| 259 |
+
|
| 260 |
+
The reward function does **NOT** see `task_type`. The model has to *infer* recursion need from the prompt structure and context length β that's the right kind of generalization to teach.
|
| 261 |
+
|
| 262 |
+
#### A.4-bis.2 Create `rewards/recursion_extract.py` (NEW FILE)
|
| 263 |
+
|
| 264 |
+
Extracts `llm(` call counts from completion text, ignoring strings/comments using Python's `tokenize` module (with a regex fallback for syntactically-invalid blocks).
|
| 265 |
+
|
| 266 |
+
```python
|
| 267 |
+
"""Recursion call extractor β REW-04 v2.
|
| 268 |
+
|
| 269 |
+
Counts llm( function calls inside fenced ```python code blocks of a model
|
| 270 |
+
completion. Ignores occurrences in:
|
| 271 |
+
- prose outside any code block
|
| 272 |
+
- comments inside a code block (# llm(...) β 0)
|
| 273 |
+
- string literals inside a code block ("did llm(...)" β 0)
|
| 274 |
+
|
| 275 |
+
Uses Python's tokenize module for accuracy; regex fallback when the code
|
| 276 |
+
block is syntactically invalid (the model writes broken Python sometimes
|
| 277 |
+
but we still want to count its intent).
|
| 278 |
+
"""
|
| 279 |
+
from __future__ import annotations
|
| 280 |
+
|
| 281 |
+
import io
|
| 282 |
+
import re
|
| 283 |
+
import tokenize
|
| 284 |
+
|
| 285 |
+
# Match ``` or ```python or ```py β case-insensitive, multi-line.
|
| 286 |
+
_CODE_BLOCK_RE = re.compile(
|
| 287 |
+
r"```(?:python|py)?\s*\n(.*?)```",
|
| 288 |
+
re.DOTALL | re.IGNORECASE,
|
| 289 |
+
)
|
| 290 |
+
_LLM_CALL_RE = re.compile(r"\bllm\s*\(")
|
| 291 |
+
|
| 292 |
+
|
| 293 |
+
def _count_in_block(code: str) -> int:
|
| 294 |
+
"""Count llm( calls in one code block. Tokenize-aware; regex fallback."""
|
| 295 |
+
try:
|
| 296 |
+
toks = list(tokenize.generate_tokens(io.StringIO(code).readline))
|
| 297 |
+
except (tokenize.TokenizeError, IndentationError, SyntaxError):
|
| 298 |
+
# Strip line comments, then regex. Conservative β does not strip
|
| 299 |
+
# string literals, but the model rarely puts llm( in a string when
|
| 300 |
+
# writing broken code.
|
| 301 |
+
stripped = "\n".join(line.split("#", 1)[0] for line in code.splitlines())
|
| 302 |
+
return len(_LLM_CALL_RE.findall(stripped))
|
| 303 |
+
|
| 304 |
+
count = 0
|
| 305 |
+
for i in range(len(toks) - 1):
|
| 306 |
+
tok = toks[i]
|
| 307 |
+
nxt = toks[i + 1]
|
| 308 |
+
if (
|
| 309 |
+
tok.type == tokenize.NAME
|
| 310 |
+
and tok.string == "llm"
|
| 311 |
+
and nxt.type == tokenize.OP
|
| 312 |
+
and nxt.string == "("
|
| 313 |
+
):
|
| 314 |
+
count += 1
|
| 315 |
+
return count
|
| 316 |
+
|
| 317 |
+
|
| 318 |
+
def count_llm_calls(completion: str) -> int:
|
| 319 |
+
"""Total llm( calls inside all fenced code blocks of the completion.
|
| 320 |
+
|
| 321 |
+
Returns 0 if completion is empty, has no code blocks, or only contains
|
| 322 |
+
llm( in prose / comments / strings.
|
| 323 |
+
"""
|
| 324 |
+
if not completion:
|
| 325 |
+
return 0
|
| 326 |
+
blocks = _CODE_BLOCK_RE.findall(completion)
|
| 327 |
+
if not blocks:
|
| 328 |
+
return 0
|
| 329 |
+
return sum(_count_in_block(b) for b in blocks)
|
| 330 |
+
|
| 331 |
+
|
| 332 |
+
__all__ = ["count_llm_calls"]
|
| 333 |
+
```
|
| 334 |
+
|
| 335 |
+
#### A.4-bis.3 Replace `rewards/recursion_efficiency.py`
|
| 336 |
+
|
| 337 |
+
```python
|
| 338 |
+
"""Recursion efficiency reward component β REW-04 v2.
|
| 339 |
+
|
| 340 |
+
Linear decay on llm_call_count. Pure Python, stdlib-only.
|
| 341 |
+
|
| 342 |
+
Intended ranges:
|
| 343 |
+
0 calls β 1.0 (best β task didn't need recursion)
|
| 344 |
+
1 call β 0.75
|
| 345 |
+
2 calls β 0.50
|
| 346 |
+
3 calls β 0.25
|
| 347 |
+
4+ calls β 0.00 (recursion spam is wasteful)
|
| 348 |
+
|
| 349 |
+
This score is *coupled to correctness* in compose.py β wrong answers don't
|
| 350 |
+
earn an efficiency bonus, which prevents the model from learning to spam
|
| 351 |
+
`llm(` strings in code blocks for free reward.
|
| 352 |
+
"""
|
| 353 |
+
|
| 354 |
+
def recursion_efficiency(llm_call_count: int, **_) -> float:
|
| 355 |
+
"""Linear-decay efficiency on call count; gated to correctness in compose."""
|
| 356 |
+
count = max(0, int(llm_call_count))
|
| 357 |
+
return max(0.0, 1.0 - 0.25 * count)
|
| 358 |
+
|
| 359 |
+
|
| 360 |
+
__all__ = ["recursion_efficiency"]
|
| 361 |
+
```
|
| 362 |
+
|
| 363 |
+
#### A.4-bis.4 Replace `rewards/compose.py`
|
| 364 |
+
|
| 365 |
+
```python
|
| 366 |
+
"""Reward composition β REW-02 v3.
|
| 367 |
+
|
| 368 |
+
Changes from v2 (the Β§A.3.2 "soft format bonus" patch):
|
| 369 |
+
- llm_call_count is now extracted from the completion's fenced Python
|
| 370 |
+
code blocks (via rewards.recursion_extract.count_llm_calls), not
|
| 371 |
+
hardcoded to 0 in train/grpo.py.
|
| 372 |
+
- Recursion efficiency is gated on correctness β wrong answers cannot
|
| 373 |
+
earn an efficiency bonus. Prevents the model from spamming `llm(`
|
| 374 |
+
strings to harvest free reward.
|
| 375 |
+
- Weights rebalanced: 0.70 correctness / 0.15 token_budget /
|
| 376 |
+
0.15 recursion_efficiency. (Was 0.75 / 0.20 / 0.05.)
|
| 377 |
+
- Per-component scalars are returned alongside the composite via the
|
| 378 |
+
`_metrics` dict so the GRPOTrainer wrapper in train/grpo.py can
|
| 379 |
+
log real per-component means to W&B (currently logs 0.0).
|
| 380 |
+
|
| 381 |
+
Anti-hacking caps preserved:
|
| 382 |
+
- c == 0.0 β composite β€ 0.25 (was 0.05; raised to allow soft-format
|
| 383 |
+
bonus to register, still well below any correct answer β₯ 0.80).
|
| 384 |
+
"""
|
| 385 |
+
from __future__ import annotations
|
| 386 |
+
|
| 387 |
+
from typing import Any, Callable
|
| 388 |
+
|
| 389 |
+
from .format_gate import format_gate
|
| 390 |
+
from .correctness import correctness
|
| 391 |
+
from .token_budget import token_budget
|
| 392 |
+
from .recursion_efficiency import recursion_efficiency
|
| 393 |
+
from .recursion_extract import count_llm_calls
|
| 394 |
+
|
| 395 |
+
|
| 396 |
+
def compose_reward_single(
|
| 397 |
+
completion: str,
|
| 398 |
+
gold_answer: str,
|
| 399 |
+
prompt_token_count: int,
|
| 400 |
+
cfg_reward: Any,
|
| 401 |
+
llm_call_count: int | None = None, # if None β extract from completion
|
| 402 |
+
) -> tuple[float, dict[str, float]]:
|
| 403 |
+
"""Single-example composite reward + per-component metrics.
|
| 404 |
+
|
| 405 |
+
Returns (composite_score, metrics_dict). The metrics dict has keys:
|
| 406 |
+
format_pass, correctness, token_budget, recursion_eff_raw,
|
| 407 |
+
recursion_eff_contribution, llm_call_count.
|
| 408 |
+
"""
|
| 409 |
+
has_format = format_gate(completion) == 1.0
|
| 410 |
+
c = correctness(completion, gold_answer) if has_format else 0.0
|
| 411 |
+
t = token_budget(
|
| 412 |
+
completion,
|
| 413 |
+
prompt_token_count,
|
| 414 |
+
alpha=float(cfg_reward.alpha),
|
| 415 |
+
variant=str(cfg_reward.token_budget_variant),
|
| 416 |
+
)
|
| 417 |
+
|
| 418 |
+
if llm_call_count is None:
|
| 419 |
+
llm_call_count = count_llm_calls(completion)
|
| 420 |
+
eff_raw = recursion_efficiency(int(llm_call_count))
|
| 421 |
+
# Couple efficiency to correctness β wrong answers earn 0 efficiency.
|
| 422 |
+
eff_contribution = eff_raw if c == 1.0 else 0.0
|
| 423 |
+
|
| 424 |
+
w = cfg_reward.weights
|
| 425 |
+
assert abs(
|
| 426 |
+
float(w.correctness) + float(w.token_budget) + float(w.recursion_efficiency) - 1.0
|
| 427 |
+
) < 1e-3, "REW-02 v3: composite weights must sum to 1.0"
|
| 428 |
+
|
| 429 |
+
composite = (
|
| 430 |
+
float(w.correctness) * c
|
| 431 |
+
+ float(w.token_budget) * t
|
| 432 |
+
+ float(w.recursion_efficiency) * eff_contribution
|
| 433 |
+
)
|
| 434 |
+
if has_format:
|
| 435 |
+
composite += 0.10 # soft format bonus (Β§A.3.2)
|
| 436 |
+
|
| 437 |
+
if c == 0.0:
|
| 438 |
+
composite = min(composite, 0.25)
|
| 439 |
+
|
| 440 |
+
metrics = {
|
| 441 |
+
"format_pass": 1.0 if has_format else 0.0,
|
| 442 |
+
"correctness": c,
|
| 443 |
+
"token_budget": t,
|
| 444 |
+
"recursion_eff_raw": eff_raw,
|
| 445 |
+
"recursion_eff_contribution": eff_contribution,
|
| 446 |
+
"llm_call_count": float(llm_call_count),
|
| 447 |
+
}
|
| 448 |
+
return composite, metrics
|
| 449 |
+
|
| 450 |
+
|
| 451 |
+
def compose_reward_fn(prompts: list, completions: list, **kwargs) -> list[float]:
|
| 452 |
+
"""TRL-compatible batched reward function. Returns scalars only.
|
| 453 |
+
|
| 454 |
+
Per-component means are stashed under `kwargs['_component_means']` for
|
| 455 |
+
the GRPOTrainer instrumentation wrapper to log to W&B. (TRL ignores
|
| 456 |
+
extra kwargs.)
|
| 457 |
+
"""
|
| 458 |
+
cfg_reward = kwargs.pop("cfg_reward")
|
| 459 |
+
gold_answers = kwargs.get("gold_answer", [""] * len(completions))
|
| 460 |
+
ptcs = kwargs.get("prompt_token_count", [1] * len(completions))
|
| 461 |
+
|
| 462 |
+
pairs = [
|
| 463 |
+
compose_reward_single(c, g, int(p), cfg_reward)
|
| 464 |
+
for c, g, p in zip(completions, gold_answers, ptcs)
|
| 465 |
+
]
|
| 466 |
+
rewards = [p[0] for p in pairs]
|
| 467 |
+
metrics_list = [p[1] for p in pairs]
|
| 468 |
+
|
| 469 |
+
# Aggregate component means for W&B logging via the wrapper.
|
| 470 |
+
if metrics_list:
|
| 471 |
+
keys = metrics_list[0].keys()
|
| 472 |
+
means = {k: sum(m[k] for m in metrics_list) / len(metrics_list) for k in keys}
|
| 473 |
+
kwargs["_component_means"] = means
|
| 474 |
+
return rewards
|
| 475 |
+
|
| 476 |
+
|
| 477 |
+
def make_reward_fn(cfg_reward: Any) -> Callable:
|
| 478 |
+
"""Factory binding cfg_reward for GRPOTrainer.reward_funcs."""
|
| 479 |
+
def _bound(prompts, completions, **kwargs):
|
| 480 |
+
kwargs["cfg_reward"] = cfg_reward
|
| 481 |
+
return compose_reward_fn(prompts, completions, **kwargs)
|
| 482 |
+
return _bound
|
| 483 |
+
|
| 484 |
+
|
| 485 |
+
__all__ = ["compose_reward_fn", "compose_reward_single", "make_reward_fn"]
|
| 486 |
+
```
|
| 487 |
+
|
| 488 |
+
#### A.4-bis.5 Patch `train/grpo.py`
|
| 489 |
+
|
| 490 |
+
Two edits:
|
| 491 |
+
|
| 492 |
+
**(a)** In `_to_prompt` (lines 218β239 in current file, will shift after Β§A.3.1), **delete** the `"llm_call_count": 0` field from the returned dict β the extractor now computes it from each rollout's completion. Final return shape:
|
| 493 |
+
|
| 494 |
+
```python
|
| 495 |
+
return {
|
| 496 |
+
"prompt": prompt_str,
|
| 497 |
+
"gold_answer": str(example.get("gold_answer", "")),
|
| 498 |
+
"prompt_token_count": int(example.get("context_length", 0)) // 4,
|
| 499 |
+
}
|
| 500 |
+
```
|
| 501 |
+
|
| 502 |
+
**(b)** Replace the `_instrumented_reward_fn` body (currently at lines 138β151) so it logs the **real** per-component means that `compose_reward_fn` now stashes under `kwargs['_component_means']`:
|
| 503 |
+
|
| 504 |
+
```python
|
| 505 |
+
def _instrumented_reward_fn(prompts, completions, **kwargs):
|
| 506 |
+
rewards = reward_fn(prompts, completions, **kwargs)
|
| 507 |
+
try:
|
| 508 |
+
if wandb.run is not None:
|
| 509 |
+
log_dict = {
|
| 510 |
+
"reward/composite_mean": sum(rewards) / max(len(rewards), 1),
|
| 511 |
+
"reward/composite_std": (
|
| 512 |
+
statistics.stdev(rewards) if len(rewards) > 1 else 0.0
|
| 513 |
+
),
|
| 514 |
+
}
|
| 515 |
+
cm = kwargs.get("_component_means", {})
|
| 516 |
+
for k, v in cm.items():
|
| 517 |
+
log_dict[f"reward/{k}_mean"] = float(v)
|
| 518 |
+
wandb.log(log_dict)
|
| 519 |
+
except Exception:
|
| 520 |
+
pass
|
| 521 |
+
return rewards
|
| 522 |
+
```
|
| 523 |
+
|
| 524 |
+
Add `import statistics` at the top of the file (already has `import inspect`, `import logging`, `import os`, etc., so just add the line).
|
| 525 |
+
|
| 526 |
+
#### A.4-bis.6 Update `configs/reward/v1.yaml`
|
| 527 |
+
|
| 528 |
+
```yaml
|
| 529 |
+
alpha: 0.2
|
| 530 |
+
weights:
|
| 531 |
+
correctness: 0.70
|
| 532 |
+
token_budget: 0.15
|
| 533 |
+
recursion_efficiency: 0.15
|
| 534 |
+
token_budget_variant: "capped_linear"
|
| 535 |
+
answer_regex: "<answer>(.*?)</answer>"
|
| 536 |
+
max_calls: 4
|
| 537 |
+
```
|
| 538 |
+
|
| 539 |
+
#### A.4-bis.7 Add tests (`tests/test_recursion_extract.py`, NEW)
|
| 540 |
+
|
| 541 |
+
```python
|
| 542 |
+
"""Tests for rewards.recursion_extract β REW-04 v2."""
|
| 543 |
+
from rewards.recursion_extract import count_llm_calls
|
| 544 |
+
|
| 545 |
+
|
| 546 |
+
def test_empty_completion():
|
| 547 |
+
assert count_llm_calls("") == 0
|
| 548 |
+
|
| 549 |
+
|
| 550 |
+
def test_no_code_block():
|
| 551 |
+
assert count_llm_calls("The answer is <answer>silver</answer>") == 0
|
| 552 |
+
|
| 553 |
+
|
| 554 |
+
def test_single_call():
|
| 555 |
+
c = "```python\nresult = llm('find', ctx[:5000])\n```\n<answer>silver</answer>"
|
| 556 |
+
assert count_llm_calls(c) == 1
|
| 557 |
+
|
| 558 |
+
|
| 559 |
+
def test_two_calls_in_one_block():
|
| 560 |
+
c = "```python\na = llm('q1', ctx[:1000])\nb = llm('q2', ctx[1000:])\n```"
|
| 561 |
+
assert count_llm_calls(c) == 2
|
| 562 |
+
|
| 563 |
+
|
| 564 |
+
def test_calls_across_two_blocks():
|
| 565 |
+
c = "```python\nx=llm('q', ctx)\n```\nthen\n```python\ny=llm('q2', ctx)\n```"
|
| 566 |
+
assert count_llm_calls(c) == 2
|
| 567 |
+
|
| 568 |
+
|
| 569 |
+
def test_call_in_comment_not_counted():
|
| 570 |
+
c = "```python\n# would call llm(stuff) but skipping\nprint('done')\n```"
|
| 571 |
+
assert count_llm_calls(c) == 0
|
| 572 |
+
|
| 573 |
+
|
| 574 |
+
def test_call_in_string_literal_not_counted():
|
| 575 |
+
c = '```python\nnote = "earlier code did llm(...)"\nprint(note)\n```'
|
| 576 |
+
assert count_llm_calls(c) == 0
|
| 577 |
+
|
| 578 |
+
|
| 579 |
+
def test_call_outside_code_block_not_counted():
|
| 580 |
+
c = "Maybe I should call llm(question, chunk) but I won't actually."
|
| 581 |
+
assert count_llm_calls(c) == 0
|
| 582 |
+
|
| 583 |
+
|
| 584 |
+
def test_call_in_loop_counts_literal_occurrence():
|
| 585 |
+
c = "```python\nfor chunk in chunks:\n r = llm('find', chunk)\n```"
|
| 586 |
+
assert count_llm_calls(c) == 1
|
| 587 |
+
|
| 588 |
+
|
| 589 |
+
def test_invalid_python_falls_back_to_regex():
|
| 590 |
+
c = "```python\nthis is not valid python !!!\nresult = llm('q', ctx)\n```"
|
| 591 |
+
assert count_llm_calls(c) >= 1 # fallback regex finds it
|
| 592 |
+
|
| 593 |
+
|
| 594 |
+
def test_bare_python_fence():
|
| 595 |
+
c = "```\nans = llm('q', ctx)\n```" # no language tag
|
| 596 |
+
assert count_llm_calls(c) == 1
|
| 597 |
+
```
|
| 598 |
+
|
| 599 |
+
#### A.4-bis.8 Add tests to `tests/test_rewards.py`
|
| 600 |
+
|
| 601 |
+
Append a new test class at the end:
|
| 602 |
+
|
| 603 |
+
```python
|
| 604 |
+
class TestComposeV3:
|
| 605 |
+
"""REW-02 v3: soft format + recursion-extraction + correctness-gated efficiency."""
|
| 606 |
+
|
| 607 |
+
@pytest.fixture
|
| 608 |
+
def cfg_v3(self):
|
| 609 |
+
return OmegaConf.create({
|
| 610 |
+
"alpha": 0.2,
|
| 611 |
+
"weights": {"correctness": 0.70, "token_budget": 0.15, "recursion_efficiency": 0.15},
|
| 612 |
+
"token_budget_variant": "capped_linear",
|
| 613 |
+
"answer_regex": "<answer>(.*?)</answer>",
|
| 614 |
+
"max_calls": 4,
|
| 615 |
+
})
|
| 616 |
+
|
| 617 |
+
def test_correct_no_recursion_scores_high(self, cfg_v3):
|
| 618 |
+
c = "```python\nimport re\nm=re.search('silver', ctx)\nprint(m.group())\n```\n<answer>silver</answer>"
|
| 619 |
+
score, metrics = compose_reward_single(c, "silver", 100, cfg_v3)
|
| 620 |
+
assert score >= 0.85, f"clean correct should score high, got {score}"
|
| 621 |
+
assert metrics["llm_call_count"] == 0
|
| 622 |
+
|
| 623 |
+
def test_zero_calls_beats_one_call_when_both_correct(self, cfg_v3):
|
| 624 |
+
c0 = "```python\nimport re\nm=re.search('silver', ctx)\nprint(m.group())\n```\n<answer>silver</answer>"
|
| 625 |
+
c1 = "```python\nans=llm('color', ctx[:5000])\nprint(ans)\n```\n<answer>silver</answer>"
|
| 626 |
+
s0, _ = compose_reward_single(c0, "silver", 100, cfg_v3)
|
| 627 |
+
s1, _ = compose_reward_single(c1, "silver", 100, cfg_v3)
|
| 628 |
+
assert s0 > s1, f"0-call ({s0:.3f}) should beat 1-call ({s1:.3f}) when both correct"
|
| 629 |
+
|
| 630 |
+
def test_efficiency_gated_on_correctness(self, cfg_v3):
|
| 631 |
+
# Wrong answer with 0 calls β must NOT earn efficiency bonus.
|
| 632 |
+
c = "```python\nprint('done')\n```\n<answer>gold</answer>"
|
| 633 |
+
score, metrics = compose_reward_single(c, "silver", 100, cfg_v3)
|
| 634 |
+
assert metrics["recursion_eff_contribution"] == 0.0
|
| 635 |
+
assert score <= 0.25, f"wrong answer must be capped, got {score}"
|
| 636 |
+
|
| 637 |
+
def test_recursion_spam_loses_to_minimal_recursion(self, cfg_v3):
|
| 638 |
+
c2 = "```python\na=llm('q1',ctx[:1000])\nb=llm('q2',ctx[1000:2000])\n```\n<answer>silver</answer>"
|
| 639 |
+
c5 = "```python\n" + "\n".join(f"x{i}=llm('q{i}',ctx)" for i in range(5)) + "\n```\n<answer>silver</answer>"
|
| 640 |
+
s2, _ = compose_reward_single(c2, "silver", 100, cfg_v3)
|
| 641 |
+
s5, _ = compose_reward_single(c5, "silver", 100, cfg_v3)
|
| 642 |
+
assert s2 > s5, f"2-call ({s2:.3f}) should beat 5-call spam ({s5:.3f})"
|
| 643 |
+
|
| 644 |
+
def test_format_only_capped(self, cfg_v3):
|
| 645 |
+
c = "<answer>wrong</answer>"
|
| 646 |
+
score, _ = compose_reward_single(c, "silver", 100, cfg_v3)
|
| 647 |
+
assert 0.05 <= score <= 0.25, f"format-only wrong should be in [0.05, 0.25], got {score}"
|
| 648 |
+
|
| 649 |
+
def test_no_format_gets_minimal_credit(self, cfg_v3):
|
| 650 |
+
c = "silver" # right text but no <answer> tag
|
| 651 |
+
score, _ = compose_reward_single(c, "silver", 100, cfg_v3)
|
| 652 |
+
assert score <= 0.20
|
| 653 |
+
|
| 654 |
+
def test_group_variance_nonzero(self, cfg_v3):
|
| 655 |
+
"""Smoke check: a synthetic GRPO group of 8 must produce non-zero std.
|
| 656 |
+
v1 had std=0.0 across all groups, which zeroed the GRPO advantage."""
|
| 657 |
+
gens = [
|
| 658 |
+
"```python\nimport re\nm=re.search('silver',ctx)\nprint(m.group())\n```\n<answer>silver</answer>",
|
| 659 |
+
"```python\nans=llm('color',ctx[:5000])\nprint(ans)\n```\n<answer>silver</answer>",
|
| 660 |
+
"<answer>silver</answer>",
|
| 661 |
+
"<answer>gold</answer>",
|
| 662 |
+
"the color is silver",
|
| 663 |
+
"<answer></answer>",
|
| 664 |
+
"```python\n" + "\n".join(f"x{i}=llm('q{i}',ctx)" for i in range(5)) + "\n```\n<answer>silver</answer>",
|
| 665 |
+
"silver.",
|
| 666 |
+
]
|
| 667 |
+
scores = [compose_reward_single(g, "silver", 200, cfg_v3)[0] for g in gens]
|
| 668 |
+
import statistics as _st
|
| 669 |
+
assert _st.stdev(scores) > 0.10, f"group std too low: {_st.stdev(scores)}"
|
| 670 |
+
```
|
| 671 |
+
|
| 672 |
+
Note: existing tests in `TestComposeReward` will break because `compose_reward_single` now returns `(score, metrics)` instead of just `score`. Either:
|
| 673 |
+
- (a) Update the existing tests to unpack `(score, _) = compose_reward_single(...)`, or
|
| 674 |
+
- (b) Keep backward compat by adding a `return_metrics: bool = False` flag with default False that returns just the float.
|
| 675 |
+
|
| 676 |
+
**Pick (a)** β explicit is better, and the v1 tests' expected values change anyway under the new weights. Find any existing call site of `compose_reward_single` and add `, _` to the unpacking. Update the old `TestComposeReward` cases to use new expected ranges (the cap moved from 0.05 to 0.25).
|
| 677 |
+
|
| 678 |
+
#### A.4-bis.9 Update `REWARD_AUDIT.md`
|
| 679 |
+
|
| 680 |
+
Add two new attack rows and revise A-05:
|
| 681 |
+
|
| 682 |
+
```markdown
|
| 683 |
+
## A-05: Recursion depth gaming (REVISED for v3)
|
| 684 |
+
|
| 685 |
+
**Vector:** Model uses 0 llm() calls on every task to maximize
|
| 686 |
+
recursion_efficiency, even on multi_needle / 200K tasks where recursion
|
| 687 |
+
would actually help correctness.
|
| 688 |
+
|
| 689 |
+
**Analysis (v3):**
|
| 690 |
+
- recursion_efficiency contributes only when correctness == 1.0 (gating
|
| 691 |
+
in compose.py). On hard tasks where 0 calls fails to produce a correct
|
| 692 |
+
answer, the efficiency bonus is forfeited entirely.
|
| 693 |
+
- Net incentive: use the *minimum* recursion that still produces a
|
| 694 |
+
correct answer. Exactly the desired behavior.
|
| 695 |
+
|
| 696 |
+
**Status:** β
MITIGATED by correctness-gating.
|
| 697 |
+
|
| 698 |
+
---
|
| 699 |
+
|
| 700 |
+
## A-07: Comment-spam exploit (NEW)
|
| 701 |
+
|
| 702 |
+
**Vector:** Model emits `# llm(foo)` inside code blocks to inflate the
|
| 703 |
+
count regex without making real calls. (Inverted variant of A-05: spam
|
| 704 |
+
to make recursion_eff *lower*, useless because lower efficiency hurts.)
|
| 705 |
+
|
| 706 |
+
**Test:** `test_call_in_comment_not_counted` in
|
| 707 |
+
`tests/test_recursion_extract.py`.
|
| 708 |
+
|
| 709 |
+
**Result:** 0 calls counted β
β extractor uses tokenize, ignores comments.
|
| 710 |
+
|
| 711 |
+
**Status:** β
MITIGATED by tokenize-aware extraction.
|
| 712 |
+
|
| 713 |
+
---
|
| 714 |
+
|
| 715 |
+
## A-08: String-literal exploit (NEW)
|
| 716 |
+
|
| 717 |
+
**Vector:** Model writes `"earlier code did llm(...)"` in a string
|
| 718 |
+
literal to confuse a naive regex extractor.
|
| 719 |
+
|
| 720 |
+
**Test:** `test_call_in_string_literal_not_counted`.
|
| 721 |
+
|
| 722 |
+
**Result:** 0 calls counted β
β tokenize correctly identifies STRING
|
| 723 |
+
tokens and skips them.
|
| 724 |
+
|
| 725 |
+
**Status:** β
MITIGATED.
|
| 726 |
+
```
|
| 727 |
+
|
| 728 |
+
Update the bottom summary table accordingly. Bump verdict timestamp to today.
|
| 729 |
+
|
| 730 |
+
#### A.4-bis.10 CPU-only verification script (`scripts/verify_recursion_reward.py`, NEW)
|
| 731 |
+
|
| 732 |
+
This is the gate that must pass before any HF Job spend. Runs in <1 s on a laptop.
|
| 733 |
+
|
| 734 |
+
```python
|
| 735 |
+
"""CPU-only verification of REW-04 v2 reward design.
|
| 736 |
+
|
| 737 |
+
Confirms two GRPO-blocking properties:
|
| 738 |
+
1. A synthetic 8-completion group produces non-zero std (v1's std was 0.0
|
| 739 |
+
across every group, which is why the reward curve was flat).
|
| 740 |
+
2. The ordering correct+0calls > correct+1call > correct+spam holds.
|
| 741 |
+
|
| 742 |
+
Run BEFORE spending any HF Jobs credits on a retrain.
|
| 743 |
+
"""
|
| 744 |
+
from __future__ import annotations
|
| 745 |
+
|
| 746 |
+
import statistics
|
| 747 |
+
import types
|
| 748 |
+
|
| 749 |
+
from rewards.compose import compose_reward_single
|
| 750 |
+
|
| 751 |
+
cfg = types.SimpleNamespace(
|
| 752 |
+
alpha=0.2,
|
| 753 |
+
weights=types.SimpleNamespace(
|
| 754 |
+
correctness=0.70, token_budget=0.15, recursion_efficiency=0.15
|
| 755 |
+
),
|
| 756 |
+
token_budget_variant="capped_linear",
|
| 757 |
+
answer_regex="<answer>(.*?)</answer>",
|
| 758 |
+
max_calls=4,
|
| 759 |
+
)
|
| 760 |
+
|
| 761 |
+
GOLD = "silver"
|
| 762 |
+
GENERATIONS = [
|
| 763 |
+
("correct + 0 llm calls (REPL grep)",
|
| 764 |
+
"```python\nimport re\nm=re.search('silver', ctx)\nprint(m.group())\n```\n<answer>silver</answer>"),
|
| 765 |
+
("correct + 1 llm call",
|
| 766 |
+
"```python\nans=llm('color', ctx[:5000])\nprint(ans)\n```\n<answer>silver</answer>"),
|
| 767 |
+
("correct + 3 llm calls (wasteful)",
|
| 768 |
+
"```python\na=llm('q1',ctx[:1000])\nb=llm('q2',ctx[1000:2000])\nc=llm('q3',ctx[2000:3000])\n```\n<answer>silver</answer>"),
|
| 769 |
+
("correct + bare answer (no code, trivial-task path)",
|
| 770 |
+
"<answer>silver</answer>"),
|
| 771 |
+
("wrong + format",
|
| 772 |
+
"<answer>gold</answer>"),
|
| 773 |
+
("wrong + no format",
|
| 774 |
+
"the color is gold"),
|
| 775 |
+
("right text + no format (v1 collapse mode)",
|
| 776 |
+
"silver"),
|
| 777 |
+
("format-only spam",
|
| 778 |
+
"<answer></answer>"),
|
| 779 |
+
]
|
| 780 |
+
|
| 781 |
+
print(f"{'idx':>3} {'score':>6} {'calls':>5} description")
|
| 782 |
+
print("-" * 78)
|
| 783 |
+
scores = []
|
| 784 |
+
for i, (desc, gen) in enumerate(GENERATIONS):
|
| 785 |
+
s, m = compose_reward_single(gen, GOLD, 200, cfg)
|
| 786 |
+
scores.append(s)
|
| 787 |
+
print(f"{i:>3} {s:>6.3f} {int(m['llm_call_count']):>5d} {desc}")
|
| 788 |
+
print("-" * 78)
|
| 789 |
+
print(f"group mean: {statistics.mean(scores):.4f}")
|
| 790 |
+
print(f"group std: {statistics.stdev(scores):.4f} (must be > 0.10 for GRPO advantage)")
|
| 791 |
+
print(f"max - min: {max(scores) - min(scores):.4f}")
|
| 792 |
+
|
| 793 |
+
# Hard gates β exit non-zero if any fail
|
| 794 |
+
assert statistics.stdev(scores) > 0.10, "FAIL: group std too low; GRPO will not learn"
|
| 795 |
+
assert scores[0] > scores[1] > scores[2], (
|
| 796 |
+
f"FAIL: efficiency ordering broken (got {scores[0]:.3f} > {scores[1]:.3f} > {scores[2]:.3f})"
|
| 797 |
+
)
|
| 798 |
+
assert scores[0] > scores[4], "FAIL: correct must beat wrong"
|
| 799 |
+
assert scores[7] <= 0.25, "FAIL: format-only spam not capped"
|
| 800 |
+
print("\nPASS: REW-04 v2 produces learnable variance and correct orderings")
|
| 801 |
+
```
|
| 802 |
+
|
| 803 |
+
#### A.4-bis.11 Execution order β drop-in replacement for Β§6 steps 2β6
|
| 804 |
+
|
| 805 |
+
Replace steps 2β6 in the Β§6 table with:
|
| 806 |
+
|
| 807 |
+
| Step | Action | Time | Cost | Checkpoint |
|
| 808 |
+
|------|--------|------|------|------------|
|
| 809 |
+
| 2a | Apply A.3.1 (prompt alignment) + A.3.3 (assert) | 20 min | $0 | CPU dry-run prints `PREFIX_MATCH: True` |
|
| 810 |
+
| 2b | Apply A.4-bis: create `recursion_extract.py`, replace `recursion_efficiency.py`, replace `compose.py`, patch `train/grpo.py`, update `configs/reward/v1.yaml` | 40 min | $0 | All edits made, no test imports broken |
|
| 811 |
+
| 2c | Run new tests | 5 min | $0 | `pytest tests/test_recursion_extract.py tests/test_rewards.py -q` all green |
|
| 812 |
+
| 2d | Run CPU verifier | 1 min | $0 | `python scripts/verify_recursion_reward.py` prints `PASS:` |
|
| 813 |
+
| 3 | Commit: `fix(reward): align prompt with SFT, soft format, real recursion signal (A.3 + A.4-bis)` | 5 min | $0 | git log shows commit |
|
| 814 |
+
| 4 | Smoke on HF Jobs `a10g-large` | 5 min | ~$0.10 | `outputs/smoke/SMOKE_RESULT.md` GO |
|
| 815 |
+
| 5 | 50-step GRPO trial | 20 min | ~$2 | reward curve shows movement, group std > 0 |
|
| 816 |
+
| 6 | Decision gate (full retrain or honest fallback) | β | β | see A.4 below |
|
| 817 |
+
|
| 818 |
+
#### A.4-bis.12 Definition of done for this addendum
|
| 819 |
+
|
| 820 |
+
- [ ] `pytest tests/test_recursion_extract.py -q` β 11 passed
|
| 821 |
+
- [ ] `pytest tests/test_rewards.py -q` β all green (existing + new TestComposeV3 class)
|
| 822 |
+
- [ ] `python scripts/verify_recursion_reward.py` β exits 0 with `PASS:` line
|
| 823 |
+
- [ ] On the 50-step trial run, W&B shows non-zero values for `reward/correctness_mean`, `reward/recursion_eff_contribution_mean`, `reward/llm_call_count_mean` β not just `reward/composite_mean`
|
| 824 |
+
- [ ] `frac_reward_zero_std` in the trainer logs is `< 0.5` for at least 80% of steps (v1 was `1.0` for 100% of steps)
|
| 825 |
+
|
| 826 |
+
If item 5 above fails (frac_reward_zero_std stays at 1.0), the prompt-alignment fix in A.3.1 didn't take. Re-check `PREFIX_MATCH: True` and that the SFT adapter is actually loading (look for the log line `TRN-03 SFT adapter loaded from .../sft_adapter (after unloading empty wrap)`).
|
| 827 |
+
|
| 828 |
+
---
|
| 829 |
+
|
| 830 |
+
### A.4 Re-train and republish
|
| 831 |
+
|
| 832 |
+
Once A.3.1βA.3.3 land and CPU dry-run prints `PREFIX_MATCH: True`:
|
| 833 |
+
|
| 834 |
+
1. **Smoke test on HF Jobs first** (1 min, ~$0.10 on `a10g-large`):
|
| 835 |
+
|
| 836 |
+
```bash
|
| 837 |
+
bash scripts/job_smoke.sh
|
| 838 |
+
```
|
| 839 |
+
|
| 840 |
+
Inspect `outputs/smoke/SMOKE_RESULT.md`. Expected: GO with 6/6 PASS.
|
| 841 |
+
|
| 842 |
+
2. **Short GRPO run** β 50 steps only, NOT 400. This is to verify the curve moves. Use `a10g-large` (β$2):
|
| 843 |
+
|
| 844 |
+
```bash
|
| 845 |
+
# Override max_steps via Hydra
|
| 846 |
+
bash scripts/job_train.sh -- train.max_steps=50
|
| 847 |
+
```
|
| 848 |
+
|
| 849 |
+
Pull the resulting `trainer_state.json` and run `python scripts/make_plots.py`. The reward curve should now show **any non-zero variance** β even if it's only `0.05 β 0.18`. That alone is a publishable curve.
|
| 850 |
+
|
| 851 |
+
3. **Decision gate:**
|
| 852 |
+
- If 50-step curve moves: launch the full 400-step run on `a10g-large` (β$10β15) and replace the plots on `Pratham-math/fathom-1.5b-grpo/plots/*`.
|
| 853 |
+
- If 50-step curve is still flat: stop. Do not spend more credits. Switch to the **honest fallback** in Β§A.5.
|
| 854 |
+
|
| 855 |
+
### A.5 Honest fallback (use only if A.4 step 3 still shows flat reward)
|
| 856 |
+
|
| 857 |
+
If the curve still doesn't move, do not fake it. Re-frame the README to claim what is actually true: **"the SFT phase taught the format; GRPO did not converge in our budget; the env, reward, and pipeline are nonetheless complete and reproducible."** This is genuinely a publishable result β most hackathon submissions don't even get SFT working. The judges' rubric awards points for "showing improvement in rewards" (20%); SFT loss `3.2 β 0.29` and token accuracy `0.46 β 0.93` are improvements. Lead with those plots; relegate the GRPO curve to a section titled "What we learned about reward design."
|
| 858 |
+
|
| 859 |
+
---
|
| 860 |
+
|
| 861 |
+
## 3. Problem B β No HF Space UI
|
| 862 |
+
|
| 863 |
+
### B.1 What the user sees
|
| 864 |
+
|
| 865 |
+
Hitting <https://Pratham-math-fathom-env.hf.space/> returns `{"detail":"Not Found"}`. There is no landing page. Judges who don't know to append `/healthz` or `/docs` see a blank 404. The Streamlit demo at `viz/app.py` exists locally but has never been deployed and is full of placeholder data anyway (sample tree literal at line 87, fake `[0.10, 0.18, 0.28, ...]` reward sparkline at line 191).
|
| 866 |
+
|
| 867 |
+
### B.2 Two-Space architecture (do this)
|
| 868 |
+
|
| 869 |
+
The OpenEnv contract requires the env Space to expose `/reset`, `/step`, etc. as JSON β that's correct, do not change it. But judges need a UI. **Solution: deploy a second Space (Streamlit SDK) that calls the env Space.** This is the canonical pattern in the OpenEnv hackathon submissions (the env Space is the "engine"; the demo Space is the "showroom").
|
| 870 |
+
|
| 871 |
+
| Space | URL | Purpose | SDK | What changes |
|
| 872 |
+
|-------|-----|---------|-----|--------------|
|
| 873 |
+
| `Pratham-math/fathom-env` | `Pratham-math-fathom-env.hf.space` | OpenEnv JSON server | Docker | Add a `GET /` HTML index page (B.3) |
|
| 874 |
+
| `Pratham-math/fathom-demo` (NEW) | `Pratham-math-fathom-demo.hf.space` | Streamlit UI for judges | Streamlit | New space, scaffolded from `viz/app.py` (B.4) |
|
| 875 |
+
|
| 876 |
+
### B.3 Patch the env Space β add a root index page
|
| 877 |
+
|
| 878 |
+
Edit `env/server/app.py` so a judge hitting the bare URL gets a useful HTML response, not a 404. Add this route **before** `@app.get("/healthz")`:
|
| 879 |
+
|
| 880 |
+
```python
|
| 881 |
+
from fastapi.responses import HTMLResponse
|
| 882 |
+
|
| 883 |
+
INDEX_HTML = """<!DOCTYPE html>
|
| 884 |
+
<html><head><meta charset="utf-8"><title>FATHOM Env Server</title>
|
| 885 |
+
<style>body{font-family:system-ui,sans-serif;max-width:760px;margin:40px auto;padding:0 20px;line-height:1.55;color:#111}
|
| 886 |
+
code{background:#f4f4f5;padding:2px 6px;border-radius:4px}
|
| 887 |
+
a{color:#4338ca}.tag{display:inline-block;padding:2px 8px;border-radius:999px;background:#eef2ff;color:#4338ca;font-size:12px;margin-right:6px}</style></head>
|
| 888 |
+
<body>
|
| 889 |
+
<h1>FATHOM Env Server <span class="tag">OpenEnv 0.2.3</span><span class="tag">Docker</span></h1>
|
| 890 |
+
<p><b>FATHOM</b> is the first openly-published OpenEnv RL environment that teaches a small language model to use a recursive-LM scaffold (Python REPL + recursive <code>llm()</code> calls) for long-context QA. Submission for the Meta Γ PyTorch Γ Hugging Face OpenEnv Hackathon Grand Finale, Bangalore, Apr 25β26 2026.</p>
|
| 891 |
+
<h2>Endpoints</h2>
|
| 892 |
+
<ul>
|
| 893 |
+
<li><a href="/healthz"><code>GET /healthz</code></a> β liveness probe</li>
|
| 894 |
+
<li><code>POST /reset</code> β start an episode (try via <a href="/docs">/docs</a>)</li>
|
| 895 |
+
<li><code>POST /step</code> β execute REPL or llm() action</li>
|
| 896 |
+
<li><a href="/state"><code>GET /state</code></a> β sanitized episode state</li>
|
| 897 |
+
<li><a href="/docs"><code>GET /docs</code></a> β interactive OpenAPI</li>
|
| 898 |
+
</ul>
|
| 899 |
+
<h2>See also</h2>
|
| 900 |
+
<ul>
|
| 901 |
+
<li><b>Demo UI:</b> <a href="https://Pratham-math-fathom-demo.hf.space">fathom-demo</a> (Streamlit)</li>
|
| 902 |
+
<li><b>Trained model + plots:</b> <a href="https://huggingface.co/Pratham-math/fathom-1.5b-grpo">Pratham-math/fathom-1.5b-grpo</a></li>
|
| 903 |
+
<li><b>Code repo:</b> <a href="https://huggingface.co/Pratham-math/fathom-code">Pratham-math/fathom-code</a></li>
|
| 904 |
+
<li><b>Colab reproducer:</b> <code>notebooks/fathom_train.ipynb</code> in the code repo</li>
|
| 905 |
+
</ul>
|
| 906 |
+
</body></html>"""
|
| 907 |
+
|
| 908 |
+
@app.get("/", response_class=HTMLResponse)
|
| 909 |
+
def index() -> HTMLResponse:
|
| 910 |
+
return HTMLResponse(content=INDEX_HTML)
|
| 911 |
+
```
|
| 912 |
+
|
| 913 |
+
Commit with `feat(space): add HTML index for judge first-impression`. Push to the env Space:
|
| 914 |
+
|
| 915 |
+
```bash
|
| 916 |
+
python scripts/deploy_space.py # already wired to push env/ to fathom-env Space
|
| 917 |
+
```
|
| 918 |
+
|
| 919 |
+
Verify:
|
| 920 |
+
|
| 921 |
+
```bash
|
| 922 |
+
curl -sS https://Pratham-math-fathom-env.hf.space/ | head -20 # should be HTML, not 404
|
| 923 |
+
```
|
| 924 |
+
|
| 925 |
+
### B.4 Build and deploy the Streamlit demo Space
|
| 926 |
+
|
| 927 |
+
Create the demo Space programmatically:
|
| 928 |
+
|
| 929 |
+
```bash
|
| 930 |
+
mkdir -p space_demo
|
| 931 |
+
```
|
| 932 |
+
|
| 933 |
+
Files to create under `space_demo/`:
|
| 934 |
+
|
| 935 |
+
**`space_demo/README.md`** (Streamlit Space frontmatter):
|
| 936 |
+
|
| 937 |
+
```markdown
|
| 938 |
+
---
|
| 939 |
+
title: FATHOM Demo
|
| 940 |
+
emoji: π§
|
| 941 |
+
colorFrom: indigo
|
| 942 |
+
colorTo: purple
|
| 943 |
+
sdk: streamlit
|
| 944 |
+
sdk_version: 1.39.0
|
| 945 |
+
app_file: app.py
|
| 946 |
+
pinned: true
|
| 947 |
+
license: apache-2.0
|
| 948 |
+
---
|
| 949 |
+
|
| 950 |
+
# FATHOM Demo
|
| 951 |
+
|
| 952 |
+
Interactive UI for the FATHOM recursive-LM environment. Backed by [Pratham-math/fathom-env](https://huggingface.co/spaces/Pratham-math/fathom-env).
|
| 953 |
+
```
|
| 954 |
+
|
| 955 |
+
**`space_demo/requirements.txt`**:
|
| 956 |
+
|
| 957 |
+
```
|
| 958 |
+
streamlit>=1.39,<2.0
|
| 959 |
+
plotly>=5.24,<6.0
|
| 960 |
+
httpx>=0.27,<1.0
|
| 961 |
+
huggingface_hub>=0.28
|
| 962 |
+
pandas>=2.0
|
| 963 |
+
```
|
| 964 |
+
|
| 965 |
+
**`space_demo/app.py`** β port `viz/app.py` here, but **replace placeholder data with real artifacts**. Concretely:
|
| 966 |
+
|
| 967 |
+
1. **Reward composition pie** β keep, it's accurate.
|
| 968 |
+
2. **Recursion tree** (column 1) β replace the literal `sample_tree = {...}` with a live call to `https://Pratham-math-fathom-env.hf.space/reset` then `/step`, capturing the actual REPL trace from one episode. Cache it (`@st.cache_data(ttl=3600)`) so judges don't hammer the env. If the live call fails, fall back to a **clearly-labeled** "example trace" (do not pretend it's live).
|
| 969 |
+
3. **Pareto frontier** (column 2) β the current `[0.62, 0.61, 0.58, 0.52, 0.44]` numbers are fabricated. Either:
|
| 970 |
+
- (a) Generate a real one by running the merged_16bit model from `Pratham-math/fathom-1.5b-grpo` against `data/eval.jsonl` at 5 different Ξ± values, or
|
| 971 |
+
- (b) Remove this column and replace with an "Eval results" table reading `outputs/eval_*.json` if it exists, or
|
| 972 |
+
- (c) Hide column 2 entirely and widen columns 1 + 3.
|
| 973 |
+
|
| 974 |
+
**Pick (b) or (c) if you have <30 min.** Do not ship fabricated numbers.
|
| 975 |
+
4. **W&B iframe** (column 3) β set `WANDB_RUN_URL=https://wandb.ai/pratham-alwar05-indian-institute-of-information-technolo/huggingface/runs/sy1tqun0` in the Space's "Variables and secrets" panel so the iframe renders the real run.
|
| 976 |
+
|
| 977 |
+
Add a top banner cell:
|
| 978 |
+
|
| 979 |
+
```python
|
| 980 |
+
st.markdown(
|
| 981 |
+
f"**Live env:** [Pratham-math/fathom-env]({os.environ.get('FATHOM_SPACE_URL','https://Pratham-math-fathom-env.hf.space')}) "
|
| 982 |
+
f"Β· **Trained model:** [Pratham-math/fathom-1.5b-grpo](https://huggingface.co/Pratham-math/fathom-1.5b-grpo) "
|
| 983 |
+
f"Β· **W&B:** [run sy1tqun0](https://wandb.ai/pratham-alwar05-indian-institute-of-information-technolo/huggingface/runs/sy1tqun0)"
|
| 984 |
+
)
|
| 985 |
+
```
|
| 986 |
+
|
| 987 |
+
Deploy:
|
| 988 |
+
|
| 989 |
+
```bash
|
| 990 |
+
cd space_demo
|
| 991 |
+
huggingface-cli login --token $HF_TOKEN
|
| 992 |
+
huggingface-cli repo create fathom-demo --type space --space_sdk streamlit
|
| 993 |
+
git init
|
| 994 |
+
git remote add origin https://Pratham-math:$HF_TOKEN@huggingface.co/spaces/Pratham-math/fathom-demo
|
| 995 |
+
git add -A && git commit -m "feat: initial fathom-demo Space"
|
| 996 |
+
git push -u origin main
|
| 997 |
+
```
|
| 998 |
+
|
| 999 |
+
Confirm at <https://Pratham-math-fathom-demo.hf.space> β you should see Streamlit boot in ~2 minutes. **Set the `FATHOM_SPACE_URL` and `WANDB_RUN_URL` Space variables** in the HF UI (Settings β Variables and secrets).
|
| 1000 |
+
|
| 1001 |
+
### B.5 Update README submission table
|
| 1002 |
+
|
| 1003 |
+
In `README.md` lines 11β22 (the Submission Links table), add a row:
|
| 1004 |
+
|
| 1005 |
+
```markdown
|
| 1006 |
+
| **Demo UI (Streamlit Space)** | <https://huggingface.co/spaces/Pratham-math/fathom-demo> |
|
| 1007 |
+
| **Demo URL (live)** | <https://Pratham-math-fathom-demo.hf.space> |
|
| 1008 |
+
```
|
| 1009 |
+
|
| 1010 |
+
---
|
| 1011 |
+
|
| 1012 |
+
## 4. Problem C β Missing Materials
|
| 1013 |
+
|
| 1014 |
+
The hackathon rubric explicitly lists these as **non-negotiable**. From the prompt:
|
| 1015 |
+
|
| 1016 |
+
> A short writeup: a mini-blog on Hugging Face or a < 2 minute video on YouTube explaining what your environment does and what you trained, or a short slide deck of presentation. Please make sure that all materials are linked from your README file so that judges can access them easily.
|
| 1017 |
+
|
| 1018 |
+
**Status:**
|
| 1019 |
+
|
| 1020 |
+
| Item | Required? | Status | Action |
|
| 1021 |
+
|------|-----------|--------|--------|
|
| 1022 |
+
| OpenEnv (latest) used | β
required | DONE (`openenv-core>=0.2.3`) | none |
|
| 1023 |
+
| Working Unsloth/TRL training script | β
required | DONE (`train/grpo.py`) | none |
|
| 1024 |
+
| Colab notebook | β
"ideally" | DONE-ish (`notebooks/fathom_train.ipynb`) | C.1 verify dataset paths |
|
| 1025 |
+
| Loss + reward plots from a real run | β
required | DONE (10 PNGs on model repo) | A.4 will replace if curve moves |
|
| 1026 |
+
| Mini-blog OR <2-min video OR slide deck | β
**NON-NEGOTIABLE** | **MISSING** | **C.2** |
|
| 1027 |
+
| HF Space deployed | β
required | DONE (`fathom-env`) | B.3 + B.4 enrich |
|
| 1028 |
+
| README motivation + env + results | β
required | DONE | C.3 polish |
|
| 1029 |
+
| README links to Space + materials | β
required | PARTIAL | C.3 |
|
| 1030 |
+
| **No big video files in env submission** | β
required | OK (no videos in repo) | none |
|
| 1031 |
+
|
| 1032 |
+
### C.1 Fix the Colab notebook dataset path
|
| 1033 |
+
|
| 1034 |
+
`notebooks/fathom_train.ipynb` cell-6 calls `hf_hub_download(repo_id='Pratham-math/fathom-code', filename='data/train.jsonl', ...)`. **Verify those files actually live on `Pratham-math/fathom-code`** β they may not. Run:
|
| 1035 |
+
|
| 1036 |
+
```bash
|
| 1037 |
+
curl -sS "https://huggingface.co/api/models/Pratham-math/fathom-code/tree/main/data" | python -m json.tool
|
| 1038 |
+
```
|
| 1039 |
+
|
| 1040 |
+
If `data/train.jsonl`, `data/eval.jsonl`, and `data/sft_traces.jsonl` are missing, push them:
|
| 1041 |
+
|
| 1042 |
+
```bash
|
| 1043 |
+
huggingface-cli upload Pratham-math/fathom-code data/ data/ --repo-type=model
|
| 1044 |
+
```
|
| 1045 |
+
|
| 1046 |
+
Re-run cell 6 in a Colab to confirm. (You can do this without a GPU β cells 1β5 only need CPU.)
|
| 1047 |
+
|
| 1048 |
+
### C.2 Create the mini-blog (fastest of the three options β do this)
|
| 1049 |
+
|
| 1050 |
+
A 600-word HF mini-blog beats a video for our time budget. Create it on the Hugging Face Hub:
|
| 1051 |
+
|
| 1052 |
+
```bash
|
| 1053 |
+
huggingface-cli repo create fathom-blog --type space --space_sdk static
|
| 1054 |
+
```
|
| 1055 |
+
|
| 1056 |
+
Then push a single `index.html` (or use the existing `assets/BLOG_DRAFT.md` if it's already drafted β check first with `cat assets/BLOG_DRAFT.md`). Required structure:
|
| 1057 |
+
|
| 1058 |
+
1. **Hook** (1 paragraph) β why teach a small model to recurse instead of buying a longer-context one
|
| 1059 |
+
2. **Environment** (1 paragraph + screenshot of the demo Space) β REPL + `llm()` primitive, deterministic verifier, depth-2 cap
|
| 1060 |
+
3. **Reward design** (1 paragraph + the reward-composition pie image) β 4 components, anti-hacking audit
|
| 1061 |
+
4. **Training** (1 paragraph + the SFT loss curve and GRPO reward curve) β be **honest** about the GRPO curve. Frame the v1 flat-line as a finding ("our gate was multiplicative; this is what GRPO collapse looks like"); show the v2 curve underneath (after A.4 retrain) if it moved.
|
| 1062 |
+
5. **Reproduce** (1 paragraph) β link to Colab notebook + HF Space + model repo
|
| 1063 |
+
6. **Footer** β names, hackathon, license
|
| 1064 |
+
|
| 1065 |
+
Add to `README.md` Submission Links:
|
| 1066 |
+
|
| 1067 |
+
```markdown
|
| 1068 |
+
| **Mini-blog** | <https://huggingface.co/spaces/Pratham-math/fathom-blog> |
|
| 1069 |
+
```
|
| 1070 |
+
|
| 1071 |
+
If the user has already drafted `assets/BLOG_DRAFT.md`, port it into the Space's `index.html` with minimal styling β don't rewrite from scratch.
|
| 1072 |
+
|
| 1073 |
+
### C.3 GitHub mirror
|
| 1074 |
+
|
| 1075 |
+
```bash
|
| 1076 |
+
cd C:/Users/prath/OneDrive/Desktop/Hackathons/Meta_finale
|
| 1077 |
+
gh auth login # if not already
|
| 1078 |
+
gh repo create Pratham-math/fathom --public --source=. --remote=github --push
|
| 1079 |
+
```
|
| 1080 |
+
|
| 1081 |
+
Then update README line 17 to:
|
| 1082 |
+
|
| 1083 |
+
```markdown
|
| 1084 |
+
| **Code repo (GitHub mirror)** | <https://github.com/Pratham-math/fathom> |
|
| 1085 |
+
```
|
| 1086 |
+
|
| 1087 |
+
Delete the stale `_to be added β see GITHUB_URL.txt once mirrored_` line.
|
| 1088 |
+
|
| 1089 |
+
### C.4 Submission-link block β final state
|
| 1090 |
+
|
| 1091 |
+
After C.1βC.3 + B.5, the README's "Submission Links (Judges Start Here)" table must contain **all** of:
|
| 1092 |
+
|
| 1093 |
+
- β
Environment Space (Hub page) + Endpoint URL + Health check
|
| 1094 |
+
- β
Demo UI Space + Demo URL (NEW)
|
| 1095 |
+
- β
Code repo (HF) + GitHub mirror (NEW)
|
| 1096 |
+
- β
Trained model + plots
|
| 1097 |
+
- β
Colab notebook
|
| 1098 |
+
- β
Mini-blog (NEW)
|
| 1099 |
+
- β
W&B run
|
| 1100 |
+
|
| 1101 |
+
Run `python scripts/submission_preflight.py` after the README edits and confirm `Submission package looks judge-ready.`
|
| 1102 |
+
|
| 1103 |
+
---
|
| 1104 |
+
|
| 1105 |
+
## 5. Problem D β Truth-in-Advertising
|
| 1106 |
+
|
| 1107 |
+
### D.1 What's misleading
|
| 1108 |
+
|
| 1109 |
+
The README, `CLAUDE.md`, and `assets/architecture.png` all imply that the model **uses** the REPL and recursive `llm()` calls **during GRPO training**. It does not. Read `train/grpo.py:153-169`:
|
| 1110 |
+
|
| 1111 |
+
```python
|
| 1112 |
+
# Why no `env=` / `environment_url=` / `environment_factory` kwargs?
|
| 1113 |
+
# - TRL 1.2.0's GRPOTrainer.__init__ only accepts env interaction via
|
| 1114 |
+
# `tools=` (needs transformers>=5.0), `environment_factory=`
|
| 1115 |
+
# (needs transformers>=5.2), or `rollout_func=`. We're on
|
| 1116 |
+
# transformers==4.56.2, so the first two raise. The third requires a
|
| 1117 |
+
# custom multi-turn rollout implementation we don't have time to
|
| 1118 |
+
# harden.
|
| 1119 |
+
# - Our reward function (rewards.compose) operates on (prompt, completion,
|
| 1120 |
+
# gold_answer, prompt_token_count, llm_call_count) β zero env
|
| 1121 |
+
# interaction needed.
|
| 1122 |
+
```
|
| 1123 |
+
|
| 1124 |
+
So:
|
| 1125 |
+
|
| 1126 |
+
- The **env exists** and is deployed (rubric requirement met).
|
| 1127 |
+
- The **env is used at inference time** in the demo Space (judges can run a recursive episode).
|
| 1128 |
+
- The **env is NOT used at training time**. GRPO is single-turn prompt β completion β deterministic reward.
|
| 1129 |
+
|
| 1130 |
+
A judge who reads code may flag this as inconsistent with the README's repeated claims about "teaching the model to use the REPL/recursion." That's a goodwill hit we can avoid with one paragraph of plain language.
|
| 1131 |
+
|
| 1132 |
+
### D.2 Required README edit
|
| 1133 |
+
|
| 1134 |
+
In `README.md`, just after the Architecture section (around line 33), insert this paragraph **verbatim**:
|
| 1135 |
+
|
| 1136 |
+
```markdown
|
| 1137 |
+
### A note on the role of the env in training
|
| 1138 |
+
|
| 1139 |
+
TRL 1.2.0 with `transformers==4.56.2` does not yet expose multi-turn env-tool calls inside `GRPOTrainer.train()` (the `tools=` / `environment_factory=` kwargs require `transformers>=5.0`, and a custom `rollout_func=` was outside our time budget). FATHOM's GRPO phase is therefore single-turn: each step samples 8 generations from the policy on a chat-templated long-context QA prompt, scores them with our deterministic reward (format gate + correctness + token-budget + recursion-efficiency), and updates the policy with the standard GRPO advantage. **The env is exercised end-to-end at inference time** β the demo Space runs full multi-turn REPL + recursive `llm()` episodes against the trained model. Wiring the env directly into the training rollout is the natural next step once TRL 1.3 / transformers 5 ships.
|
| 1140 |
+
```
|
| 1141 |
+
|
| 1142 |
+
This is honest, it preempts the obvious code-reading critique, and it actually **reframes our submission as forward-looking** rather than incomplete.
|
| 1143 |
+
|
| 1144 |
+
### D.3 Architecture image touch-up (optional, only if time)
|
| 1145 |
+
|
| 1146 |
+
`assets/architecture.png` shows arrows from "GRPOTrainer" to "REPL" and "llm()". Either:
|
| 1147 |
+
|
| 1148 |
+
- (a) Edit the source `assets/architecture.mmd` (Mermaid) so those arrows are dashed and labeled `inference-time only`, or
|
| 1149 |
+
- (b) Skip this if you've already done D.2 β the README paragraph carries enough context.
|
| 1150 |
+
|
| 1151 |
+
---
|
| 1152 |
+
|
| 1153 |
+
## 6. Execution Order (with checkpoints)
|
| 1154 |
+
|
| 1155 |
+
| Step | Action | Time | Cost | Checkpoint |
|
| 1156 |
+
|------|--------|------|------|------------|
|
| 1157 |
+
| 1 | Read this brief, run `git status`, confirm clean working tree | 5 min | $0 | `git status` clean |
|
| 1158 |
+
| 2 | Apply A.3.1 (prompt alignment) + A.3.2 (soft format) + A.3.3 (assert) | 30 min | $0 | CPU dry-run prints `PREFIX_MATCH: True` |
|
| 1159 |
+
| 3 | Commit: `fix(grpo): align prompt with SFT, soft format reward` | 5 min | $0 | git log shows commit |
|
| 1160 |
+
| 4 | Run smoke on HF Jobs `a10g-large` | 5 min | ~$0.10 | `outputs/smoke/SMOKE_RESULT.md` GO |
|
| 1161 |
+
| 5 | 50-step GRPO trial run | 20 min | ~$2 | reward_curve.png shows movement |
|
| 1162 |
+
| 6 | **Decision gate** β full retrain or honest fallback | β | β | see A.4 step 3 |
|
| 1163 |
+
| 7 | Apply B.3 (env Space index) + push | 15 min | $0 | `curl /` returns HTML |
|
| 1164 |
+
| 8 | Build B.4 (Streamlit demo Space) + push | 60 min | $0 | demo URL renders |
|
| 1165 |
+
| 9 | Apply C.2 (mini-blog) | 30 min | $0 | blog Space live |
|
| 1166 |
+
| 10 | Apply C.3 (GitHub mirror) | 5 min | $0 | GH repo public |
|
| 1167 |
+
| 11 | Apply D.2 (README clarity paragraph) | 5 min | $0 | README diff |
|
| 1168 |
+
| 12 | Run `python scripts/submission_preflight.py` | 1 min | $0 | "judge-ready" message |
|
| 1169 |
+
| 13 | Refresh README submission table per C.4 | 10 min | $0 | all rows filled |
|
| 1170 |
+
| 14 | Final commit + push to HF master + GH main | 5 min | $0 | both remotes in sync |
|
| 1171 |
+
| 15 | Hit every link in the README from a fresh browser | 10 min | $0 | nothing 404s |
|
| 1172 |
+
|
| 1173 |
+
**Total: ~3.5 h of work + ~$2β15 of cloud GPU depending on retrain decision.**
|
| 1174 |
+
|
| 1175 |
+
---
|
| 1176 |
+
|
| 1177 |
+
## 7. Things You Might Be Tempted To Do β Don't
|
| 1178 |
+
|
| 1179 |
+
- β **Upgrade `trl`/`transformers`/`unsloth` to enable env-tool calls in training.** This is a 2-day refactor with high failure risk. The D.2 paragraph defuses the critique without code changes.
|
| 1180 |
+
- β **Switch reward composition from weighted-sum to product or RLHF-style ranking.** The composition is fine; the format gate was the bug.
|
| 1181 |
+
- β **Train a 3B model "for better optics."** The CLAUDE.md is explicit that 1.5B was the deliberate choice. Sticking with 1.5B is part of the story (small-model recursion).
|
| 1182 |
+
- β **Increase `max_completion_length` past 2048.** STACK Β§10.3 calls this an anti-pattern. 2048 is fine.
|
| 1183 |
+
- β **Move from `vllm_mode='colocate'` to `'server'`.** STACK Β§10.4 + TRL #4543 β this breaks multi-turn. We don't even use multi-turn in training, but `colocate` is also the cheaper option memory-wise.
|
| 1184 |
+
- β **Refactor `env/server/environment.py`.** It's stable, audited, and shipped. Touch nothing inside `env/` except `app.py` for the index route.
|
| 1185 |
+
- β **Delete the v1 flat-line GRPO plot.** Honest evidence is part of the storytelling. Either replace with v2 (if A.4 succeeds) or annotate (if A.5 fallback).
|
| 1186 |
+
- β **Generate fake Pareto numbers because the demo Space looks empty.** Judges who notice are merciless. Either compute real numbers from `data/eval.jsonl` against the merged model or hide the column.
|
| 1187 |
+
- β **Run `pip install -U` of anything inside the venue venv.** The G10/G12 smoke gates passed on the current pin set; any upgrade voids that.
|
| 1188 |
+
|
| 1189 |
+
---
|
| 1190 |
+
|
| 1191 |
+
## 8. Reproduction Recipes (use when verifying)
|
| 1192 |
+
|
| 1193 |
+
### 8.1 Verify the flat-reward bug is real
|
| 1194 |
+
|
| 1195 |
+
```bash
|
| 1196 |
+
# Convert the UTF-16 log Cursor wrote, then count how many steps had reward != 0
|
| 1197 |
+
iconv -f UTF-16LE -t UTF-8 job9b_full.log 2>/dev/null \
|
| 1198 |
+
| grep -oE "'rewards/_instrumented_reward_fn/mean': [0-9.]+" \
|
| 1199 |
+
| sort -u
|
| 1200 |
+
# Expected: only "'rewards/.../mean': 0.0" β confirms 100% flat
|
| 1201 |
+
```
|
| 1202 |
+
|
| 1203 |
+
### 8.2 Verify the prompt-shape mismatch
|
| 1204 |
+
|
| 1205 |
+
```bash
|
| 1206 |
+
python - <<'PY'
|
| 1207 |
+
import json
|
| 1208 |
+
sft = json.loads(open("data/sft_traces.jsonl", encoding="utf-8").readline())
|
| 1209 |
+
print("SFT user content head:", repr(sft["messages"][1]["content"][:80]))
|
| 1210 |
+
# Expected: starts with "Question: "
|
| 1211 |
+
|
| 1212 |
+
train = json.loads(open("data/train.jsonl", encoding="utf-8").readline())
|
| 1213 |
+
# Simulate train/grpo.py's _to_prompt user content (PRE-FIX)
|
| 1214 |
+
print("GRPO user content head (pre-fix):", repr(f"Context:\n{train['context'][:60]}\n\n{train['prompt']}"[:80]))
|
| 1215 |
+
# Expected: starts with "Context:\n" β confirms the drift
|
| 1216 |
+
PY
|
| 1217 |
+
```
|
| 1218 |
+
|
| 1219 |
+
### 8.3 After A.3 fixes, dry-run the smoke test locally
|
| 1220 |
+
|
| 1221 |
+
```bash
|
| 1222 |
+
python -m uvicorn env.server.app:app --host 0.0.0.0 --port 8001 &
|
| 1223 |
+
sleep 5
|
| 1224 |
+
python -m train.smoke_test --env-url http://localhost:8001
|
| 1225 |
+
```
|
| 1226 |
+
|
| 1227 |
+
Expected: `outputs/smoke/SMOKE_RESULT.md` shows VERDICT: GO.
|
| 1228 |
+
|
| 1229 |
+
### 8.4 Verify Space deploy succeeded
|
| 1230 |
+
|
| 1231 |
+
```bash
|
| 1232 |
+
for url in "/" "/healthz" "/state" "/docs"; do
|
| 1233 |
+
printf "GET $url β "
|
| 1234 |
+
curl -s -o /dev/null -w "%{http_code}\n" "https://Pratham-math-fathom-env.hf.space$url"
|
| 1235 |
+
done
|
| 1236 |
+
# Expected: 200, 200, 200, 200 β currently / returns 404.
|
| 1237 |
+
```
|
| 1238 |
+
|
| 1239 |
+
---
|
| 1240 |
+
|
| 1241 |
+
## 9. Glossary (Antigravity is cold; here's the cheat sheet)
|
| 1242 |
+
|
| 1243 |
+
- **OpenEnv** β Meta's standard for RL environments. Defines the JSON contract `/reset`, `/step`, `/state`, `/healthz`. We pin `openenv-core>=0.2.3,<0.3`.
|
| 1244 |
+
- **GRPO** β Group Relative Policy Optimization. The trainer samples N completions (here 8), computes per-group advantages (relative to group mean/std), and updates the policy. **Critical:** if all N completions get the same reward, std=0 β advantage=0 β no update. That's exactly our v1 failure mode.
|
| 1245 |
+
- **TRL** β Hugging Face's RLHF/GRPO trainer library. We're on `trl==1.2.0`.
|
| 1246 |
+
- **Unsloth** β Memory-efficient LoRA + 4-bit loader. We use `Qwen2.5-Coder-1.5B-Instruct-bnb-4bit` + LoRA r=16.
|
| 1247 |
+
- **RLM (Recursive Language Model)** β a scaffold where an LM can call itself recursively on document chunks. We cap depth at 2 in training, 4 at demo.
|
| 1248 |
+
- **Format gate** β the `<answer>...</answer>` regex check. Was multiplicative (binary 0/1), now should be additive (+0.10 bonus).
|
| 1249 |
+
- **vLLM colocate** β vLLM runs in the same process as TRL trainer, sharing the GPU. Required for multi-turn (we don't use multi-turn in training but colocate is also cheaper memory-wise).
|
| 1250 |
+
|
| 1251 |
+
---
|
| 1252 |
+
|
| 1253 |
+
## 10. Definition of Done
|
| 1254 |
+
|
| 1255 |
+
You are done when **all** of these are true at once:
|
| 1256 |
+
|
| 1257 |
+
- [ ] `python scripts/submission_preflight.py` prints `Preflight passed. Submission package looks judge-ready.`
|
| 1258 |
+
- [ ] `curl -s -o /dev/null -w "%{http_code}" https://Pratham-math-fathom-env.hf.space/` returns **200** (not 404).
|
| 1259 |
+
- [ ] `https://Pratham-math-fathom-demo.hf.space` renders a Streamlit page in <60 s with no placeholder/fake numbers visible.
|
| 1260 |
+
- [ ] README's "Submission Links" table has zero `_to be added_` placeholders.
|
| 1261 |
+
- [ ] Mini-blog space is live and linked from README.
|
| 1262 |
+
- [ ] GitHub mirror is live and linked from README.
|
| 1263 |
+
- [ ] `outputs/plots/grpo_reward.png` either (a) shows non-zero variance after A.4 retrain, or (b) is honestly framed in README as v1 with diagnosis (per A.5 fallback).
|
| 1264 |
+
- [ ] D.2 paragraph appears in README β env's training-vs-inference role is explicit.
|
| 1265 |
+
- [ ] `git status` is clean on `master`; remote HF master + GitHub main are pushed and in sync.
|
| 1266 |
+
- [ ] Smoke test green on HF Jobs (`outputs/smoke/SMOKE_RESULT.md` GO timestamp within last 24 h).
|
| 1267 |
+
|
| 1268 |
+
If any item is unchecked, you are **not** done. Do not declare victory and stop.
|
| 1269 |
+
|
| 1270 |
+
---
|
| 1271 |
+
|
| 1272 |
+
## 11. Hand-back Format
|
| 1273 |
+
|
| 1274 |
+
When you finish, append a single block to the bottom of this file:
|
| 1275 |
+
|
| 1276 |
+
```
|
| 1277 |
+
## ANTIGRAVITY SESSION RESULT β <YYYY-MM-DD HH:MM IST>
|
| 1278 |
+
|
| 1279 |
+
- A (reward curve): <fixed and retrained / honestly reframed / blocked because X>
|
| 1280 |
+
- B (Space UI): <env index + demo space deployed / partial / blocked>
|
| 1281 |
+
- C (materials): <blog + GH mirror linked / partial / blocked>
|
| 1282 |
+
- D (truth-in-advertising): <paragraph in / not in>
|
| 1283 |
+
- Final preflight: <PASS / FAIL+reason>
|
| 1284 |
+
- Cloud spend: $<x>
|
| 1285 |
+
- Commits: <list of commit short-shas>
|
| 1286 |
+
|
| 1287 |
+
Open risks for the user before submission:
|
| 1288 |
+
- ...
|
| 1289 |
+
```
|
| 1290 |
+
|
| 1291 |
+
Stop after that block. Do not push the brief itself; the user committed it.
|
SMOKE_RESULT.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
# Smoke Test Result - TRN-04
|
| 2 |
|
| 3 |
-
**VERDICT: GO** | Mode: quick | Elapsed:
|
| 4 |
|
| 5 |
|
| 6 |
| Check | Result |
|
|
@@ -13,7 +13,7 @@
|
|
| 13 |
| forward_pass | PASS |
|
| 14 |
|
| 15 |
- Model: unsloth/Qwen2.5-Coder-0.5B-Instruct-bnb-4bit
|
| 16 |
-
- Env URL:
|
| 17 |
|
| 18 |
## Phase 1 Exit Gate
|
| 19 |
PASS - Phase 2 training can proceed.
|
|
|
|
| 1 |
# Smoke Test Result - TRN-04
|
| 2 |
|
| 3 |
+
**VERDICT: GO** | Mode: quick | Elapsed: 182.0s
|
| 4 |
|
| 5 |
|
| 6 |
| Check | Result |
|
|
|
|
| 13 |
| forward_pass | PASS |
|
| 14 |
|
| 15 |
- Model: unsloth/Qwen2.5-Coder-0.5B-Instruct-bnb-4bit
|
| 16 |
+
- Env URL: http://localhost:8001
|
| 17 |
|
| 18 |
## Phase 1 Exit Gate
|
| 19 |
PASS - Phase 2 training can proceed.
|
scripts/deploy_demo_space.py
CHANGED
|
@@ -30,7 +30,7 @@ def deploy():
|
|
| 30 |
api.create_repo(
|
| 31 |
repo_id=SPACE_NAME,
|
| 32 |
repo_type="space",
|
| 33 |
-
space_sdk="
|
| 34 |
private=False,
|
| 35 |
exist_ok=True,
|
| 36 |
)
|
|
|
|
| 30 |
api.create_repo(
|
| 31 |
repo_id=SPACE_NAME,
|
| 32 |
repo_type="space",
|
| 33 |
+
space_sdk="docker",
|
| 34 |
private=False,
|
| 35 |
exist_ok=True,
|
| 36 |
)
|