"""CPU-only verification of REW-04 v2 reward design.
Confirms two GRPO-blocking properties:
1. A synthetic 8-completion group produces non-zero std (v1's std was 0.0
across every group, which is why the reward curve was flat).
2. The ordering correct+0calls > correct+1call > correct+spam holds.
Run BEFORE spending any HF Jobs credits on a retrain.
"""
from __future__ import annotations
import statistics
import types
from rewards.compose import compose_reward_single
cfg = types.SimpleNamespace(
alpha=0.2,
weights=types.SimpleNamespace(
correctness=0.70, token_budget=0.15, recursion_efficiency=0.15
),
token_budget_variant="capped_linear",
answer_regex="(.*?)",
max_calls=4,
)
GOLD = "silver"
GENERATIONS = [
("correct + 0 llm calls (REPL grep)",
"```python\nimport re\nm=re.search('silver', ctx)\nprint(m.group())\n```\nsilver"),
("correct + 1 llm call",
"```python\nans=llm('color', ctx[:5000])\nprint(ans)\n```\nsilver"),
("correct + 3 llm calls (wasteful)",
"```python\na=llm('q1',ctx[:1000])\nb=llm('q2',ctx[1000:2000])\nc=llm('q3',ctx[2000:3000])\n```\nsilver"),
("correct + bare answer (no code, trivial-task path)",
"silver"),
("wrong + format",
"gold"),
("wrong + no format",
"the color is gold"),
("right text + no format (v1 collapse mode)",
"silver"),
("format-only spam",
""),
]
print(f"{'idx':>3} {'score':>6} {'calls':>5} description")
print("-" * 78)
scores = []
for i, (desc, gen) in enumerate(GENERATIONS):
s, m = compose_reward_single(gen, GOLD, 200, cfg)
scores.append(s)
print(f"{i:>3} {s:>6.3f} {int(m['llm_call_count']):>5d} {desc}")
print("-" * 78)
print(f"group mean: {statistics.mean(scores):.4f}")
print(f"group std: {statistics.stdev(scores):.4f} (must be > 0.10 for GRPO advantage)")
print(f"max - min: {max(scores) - min(scores):.4f}")
# Hard gates — exit non-zero if any fail
assert statistics.stdev(scores) > 0.10, "FAIL: group std too low; GRPO will not learn"
assert scores[0] > scores[1] > scores[2], (
f"FAIL: efficiency ordering broken (got {scores[0]:.3f} > {scores[1]:.3f} > {scores[2]:.3f})"
)
assert scores[0] > scores[4], "FAIL: correct must beat wrong"
assert scores[7] <= 0.25, "FAIL: format-only spam not capped"
print("\nPASS: REW-04 v2 produces learnable variance and correct orderings")