Spaces:
Sleeping
Sleeping
hschinmaybharadwaj
feat(submission): finalize hackathon deliverables β 24-turn 6/6 eval, hybrid policy, gitignored summary
565ba48 Download scripts/evaluate_and_plot.py from hsbharadwaj/ev: direct link, hf CLI and curl.
- Browser
- Download file 14 kB
-
https://huggingface.co/spaces/hsbharadwaj/ev/resolve/main/scripts/evaluate_and_plot.py
- Command line
-
hf download hf://spaces/hsbharadwaj/ev/scripts/evaluate_and_plot.py
-
curl -L -o evaluate_and_plot.py https://huggingface.co/spaces/hsbharadwaj/ev/resolve/main/scripts/evaluate_and_plot.py
14 kB
| """Evaluate Random / Heuristic / Expert policies against the Neon Syndicate | |
| environment and write judge-friendly plots and metrics. | |
| Why three policies on the same axes: | |
| * ``random`` is the absolute floor β it's what an untrained policy looks | |
| like. Reward curves should sit near 0 and success should be ~0%. | |
| * ``heuristic`` is the curriculum-style fallback baked into the codebase. | |
| It approximates a *partially trained* agent: it can solve the easy | |
| missions but is brittle on hard ones. | |
| * ``expert`` is the target-aware oracle β it represents the converged | |
| trained policy. After PPO has done its job, the trained model should | |
| match (or approach) this curve. ``expert`` reaches **6/6 success** on | |
| every bundled mission. | |
| The script runs in-process (no HTTP, no Space hop) so judges can | |
| reproduce the plots locally in seconds. It writes: | |
| * ``artifacts/eval_metrics.jsonl`` per-episode metrics (one JSON per line) | |
| * ``artifacts/results_summary.json`` averaged scoreboard | |
| * ``artifacts/reward_curves.png`` cumulative-reward curves, three lanes | |
| * ``artifacts/loss_curve.png`` per-episode (1 - normalised score) "loss" | |
| * ``artifacts/expert_baseline.jsonl`` expert-oracle trajectory (target reference) | |
| The trained-policy ``training_summary.jsonl`` is intentionally NOT | |
| written by this script -- it is produced when you execute the Colab | |
| notebook (``notebooks/trl_training_colab.ipynb``). | |
| If matplotlib isn't installed (smoke-tests, CI), the script still writes | |
| the JSONL so downstream pipelines stay green; PNGs are skipped. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import random | |
| import sys | |
| import os | |
| from dataclasses import asdict, dataclass | |
| from pathlib import Path | |
| from typing import Any, Dict, List, Optional | |
| # Allow running from repo root without installing the package. | |
| _REPO_ROOT = Path(__file__).resolve().parent.parent | |
| if str(_REPO_ROOT) not in sys.path: | |
| sys.path.insert(0, str(_REPO_ROOT)) | |
| # pylint: disable=wrong-import-position | |
| from models import Action # noqa: E402 | |
| from server.agent import ( # noqa: E402 | |
| expert_action, | |
| get_task_target, | |
| heuristic_action, | |
| random_action, | |
| ) | |
| from server.environment import NeonSyndicateEnvironment # noqa: E402 | |
| class EpisodeMetric: | |
| policy: str | |
| episode: int | |
| task_id: str | |
| difficulty: str | |
| total_reward: float | |
| final_task_score: float | |
| success: bool | |
| steps: int | |
| def _policy_act(policy: str, obs: Dict[str, Any], target: Any) -> Dict[str, Any]: | |
| if policy == "random": | |
| return random_action(obs) | |
| if policy == "heuristic": | |
| return heuristic_action(obs) | |
| if policy == "expert": | |
| return expert_action(obs, target) | |
| raise ValueError(f"Unknown policy: {policy}") | |
| def run_policy(policy: str, episodes: int, max_steps: int, seed: int) -> List[EpisodeMetric]: | |
| rng = random.Random(seed) | |
| metrics: List[EpisodeMetric] = [] | |
| env = NeonSyndicateEnvironment() | |
| task_order = list(env._task_order) # noqa: SLF001 -- intentional read | |
| for ep in range(episodes): | |
| task_id = task_order[ep % len(task_order)] | |
| obs = env.reset(task_id=task_id) | |
| target = env._state.active_task.target # noqa: SLF001 -- needed for expert | |
| total_reward = 0.0 | |
| final_task_score = 0.0 | |
| success = False | |
| steps = 0 | |
| for t in range(max_steps): | |
| obs_dict = obs.model_dump() | |
| try: | |
| action_dict = _policy_act(policy, obs_dict, target) | |
| except Exception: | |
| action_dict = {"action_type": "noop"} | |
| try: | |
| action = Action(**action_dict) | |
| except Exception: | |
| action = Action(action_type="noop") | |
| response = env.step(action) | |
| total_reward += float(response.reward.score) | |
| steps = t + 1 | |
| if response.done: | |
| info = response.info or {} | |
| final_task_score = float(info.get("task_score", 0.0)) | |
| success = bool(info.get("success", False)) | |
| break | |
| obs = response.observation | |
| metrics.append( | |
| EpisodeMetric( | |
| policy=policy, | |
| episode=ep, | |
| task_id=task_id, | |
| difficulty=env._tasks[task_id].difficulty, # noqa: SLF001 | |
| total_reward=total_reward, | |
| final_task_score=final_task_score, | |
| success=success, | |
| steps=steps, | |
| ) | |
| ) | |
| # Re-seed RNG between episodes so random policy is reproducible. | |
| rng.random() | |
| return metrics | |
| def save_metrics(metrics: List[EpisodeMetric], output_jsonl: Path) -> None: | |
| output_jsonl.parent.mkdir(parents=True, exist_ok=True) | |
| with output_jsonl.open("w", encoding="utf-8") as fh: | |
| for row in metrics: | |
| fh.write(json.dumps(asdict(row)) + "\n") | |
| def summarize(metrics: List[EpisodeMetric]) -> Dict[str, Dict[str, float]]: | |
| by_policy: Dict[str, List[EpisodeMetric]] = {} | |
| for m in metrics: | |
| by_policy.setdefault(m.policy, []).append(m) | |
| summary: Dict[str, Dict[str, float]] = {} | |
| for policy, rows in by_policy.items(): | |
| n = max(1, len(rows)) | |
| summary[policy] = { | |
| "episodes": float(len(rows)), | |
| "avg_total_reward": sum(r.total_reward for r in rows) / n, | |
| "avg_task_score": sum(r.final_task_score for r in rows) / n, | |
| "success_rate": sum(1.0 for r in rows if r.success) / n, | |
| "successes": float(sum(1 for r in rows if r.success)), | |
| "avg_steps": sum(r.steps for r in rows) / n, | |
| } | |
| return summary | |
| def save_summary(summary: Dict[str, Dict[str, float]], path: Path) -> None: | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| path.write_text(json.dumps(summary, indent=2) + "\n", encoding="utf-8") | |
| def _try_import_pyplot(): | |
| try: | |
| os.environ.setdefault("MPLCONFIGDIR", "/tmp/mpl") | |
| import matplotlib | |
| matplotlib.use("Agg") | |
| import matplotlib.pyplot as plt | |
| return plt | |
| except Exception: | |
| return None | |
| _POLICY_STYLES = { | |
| "random": {"color": "#9ca3af", "marker": "o", "linestyle": ":"}, | |
| "heuristic": {"color": "#f59e0b", "marker": "s", "linestyle": "--"}, | |
| "expert": {"color": "#22d3ee", "marker": "*", "linestyle": "-"}, | |
| } | |
| def plot_reward_curves(metrics: List[EpisodeMetric], output_png: Path) -> bool: | |
| plt = _try_import_pyplot() | |
| if plt is None: | |
| return False | |
| output_png.parent.mkdir(parents=True, exist_ok=True) | |
| by_policy: Dict[str, List[EpisodeMetric]] = {} | |
| for m in metrics: | |
| by_policy.setdefault(m.policy, []).append(m) | |
| fig, ax = plt.subplots(figsize=(11, 6)) | |
| fig.patch.set_facecolor("#0b1020") | |
| ax.set_facecolor("#0b1020") | |
| for policy in ("random", "heuristic", "expert"): | |
| rows = sorted(by_policy.get(policy, []), key=lambda r: r.episode) | |
| if not rows: | |
| continue | |
| xs = [r.episode + 1 for r in rows] | |
| ys = [r.total_reward for r in rows] | |
| style = _POLICY_STYLES.get(policy, {}) | |
| ax.plot( | |
| xs, | |
| ys, | |
| label=f"{policy} (avg={sum(ys)/len(ys):.2f}, " | |
| f"successes={sum(1 for r in rows if r.success)}/{len(rows)})", | |
| linewidth=2.4, | |
| **style, | |
| ) | |
| ax.set_title( | |
| "Neon Syndicate β Per-Episode Total Reward by Policy\n" | |
| "Random < Heuristic < Expert (target the trained agent should match)", | |
| color="#f8fafc", | |
| fontsize=14, | |
| ) | |
| ax.set_xlabel("Episode (cycling through 6 missions, easyβhard)", color="#cbd5e1") | |
| ax.set_ylabel("Total reward (sum of dense per-step rewards, max β 1.0)", color="#cbd5e1") | |
| ax.tick_params(colors="#cbd5e1") | |
| ax.grid(alpha=0.2, color="#475569") | |
| ax.legend(facecolor="#0f172a", edgecolor="#334155", labelcolor="#f8fafc", loc="lower right") | |
| ax.set_ylim(bottom=0.0) | |
| fig.tight_layout() | |
| fig.savefig(output_png, dpi=160, facecolor=fig.get_facecolor()) | |
| plt.close(fig) | |
| return True | |
| def plot_loss_curve(metrics: List[EpisodeMetric], output_png: Path) -> bool: | |
| """Plot per-policy "loss" = 1 - task_score (lower is better). | |
| For RL this isn't a true cross-entropy loss but it gives reviewers a | |
| monotonically decreasing trace as the policy improves, on the same x | |
| axis as the reward curve. | |
| """ | |
| plt = _try_import_pyplot() | |
| if plt is None: | |
| return False | |
| output_png.parent.mkdir(parents=True, exist_ok=True) | |
| by_policy: Dict[str, List[EpisodeMetric]] = {} | |
| for m in metrics: | |
| by_policy.setdefault(m.policy, []).append(m) | |
| fig, ax = plt.subplots(figsize=(11, 6)) | |
| fig.patch.set_facecolor("#0b1020") | |
| ax.set_facecolor("#0b1020") | |
| for policy in ("random", "heuristic", "expert"): | |
| rows = sorted(by_policy.get(policy, []), key=lambda r: r.episode) | |
| if not rows: | |
| continue | |
| xs = [r.episode + 1 for r in rows] | |
| ys = [max(0.0, 1.0 - r.final_task_score) for r in rows] | |
| # Running mean smooths the per-task variance. | |
| window = 3 | |
| smoothed: List[float] = [] | |
| for i in range(len(ys)): | |
| lo = max(0, i - window + 1) | |
| smoothed.append(sum(ys[lo : i + 1]) / (i - lo + 1)) | |
| style = _POLICY_STYLES.get(policy, {}) | |
| ax.plot( | |
| xs, | |
| smoothed, | |
| label=f"{policy} (mean={sum(ys)/len(ys):.3f})", | |
| linewidth=2.4, | |
| **style, | |
| ) | |
| ax.set_title( | |
| "Neon Syndicate β Episode Loss (1 β task_score, smoothed)\n" | |
| "Lower is better. Expert β 0 means the policy clears every gate.", | |
| color="#f8fafc", | |
| fontsize=14, | |
| ) | |
| ax.set_xlabel("Episode", color="#cbd5e1") | |
| ax.set_ylabel("1 β task_score (rolling mean, window = 3)", color="#cbd5e1") | |
| ax.tick_params(colors="#cbd5e1") | |
| ax.grid(alpha=0.2, color="#475569") | |
| ax.legend(facecolor="#0f172a", edgecolor="#334155", labelcolor="#f8fafc", loc="upper right") | |
| ax.set_ylim(0.0, 1.05) | |
| fig.tight_layout() | |
| fig.savefig(output_png, dpi=160, facecolor=fig.get_facecolor()) | |
| plt.close(fig) | |
| return True | |
| def write_expert_baseline(metrics: List[EpisodeMetric], path: Path) -> None: | |
| """Persist the *expert oracle* trajectory as a separate baseline artifact. | |
| This is **not** ``training_summary.jsonl`` β that file is owned by the | |
| Colab notebook (and ``scripts/run_notebook_eval.py``) so it always | |
| reflects the trained-runtime stack the HF Space serves. The expert | |
| baseline below is what the trained agent should approach; it lives at | |
| ``artifacts/expert_baseline.jsonl`` and is referenced by the README | |
| results table. | |
| """ | |
| expert_rows = [m for m in metrics if m.policy == "expert"] | |
| payload_lines = [ | |
| json.dumps( | |
| { | |
| "episode_id": r.episode, | |
| "task_id": r.task_id, | |
| "difficulty": r.difficulty, | |
| "total_reward": r.total_reward, | |
| "final_task_score": r.final_task_score, | |
| "success": r.success, | |
| "steps": r.steps, | |
| "policy": "expert", | |
| } | |
| ) | |
| for r in expert_rows | |
| ] | |
| body = "\n".join(payload_lines) + ("\n" if payload_lines else "") | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| path.write_text(body, encoding="utf-8") | |
| def parse_args() -> argparse.Namespace: | |
| parser = argparse.ArgumentParser( | |
| description="Evaluate Random/Heuristic/Expert policies and emit reward+loss plots." | |
| ) | |
| parser.add_argument("--episodes", type=int, default=18, help="Episodes per policy (cycles tasks).") | |
| parser.add_argument("--max-steps", type=int, default=24) | |
| parser.add_argument("--output-dir", type=str, default="artifacts") | |
| parser.add_argument("--seed", type=int, default=42) | |
| return parser.parse_args() | |
| def main() -> None: | |
| args = parse_args() | |
| out_dir = Path(args.output_dir) | |
| print(f"[EVAL] running random / heuristic / expert episodes={args.episodes}") | |
| all_metrics: List[EpisodeMetric] = [] | |
| for policy in ("random", "heuristic", "expert"): | |
| metrics = run_policy(policy, args.episodes, args.max_steps, args.seed) | |
| all_metrics.extend(metrics) | |
| save_metrics(all_metrics, out_dir / "eval_metrics.jsonl") | |
| summary = summarize(all_metrics) | |
| save_summary(summary, out_dir / "results_summary.json") | |
| for policy, stats in summary.items(): | |
| print( | |
| f"[EVAL] policy={policy:<9} " | |
| f"episodes={int(stats['episodes']):>3} " | |
| f"avg_reward={stats['avg_total_reward']:.3f} " | |
| f"avg_task_score={stats['avg_task_score']:.3f} " | |
| f"success_rate={stats['success_rate']:.3f} " | |
| f"avg_steps={stats['avg_steps']:.1f}" | |
| ) | |
| reward_png = out_dir / "reward_curves.png" | |
| loss_png = out_dir / "loss_curve.png" | |
| if plot_reward_curves(all_metrics, reward_png): | |
| print(f"[ARTIFACT] {reward_png}") | |
| else: | |
| print("[WARN] matplotlib unavailable β skipped reward_curves.png") | |
| if plot_loss_curve(all_metrics, loss_png): | |
| print(f"[ARTIFACT] {loss_png}") | |
| else: | |
| print("[WARN] matplotlib unavailable β skipped loss_curve.png") | |
| # The expert oracle trajectory is the *target* the trained agent should | |
| # approach; persist it as a side-by-side baseline artifact. Note: the | |
| # canonical ``training_summary.jsonl`` is intentionally NOT written here | |
| # -- it is owned by the Colab notebook / scripts/run_notebook_eval.py | |
| # so it always reflects the trained-runtime stack the HF Space serves. | |
| write_expert_baseline(all_metrics, out_dir / "expert_baseline.jsonl") | |
| print(f"[ARTIFACT] {out_dir / 'expert_baseline.jsonl'}") | |
| print(f"[ARTIFACT] {out_dir / 'eval_metrics.jsonl'}") | |
| print(f"[ARTIFACT] {out_dir / 'results_summary.json'}") | |
| if __name__ == "__main__": | |
| main() | |