Download methods.json from world-model/world-model-training-evaluation-lab: direct link, hf CLI and curl.
- Browser
- Download file 5.84 kB
-
https://huggingface.co/spaces/world-model/world-model-training-evaluation-lab/resolve/main/methods.json
- Command line
-
hf download hf://spaces/world-model/world-model-training-evaluation-lab/methods.json
-
curl -L -o methods.json https://huggingface.co/spaces/world-model/world-model-training-evaluation-lab/resolve/main/methods.json
5.84 kB
| [ | |
| { | |
| "name": "Multi-Step Prediction Loss", | |
| "phase": "Training", | |
| "category": "Long-Horizon Learning", | |
| "difficulty": "Advanced", | |
| "goal": "Reduce compounding rollout error", | |
| "summary": "Train the world model over several imagined future steps instead of only a one-step target.", | |
| "signals": [ | |
| "multi-step loss", | |
| "rollout consistency", | |
| "horizon weighting" | |
| ], | |
| "when": "Use when one-step predictions look strong but longer rollouts drift quickly." | |
| }, | |
| { | |
| "name": "Latent Overshooting", | |
| "phase": "Training", | |
| "category": "Latent Dynamics", | |
| "difficulty": "Advanced", | |
| "goal": "Stabilize latent dynamics across multiple steps", | |
| "summary": "Compare predicted latent states against future encoded targets several steps ahead.", | |
| "signals": [ | |
| "latent loss", | |
| "overshooting horizon", | |
| "representation stability" | |
| ], | |
| "when": "Useful for latent-state models and model-based reinforcement learning." | |
| }, | |
| { | |
| "name": "Scheduled Sampling", | |
| "phase": "Training", | |
| "category": "Exposure Bias", | |
| "difficulty": "Intermediate", | |
| "goal": "Reduce train-deployment mismatch", | |
| "summary": "Gradually train on model-generated states instead of always conditioning on ground truth.", | |
| "signals": [ | |
| "teacher forcing ratio", | |
| "self-conditioned steps" | |
| ], | |
| "when": "Useful when rollout behavior degrades because the model never trained on its own errors." | |
| }, | |
| { | |
| "name": "Action-Conditioned Consistency", | |
| "phase": "Training", | |
| "category": "Control", | |
| "difficulty": "Advanced", | |
| "goal": "Preserve causal action effects", | |
| "summary": "Encourage different actions to produce meaningfully different predicted futures.", | |
| "signals": [ | |
| "action sensitivity", | |
| "counterfactual consistency" | |
| ], | |
| "when": "Important for robotics, agents and interactive world models." | |
| }, | |
| { | |
| "name": "Joint Reconstruction + Prediction", | |
| "phase": "Training", | |
| "category": "Representation Learning", | |
| "difficulty": "Intermediate", | |
| "goal": "Balance state fidelity and predictive usefulness", | |
| "summary": "Combine reconstruction with future-state objectives so the latent representation remains informative and predictive.", | |
| "signals": [ | |
| "reconstruction loss", | |
| "prediction loss", | |
| "latent regularization" | |
| ], | |
| "when": "Useful when the representation becomes too compressed or too visually focused." | |
| }, | |
| { | |
| "name": "Rollout Error by Horizon", | |
| "phase": "Evaluation", | |
| "category": "Long-Horizon Evaluation", | |
| "difficulty": "Core", | |
| "goal": "Measure error accumulation", | |
| "summary": "Evaluate prediction error at each future horizon rather than reporting only an average.", | |
| "signals": [ | |
| "error@1", | |
| "error@5", | |
| "error@10", | |
| "drift curve" | |
| ], | |
| "when": "Essential for any model used beyond one-step forecasting." | |
| }, | |
| { | |
| "name": "Action Fidelity Test", | |
| "phase": "Evaluation", | |
| "category": "Control", | |
| "difficulty": "Advanced", | |
| "goal": "Verify that action changes are reflected in predicted futures", | |
| "summary": "Compare predicted consequences under different actions and validate against real transitions.", | |
| "signals": [ | |
| "action effect error", | |
| "counterfactual separation" | |
| ], | |
| "when": "Critical for robotics, control and interactive simulation." | |
| }, | |
| { | |
| "name": "Uncertainty Calibration", | |
| "phase": "Evaluation", | |
| "category": "Reliability", | |
| "difficulty": "Advanced", | |
| "goal": "Measure whether confidence tracks prediction quality", | |
| "summary": "Evaluate if uncertain predictions receive lower confidence than reliable predictions.", | |
| "signals": [ | |
| "calibration error", | |
| "coverage", | |
| "ensemble variance" | |
| ], | |
| "when": "Important for safety-sensitive or out-of-distribution settings." | |
| }, | |
| { | |
| "name": "Distribution Shift Evaluation", | |
| "phase": "Evaluation", | |
| "category": "Generalization", | |
| "difficulty": "Advanced", | |
| "goal": "Measure robustness beyond familiar training conditions", | |
| "summary": "Test new environments, dynamics, objects, actions or visual conditions not seen during training.", | |
| "signals": [ | |
| "OOD error", | |
| "relative degradation", | |
| "recovery" | |
| ], | |
| "when": "Necessary before deployment into diverse real environments." | |
| }, | |
| { | |
| "name": "Planning Utility", | |
| "phase": "Evaluation", | |
| "category": "Decision Making", | |
| "difficulty": "Advanced", | |
| "goal": "Measure whether the world model improves decisions", | |
| "summary": "Evaluate downstream planning or control performance using the learned world model.", | |
| "signals": [ | |
| "task success", | |
| "return", | |
| "planning regret", | |
| "sample efficiency" | |
| ], | |
| "when": "The strongest evaluation when the model is intended for agents or control." | |
| }, | |
| { | |
| "name": "Compute & Rollout Efficiency", | |
| "phase": "Evaluation", | |
| "category": "Efficiency", | |
| "difficulty": "Core", | |
| "goal": "Quantify operational cost", | |
| "summary": "Measure latency, memory and throughput for one-step and multi-step simulation.", | |
| "signals": [ | |
| "ms/step", | |
| "tokens/s", | |
| "VRAM", | |
| "rollouts/s" | |
| ], | |
| "when": "Important for real-time planning and large candidate-search spaces." | |
| }, | |
| { | |
| "name": "Model Error vs. Task Error", | |
| "phase": "Evaluation", | |
| "category": "Task Relevance", | |
| "difficulty": "Advanced", | |
| "goal": "Separate prediction quality from downstream usefulness", | |
| "summary": "Compare predictive error with the actual effect on planning, control or task success.", | |
| "signals": [ | |
| "prediction loss", | |
| "task success", | |
| "control regret" | |
| ], | |
| "when": "Useful when low reconstruction loss does not translate into better decisions." | |
| } | |
| ] |