[ { "name": "Multi-Step Prediction Loss", "phase": "Training", "category": "Long-Horizon Learning", "difficulty": "Advanced", "goal": "Reduce compounding rollout error", "summary": "Train the world model over several imagined future steps instead of only a one-step target.", "signals": [ "multi-step loss", "rollout consistency", "horizon weighting" ], "when": "Use when one-step predictions look strong but longer rollouts drift quickly." }, { "name": "Latent Overshooting", "phase": "Training", "category": "Latent Dynamics", "difficulty": "Advanced", "goal": "Stabilize latent dynamics across multiple steps", "summary": "Compare predicted latent states against future encoded targets several steps ahead.", "signals": [ "latent loss", "overshooting horizon", "representation stability" ], "when": "Useful for latent-state models and model-based reinforcement learning." }, { "name": "Scheduled Sampling", "phase": "Training", "category": "Exposure Bias", "difficulty": "Intermediate", "goal": "Reduce train-deployment mismatch", "summary": "Gradually train on model-generated states instead of always conditioning on ground truth.", "signals": [ "teacher forcing ratio", "self-conditioned steps" ], "when": "Useful when rollout behavior degrades because the model never trained on its own errors." }, { "name": "Action-Conditioned Consistency", "phase": "Training", "category": "Control", "difficulty": "Advanced", "goal": "Preserve causal action effects", "summary": "Encourage different actions to produce meaningfully different predicted futures.", "signals": [ "action sensitivity", "counterfactual consistency" ], "when": "Important for robotics, agents and interactive world models." }, { "name": "Joint Reconstruction + Prediction", "phase": "Training", "category": "Representation Learning", "difficulty": "Intermediate", "goal": "Balance state fidelity and predictive usefulness", "summary": "Combine reconstruction with future-state objectives so the latent representation remains informative and predictive.", "signals": [ "reconstruction loss", "prediction loss", "latent regularization" ], "when": "Useful when the representation becomes too compressed or too visually focused." }, { "name": "Rollout Error by Horizon", "phase": "Evaluation", "category": "Long-Horizon Evaluation", "difficulty": "Core", "goal": "Measure error accumulation", "summary": "Evaluate prediction error at each future horizon rather than reporting only an average.", "signals": [ "error@1", "error@5", "error@10", "drift curve" ], "when": "Essential for any model used beyond one-step forecasting." }, { "name": "Action Fidelity Test", "phase": "Evaluation", "category": "Control", "difficulty": "Advanced", "goal": "Verify that action changes are reflected in predicted futures", "summary": "Compare predicted consequences under different actions and validate against real transitions.", "signals": [ "action effect error", "counterfactual separation" ], "when": "Critical for robotics, control and interactive simulation." }, { "name": "Uncertainty Calibration", "phase": "Evaluation", "category": "Reliability", "difficulty": "Advanced", "goal": "Measure whether confidence tracks prediction quality", "summary": "Evaluate if uncertain predictions receive lower confidence than reliable predictions.", "signals": [ "calibration error", "coverage", "ensemble variance" ], "when": "Important for safety-sensitive or out-of-distribution settings." }, { "name": "Distribution Shift Evaluation", "phase": "Evaluation", "category": "Generalization", "difficulty": "Advanced", "goal": "Measure robustness beyond familiar training conditions", "summary": "Test new environments, dynamics, objects, actions or visual conditions not seen during training.", "signals": [ "OOD error", "relative degradation", "recovery" ], "when": "Necessary before deployment into diverse real environments." }, { "name": "Planning Utility", "phase": "Evaluation", "category": "Decision Making", "difficulty": "Advanced", "goal": "Measure whether the world model improves decisions", "summary": "Evaluate downstream planning or control performance using the learned world model.", "signals": [ "task success", "return", "planning regret", "sample efficiency" ], "when": "The strongest evaluation when the model is intended for agents or control." }, { "name": "Compute & Rollout Efficiency", "phase": "Evaluation", "category": "Efficiency", "difficulty": "Core", "goal": "Quantify operational cost", "summary": "Measure latency, memory and throughput for one-step and multi-step simulation.", "signals": [ "ms/step", "tokens/s", "VRAM", "rollouts/s" ], "when": "Important for real-time planning and large candidate-search spaces." }, { "name": "Model Error vs. Task Error", "phase": "Evaluation", "category": "Task Relevance", "difficulty": "Advanced", "goal": "Separate prediction quality from downstream usefulness", "summary": "Compare predictive error with the actual effect on planning, control or task success.", "signals": [ "prediction loss", "task success", "control regret" ], "when": "Useful when low reconstruction loss does not translate into better decisions." } ]