Agenten's picture
Upload 8 files
8105265 verified
Raw History Blame Contribute Delete
5.84 kB
[
{
"name": "Multi-Step Prediction Loss",
"phase": "Training",
"category": "Long-Horizon Learning",
"difficulty": "Advanced",
"goal": "Reduce compounding rollout error",
"summary": "Train the world model over several imagined future steps instead of only a one-step target.",
"signals": [
"multi-step loss",
"rollout consistency",
"horizon weighting"
],
"when": "Use when one-step predictions look strong but longer rollouts drift quickly."
},
{
"name": "Latent Overshooting",
"phase": "Training",
"category": "Latent Dynamics",
"difficulty": "Advanced",
"goal": "Stabilize latent dynamics across multiple steps",
"summary": "Compare predicted latent states against future encoded targets several steps ahead.",
"signals": [
"latent loss",
"overshooting horizon",
"representation stability"
],
"when": "Useful for latent-state models and model-based reinforcement learning."
},
{
"name": "Scheduled Sampling",
"phase": "Training",
"category": "Exposure Bias",
"difficulty": "Intermediate",
"goal": "Reduce train-deployment mismatch",
"summary": "Gradually train on model-generated states instead of always conditioning on ground truth.",
"signals": [
"teacher forcing ratio",
"self-conditioned steps"
],
"when": "Useful when rollout behavior degrades because the model never trained on its own errors."
},
{
"name": "Action-Conditioned Consistency",
"phase": "Training",
"category": "Control",
"difficulty": "Advanced",
"goal": "Preserve causal action effects",
"summary": "Encourage different actions to produce meaningfully different predicted futures.",
"signals": [
"action sensitivity",
"counterfactual consistency"
],
"when": "Important for robotics, agents and interactive world models."
},
{
"name": "Joint Reconstruction + Prediction",
"phase": "Training",
"category": "Representation Learning",
"difficulty": "Intermediate",
"goal": "Balance state fidelity and predictive usefulness",
"summary": "Combine reconstruction with future-state objectives so the latent representation remains informative and predictive.",
"signals": [
"reconstruction loss",
"prediction loss",
"latent regularization"
],
"when": "Useful when the representation becomes too compressed or too visually focused."
},
{
"name": "Rollout Error by Horizon",
"phase": "Evaluation",
"category": "Long-Horizon Evaluation",
"difficulty": "Core",
"goal": "Measure error accumulation",
"summary": "Evaluate prediction error at each future horizon rather than reporting only an average.",
"signals": [
"error@1",
"error@5",
"error@10",
"drift curve"
],
"when": "Essential for any model used beyond one-step forecasting."
},
{
"name": "Action Fidelity Test",
"phase": "Evaluation",
"category": "Control",
"difficulty": "Advanced",
"goal": "Verify that action changes are reflected in predicted futures",
"summary": "Compare predicted consequences under different actions and validate against real transitions.",
"signals": [
"action effect error",
"counterfactual separation"
],
"when": "Critical for robotics, control and interactive simulation."
},
{
"name": "Uncertainty Calibration",
"phase": "Evaluation",
"category": "Reliability",
"difficulty": "Advanced",
"goal": "Measure whether confidence tracks prediction quality",
"summary": "Evaluate if uncertain predictions receive lower confidence than reliable predictions.",
"signals": [
"calibration error",
"coverage",
"ensemble variance"
],
"when": "Important for safety-sensitive or out-of-distribution settings."
},
{
"name": "Distribution Shift Evaluation",
"phase": "Evaluation",
"category": "Generalization",
"difficulty": "Advanced",
"goal": "Measure robustness beyond familiar training conditions",
"summary": "Test new environments, dynamics, objects, actions or visual conditions not seen during training.",
"signals": [
"OOD error",
"relative degradation",
"recovery"
],
"when": "Necessary before deployment into diverse real environments."
},
{
"name": "Planning Utility",
"phase": "Evaluation",
"category": "Decision Making",
"difficulty": "Advanced",
"goal": "Measure whether the world model improves decisions",
"summary": "Evaluate downstream planning or control performance using the learned world model.",
"signals": [
"task success",
"return",
"planning regret",
"sample efficiency"
],
"when": "The strongest evaluation when the model is intended for agents or control."
},
{
"name": "Compute & Rollout Efficiency",
"phase": "Evaluation",
"category": "Efficiency",
"difficulty": "Core",
"goal": "Quantify operational cost",
"summary": "Measure latency, memory and throughput for one-step and multi-step simulation.",
"signals": [
"ms/step",
"tokens/s",
"VRAM",
"rollouts/s"
],
"when": "Important for real-time planning and large candidate-search spaces."
},
{
"name": "Model Error vs. Task Error",
"phase": "Evaluation",
"category": "Task Relevance",
"difficulty": "Advanced",
"goal": "Separate prediction quality from downstream usefulness",
"summary": "Compare predictive error with the actual effect on planning, control or task success.",
"signals": [
"prediction loss",
"task success",
"control regret"
],
"when": "Useful when low reconstruction loss does not translate into better decisions."
}
]