[ { "name": "Transition Dataset", "type": "Dataset Pattern", "focus": "One-step dynamics", "best_for": [ "Dynamics learning", "Control", "Simple environments" ], "required": [ "observation_t", "action_t", "observation_t+1" ], "optional": [ "reward", "done", "goal", "timestamp" ], "risk": "Frame-level random splits can leak nearly identical states across train and test." }, { "name": "Episode Dataset", "type": "Dataset Pattern", "focus": "Sequential dynamics", "best_for": [ "Rollouts", "Memory", "Long-horizon modeling" ], "required": [ "episode_id", "observations[]", "actions[]" ], "optional": [ "rewards[]", "terminal", "instruction", "metadata" ], "risk": "Broken episode boundaries can destroy temporal structure." }, { "name": "Multimodal Episode", "type": "Dataset Pattern", "focus": "Cross-modal world state", "best_for": [ "Robotics", "Embodied AI", "Physical AI" ], "required": [ "video", "actions", "robot_state" ], "optional": [ "depth", "audio", "language", "force", "pose" ], "risk": "Unsynchronized modalities create false transition errors." }, { "name": "One-Step Benchmark", "type": "Benchmark Pattern", "focus": "Immediate next-state prediction", "best_for": [ "Fast iteration", "Architecture debugging" ], "required": [ "held-out transitions", "prediction metric" ], "optional": [ "uncertainty", "per-task breakdown" ], "risk": "Can hide catastrophic long-horizon drift." }, { "name": "Multi-Horizon Benchmark", "type": "Benchmark Pattern", "focus": "Error growth over time", "best_for": [ "Rollouts", "Planning", "Simulation" ], "required": [ "horizons", "rollout protocol", "per-horizon metric" ], "optional": [ "drift ratio", "consistency score" ], "risk": "Averaging across horizons can conceal where failure begins." }, { "name": "Action Fidelity Benchmark", "type": "Benchmark Pattern", "focus": "Correct action consequences", "best_for": [ "Robotics", "Interactive worlds", "Agents" ], "required": [ "paired actions", "ground-truth transitions" ], "optional": [ "counterfactual actions", "action sensitivity" ], "risk": "Visual similarity alone does not prove correct causal response." }, { "name": "OOD Generalization Split", "type": "Split Strategy", "focus": "Distribution shift", "best_for": [ "Deployment", "Robotics", "Robustness" ], "required": [ "held-out environments or tasks" ], "optional": [ "held-out objects", "held-out embodiments" ], "risk": "Random splits can dramatically overestimate generalization." }, { "name": "Planning Utility Benchmark", "type": "Benchmark Pattern", "focus": "Downstream decision quality", "best_for": [ "Agents", "MPC", "Model-based RL" ], "required": [ "planner", "task success metric", "world-model rollouts" ], "optional": [ "regret", "return", "sample efficiency" ], "risk": "Prediction scores may not correlate with better decisions." }, { "name": "Control Benchmark", "type": "Benchmark Pattern", "focus": "Closed-loop task performance", "best_for": [ "Robotics", "Autonomous systems" ], "required": [ "environment", "policy/controller", "success metric" ], "optional": [ "safety violations", "energy", "path efficiency" ], "risk": "Offline prediction quality may not transfer to closed-loop control." }, { "name": "Efficiency Benchmark", "type": "Benchmark Pattern", "focus": "Operational cost", "best_for": [ "Real-time systems", "Large search spaces" ], "required": [ "latency", "memory", "throughput" ], "optional": [ "energy", "rollouts/sec", "cost" ], "risk": "A strong model may be unusable if rollouts are too slow for planning." } ]