Download patterns.json from world-model/world-model-data-benchmark-lab: direct link, hf CLI and curl.
- Browser
- Download file 4.46 kB
-
https://huggingface.co/spaces/world-model/world-model-data-benchmark-lab/resolve/main/patterns.json
- Command line
-
hf download hf://spaces/world-model/world-model-data-benchmark-lab/patterns.json
-
curl -L -o patterns.json https://huggingface.co/spaces/world-model/world-model-data-benchmark-lab/resolve/main/patterns.json
4.46 kB
| [ | |
| { | |
| "name": "Transition Dataset", | |
| "type": "Dataset Pattern", | |
| "focus": "One-step dynamics", | |
| "best_for": [ | |
| "Dynamics learning", | |
| "Control", | |
| "Simple environments" | |
| ], | |
| "required": [ | |
| "observation_t", | |
| "action_t", | |
| "observation_t+1" | |
| ], | |
| "optional": [ | |
| "reward", | |
| "done", | |
| "goal", | |
| "timestamp" | |
| ], | |
| "risk": "Frame-level random splits can leak nearly identical states across train and test." | |
| }, | |
| { | |
| "name": "Episode Dataset", | |
| "type": "Dataset Pattern", | |
| "focus": "Sequential dynamics", | |
| "best_for": [ | |
| "Rollouts", | |
| "Memory", | |
| "Long-horizon modeling" | |
| ], | |
| "required": [ | |
| "episode_id", | |
| "observations[]", | |
| "actions[]" | |
| ], | |
| "optional": [ | |
| "rewards[]", | |
| "terminal", | |
| "instruction", | |
| "metadata" | |
| ], | |
| "risk": "Broken episode boundaries can destroy temporal structure." | |
| }, | |
| { | |
| "name": "Multimodal Episode", | |
| "type": "Dataset Pattern", | |
| "focus": "Cross-modal world state", | |
| "best_for": [ | |
| "Robotics", | |
| "Embodied AI", | |
| "Physical AI" | |
| ], | |
| "required": [ | |
| "video", | |
| "actions", | |
| "robot_state" | |
| ], | |
| "optional": [ | |
| "depth", | |
| "audio", | |
| "language", | |
| "force", | |
| "pose" | |
| ], | |
| "risk": "Unsynchronized modalities create false transition errors." | |
| }, | |
| { | |
| "name": "One-Step Benchmark", | |
| "type": "Benchmark Pattern", | |
| "focus": "Immediate next-state prediction", | |
| "best_for": [ | |
| "Fast iteration", | |
| "Architecture debugging" | |
| ], | |
| "required": [ | |
| "held-out transitions", | |
| "prediction metric" | |
| ], | |
| "optional": [ | |
| "uncertainty", | |
| "per-task breakdown" | |
| ], | |
| "risk": "Can hide catastrophic long-horizon drift." | |
| }, | |
| { | |
| "name": "Multi-Horizon Benchmark", | |
| "type": "Benchmark Pattern", | |
| "focus": "Error growth over time", | |
| "best_for": [ | |
| "Rollouts", | |
| "Planning", | |
| "Simulation" | |
| ], | |
| "required": [ | |
| "horizons", | |
| "rollout protocol", | |
| "per-horizon metric" | |
| ], | |
| "optional": [ | |
| "drift ratio", | |
| "consistency score" | |
| ], | |
| "risk": "Averaging across horizons can conceal where failure begins." | |
| }, | |
| { | |
| "name": "Action Fidelity Benchmark", | |
| "type": "Benchmark Pattern", | |
| "focus": "Correct action consequences", | |
| "best_for": [ | |
| "Robotics", | |
| "Interactive worlds", | |
| "Agents" | |
| ], | |
| "required": [ | |
| "paired actions", | |
| "ground-truth transitions" | |
| ], | |
| "optional": [ | |
| "counterfactual actions", | |
| "action sensitivity" | |
| ], | |
| "risk": "Visual similarity alone does not prove correct causal response." | |
| }, | |
| { | |
| "name": "OOD Generalization Split", | |
| "type": "Split Strategy", | |
| "focus": "Distribution shift", | |
| "best_for": [ | |
| "Deployment", | |
| "Robotics", | |
| "Robustness" | |
| ], | |
| "required": [ | |
| "held-out environments or tasks" | |
| ], | |
| "optional": [ | |
| "held-out objects", | |
| "held-out embodiments" | |
| ], | |
| "risk": "Random splits can dramatically overestimate generalization." | |
| }, | |
| { | |
| "name": "Planning Utility Benchmark", | |
| "type": "Benchmark Pattern", | |
| "focus": "Downstream decision quality", | |
| "best_for": [ | |
| "Agents", | |
| "MPC", | |
| "Model-based RL" | |
| ], | |
| "required": [ | |
| "planner", | |
| "task success metric", | |
| "world-model rollouts" | |
| ], | |
| "optional": [ | |
| "regret", | |
| "return", | |
| "sample efficiency" | |
| ], | |
| "risk": "Prediction scores may not correlate with better decisions." | |
| }, | |
| { | |
| "name": "Control Benchmark", | |
| "type": "Benchmark Pattern", | |
| "focus": "Closed-loop task performance", | |
| "best_for": [ | |
| "Robotics", | |
| "Autonomous systems" | |
| ], | |
| "required": [ | |
| "environment", | |
| "policy/controller", | |
| "success metric" | |
| ], | |
| "optional": [ | |
| "safety violations", | |
| "energy", | |
| "path efficiency" | |
| ], | |
| "risk": "Offline prediction quality may not transfer to closed-loop control." | |
| }, | |
| { | |
| "name": "Efficiency Benchmark", | |
| "type": "Benchmark Pattern", | |
| "focus": "Operational cost", | |
| "best_for": [ | |
| "Real-time systems", | |
| "Large search spaces" | |
| ], | |
| "required": [ | |
| "latency", | |
| "memory", | |
| "throughput" | |
| ], | |
| "optional": [ | |
| "energy", | |
| "rollouts/sec", | |
| "cost" | |
| ], | |
| "risk": "A strong model may be unusable if rollouts are too slow for planning." | |
| } | |
| ] |