Agenten's picture
Upload 8 files
9ecc04a verified
Raw History Blame Contribute Delete
4.46 kB
[
{
"name": "Transition Dataset",
"type": "Dataset Pattern",
"focus": "One-step dynamics",
"best_for": [
"Dynamics learning",
"Control",
"Simple environments"
],
"required": [
"observation_t",
"action_t",
"observation_t+1"
],
"optional": [
"reward",
"done",
"goal",
"timestamp"
],
"risk": "Frame-level random splits can leak nearly identical states across train and test."
},
{
"name": "Episode Dataset",
"type": "Dataset Pattern",
"focus": "Sequential dynamics",
"best_for": [
"Rollouts",
"Memory",
"Long-horizon modeling"
],
"required": [
"episode_id",
"observations[]",
"actions[]"
],
"optional": [
"rewards[]",
"terminal",
"instruction",
"metadata"
],
"risk": "Broken episode boundaries can destroy temporal structure."
},
{
"name": "Multimodal Episode",
"type": "Dataset Pattern",
"focus": "Cross-modal world state",
"best_for": [
"Robotics",
"Embodied AI",
"Physical AI"
],
"required": [
"video",
"actions",
"robot_state"
],
"optional": [
"depth",
"audio",
"language",
"force",
"pose"
],
"risk": "Unsynchronized modalities create false transition errors."
},
{
"name": "One-Step Benchmark",
"type": "Benchmark Pattern",
"focus": "Immediate next-state prediction",
"best_for": [
"Fast iteration",
"Architecture debugging"
],
"required": [
"held-out transitions",
"prediction metric"
],
"optional": [
"uncertainty",
"per-task breakdown"
],
"risk": "Can hide catastrophic long-horizon drift."
},
{
"name": "Multi-Horizon Benchmark",
"type": "Benchmark Pattern",
"focus": "Error growth over time",
"best_for": [
"Rollouts",
"Planning",
"Simulation"
],
"required": [
"horizons",
"rollout protocol",
"per-horizon metric"
],
"optional": [
"drift ratio",
"consistency score"
],
"risk": "Averaging across horizons can conceal where failure begins."
},
{
"name": "Action Fidelity Benchmark",
"type": "Benchmark Pattern",
"focus": "Correct action consequences",
"best_for": [
"Robotics",
"Interactive worlds",
"Agents"
],
"required": [
"paired actions",
"ground-truth transitions"
],
"optional": [
"counterfactual actions",
"action sensitivity"
],
"risk": "Visual similarity alone does not prove correct causal response."
},
{
"name": "OOD Generalization Split",
"type": "Split Strategy",
"focus": "Distribution shift",
"best_for": [
"Deployment",
"Robotics",
"Robustness"
],
"required": [
"held-out environments or tasks"
],
"optional": [
"held-out objects",
"held-out embodiments"
],
"risk": "Random splits can dramatically overestimate generalization."
},
{
"name": "Planning Utility Benchmark",
"type": "Benchmark Pattern",
"focus": "Downstream decision quality",
"best_for": [
"Agents",
"MPC",
"Model-based RL"
],
"required": [
"planner",
"task success metric",
"world-model rollouts"
],
"optional": [
"regret",
"return",
"sample efficiency"
],
"risk": "Prediction scores may not correlate with better decisions."
},
{
"name": "Control Benchmark",
"type": "Benchmark Pattern",
"focus": "Closed-loop task performance",
"best_for": [
"Robotics",
"Autonomous systems"
],
"required": [
"environment",
"policy/controller",
"success metric"
],
"optional": [
"safety violations",
"energy",
"path efficiency"
],
"risk": "Offline prediction quality may not transfer to closed-loop control."
},
{
"name": "Efficiency Benchmark",
"type": "Benchmark Pattern",
"focus": "Operational cost",
"best_for": [
"Real-time systems",
"Large search spaces"
],
"required": [
"latency",
"memory",
"throughput"
],
"optional": [
"energy",
"rollouts/sec",
"cost"
],
"risk": "A strong model may be unusable if rollouts are too slow for planning."
}
]