diff --git a/.gitattributes b/.gitattributes
new file mode 100644
index 0000000000000000000000000000000000000000..7bf231e794542a83b16930baa79926b163ee8d78
--- /dev/null
+++ b/.gitattributes
@@ -0,0 +1,5 @@
+*.webp filter=lfs diff=lfs merge=lfs -text
+*.png filter=lfs diff=lfs merge=lfs -text
+*.jpg filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+*.bag filter=lfs diff=lfs merge=lfs -text
diff --git a/Dockerfile b/Dockerfile
new file mode 100644
index 0000000000000000000000000000000000000000..45f3f48a4b0a745a7b37398ae44b0a8d6fa3f58b
--- /dev/null
+++ b/Dockerfile
@@ -0,0 +1,54 @@
+FROM nvidia/cuda:12.4.1-devel-ubuntu22.04
+
+ENV DEBIAN_FRONTEND=noninteractive
+
+# System deps + Python 3.10 + ffmpeg
+RUN apt-get update && apt-get install -y --no-install-recommends \
+    git git-lfs wget curl ca-certificates \
+    libgl1-mesa-glx libglib2.0-0 \
+    software-properties-common \
+    && add-apt-repository ppa:deadsnakes/ppa -y \
+    && add-apt-repository ppa:ubuntuhandbook1/ffmpeg6 -y \
+    && apt-get update \
+    && apt-get install -y --no-install-recommends \
+    python3.10 python3.10-venv python3.10-dev \
+    ffmpeg \
+    && git lfs install \
+    && rm -rf /var/lib/apt/lists/*
+
+# Make python3.10 the default
+RUN update-alternatives --install /usr/bin/python python /usr/bin/python3.10 1 \
+    && update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.10 1
+
+# Install pip
+RUN curl -sS https://bootstrap.pypa.io/get-pip.py | python
+
+# Install PyTorch with CUDA 12.4 support
+RUN pip install --no-cache-dir \
+    torch torchvision torchaudio \
+    --index-url https://download.pytorch.org/whl/cu124
+
+# Clone our repo (includes our modified lerobot)
+RUN huggingface-cli download StrongRoboticsLab/pi05-so100-diverse \
+    --local-dir /workspace/pi05-so100-diverse \
+    --exclude "logs/*"
+
+# Install our modified lerobot + other deps
+RUN pip install --no-cache-dir \
+    -e /workspace/pi05-so100-diverse/lerobot \
+    "transformers==4.54.1" \
+    "accelerate>=0.34" \
+    wandb \
+    huggingface_hub \
+    hf_xet
+
+# Install Node.js + Claude Code
+RUN curl -fsSL https://deb.nodesource.com/setup_22.x | bash - \
+    && apt-get install -y nodejs \
+    && rm -rf /var/lib/apt/lists/* \
+    && npm install -g @anthropic-ai/claude-code
+
+WORKDIR /workspace/pi05-so100-diverse
+
+ENTRYPOINT ["/bin/bash", "-c"]
+CMD ["bash"]
diff --git a/bootstrap.sh b/bootstrap.sh
new file mode 100755
index 0000000000000000000000000000000000000000..248b0e4d158e46ede921af494d1d99f893f97df6
--- /dev/null
+++ b/bootstrap.sh
@@ -0,0 +1,65 @@
+#!/bin/bash
+# One-click bootstrap: builds Docker image, downloads dataset, starts training.
+# Usage: HF_TOKEN=xxx WANDB_API_KEY=xxx bash bootstrap.sh
+
+set -e
+
+if [ -z "$HF_TOKEN" ]; then echo "ERROR: export HF_TOKEN first"; exit 1; fi
+if [ -z "$WANDB_API_KEY" ]; then echo "ERROR: export WANDB_API_KEY first"; exit 1; fi
+
+DATASET_DIR="${DATASET_DIR:-/ephemeral/community_dataset_v3}"
+REPO_DIR="${REPO_DIR:-/workspace/pi05-so100-diverse}"
+NUM_GPUS="${NUM_GPUS:-1}"
+
+echo "=== Step 1: Clone repo ==="
+if [ ! -d "$REPO_DIR" ]; then
+    git clone https://huggingface.co/StrongRoboticsLab/pi05-so100-diverse "$REPO_DIR"
+else
+    echo "Repo already cloned, skipping"
+fi
+
+echo "=== Step 2: Build Docker image ==="
+cd "$REPO_DIR"
+if ! docker images pi05-training --format '{{.ID}}' | grep -q .; then
+    docker build -t pi05-training .
+else
+    echo "Image already built, skipping"
+fi
+
+echo "=== Step 3: Preflight checks ==="
+docker run --rm --runtime=nvidia \
+    -e HF_TOKEN="$HF_TOKEN" \
+    pi05-training "bash /workspace/pi05-so100-diverse/preflight.sh"
+
+if [ "${SKIP_DOWNLOAD:-0}" != "1" ]; then
+    echo "=== Step 4: Download dataset ==="
+    mkdir -p "$DATASET_DIR"
+    docker run --rm \
+        -e HF_TOKEN="$HF_TOKEN" \
+        -e HF_XET_HIGH_PERFORMANCE=1 \
+        -v "$(dirname $DATASET_DIR):$(dirname $DATASET_DIR)" \
+        pi05-training "huggingface-cli download \
+            --repo-type dataset \
+            HuggingFaceVLA/community_dataset_v3 \
+            --local-dir $DATASET_DIR \
+            --token \$HF_TOKEN"
+else
+    echo "=== Step 4: Skipped (SKIP_DOWNLOAD=1) ==="
+fi
+
+echo "=== Step 5: Start training ==="
+docker run --rm --runtime=nvidia \
+    --ipc=host \
+    --ulimit memlock=-1 \
+    --ulimit stack=67108864 \
+    -e HF_TOKEN="$HF_TOKEN" \
+    -e WANDB_API_KEY="$WANDB_API_KEY" \
+    -e NUM_GPUS="$NUM_GPUS" \
+    -e DATASET_DIR="$DATASET_DIR" \
+    -v /ephemeral:/ephemeral \
+    -v "$REPO_DIR:/workspace/pi05-so100-diverse" \
+    -e PYTHONPATH="" \
+    pi05-training "pip install -q 'transformers>=5.0.0' \
+        && SITE_PKG=\$(python3 -c \"import lerobot,os; print(os.path.dirname(lerobot.__path__[0]))\") \
+        && bash /workspace/pi05-so100-diverse/lerobot_patches/apply.sh \$SITE_PKG \
+        && bash /workspace/pi05-so100-diverse/train_cloud.sh"
diff --git a/cleanup_dataset.py b/cleanup_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..2e594180dd7214e2851315b1410ab5df60c71f8d
--- /dev/null
+++ b/cleanup_dataset.py
@@ -0,0 +1,63 @@
+#!/usr/bin/env python3
+"""Delete all files from the dataset that aren't in filtered_index.json."""
+
+import json
+import os
+import shutil
+import sys
+from collections import defaultdict
+from pathlib import Path
+
+
+def main():
+    index_path = sys.argv[1] if len(sys.argv) > 1 else "filtered_index.json"
+    dataset_dir = sys.argv[2] if len(sys.argv) > 2 else "/ephemeral/community_dataset_v3"
+
+    with open(index_path) as f:
+        index = json.load(f)
+
+    # Build set of needed directories (contributor/dataset)
+    needed_datasets = set()
+    for ep in index["episodes"]:
+        needed_datasets.add(ep["dataset"])
+
+    # Walk the dataset dir and find all contributor/dataset dirs
+    dataset_root = Path(dataset_dir)
+    deleted_bytes = 0
+    deleted_dirs = 0
+
+    for contributor_dir in sorted(dataset_root.iterdir()):
+        if not contributor_dir.is_dir() or contributor_dir.name.startswith("."):
+            continue
+
+        for ds_dir in sorted(contributor_dir.iterdir()):
+            if not ds_dir.is_dir():
+                continue
+
+            dataset_name = f"{contributor_dir.name}/{ds_dir.name}"
+            if dataset_name not in needed_datasets:
+                # Get size before deleting
+                size = sum(f.stat().st_size for f in ds_dir.rglob("*") if f.is_file())
+                shutil.rmtree(ds_dir)
+                deleted_bytes += size
+                deleted_dirs += 1
+                if deleted_dirs % 50 == 0:
+                    print(f"  Deleted {deleted_dirs} datasets, freed {deleted_bytes / 1024**3:.1f}GB", flush=True)
+
+        # Remove empty contributor dirs
+        if contributor_dir.exists() and not any(contributor_dir.iterdir()):
+            contributor_dir.rmdir()
+
+    # Also delete the .cache dir
+    cache_dir = dataset_root / ".cache"
+    if cache_dir.exists():
+        cache_size = sum(f.stat().st_size for f in cache_dir.rglob("*") if f.is_file())
+        shutil.rmtree(cache_dir)
+        deleted_bytes += cache_size
+        print(f"  Deleted .cache ({cache_size / 1024**3:.1f}GB)")
+
+    print(f"\nDone: deleted {deleted_dirs} unused datasets, freed {deleted_bytes / 1024**3:.1f}GB")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/download_dataset.sh b/download_dataset.sh
new file mode 100644
index 0000000000000000000000000000000000000000..364bee15c96b0c138ba76da0d66d11bee5cd801f
--- /dev/null
+++ b/download_dataset.sh
@@ -0,0 +1,25 @@
+#!/bin/bash
+# Download the full community_dataset_v3 using hfd (aria2c-based, resolver-only, no API rate limit issues)
+set -e
+
+if [ -z "$HF_TOKEN" ]; then echo "ERROR: export HF_TOKEN first"; exit 1; fi
+DATASET_DIR="${DATASET_DIR:-/ephemeral/community_dataset_v3}"
+
+# Install hfd if not present
+if [ ! -f /usr/local/bin/hfd ]; then
+    wget -q https://gist.githubusercontent.com/padeoe/697678ab8e528b85a2a7bddafea1fa4f/raw/hfd.sh -O /usr/local/bin/hfd
+    chmod +x /usr/local/bin/hfd
+fi
+
+echo "Downloading dataset to $DATASET_DIR..."
+echo "Using aria2c with 4 threads per file, 5 concurrent downloads"
+
+hfd HuggingFaceVLA/community_dataset_v3 \
+    --dataset \
+    --hf_token "$HF_TOKEN" \
+    --tool aria2c \
+    -x 4 \
+    -j 5 \
+    --local-dir "$DATASET_DIR"
+
+echo "Download complete!"
diff --git a/download_subset.py b/download_subset.py
new file mode 100644
index 0000000000000000000000000000000000000000..47eb8e3d211d914123cb3f5d217072288c688e30
--- /dev/null
+++ b/download_subset.py
@@ -0,0 +1,252 @@
+#!/usr/bin/env python3
+"""
+Download only the files needed for training (defined by filtered_index.json)
+from HuggingFaceVLA/community_dataset_v3.
+
+Rate-limit-aware greedy scheduler: downloads small files when we have rate limit
+headroom, swaps to large files (videos) when approaching the limit to keep
+bandwidth busy while the window recovers. Goal: never actually hit a 429.
+"""
+
+import argparse
+import json
+import os
+import sys
+import time
+import threading
+from collections import defaultdict, deque
+from concurrent.futures import ThreadPoolExecutor, as_completed, Future
+
+from huggingface_hub import hf_hub_download
+
+RATE_LIMIT = 100  # 1000 actual / ~10 API calls per hf_hub_download
+RATE_WINDOW = 300  # 5 minutes
+
+
+class RateLimitTracker:
+    """Sliding window request counter."""
+    def __init__(self):
+        self.lock = threading.Lock()
+        self.timestamps: deque[float] = deque()
+
+    def record(self):
+        now = time.time()
+        with self.lock:
+            self.timestamps.append(now)
+            self._prune(now)
+
+    def count(self) -> int:
+        now = time.time()
+        with self.lock:
+            self._prune(now)
+            return len(self.timestamps)
+
+    def headroom(self) -> int:
+        """How many more requests we can make in this window."""
+        return max(0, RATE_LIMIT - self.count())
+
+    def wait_if_needed(self):
+        """If we've exhausted the window, sleep until oldest request expires."""
+        while self.headroom() <= 0:
+            with self.lock:
+                if self.timestamps:
+                    wait = RATE_WINDOW - (time.time() - self.timestamps[0]) + 1
+                    if wait > 0:
+                        print(f"  Rate limit reached, waiting {wait:.0f}s...", flush=True)
+                        # Release lock while sleeping
+                else:
+                    wait = 0
+            if wait > 0:
+                time.sleep(wait)
+
+    def _prune(self, now):
+        cutoff = now - RATE_WINDOW
+        while self.timestamps and self.timestamps[0] < cutoff:
+            self.timestamps.popleft()
+
+
+class FileQueue:
+    """Thread-safe queue that serves small or large files on demand."""
+    def __init__(self, small_files: list[str], large_files: list[str]):
+        self.lock = threading.Lock()
+        self.small = deque(small_files)
+        self.large = deque(large_files)
+        self.total = len(small_files) + len(large_files)
+
+    def get(self, prefer_small: bool) -> str | None:
+        with self.lock:
+            if prefer_small and self.small:
+                return self.small.popleft()
+            elif self.large:
+                return self.large.popleft()
+            elif self.small:
+                return self.small.popleft()
+            return None
+
+    def remaining(self) -> int:
+        with self.lock:
+            return len(self.small) + len(self.large)
+
+    def small_remaining(self) -> int:
+        with self.lock:
+            return len(self.small)
+
+    def large_remaining(self) -> int:
+        with self.lock:
+            return len(self.large)
+
+
+def build_file_lists(index_path: str, output_dir: str) -> tuple[list[str], list[str], int]:
+    """Returns (small_files, large_files, skipped_count) from filtered_index.json.
+    Skips files already on disk."""
+    with open(index_path) as f:
+        index = json.load(f)
+
+    datasets = defaultdict(list)
+    for ep in index["episodes"]:
+        datasets[ep["dataset"]].append(ep["episode_index"])
+
+    small = []
+    large = []
+    skipped = 0
+
+    def add_if_missing(filepath, target_list):
+        nonlocal skipped
+        if os.path.exists(os.path.join(output_dir, filepath)):
+            skipped += 1
+        else:
+            target_list.append(filepath)
+
+    for dataset_name, episode_indices in datasets.items():
+        prefix = dataset_name
+        add_if_missing(f"{prefix}/meta/info.json", small)
+        add_if_missing(f"{prefix}/meta/tasks.jsonl", small)
+        add_if_missing(f"{prefix}/meta/episodes.jsonl", small)
+
+        for ep_idx in episode_indices:
+            ep_str = f"episode_{ep_idx:06d}"
+            add_if_missing(f"{prefix}/data/chunk-000/{ep_str}.parquet", small)
+            add_if_missing(f"{prefix}/videos/chunk-000/observation.images.image/{ep_str}.mp4", large)
+            add_if_missing(f"{prefix}/videos/chunk-000/observation.images.image2/{ep_str}.mp4", large)
+
+    return small, large, skipped
+
+
+# Shared state
+tracker = RateLimitTracker()
+queue: FileQueue = None
+stats_lock = threading.Lock()
+downloaded = 0
+total_bytes = 0
+failed = []
+start_time = 0
+
+# When headroom drops below this, prefer large files
+HEADROOM_THRESHOLD = 50
+
+
+def worker(output_dir, token):
+    """Worker loop: grab a file based on rate limit state, download it, repeat."""
+    global downloaded, total_bytes
+
+    while True:
+        headroom = tracker.headroom()
+        prefer_small = headroom > HEADROOM_THRESHOLD
+
+        filepath = queue.get(prefer_small)
+        if filepath is None:
+            return
+
+        for attempt in range(10):
+            tracker.wait_if_needed()
+            tracker.record()
+            try:
+                path = hf_hub_download(
+                    repo_id="HuggingFaceVLA/community_dataset_v3",
+                    repo_type="dataset",
+                    filename=filepath,
+                    local_dir=output_dir,
+                    token=token,
+                )
+                size = os.path.getsize(path)
+                with stats_lock:
+                    downloaded += 1
+                    total_bytes += size
+                    _maybe_log()
+                break
+            except Exception as e:
+                if "429" in str(e) and attempt < 9:
+                    time.sleep(30 * (attempt + 1))
+                    continue
+                with stats_lock:
+                    failed.append((filepath, str(e)))
+                    _maybe_log()
+                break
+
+
+def _maybe_log():
+    """Log progress every 100 files. Must be called with stats_lock held."""
+    total = downloaded + len(failed)
+    if total % 100 == 0 and total > 0:
+        elapsed = time.time() - start_time
+        rate = total / elapsed if elapsed > 0 else 0
+        mb_s = (total_bytes / 1024 / 1024) / elapsed if elapsed > 0 else 0
+        gb_done = total_bytes / 1024 / 1024 / 1024
+        headroom = tracker.headroom()
+        remaining = queue.remaining()
+        est_min = remaining / rate / 60 if rate > 0 else 0
+        print(f"  [{total}/{queue.total}] {gb_done:.1f}GB, "
+              f"{mb_s:.0f} MB/s, {rate:.1f} files/s, "
+              f"headroom: {headroom}/{RATE_LIMIT}, "
+              f"queued: {queue.small_remaining()}s+{queue.large_remaining()}L, "
+              f"~{est_min:.0f}min left", flush=True)
+
+
+def main():
+    global queue, start_time
+
+    parser = argparse.ArgumentParser(description="Download training subset from community_dataset_v3")
+    parser.add_argument("--index", type=str, default="filtered_index.json")
+    parser.add_argument("--output", type=str, default="/data/community_dataset_v3")
+    parser.add_argument("--token", type=str, default=os.environ.get("HF_TOKEN"))
+    parser.add_argument("--workers", type=int, default=8)
+    args = parser.parse_args()
+
+    if not args.token:
+        print("ERROR: Set HF_TOKEN or pass --token")
+        return
+
+    small, large, skipped = build_file_lists(args.index, args.output)
+    queue = FileQueue(small, large)
+
+    print(f"Files to download: {queue.total} ({skipped} already on disk, skipped)")
+    print(f"  Small (metadata+parquets): {len(small)}")
+    print(f"  Large (videos):            {len(large)}")
+    print(f"  Workers: {args.workers}")
+    print(f"  Rate limit: {RATE_LIMIT}/{RATE_WINDOW}s, "
+          f"swap to large files at <{HEADROOM_THRESHOLD} headroom")
+    print()
+
+    start_time = time.time()
+
+    with ThreadPoolExecutor(max_workers=args.workers) as pool:
+        futures = [pool.submit(worker, args.output, args.token)
+                   for _ in range(args.workers)]
+        for f in futures:
+            f.result()
+
+    elapsed = time.time() - start_time
+    gb_total = total_bytes / 1024 / 1024 / 1024
+    print(f"\nDone in {elapsed/60:.1f} min: {downloaded} files, "
+          f"{gb_total:.1f}GB, {len(failed)} failed")
+    if failed:
+        print("Failed files:")
+        for f, err in failed[:20]:
+            print(f"  {f}: {err}")
+        if len(failed) > 20:
+            print(f"  ... and {len(failed) - 20} more")
+        sys.exit(1)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/.dockerignore b/lerobot/.dockerignore
new file mode 100644
index 0000000000000000000000000000000000000000..c0d8a84b566aef1f4bf9e7009875decfccfc0125
--- /dev/null
+++ b/lerobot/.dockerignore
@@ -0,0 +1,160 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# Misc
+.git
+tmp
+wandb
+data
+outputs
+.vscode
+rl
+media
+
+
+# Logging
+logs
+
+# HPC
+nautilus/*.yaml
+*.key
+
+# Slurm
+sbatch*.sh
+
+# Byte-compiled / optimized / DLL files
+__pycache__/
+*.py[cod]
+*$py.class
+
+# C extensions
+*.so
+
+# Distribution / packaging
+.Python
+build/
+develop-eggs/
+dist/
+downloads/
+eggs/
+.eggs/
+lib/
+lib64/
+parts/
+sdist/
+var/
+wheels/
+pip-wheel-metadata/
+share/python-wheels/
+*.egg-info/
+.installed.cfg
+*.egg
+MANIFEST
+
+# PyInstaller
+#  Usually these files are written by a python script from a template
+#  before PyInstaller builds the exe, so as to inject date/other infos into it.
+*.manifest
+*.spec
+
+# Installer logs
+pip-log.txt
+pip-delete-this-directory.txt
+
+# Unit test / coverage reports
+!tests/artifacts
+htmlcov/
+.tox/
+.nox/
+.coverage
+.coverage.*
+nosetests.xml
+coverage.xml
+*.cover
+*.py,cover
+.hypothesis/
+.pytest_cache/
+
+# Ignore .cache except calibration
+.cache/*
+!.cache/calibration/
+!.cache/calibration/**
+
+# Translations
+*.mo
+*.pot
+
+# Django stuff:
+*.log
+local_settings.py
+db.sqlite3
+db.sqlite3-journal
+
+# Flask stuff:
+instance/
+.webassets-cache
+
+# Scrapy stuff:
+.scrapy
+
+# Sphinx documentation
+docs/_build/
+
+# PyBuilder
+target/
+
+# Jupyter Notebook
+.ipynb_checkpoints
+
+# IPython
+profile_default/
+ipython_config.py
+
+# pyenv
+.python-version
+
+# pipenv
+#   According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
+#   However, in case of collaboration, if having platform-specific dependencies or dependencies
+#   having no cross-platform support, pipenv may install dependencies that don't work, or not
+#   install all needed dependencies.
+#Pipfile.lock
+
+# PEP 582; used by e.g. github.com/David-OConnor/pyflow
+__pypackages__/
+
+# Celery stuff
+celerybeat-schedule
+celerybeat.pid
+
+# SageMath parsed files
+*.sage.py
+
+# Spyder project settings
+.spyderproject
+.spyproject
+
+# Rope project settings
+.ropeproject
+
+# mkdocs documentation
+/site
+
+# mypy
+.mypy_cache/
+.dmypy.json
+dmypy.json
+
+# Pyre type checker
+.pyre/
diff --git a/lerobot/.gitattributes b/lerobot/.gitattributes
new file mode 100644
index 0000000000000000000000000000000000000000..7d89f37b2bf0c1443508914a62697c1c69f8920f
--- /dev/null
+++ b/lerobot/.gitattributes
@@ -0,0 +1,21 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+*.memmap filter=lfs diff=lfs merge=lfs -text
+*.stl filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+*.mp4 filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.json !text !filter !merge !diff
+tests/artifacts/cameras/*.png filter=lfs diff=lfs merge=lfs -text
+*.bag filter=lfs diff=lfs merge=lfs -text
diff --git a/lerobot/.github/ISSUE_TEMPLATE/bug-report.yml b/lerobot/.github/ISSUE_TEMPLATE/bug-report.yml
new file mode 100644
index 0000000000000000000000000000000000000000..9f602de30960a5ddaed9ed04272a2981718c1f2a
--- /dev/null
+++ b/lerobot/.github/ISSUE_TEMPLATE/bug-report.yml
@@ -0,0 +1,94 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+name: "🚀 Issue / Bug / Request"
+description: Report a bug, suggest an improvement, or ask a technical question.
+body:
+  - type: markdown
+    attributes:
+      value: |
+        ### Thanks for contributing to LeRobot! 🙌
+        Please choose the most relevant sections below. If this is a general "how-to" question, consider our [Discord](https://discord.gg/s3KuuzsPFb) for faster community support.
+
+  - type: dropdown
+    id: issue-type
+    attributes:
+      label: Ticket Type
+      description: What kind of ticket are you opening?
+      options:
+        - "🐛 Bug Report (Something isn't working)"
+        - "💡 Feature Request / Improvement"
+        - "❓ Technical Question"
+        - "🧹 Maintenance / Documentation"
+    validations:
+      required: true
+
+  - type: textarea
+    id: system-info
+    attributes:
+      label: Environment & System Info
+      description: |
+        For bugs or technical questions, please run `lerobot-info` and paste the output.
+        (Optional for feature requests).
+      render: Shell
+      placeholder: lerobot version, OS, python version, etc.
+
+  - type: textarea
+    id: description
+    validations:
+      required: true
+    attributes:
+      label: Description
+      description: |
+        Provide a clear summary of the issue or your proposal.
+        - **Bugs:** What is happening?
+        - **Features:** What is the goal/use case?
+        - **Questions:** What are you trying to achieve?
+      placeholder: |
+        A clear and concise description of the issue or suggestion.
+
+  - type: textarea
+    id: context-repro
+    attributes:
+      label: Context & Reproduction
+      description: |
+        Provide a code snippet, steps to reproduce a bug, or technical details about your proposal.
+        Please use code blocks for scripts and CLI commands.
+      placeholder: |
+        Steps to reproduce / Usage example:
+        1.
+        2.
+        3.
+
+  - type: textarea
+    id: logs
+    attributes:
+      label: Relevant logs or stack trace
+      description: If applicable, paste relevant error logs here.
+      render: Shell
+
+  - type: checkboxes
+    id: extras
+    attributes:
+      label: Checklist
+      options:
+        - label: I have searched existing tickets to ensure this isn't a duplicate.
+        - label: I am using the latest version of the `main` branch.
+        - label: I have verified this is not an environment-specific problem.
+
+  - type: textarea
+    id: workaround
+    attributes:
+      label: Additional Info / Workarounds
+      description: Anything else we should know? If you have a workaround, please share it!
diff --git a/lerobot/.github/PULL_REQUEST_TEMPLATE.md b/lerobot/.github/PULL_REQUEST_TEMPLATE.md
new file mode 100644
index 0000000000000000000000000000000000000000..43e2442d323059f04a23eaf181ab18750da08acf
--- /dev/null
+++ b/lerobot/.github/PULL_REQUEST_TEMPLATE.md
@@ -0,0 +1,55 @@
+## Title
+
+Short, imperative summary (e.g., "fix(robots): handle None in sensor parser"). See [CONTRIBUTING.md](../CONTRIBUTING.md) for PR conventions.
+
+## Type / Scope
+
+- **Type**: (Bug | Feature | Docs | Performance | Test | CI | Chore)
+- **Scope**: (optional — name of module or package affected)
+
+## Summary / Motivation
+
+- One-paragraph description of what changes and why.
+- Why this change is needed and any trade-offs or design notes.
+
+## Related issues
+
+- Fixes / Closes: # (if any)
+- Related: # (if any)
+
+## What changed
+
+- Short, concrete bullets of the modifications (files/behaviour).
+- Short note if this introduces breaking changes and migration steps.
+
+## How was this tested (or how to run locally)
+
+- Tests added: list new tests or test files.
+- Manual checks / dataset runs performed.
+- Instructions for the reviewer
+
+Example:
+
+- Ran the relevant tests:
+
+  ```bash
+  pytest -q tests/ -k <keyword>
+  ```
+
+- Reproduce with a quick example or CLI (if applicable):
+
+  ```bash
+  lerobot-train --some.option=true
+  ```
+
+## Checklist (required before merge)
+
+- [ ] Linting/formatting run (`pre-commit run -a`)
+- [ ] All tests pass locally (`pytest`)
+- [ ] Documentation updated
+- [ ] CI is green
+
+## Reviewer notes
+
+- Anything the reviewer should focus on (performance, edge-cases, specific files) or general notes.
+- Anyone in the community is free to review the PR.
diff --git a/lerobot/.github/labeler.yml b/lerobot/.github/labeler.yml
new file mode 100644
index 0000000000000000000000000000000000000000..d3c5cc622c89e8260df64353e55bdc1f9af9ade8
--- /dev/null
+++ b/lerobot/.github/labeler.yml
@@ -0,0 +1,69 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+CI:
+  - changed-files:
+      - any-glob-to-any-file:
+        - '.github/**'
+        - 'docker/**'
+
+github_actions:
+  - changed-files:
+      - any-glob-to-any-file: '.github/**'
+
+documentation:
+  - changed-files:
+      - any-glob-to-any-file:
+          - '**/*.md'
+          - '**/*.mdx'
+          - 'docs/**'
+
+examples:
+  - changed-files:
+      - any-glob-to-any-file: 'examples/**'
+
+tests:
+  - changed-files:
+      - any-glob-to-any-file: 'tests/**'
+
+sensors:
+  - changed-files:
+      - any-glob-to-any-file: 'src/lerobot/cameras/**'
+
+configuration:
+  - changed-files:
+      - any-glob-to-any-file: 'src/lerobot/configs/**'
+
+dataset:
+  - changed-files:
+      - any-glob-to-any-file: 'src/lerobot/datasets/**'
+
+evaluation:
+  - changed-files:
+      - any-glob-to-any-file: 'src/lerobot/envs/**'
+
+robots:
+  - changed-files:
+      - any-glob-to-any-file:
+          - 'src/lerobot/teleoperators/**'
+          - 'src/lerobot/robots/**'
+          - 'src/lerobot/motors/**'
+
+policies:
+  - changed-files:
+      - any-glob-to-any-file: 'src/lerobot/policies/**'
+
+processor:
+  - changed-files:
+      - any-glob-to-any-file: 'src/lerobot/processor/**'
diff --git a/lerobot/.github/workflows/documentation-upload-pr.yml b/lerobot/.github/workflows/documentation-upload-pr.yml
new file mode 100644
index 0000000000000000000000000000000000000000..6ee2a5caad0618009ec49156d67844ded9dc29ba
--- /dev/null
+++ b/lerobot/.github/workflows/documentation-upload-pr.yml
@@ -0,0 +1,41 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow uploads the documentation preview built for a PR and comments the link on the PR.
+name: Documentation PR Upload
+permissions:
+  contents: read
+  pull-requests: write
+
+on:
+  # Triggered by the completion of the main 'Documentation' workflow.
+  workflow_run: # zizmor: ignore[dangerous-triggers] We follow the same pattern as in Transformers
+    workflows: ["Documentation"]
+    types:
+      - completed
+
+jobs:
+  # This job uploads a preview of the documentation for a pull request.
+  upload_and_comment:
+    name: Upload Preview and Comment
+    if: >
+      github.event.workflow_run.event == 'pull_request' &&
+      github.event.workflow_run.conclusion == 'success' &&
+      github.repository == 'huggingface/lerobot'
+    uses: huggingface/doc-builder/.github/workflows/upload_pr_documentation.yml@main
+    with:
+      package_name: lerobot
+    secrets:
+      hf_token: ${{ secrets.HF_DOC_BUILD_PUSH }}
+      comment_bot_token: ${{ secrets.COMMENT_BOT_TOKEN }}
diff --git a/lerobot/.github/workflows/documentation.yml b/lerobot/.github/workflows/documentation.yml
new file mode 100644
index 0000000000000000000000000000000000000000..c7926c54233baddd2d29685e777332bbf07db747
--- /dev/null
+++ b/lerobot/.github/workflows/documentation.yml
@@ -0,0 +1,86 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow handles building documentation for both main branches and PRs.
+name: Documentation
+
+on:
+  # Allows running this workflow manually from the Actions tab
+  workflow_dispatch:
+    inputs:
+      version:
+        description: 'Version tag (e.g. v0.1.2) - Leave empty for standard main build'
+        required: false
+        type: string
+
+  # Triggers the workflow on push events to main for the docs folder
+  push:
+    branches:
+      - main
+    paths:
+      - "docs/**"
+
+  # Triggers the workflow on pull request events targeting main for the docs folder
+  pull_request:
+    branches:
+      - main
+    paths:
+      - "docs/**"
+
+  release:
+    types: [published]
+
+# Ensures that only the latest commit for a PR or branch is built, canceling older runs.
+concurrency:
+  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
+  cancel-in-progress: true
+
+jobs:
+  # This job builds and deploys the official documentation.
+  build_main_docs:
+    name: Build Main Docs
+    if: >
+      (github.event_name == 'push' || github.event_name == 'workflow_dispatch' || github.event_name == 'release') &&
+      github.repository == 'huggingface/lerobot'
+    permissions:
+      contents: read
+    uses: huggingface/doc-builder/.github/workflows/build_main_documentation.yml@main
+    with:
+      commit_sha: ${{ github.sha }}
+      package: lerobot
+      additional_args: >-
+        --not_python_module
+        ${{
+          (github.event_name == 'release' && format('--version {0}', github.event.release.tag_name)) ||
+          (inputs.version != '' && format('--version {0}', inputs.version)) ||
+          ''
+        }}
+    secrets:
+      token: ${{ secrets.HUGGINGFACE_PUSH }}
+      hf_token: ${{ secrets.HF_DOC_BUILD_PUSH }}
+
+  # This job builds a preview of the documentation for a pull request.
+  # The result of this job triggers the 'Upload PR Documentation' workflow.
+  build_pr_docs:
+    name: Build PR Docs
+    if: github.event_name == 'pull_request' && github.repository == 'huggingface/lerobot'
+    permissions:
+      contents: read
+      pull-requests: write
+    uses: huggingface/doc-builder/.github/workflows/build_pr_documentation.yml@main
+    with:
+      commit_sha: ${{ github.event.pull_request.head.sha }}
+      pr_number: ${{ github.event.number }}
+      package: lerobot
+      additional_args: --not_python_module
diff --git a/lerobot/.github/workflows/fast_tests.yml b/lerobot/.github/workflows/fast_tests.yml
new file mode 100644
index 0000000000000000000000000000000000000000..fc169e25341163a395d95aeecfc8a9ec59b643ff
--- /dev/null
+++ b/lerobot/.github/workflows/fast_tests.yml
@@ -0,0 +1,100 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow handles fast testing.
+name: Fast Tests
+
+on:
+  # Allows running this workflow manually from the Actions tab
+  workflow_dispatch:
+
+  pull_request:
+    branches:
+      - main
+    paths:
+      - "src/**"
+      - "tests/**"
+      - ".github/workflows/**"
+      - "pyproject.toml"
+      - "Makefile"
+  push:
+    branches:
+      - main
+    paths:
+      - "src/**"
+      - "tests/**"
+      - ".github/workflows/**"
+      - "pyproject.toml"
+      - "Makefile"
+
+permissions:
+  contents: read
+
+# Sets up the environment variables
+env:
+  UV_VERSION: "0.8.0"
+  PYTHON_VERSION: "3.12"
+
+# Ensures that only the latest commit for a PR or branch is built, canceling older runs.
+concurrency:
+  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
+  cancel-in-progress: true
+
+jobs:
+  # This job runs pytests with the default dependencies.
+  # It runs everytime we commit to a PR or push to main
+  fast-pytest-tests:
+    name: Fast Pytest Tests
+    runs-on: ubuntu-latest
+    env:
+      MUJOCO_GL: egl
+      HF_HOME: /mnt/cache/.cache/huggingface
+      HF_LEROBOT_HOME: /mnt/cache/.cache/huggingface/lerobot
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
+    steps:
+      - uses: actions/checkout@v6
+        with:
+          persist-credentials: false
+          lfs: true
+
+      # NOTE(Steven): Mount to `/mnt` to avoid the limited storage on `/home`. Consider cleaning default SDKs or using self-hosted runners for more space.
+      # (As of 2024-06-10, the runner's `/home` has only 6.2 GB free—8% of its 72 GB total.)
+      - name: Setup /mnt storage
+        run: sudo chown -R $USER:$USER /mnt
+
+      # TODO(Steven): Evaluate the need of these dependencies
+      - name: Install apt dependencies
+        run: |
+          sudo apt-get update && sudo apt-get install -y build-essential git \
+          curl libglib2.0-0 libegl1-mesa-dev ffmpeg \
+          libusb-1.0-0-dev speech-dispatcher libgeos-dev portaudio19-dev
+
+      - name: Setup uv and Python
+        uses: astral-sh/setup-uv@v6 # zizmor: ignore[unpinned-uses]
+        with:
+          enable-cache: true
+          version: ${{ env.UV_VERSION }}
+          python-version: ${{ env.PYTHON_VERSION }}
+
+      - name: Install lerobot with test extras
+        run: uv sync --extra "test"
+
+      - name: Login to Hugging Face
+        if: env.HF_USER_TOKEN != ''
+        run: |
+          uv run hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          uv run hf auth whoami
+
+      - name: Run pytest
+        run: uv run pytest tests -vv --maxfail=10
diff --git a/lerobot/.github/workflows/full_tests.yml b/lerobot/.github/workflows/full_tests.yml
new file mode 100644
index 0000000000000000000000000000000000000000..8b7d28123fb9d1ab6641588c77d6994b3b13d5bb
--- /dev/null
+++ b/lerobot/.github/workflows/full_tests.yml
@@ -0,0 +1,237 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow handles full testing.
+name: Full Tests
+
+on:
+  # Allows running this workflow manually from the Actions tab
+  workflow_dispatch:
+
+  pull_request_review:
+    types: [submitted]
+  push:
+    branches:
+      - main
+    paths:
+      - "src/**"
+      - "tests/**"
+      - ".github/workflows/**"
+      - "pyproject.toml"
+      - "Makefile"
+
+permissions:
+  contents: read
+
+# Sets up the environment variables
+env:
+  UV_VERSION: "0.8.0"
+  PYTHON_VERSION: "3.12"
+  DOCKER_IMAGE_NAME: huggingface/lerobot-gpu
+
+# Ensures that only the latest action is built, canceling older runs.
+concurrency:
+  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
+  cancel-in-progress: true
+
+jobs:
+
+  # This job runs the E2E tests + pytest with all extras
+  # It runs everytime a PR is approved or a push to main
+  full-tests:
+    name: Full Tests
+    runs-on: ubuntu-latest
+    if: |
+      (github.event_name == 'pull_request_review' && github.event.review.state == 'approved') ||
+      github.event_name == 'push' ||
+      github.event_name == 'workflow_dispatch'
+    env:
+      MUJOCO_GL: egl
+      HF_HOME: /mnt/cache/.cache/huggingface
+      HF_LEROBOT_HOME: /mnt/cache/.cache/huggingface/lerobot
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
+    steps:
+      - uses: actions/checkout@v6
+        with:
+          lfs: true
+          persist-credentials: false
+
+      # NOTE(Steven): Mount to `/mnt` to avoid the limited storage on `/home`. Consider cleaning default SDKs or using self-hosted runners for more space.
+      # (As of 2024-06-10, the runner's `/home` has only 6.2 GB free—8% of its 72 GB total.)
+      - name: Setup /mnt storage
+        run: sudo chown -R $USER:$USER /mnt
+
+      - name: Install apt dependencies
+        run: |
+          sudo apt-get update && sudo apt-get install -y build-essential \
+          git curl libglib2.0-0 libegl1-mesa-dev ffmpeg libusb-1.0-0-dev \
+          speech-dispatcher libgeos-dev portaudio19-dev
+
+      - name: Setup uv and Python
+        uses: astral-sh/setup-uv@v6 # zizmor: ignore[unpinned-uses]
+        with:
+          enable-cache: true
+          version: ${{ env.UV_VERSION }}
+          python-version: ${{ env.PYTHON_VERSION }}
+
+      - name: Install lerobot with all extras
+        run: uv sync --extra all # TODO(Steven): Make flash-attn optional
+
+      - name: Login to Hugging Face
+        if: env.HF_USER_TOKEN != ''
+        run: |
+          uv run hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          uv run hf auth whoami
+
+      - name: Run pytest (all extras)
+        run: uv run pytest tests -vv --maxfail=10
+
+      - name: Run end-to-end tests
+        run: uv run make test-end-to-end
+
+  # This job builds a GPU enabled image for testing
+  # It runs everytime a PR is approved or a push to main
+  # TODO(Steven): For now we skip this job for community PRs
+  build-and-push-docker:
+    name: Build and Push Docker
+    runs-on:
+      group: aws-general-8-plus
+    if: |
+      github.repository == 'huggingface/lerobot' && (
+        (github.event_name == 'pull_request_review' && github.event.review.state == 'approved' && github.event.pull_request.head.repo.fork == false) ||
+        github.event_name == 'push' ||
+        github.event_name == 'workflow_dispatch'
+      )
+    outputs:
+      image_tag: ${{ steps.set_tag.outputs.image_tag }}
+    env:
+      GITHUB_EVENT_NAME: ${{ github.event_name }}
+      GITHUB_REF: ${{ github.ref }}
+      GITHUB_PR_NUMBER: ${{ github.event.pull_request.number }}
+    steps:
+      - name: Set Docker image tag
+        id: set_tag
+        run: |
+          if [[ "${GITHUB_EVENT_NAME}" == "push" ]]; then
+            TAG="${DOCKER_IMAGE_NAME}:latest"
+          elif [[ -n "${GITHUB_PR_NUMBER}" ]]; then
+            TAG="${DOCKER_IMAGE_NAME}:pr-${GITHUB_PR_NUMBER}"
+          else
+            TAG="${DOCKER_IMAGE_NAME}:pr-${GITHUB_REF##*/}"
+          fi
+          echo "image_tag=$TAG" >> $GITHUB_OUTPUT
+      - name: Install Git LFS
+        run: |
+          sudo apt-get update
+          sudo apt-get install git-lfs
+          git lfs install
+      - uses: actions/checkout@v6
+        with:
+          lfs: true
+          persist-credentials: false
+      - name: Set up Docker Buildx
+        uses: docker/setup-buildx-action@v3 # zizmor: ignore[unpinned-uses]
+        with:
+          cache-binary: false
+      - name: Login to Docker Hub
+        uses: docker/login-action@v3 # zizmor: ignore[unpinned-uses]
+        with:
+          username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+          password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+      - name: Build and push Docker image
+        uses: docker/build-push-action@v6 # zizmor: ignore[unpinned-uses]
+        with:
+          context: .
+          file: ./docker/Dockerfile.internal
+          push: true
+          tags: ${{ steps.set_tag.outputs.image_tag }}
+
+  # This job runs pytest with all extras in a GPU enabled host
+  # It runs everytime a test image is created
+  gpu-tests:
+    name: GPU Tests
+    needs: [build-and-push-docker]
+    runs-on:
+      group: aws-g6-4xlarge-plus
+    env:
+      HF_HOME: /home/user_lerobot/.cache/huggingface
+      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
+      TORCH_HOME: /home/user_lerobot/.cache/torch
+      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
+    container:
+      image: ${{ needs.build-and-push-docker.outputs.image_tag }} # zizmor: ignore[unpinned-images]
+      options: --gpus all --shm-size "16gb"
+      credentials:
+        username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+        password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+    defaults:
+      run:
+        shell: bash
+        working-directory: /lerobot
+    steps:
+      - name: Login to Hugging Face
+        if: env.HF_USER_TOKEN != ''
+        run: |
+          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          hf auth whoami
+      - name: Fix ptxas permissions
+        run: chmod +x /lerobot/.venv/lib/python3.12/site-packages/triton/backends/nvidia/bin/ptxas
+      - name: Run pytest on GPU
+        run: pytest tests -vv --maxfail=10
+      - name: Run end-to-end tests
+        run: make test-end-to-end
+
+  # This job deletes the test image recently created
+  # It runs everytime after the gpu-tests have finished
+  delete-pr-image:
+    name: Delete PR Image
+    needs: [gpu-tests, build-and-push-docker]
+    if: always() && ((github.event.review.state == 'approved') || (github.event_name == 'workflow_dispatch')) && needs.build-and-push-docker.result == 'success'
+    runs-on: ubuntu-latest
+    steps:
+      - name: Get Docker Hub Token and Delete Image
+        # zizmor: ignore[template-injection]
+        env:
+          DOCKERHUB_LEROBOT_USERNAME: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+          DOCKERHUB_LEROBOT_PASSWORD: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+          IMAGE_FULL: ${{ needs.build-and-push-docker.outputs.image_tag }}
+        run: |
+          IMAGE_NAME=$(echo "$IMAGE_FULL" | cut -d':' -f1)
+          IMAGE_TAG=$(echo "$IMAGE_FULL" | cut -d':' -f2-)
+          echo "Attempting to delete image: $IMAGE_NAME:$IMAGE_TAG"
+
+          TOKEN=$(curl -s -H "Content-Type: application/json" \
+                       -X POST \
+                       -d "{\"username\": \"$DOCKERHUB_LEROBOT_USERNAME\", \"password\": \"$DOCKERHUB_LEROBOT_PASSWORD\"}" \
+                       https://hub.docker.com/v2/users/login/ | jq -r .token)
+
+          if [ "$TOKEN" == "null" ] || [ -z "$TOKEN" ]; then
+            echo "::error::Failed to get Docker Hub token."
+            exit 1
+          fi
+
+          HTTP_RESPONSE=$(curl -s -o /dev/null -w "%{http_code}" \
+                               -H "Authorization: JWT ${TOKEN}" \
+                               -X DELETE \
+                               https://hub.docker.com/v2/repositories/${IMAGE_NAME}/tags/$IMAGE_TAG)
+
+          if [ "$HTTP_RESPONSE" -eq 204 ]; then
+            echo "Successfully deleted Docker image tag: $IMAGE_NAME:$IMAGE_TAG"
+          else
+            echo "::error::Failed to delete Docker image. HTTP status: $HTTP_RESPONSE"
+            exit 1
+          fi
+
+# TODO(Steven): Check dockerimages pull in ubuntu
diff --git a/lerobot/.github/workflows/issue_labeler.yml b/lerobot/.github/workflows/issue_labeler.yml
new file mode 100644
index 0000000000000000000000000000000000000000..438184e3f91ae41d2bed7c2544e63d4a780cf4c3
--- /dev/null
+++ b/lerobot/.github/workflows/issue_labeler.yml
@@ -0,0 +1,77 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow automatically labels issues based on their content.
+name: Issue Labeler
+on:
+  # Trigger on new issues and edits to existing issues
+  issues:
+    types: [opened, edited]
+
+permissions:
+  contents: read
+  issues: write
+
+jobs:
+  label-issue:
+    name: Auto Label Issue
+    runs-on: ubuntu-latest
+    if: github.repository == 'huggingface/lerobot'
+    steps:
+      - uses: actions/github-script@v8
+        with:
+          script: |
+            // Setup Input Text
+            const body = (context.payload.issue.body || '');
+            const title = (context.payload.issue.title || '');
+            const cleanBody = body.replace(/```[\s\S]*?```/g, '');
+            const text = `${title}\n${cleanBody}`.toLowerCase();
+            const labelsToAdd = new Set();
+            const matches = (re) => re.test(text);
+
+            // Keyword Heuristics
+
+            if (matches(/\b(bug|error|crash|exception)\b/i)) labelsToAdd.add('bug');
+            if (matches(/\b(new feature|enhancement|improvement|proposal|feature request)\b/i)) labelsToAdd.add('enhancement');
+            if (matches(/\b(question|how to|clarify|explain|how do i|help me|question about)\b/i)) labelsToAdd.add('question');
+            if (matches(/\b(documentation|docs?|readme|tutorial|wiki|typo|docstring)\b/i)) labelsToAdd.add('documentation');
+            if (matches(/\b(example|sample|demo|notebook)s?\b/i)) labelsToAdd.add('examples');
+            if (matches(/\b(datasets?|data loader|data augmentation|data preprocessing)\b/i)) labelsToAdd.add('dataset');
+            if (matches(/\b(mujoco|isaac|simulation|sim)\b/i)) labelsToAdd.add('simulation');
+            if (matches(/\b(train|training|optimizer|gradient|wandb|sac)\b/i)) labelsToAdd.add('training');
+            if (matches(/\b(rerun|plot|render|rendering|visualizer)/i)) labelsToAdd.add('visualization');
+            if (matches(/\b(cameras?|opencv|realsense|lidars?|sensors?|imus?|microphones?|rgbd|encoders?)\b/i)) labelsToAdd.add('sensors');
+            if (matches(/\b(urdf|actuators?|calibration|end-effector|kinematics)\b/i)) labelsToAdd.add('robots');
+            if (matches(/\b(teleop|teleoperator|controller|leader|follower|joystick|gamepad)\b/i)) labelsToAdd.add('teleoperators');
+            if (matches(/\b(policy|policies|model?)\b/i)) labelsToAdd.add('policies');
+            if (matches(/\b(processor|pipeline|preprocessor|postprocessor)s?\b/i)) labelsToAdd.add('processor');
+            if (matches(/\b(eval|evaluate|evaluation|metrics?|score|benchmarks?)\b/i)) labelsToAdd.add('evaluation');
+            if (matches(/\b(tests?|pytest|unittest|failing test)\b/i)) labelsToAdd.add('tests');
+            if (matches(/\b(ci|github actions?|github workflows?|gha|docker|pypi)\b/i)) labelsToAdd.add('CI');
+            if (matches(/\b(perf|latency|throughput|fps|speed|performance|slow|fast|slower|faster|memory usage)\b/i)) labelsToAdd.add('performance');
+            if (matches(/\b(dependency|dependencies|pip|install error|importerror|package not found|pyproject)\b/i)) labelsToAdd.add('dependencies');
+            if (matches(/\b(configuration|config|arguments?|input feature|dracuss)\b/i)) labelsToAdd.add('configuration');
+
+            // Apply Labels
+            const labels = Array.from(labelsToAdd).filter(Boolean);
+
+            if (labels.length > 0) {
+              console.log(`Adding labels: ${labels.join(', ')}`);
+              await github.rest.issues.addLabels({
+                owner: context.repo.owner,
+                repo: context.repo.repo,
+                issue_number: context.issue.number,
+                labels,
+              });
+            }
diff --git a/lerobot/.github/workflows/nightly.yml b/lerobot/.github/workflows/nightly.yml
new file mode 100644
index 0000000000000000000000000000000000000000..5bc86857a8022ec6554d243f5683b6089993f157
--- /dev/null
+++ b/lerobot/.github/workflows/nightly.yml
@@ -0,0 +1,212 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow handles nightly testing & docker images publishing.
+name: Nightly
+permissions:
+  contents: read
+
+on:
+  # Allows running this workflow manually from the Actions tab
+  workflow_dispatch:
+
+  # Runs at 02:00
+  schedule:
+    - cron: "0 2 * * *"
+
+# Sets up the environment variables
+env:
+  UV_VERSION: "0.8.0"
+  PYTHON_VERSION: "3.12"
+  DOCKER_IMAGE_NAME_CPU: huggingface/lerobot-cpu:latest
+  DOCKER_IMAGE_NAME_GPU: huggingface/lerobot-gpu:latest
+
+# Ensures that only the latest commit is built, canceling older runs.
+concurrency:
+  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
+  cancel-in-progress: true
+
+jobs:
+  # This job builds a CPU image for testing & distribution
+  build-docker-cpu-nightly:
+    name: Build CPU Docker for Nightly
+    runs-on:
+      group: aws-general-8-plus
+    if: github.repository == 'huggingface/lerobot'
+    outputs:
+      image_tag: ${{ env.DOCKER_IMAGE_NAME_CPU }}
+    steps:
+      - name: Install Git LFS
+        run: |
+          sudo apt-get update
+          sudo apt-get install git-lfs
+          git lfs install
+      - uses: actions/checkout@v6
+        with:
+          lfs: true
+          persist-credentials: false
+      - name: Set up Docker Buildx
+        uses: docker/setup-buildx-action@v3 # zizmor: ignore[unpinned-uses]
+        with:
+          cache-binary: false
+      - name: Login to Docker Hub
+        uses: docker/login-action@v3 # zizmor: ignore[unpinned-uses]
+        with:
+          username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+          password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+      - name: Build and push Docker image CPU
+        uses: docker/build-push-action@v6 # zizmor: ignore[unpinned-uses]
+        with:
+          context: .
+          file: ./docker/Dockerfile.user
+          push: true
+          tags: ${{ env.DOCKER_IMAGE_NAME_CPU }}
+
+  # This job builds a GPU image for testing & distribution
+  build-docker-gpu-nightly:
+    name: Build GPU Docker for Nightly
+    runs-on:
+      group: aws-general-8-plus
+    if: github.repository == 'huggingface/lerobot'
+    outputs:
+      image_tag: ${{ env.DOCKER_IMAGE_NAME_GPU }}
+    steps:
+      - name: Install Git LFS
+        run: |
+          sudo apt-get update
+          sudo apt-get install git-lfs
+          git lfs install
+      - uses: actions/checkout@v6
+        with:
+          lfs: true
+          persist-credentials: false
+      - name: Set up Docker Buildx
+        uses: docker/setup-buildx-action@v3 # zizmor: ignore[unpinned-uses]
+        with:
+          cache-binary: false
+      - name: Login to Docker Hub
+        uses: docker/login-action@v3 # zizmor: ignore[unpinned-uses]
+        with:
+          username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+          password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+      - name: Build and push Docker image GPU
+        uses: docker/build-push-action@v6 # zizmor: ignore[unpinned-uses]
+        with:
+          context: .
+          file: ./docker/Dockerfile.internal
+          push: true
+          tags: ${{ env.DOCKER_IMAGE_NAME_GPU }}
+
+  # This job runs the E2E tests + pytest with all extras in the CPU image
+  nightly-cpu-tests:
+    name: Nightly CPU Tests
+    needs: [build-docker-cpu-nightly]
+    runs-on:
+      group: aws-g6-4xlarge-plus
+    env:
+      HF_HOME: /home/user_lerobot/.cache/huggingface
+      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
+      TORCH_HOME: /home/user_lerobot/.cache/torch
+      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
+    container:
+      image: ${{ needs.build-docker-cpu-nightly.outputs.image_tag }} # zizmor: ignore[unpinned-images]
+      options: --shm-size "16gb"
+      credentials:
+        username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+        password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+    defaults:
+      run:
+        shell: bash
+        working-directory: /lerobot
+    steps:
+      - name: Login to Hugging Face
+        if: env.HF_USER_TOKEN != ''
+        run: |
+          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          hf auth whoami
+      - name: Run pytest on CPU
+        run: pytest tests -vv --maxfail=10
+      - name: Run end-to-end tests
+        run: make test-end-to-end
+
+  # This job runs the E2E tests + pytest with all extras in the GPU image
+  nightly-gpu-tests:
+    name: Nightly GPU Tests
+    needs: [build-docker-gpu-nightly]
+    runs-on:
+      group: aws-g6-4xlarge-plus
+    env:
+      HF_HOME: /home/user_lerobot/.cache/huggingface
+      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
+      TORCH_HOME: /home/user_lerobot/.cache/torch
+      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
+    container:
+      image: ${{ needs.build-docker-gpu-nightly.outputs.image_tag }} # zizmor: ignore[unpinned-images]
+      options: --gpus all --shm-size "16gb"
+      credentials:
+        username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+        password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+    defaults:
+      run:
+        shell: bash
+        working-directory: /lerobot
+    steps:
+      - name: Login to Hugging Face
+        if: env.HF_USER_TOKEN != ''
+        run: |
+          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          hf auth whoami
+      - name: Run pytest on GPU
+        run: pytest tests -vv --maxfail=10
+      - name: Run end-to-end tests
+        run: make test-end-to-end
+
+  # This job runs multi-GPU training tests with 4 GPUs
+  nightly-multi-gpu-tests:
+    name: Nightly Multi-GPU Tests
+    needs: [build-docker-gpu-nightly]
+    runs-on:
+      group: aws-g4dn-12xlarge  # Instance with 4 GPUs
+    env:
+      HF_HOME: /home/user_lerobot/.cache/huggingface
+      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
+      TORCH_HOME: /home/user_lerobot/.cache/torch
+      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
+      CUDA_VISIBLE_DEVICES: "0,1,2,3"
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
+    container:
+      image: ${{ needs.build-docker-gpu-nightly.outputs.image_tag }} # zizmor: ignore[unpinned-images]
+      options: --gpus all --shm-size "16gb"
+      credentials:
+        username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+        password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+    defaults:
+      run:
+        shell: bash
+        working-directory: /lerobot
+    steps:
+      - name: Login to Hugging Face
+        if: env.HF_USER_TOKEN != ''
+        run: |
+          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          hf auth whoami
+      - name: Verify GPU availability
+        run: |
+          nvidia-smi
+          python -c "import torch; print(f'PyTorch CUDA available: {torch.cuda.is_available()}'); print(f'Number of GPUs: {torch.cuda.device_count()}')"
+
+      - name: Run multi-GPU training tests
+        run: pytest -vv tests/training/
diff --git a/lerobot/.github/workflows/pr_labeler.yml b/lerobot/.github/workflows/pr_labeler.yml
new file mode 100644
index 0000000000000000000000000000000000000000..177c209596d933a660d03bf108de4e3397bbdf10
--- /dev/null
+++ b/lerobot/.github/workflows/pr_labeler.yml
@@ -0,0 +1,39 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow labels pull requests based on the files that were changed.
+name: Pull Request Labeler
+
+on:
+  # Allows labeling pull requests when they are opened or updated
+  # zizmor: ignore[dangerous-triggers] Needed to label PRs from forks
+  pull_request_target:
+    branches:
+      - main
+    types: [opened, synchronize, reopened, ready_for_review]
+
+permissions:
+  contents: read
+  pull-requests: write
+
+jobs:
+  triage:
+    name: Label PR
+    runs-on: ubuntu-latest
+    if: github.repository == 'huggingface/lerobot' && !github.event.pull_request.draft
+    steps:
+      - uses: actions/labeler@v6
+        with:
+          repo-token: ${{ secrets.GITHUB_TOKEN }}
+          sync-labels: true # Removes labels if files are removed from the PR
diff --git a/lerobot/.github/workflows/quality.yml b/lerobot/.github/workflows/quality.yml
new file mode 100644
index 0000000000000000000000000000000000000000..a84e9c17ed3771bbe57647bc3ce7f6632f731162
--- /dev/null
+++ b/lerobot/.github/workflows/quality.yml
@@ -0,0 +1,58 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow handles linting, formatting, and static analysis checks for the codebase.
+name: Quality
+permissions:
+  contents: read
+
+on:
+  # Allows running this workflow manually from the Actions tab
+  workflow_dispatch:
+
+  # Triggers the workflow on push events to main
+  push:
+    branches:
+      - main
+
+  # Triggers the workflow on pull request events targeting main
+  pull_request:
+    branches:
+      - main
+
+# Ensures that only the latest commit for a PR or branch is built, canceling older runs.
+concurrency:
+  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
+  cancel-in-progress: true
+
+jobs:
+  # This job runs pre-commit hooks to check code style and formatting.
+  pre-commit-checks:
+    name: Run Pre-commit Hooks (Lint, Format & Static Analysis)
+    runs-on: ubuntu-latest
+    steps:
+      - name: Checkout code
+        uses: actions/checkout@v6
+        with:
+          persist-credentials: false
+
+      - name: Set up Python
+        uses: actions/setup-python@v6
+        with:
+          python-version: '3.12'
+
+      - name: Run pre-commit hooks
+        uses: pre-commit/action@v3.0.1 # zizmor: ignore[unpinned-uses]
+        with:
+          extra_args: --all-files --show-diff-on-failure --color=always
diff --git a/lerobot/.github/workflows/release.yml b/lerobot/.github/workflows/release.yml
new file mode 100644
index 0000000000000000000000000000000000000000..f7bd2be6c565e2030ac297f7e382e76da1c45a3e
--- /dev/null
+++ b/lerobot/.github/workflows/release.yml
@@ -0,0 +1,171 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+name: Create Release and Publish to PyPI
+
+on:
+  push:
+    tags:
+      - 'v*.*.*' # Trigger on tags like v0.1.0, v1.0.0
+
+# Sets up the environment variables
+env:
+  UV_VERSION: "0.8.0"
+  PYTHON_VERSION: "3.12"
+
+jobs:
+  # This job builds the Python package and publishes it to PyPI
+  build-and-publish:
+    name: Build and publish Python distributions
+    runs-on: ubuntu-latest
+    if: github.repository == 'huggingface/lerobot'
+    outputs:
+      version: ${{ steps.extract_info.outputs.tag_version }}
+    permissions:
+      contents: write
+      id-token: write
+
+    steps:
+      - name: Checkout code
+        uses: actions/checkout@v6
+        with:
+          persist-credentials: false
+
+      - name: Set up Python
+        uses: actions/setup-python@v6
+        with:
+          python-version: '3.12'
+
+      - name: Extract Version
+        id: extract_info
+        # Extract version from tag (e.g., v0.1.0 -> 0.1.0)
+        # zizmor: ignore[template-injection]
+        run: |
+          VERSION=${{ github.ref_name }}
+          VERSION_NUMBER=${VERSION#v}
+          echo "tag_version=$VERSION_NUMBER" >> $GITHUB_OUTPUT
+      - name: Check if version matches pyproject.toml
+        if: startsWith(github.ref, 'refs/tags/v') && !contains(github.ref, '-')
+        # zizmor: ignore[template-injection]
+        run: |
+          TAG_VERSION=${{ steps.extract_info.outputs.tag_version }}
+
+          PYPROJECT_VERSION=$(grep '^version = ' pyproject.toml | awk -F' = ' '{print $2}' | tr -d '"')
+
+          if [[ "$TAG_VERSION" != "$PYPROJECT_VERSION" ]]; then
+            echo "Error: Tag version ($TAG_VERSION) does not match pyproject.toml version ($PYPROJECT_VERSION)." >&2
+            exit 1
+          else
+            echo "Tag version matches pyproject.toml version: $TAG_VERSION. Proceeding with release."
+          fi
+
+      - name: Check if version exists on PyPI
+      # zizmor: ignore[template-injection]
+        run: |
+          NEW_VERSION=${{ steps.extract_info.outputs.tag_version }}
+
+          response=$(curl -s "https://pypi.org/pypi/lerobot/$NEW_VERSION/json")
+          if echo "$response" | grep -q "message"; then
+            echo "Version $NEW_VERSION is available on PyPI. Proceeding with release."
+          else
+            echo "Error: Version $NEW_VERSION already exists on PyPI. Aborting."
+            exit 1
+          fi
+
+      - name: Install build dependencies
+        run: python -m pip install build
+
+      - name: Build package
+        run: python -m build
+
+      - name: Create GitHub Release
+        env:
+          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+        # zizmor: ignore[template-injection]
+        run: |
+          gh release create ${{ github.ref_name }} \
+            --title "Release ${{ github.ref_name }}" \
+            --generate-notes \
+            --draft=$([[ "${{ github.ref_name }}" == *-* ]] && echo true || echo false) \
+            --prerelease=$([[ "${{ github.ref_name }}" == *-* ]] && echo true || echo false) \
+            ./dist/*
+
+      - name: Publish to TestPyPI for pre-releases
+        # True for tags like 'v0.2.0-rc1'
+        if: startsWith(github.ref, 'refs/tags/v') && contains(github.ref, '-')
+        uses: pypa/gh-action-pypi-publish@v1.13.0 # zizmor: ignore[unpinned-uses, use-trusted-publishing]
+        with:
+          repository-url: https://test.pypi.org/legacy/
+          verbose: true
+          print-hash: true
+
+      - name: Publish to PyPI
+        if: startsWith(github.ref, 'refs/tags/v') && !contains(github.ref, '-')
+        uses: pypa/gh-action-pypi-publish@v1.13.0 # zizmor: ignore[unpinned-uses, use-trusted-publishing]
+        with:
+          verbose: true
+          print-hash: true
+
+  # This job runs end-to-end tests on the release
+  test-release:
+    name: Test Release
+    needs: [build-and-publish]
+    runs-on: ubuntu-latest
+    permissions:
+      contents: read
+    env:
+      MUJOCO_GL: egl
+    steps:
+      - uses: actions/checkout@v6
+        with:
+          lfs: true
+          persist-credentials: false
+      - name: Install apt dependencies
+        run: |
+          sudo apt-get update && sudo apt-get install -y build-essential \
+          git curl libglib2.0-0 libegl1-mesa-dev ffmpeg libusb-1.0-0-dev \
+          speech-dispatcher libgeos-dev portaudio19-dev
+      - name: Setup uv and Python
+        uses: astral-sh/setup-uv@v6 # zizmor: ignore[unpinned-uses]
+        with:
+          enable-cache: true # zizmor: ignore[cache-poisoning]
+          version: ${{ env.UV_VERSION }}
+          python-version: ${{ env.PYTHON_VERSION }}
+      - name: Create uv virtual environment
+        run: uv venv
+      - name: Install lerobot release
+        # zizmor: ignore[template-injection]
+        run: |
+          VERSION="${{ needs.build-and-publish.outputs.version }}"
+          if [[ "$VERSION" == *-* ]]; then
+            BASE_VERSION="${VERSION%%-*}"
+            echo "Installing pre-release version $BASE_VERSION from TestPyPI..."
+            uv pip install \
+              --index-url https://test.pypi.org/simple/ \
+              --extra-index-url https://pypi.org/simple \
+              --index-strategy unsafe-best-match \
+               "lerobot[all]==$BASE_VERSION"
+          else
+            echo "Installing release version $VERSION from PyPI..."
+            uv pip install "lerobot[all]==$VERSION"
+          fi
+      - name: Check lerobot version
+        run: uv run python -c "import lerobot; print(lerobot.__version__)"
+
+      - name: Run end-to-end tests
+        run: uv run make test-end-to-end
+
+
+# TODO(Steven): Publish draft/pre-release and to test pypi weekly
+# TODO(Steven): Separate build and publish job
diff --git a/lerobot/.github/workflows/security.yml b/lerobot/.github/workflows/security.yml
new file mode 100644
index 0000000000000000000000000000000000000000..50c0c1fc3df05b3dacdcc2b8896c2b98d68ab973
--- /dev/null
+++ b/lerobot/.github/workflows/security.yml
@@ -0,0 +1,54 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow handles secret scanning using TruffleHog to detect sensitive information in the codebase.
+name: Security
+permissions:
+  contents: read
+
+on:
+  # Allows running this workflow manually from the Actions tab
+  workflow_dispatch:
+
+  # Triggers the workflow on push events to main
+  push:
+    branches:
+      - main
+
+  # Triggers the workflow on pull request events targeting main
+  pull_request:
+    branches:
+      - main
+
+# Ensures that only the latest commit for a PR or branch is built, canceling older runs.
+concurrency:
+  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
+  cancel-in-progress: true
+
+jobs:
+  # This job runs TruffleHog to scan the full history of the repository for secrets.
+  trufflehog:
+    name: Secret Leaks Scan
+    runs-on: ubuntu-latest
+    steps:
+      - name: Checkout code
+        uses: actions/checkout@v6 # zizmor: ignore[unpinned-uses]
+        with:
+          fetch-depth: 0
+          persist-credentials: false
+
+      - name: Secret Scanning
+        uses: trufflesecurity/trufflehog@v3.90.0  # zizmor: ignore[unpinned-uses]
+        with:
+          extra_args: --only-verified
diff --git a/lerobot/.github/workflows/stale.yml b/lerobot/.github/workflows/stale.yml
new file mode 100644
index 0000000000000000000000000000000000000000..4dc119b5efe5448e88cdc9555db07284cbbf3689
--- /dev/null
+++ b/lerobot/.github/workflows/stale.yml
@@ -0,0 +1,71 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow handles closing stale issues and PRs.
+name: Stale
+on:
+  # Allows running this workflow manually from the Actions tab
+  workflow_dispatch:
+
+  # Runs at 02:00
+  schedule:
+    - cron: "0 2 * * *"
+
+env:
+  CLOSE_ISSUE_MESSAGE: >
+    This issue was closed because it has been stalled for 14 days with no activity.
+    Feel free to reopen if is still relevant, or to ping a collaborator if you have any questions.
+  CLOSE_PR_MESSAGE: >
+    This PR was closed because it has been stalled for 21 days with no activity.
+    Feel free to reopen if is still relevant, or to ping a collaborator if you have any questions.
+  WARN_ISSUE_MESSAGE: >
+    This issue has been automatically marked as stale because it has not had
+    recent activity (6 months). It will be closed if no further activity occurs.
+    Any change, comment or update to this issue will reset this count.
+    Thank you for your contributions.
+  WARN_PR_MESSAGE: >
+    This PR has been automatically marked as stale because it has not had
+    recent activity (1 year). It will be closed if no further activity occurs.
+    Any change, comment or update to this PR will reset this count.
+    Thank you for your contributions.
+
+jobs:
+  # This job runs the actions/stale action to close stale issues and PRs.
+  stale:
+    name: Close Stale Issues and PRs
+    runs-on: ubuntu-latest
+    if: github.repository == 'huggingface/lerobot'
+    permissions:
+      actions: write
+      contents: write # only for delete-branch option
+      issues: write
+      pull-requests: write
+    steps:
+      - uses: actions/stale@v10
+        with:
+          repo-token: ${{ secrets.GITHUB_TOKEN }}
+          stale-issue-label: stale
+          stale-pr-label: stale
+          exempt-issue-labels: never-stale
+          exempt-pr-labels: never-stale
+          days-before-issue-stale: 180
+          days-before-issue-close: 14
+          days-before-pr-stale: 365
+          days-before-pr-close: 21
+          delete-branch: true
+          close-issue-message: ${{ env.CLOSE_ISSUE_MESSAGE }}
+          close-pr-message: ${{ env.CLOSE_PR_MESSAGE }}
+          stale-issue-message: ${{ env.WARN_ISSUE_MESSAGE }}
+          stale-pr-message: ${{ env.WARN_PR_MESSAGE }}
+          operations-per-run: 500
diff --git a/lerobot/.github/workflows/unbound_deps_tests.yml b/lerobot/.github/workflows/unbound_deps_tests.yml
new file mode 100644
index 0000000000000000000000000000000000000000..404816c5204a48d88eebac4366140642ce2b813b
--- /dev/null
+++ b/lerobot/.github/workflows/unbound_deps_tests.yml
@@ -0,0 +1,207 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This workflow handles full testing with unboud dependencies versions.
+name: Unbound Dependency Tests
+
+on:
+  # Allows running this workflow manually from the Actions tab
+  workflow_dispatch:
+
+  # Run on the 1st and 15th of every month at 09:00 UTC
+  # schedule:
+  #  - cron: '0 2 1,15 * *'
+
+permissions:
+  contents: read
+
+# Sets up the environment variables
+env:
+  UV_VERSION: "0.8.0"
+  PYTHON_VERSION: "3.12"
+  DOCKER_IMAGE_NAME: huggingface/lerobot-gpu:unbound
+
+# Ensures that only the latest action is built, canceling older runs.
+concurrency:
+  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
+  cancel-in-progress: true
+
+jobs:
+
+  # This job runs the E2E tests + pytest with all unbound extras
+  full-tests:
+    name: Full Unbound Tests
+    runs-on: ubuntu-latest
+    if: github.repository == 'huggingface/lerobot'
+    env:
+      MUJOCO_GL: egl
+      HF_HOME: /mnt/cache/.cache/huggingface
+      HF_LEROBOT_HOME: /mnt/cache/.cache/huggingface/lerobot
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
+    steps:
+      - uses: actions/checkout@v6
+        with:
+          lfs: true
+          persist-credentials: false
+
+      # NOTE(Steven): Mount to `/mnt` to avoid the limited storage on `/home`. Consider cleaning default SDKs or using self-hosted runners for more space.
+      # (As of 2024-06-10, the runner's `/home` has only 6.2 GB free—8% of its 72 GB total.)
+      - name: Setup /mnt storage
+        run: sudo chown -R $USER:$USER /mnt
+
+      - name: Install apt dependencies
+        run: |
+          sudo apt-get update && sudo apt-get install -y build-essential \
+          git curl libglib2.0-0 libegl1-mesa-dev ffmpeg libusb-1.0-0-dev \
+          speech-dispatcher libgeos-dev portaudio19-dev
+
+      - name: Setup uv and Python
+        uses: astral-sh/setup-uv@v6 # zizmor: ignore[unpinned-uses]
+        with:
+          enable-cache: true
+          version: ${{ env.UV_VERSION }}
+          python-version: ${{ env.PYTHON_VERSION }}
+
+      - name: Unbound dependencies
+        run: |
+          sed -i 's/,[[:space:]]*<[0-9\.]*//g' pyproject.toml
+          echo "Dependencies unbound:" && cat pyproject.toml
+
+      - name: Install lerobot with all extras
+        run: uv sync --extra all # TODO(Steven): Make flash-attn optional
+      - name: Login to Hugging Face
+        if: env.HF_USER_TOKEN != ''
+        run: |
+          uv run hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          uv run hf auth whoami
+      - name: Run pytest (all extras)
+        run: uv run pytest tests -vv
+
+      - name: Run end-to-end tests
+        run: uv run make test-end-to-end
+
+  # This job builds a GPU enabled image for testing
+  build-and-push-docker:
+    name: Build and Push Docker
+    runs-on:
+      group: aws-general-8-plus
+    if: github.repository == 'huggingface/lerobot'
+    outputs:
+      image_tag: ${{ env.DOCKER_IMAGE_NAME }}
+    env:
+      GITHUB_REF: ${{ github.ref }}
+    steps:
+      - name: Install Git LFS
+        run: |
+          sudo apt-get update
+          sudo apt-get install git-lfs
+          git lfs install
+      - uses: actions/checkout@v6
+        with:
+          lfs: true
+          persist-credentials: false
+      - name: Set up Docker Buildx
+        uses: docker/setup-buildx-action@v3 # zizmor: ignore[unpinned-uses]
+        with:
+          cache-binary: false
+      - name: Login to Docker Hub
+        uses: docker/login-action@v3 # zizmor: ignore[unpinned-uses]
+        with:
+          username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+          password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+      - name: Build and push Docker image
+        uses: docker/build-push-action@v6 # zizmor: ignore[unpinned-uses]
+        with:
+          context: .
+          file: ./docker/Dockerfile.internal
+          push: true
+          tags: ${{ env.DOCKER_IMAGE_NAME }}
+          build-args: |
+            UNBOUND_DEPS=true
+
+  # This job runs pytest with all unbound extras in a GPU enabled host
+  # It runs everytime a test image is created
+  gpu-tests:
+    name: GPU Unbound Tests
+    needs: [build-and-push-docker]
+    runs-on:
+      group: aws-g6-4xlarge-plus
+    env:
+      HF_HOME: /home/user_lerobot/.cache/huggingface
+      HF_LEROBOT_HOME: /home/user_lerobot/.cache/huggingface/lerobot
+      TORCH_HOME: /home/user_lerobot/.cache/torch
+      TRITON_CACHE_DIR: /home/user_lerobot/.cache/triton
+      HF_USER_TOKEN: ${{ secrets.LEROBOT_HF_USER }}
+    container:
+      image: ${{ needs.build-and-push-docker.outputs.image_tag }} # zizmor: ignore[unpinned-images]
+      options: --gpus all --shm-size "16gb"
+      credentials:
+        username: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+        password: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+    defaults:
+      run:
+        shell: bash
+        working-directory: /lerobot
+    steps:
+      - name: Login to Hugging Face
+        if: env.HF_USER_TOKEN != ''
+        run: |
+          hf auth login --token "$HF_USER_TOKEN" --add-to-git-credential
+          hf auth whoami
+      - name: Run pytest on GPU
+        run: pytest tests -vv
+      - name: Run end-to-end tests
+        run: make test-end-to-end
+
+  # This job deletes the test image recently created
+  # It runs everytime after the gpu-tests have finished
+  delete-unbound-image:
+    name: Delete Unbound Image
+    needs: [gpu-tests, build-and-push-docker]
+    if: always() && needs.build-and-push-docker.result == 'success'
+    runs-on: ubuntu-latest
+    steps:
+      - name: Get Docker Hub Token and Delete Image
+        # zizmor: ignore[template-injection]
+        env:
+          DOCKERHUB_LEROBOT_USERNAME: ${{ secrets.DOCKERHUB_LEROBOT_USERNAME }}
+          DOCKERHUB_LEROBOT_PASSWORD: ${{ secrets.DOCKERHUB_LEROBOT_PASSWORD }}
+          IMAGE_FULL: ${{ needs.build-and-push-docker.outputs.image_tag }}
+        run: |
+          IMAGE_NAME=$(echo "$IMAGE_FULL" | cut -d':' -f1)
+          IMAGE_TAG=$(echo "$IMAGE_FULL" | cut -d':' -f2)
+
+          echo "Attempting to delete image: $IMAGE_NAME:$IMAGE_TAG"
+
+          TOKEN=$(curl -s -H "Content-Type: application/json" \
+                       -X POST \
+                       -d "{\"username\": \"$DOCKERHUB_LEROBOT_USERNAME\", \"password\": \"$DOCKERHUB_LEROBOT_PASSWORD\"}" \
+                       https://hub.docker.com/v2/users/login/ | jq -r .token)
+
+          if [ "$TOKEN" == "null" ] || [ -z "$TOKEN" ]; then
+            echo "::error::Failed to get Docker Hub token."
+            exit 1
+          fi
+
+          HTTP_RESPONSE=$(curl -s -o /dev/null -w "%{http_code}" \
+                               -H "Authorization: JWT ${TOKEN}" \
+                               -X DELETE \
+                               https://hub.docker.com/v2/repositories/${IMAGE_NAME}/tags/$IMAGE_TAG)
+
+          if [ "$HTTP_RESPONSE" -eq 204 ]; then
+            echo "Successfully deleted Docker image tag: $IMAGE_NAME:$IMAGE_TAG"
+          else
+            echo "::error::Failed to delete Docker image. HTTP status: $HTTP_RESPONSE"
+            exit 1
+          fi
diff --git a/lerobot/.gitignore b/lerobot/.gitignore
new file mode 100644
index 0000000000000000000000000000000000000000..b47e22cbfadc53e203d36f95d11b76b2666ea2ff
--- /dev/null
+++ b/lerobot/.gitignore
@@ -0,0 +1,179 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+### Environments & Dependencies ###
+.env
+.venv
+env/
+venv/
+env.bak/
+venv.bak/
+.python-version
+__pypackages__/
+node_modules/
+
+# Lock files
+poetry.lock
+uv.lock
+Pipfile.lock
+
+### Build & Distribution ###
+build/
+dist/
+sdist/
+wheels/
+downloads/
+eggs/
+.eggs/
+parts/
+var/
+pip-wheel-metadata/
+share/python-wheels/
+develop-eggs/
+*.egg-info/
+.installed.cfg
+*.egg
+MANIFEST
+lib/
+lib64/
+
+# PyInstaller
+*.manifest
+*.spec
+
+### Compiled & Cached Files ###
+__pycache__/
+*.py[cod]
+*$py.class
+*.so
+*.sage.py
+.cache/
+.ruff_cache/
+.mypy_cache/
+.pyre/
+.pytype/
+cython_debug/
+
+### Testing & Coverage ###
+htmlcov/
+.tox/
+.nox/
+.coverage
+.coverage.*
+.pytest_cache/
+.hypothesis/
+nosetests.xml
+coverage.xml
+*.cover
+*.py,cover
+!tests/artifacts
+
+### Logs & Temporary Files ###
+logs/
+tmp/
+*.log
+pip-log.txt
+pip-delete-this-directory.txt
+celerybeat-schedule
+celerybeat.pid
+
+### IDE & Editor Config ###
+# VS Code
+.vscode/
+.devcontainer/
+
+# JetBrains / PyCharm
+.idea/
+
+# Spyder
+.spyderproject
+.spyproject
+
+# Rope
+.ropeproject
+
+# Vim
+*.swp
+
+# Other
+*~
+
+### OS Specific ###
+# macOS
+.DS_Store
+
+# Windows
+Thumbs.db
+
+### Framework & Tool Specific ###
+
+.Python
+
+# Django
+local_settings.py
+db.sqlite3
+db.sqlite3-journal
+
+# Flask
+instance/
+.webassets-cache
+
+# Scrapy
+.scrapy
+
+# Jupyter
+.ipynb_checkpoints/
+profile_default/
+ipython_config.py
+
+# Sphinx
+docs/_build/
+
+# MkDocs
+/site
+
+# PyBuilder
+.pybuilder/
+target/
+
+# mypy
+.dmypy.json
+dmypy.json
+
+### HPC & Slurm ###
+nautilus/*.yaml
+*.key
+sbatch*.sh
+
+### Miscellaneous ###
+# W&B
+wandb/
+
+# Dev scripts
+.dev/
+
+# Data folders
+data/
+outputs/
+
+# Translations
+*.mo
+*.pot
+
+# Dev folders
+.cache/*
+*.stl
+*.urdf
+*.xml
+*.part
diff --git a/lerobot/.pre-commit-config.yaml b/lerobot/.pre-commit-config.yaml
new file mode 100644
index 0000000000000000000000000000000000000000..dff7416f41d157cd1be5e4bd35f1f68fc21f20f9
--- /dev/null
+++ b/lerobot/.pre-commit-config.yaml
@@ -0,0 +1,108 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+default_language_version:
+    python: python3.12
+
+exclude: "tests/artifacts/.*\\.safetensors$"
+
+repos:
+  ##### Meta #####
+  - repo: meta
+    hooks:
+      - id: check-useless-excludes
+      - id: check-hooks-apply
+
+   ##### General Code Quality & Formatting #####
+  - repo: https://github.com/pre-commit/pre-commit-hooks
+    rev: v6.0.0
+    hooks:
+      - id: check-added-large-files
+        args: ['--maxkb=1024']
+      - id: debug-statements
+      - id: check-merge-conflict
+      - id: check-case-conflict
+      - id: check-yaml
+      - id: check-toml
+      - id: end-of-file-fixer
+      - id: trailing-whitespace
+
+  - repo: https://github.com/astral-sh/ruff-pre-commit
+    rev: v0.14.1
+    hooks:
+      - id: ruff-format
+      - id: ruff
+        args: [--fix, --exit-non-zero-on-fix]
+
+  - repo: https://github.com/adhtruong/mirrors-typos
+    rev: v1.38.1
+    hooks:
+      - id: typos
+        args: [--force-exclude]
+
+  - repo: https://github.com/asottile/pyupgrade
+    rev: v3.21.0
+    hooks:
+    -   id: pyupgrade
+        args: [--py312-plus]
+
+  ##### Markdown Quality #####
+  - repo: https://github.com/rbubley/mirrors-prettier
+    rev: v3.6.2
+    hooks:
+      - id: prettier
+        name: Format Markdown with Prettier
+        types_or: [markdown, mdx]
+        args: [--prose-wrap=preserve]
+
+  ##### Security #####
+  - repo: https://github.com/gitleaks/gitleaks
+    rev: v8.28.0
+    hooks:
+      - id: gitleaks
+
+  - repo: https://github.com/woodruffw/zizmor-pre-commit
+    rev: v1.15.2
+    hooks:
+      - id: zizmor
+
+  - repo: https://github.com/PyCQA/bandit
+    rev: 1.8.6
+    hooks:
+    - id: bandit
+      args: ["-c", "pyproject.toml"]
+      additional_dependencies: ["bandit[toml]"]
+
+  # TODO(Steven): Uncomment when ready to use
+  ##### Static Analysis & Typing #####
+  - repo: https://github.com/pre-commit/mirrors-mypy
+    rev: v1.19.1
+    hooks:
+      - id: mypy
+        args: [--config-file=pyproject.toml]
+        exclude: ^(examples|benchmarks|tests)/
+
+  ##### Docstring Checks #####
+  # - repo: https://github.com/akaihola/darglint2
+  #   rev: v1.8.2
+  #   hooks:
+  #     - id: darglint2
+  #       args: ["--docstring-style", "google", "-v", "2"]
+  #       exclude: ^tests/.*$
+
+  # - repo: https://github.com/econchick/interrogate
+  #   rev: 1.7.0
+  #   hooks:
+  #     - id: interrogate
+  #       args: ["-vv", "--config=pyproject.toml"]
diff --git a/lerobot/AI_POLICY.md b/lerobot/AI_POLICY.md
new file mode 100644
index 0000000000000000000000000000000000000000..272ee8c120072bf326f129c9f48c390c0a7c9e2e
--- /dev/null
+++ b/lerobot/AI_POLICY.md
@@ -0,0 +1,25 @@
+# AI Usage Policy
+
+The LeRobot project welcomes contributions from everyone, and we have a few guidelines regarding AI usage to ensure high code quality, clear communication, and a healthy open-source ecosystem:
+
+- **Please disclose significant AI assistance.** If you used AI tools (e.g., Copilot, Claude, Cursor, ChatGPT) to generate a substantial portion of your code or text, let us know in your PR description. Transparency helps us review your changes more effectively.
+- **Own your code (The Human-in-the-Loop).** You must fully understand all the changes you are proposing. If you cannot explain what your AI-assisted code does or how it interacts with LeRobot's broader architecture, please take the time to learn and test it before submitting.
+- **Keep issues and discussions focused.** You are welcome to use AI to help draft issues or PR descriptions, but please review and edit them carefully before posting. AI can often be overly verbose; trimming the noise and getting straight to the point helps our maintainers address your needs faster.
+
+Our core maintainers also use AI tools to aid their workflows, but they do so while bringing deep contextual knowledge of the LeRobot codebase to validate the output. We ask all contributors to apply that same level of rigor.
+
+## Remember the Human Maintainers
+
+Please remember that LeRobot is maintained by a dedicated team of humans.
+
+Every discussion, issue, and pull request is read and reviewed by real people. While AI tools can generate thousands of lines of code in seconds, reviewing that code still takes human time and energy. Submitting unverified or low-effort AI output puts an unfair burden on our maintainers.
+
+Today, the quality of the AI output still heavily depends on the developer driving the tool. We ask that you respect our maintainers' time by thoroughly vetting, testing, and refining your submissions.
+
+## AI is Welcome Here
+
+LeRobot operates at the cutting edge of AI and robotics, and many of our maintainers actively embrace AI coding assistants as valuable productivity tools. We are a pro-AI project!
+
+Our reason for having an AI policy is not an anti-AI stance. Rather, it exists to ensure that AI is used to enhance human contributions, not replace them with unverified noise. It's about how the tools are used, not the tools themselves.
+
+We value the unique human insight you bring to the LeRobot community. Let AI empower your workflow, but always let your own judgment take the wheel.
diff --git a/lerobot/CODE_OF_CONDUCT.md b/lerobot/CODE_OF_CONDUCT.md
new file mode 100644
index 0000000000000000000000000000000000000000..305ffa276c5c42a15812cb2ddcfd2df3c992acf5
--- /dev/null
+++ b/lerobot/CODE_OF_CONDUCT.md
@@ -0,0 +1,132 @@
+# Contributor Covenant Code of Conduct
+
+## Our Pledge
+
+We as members, contributors, and leaders pledge to make participation in our
+community a harassment-free experience for everyone, regardless of age, body
+size, visible or invisible disability, ethnicity, sex characteristics, gender
+identity and expression, level of experience, education, socio-economic status,
+nationality, personal appearance, race, caste, color, religion, or sexual
+identity and orientation.
+
+We pledge to act and interact in ways that contribute to an open, welcoming,
+diverse, inclusive, and healthy community.
+
+## Our Standards
+
+Examples of behavior that contributes to a positive environment for our
+community include:
+
+- Demonstrating empathy and kindness toward other people
+- Being respectful of differing opinions, viewpoints, and experiences
+- Giving and gracefully accepting constructive feedback
+- Accepting responsibility and apologizing to those affected by our mistakes,
+  and learning from the experience
+- Focusing on what is best not just for us as individuals, but for the overall
+  community
+
+Examples of unacceptable behavior include:
+
+- The use of sexualized language or imagery, and sexual attention or advances of
+  any kind
+- Trolling, insulting or derogatory comments, and personal or political attacks
+- Public or private harassment
+- Publishing others' private information, such as a physical or email address,
+  without their explicit permission
+- Other conduct which could reasonably be considered inappropriate in a
+  professional setting
+
+## Enforcement Responsibilities
+
+Community leaders are responsible for clarifying and enforcing our standards of
+acceptable behavior and will take appropriate and fair corrective action in
+response to any behavior that they deem inappropriate, threatening, offensive,
+or harmful.
+
+Community leaders have the right and responsibility to remove, edit, or reject
+comments, commits, code, wiki edits, issues, and other contributions that are
+not aligned to this Code of Conduct, and will communicate reasons for moderation
+decisions when appropriate.
+
+## Scope
+
+This Code of Conduct applies within all community spaces, and also applies when
+an individual is officially representing the community in public spaces.
+Examples of representing our community include using an official e-mail address,
+posting via an official social media account, or acting as an appointed
+representative at an online or offline event.
+
+## Enforcement
+
+Instances of abusive, harassing, or otherwise unacceptable behavior may be
+reported to the community leaders responsible for enforcement at
+feedback@huggingface.co.
+All complaints will be reviewed and investigated promptly and fairly.
+
+All community leaders are obligated to respect the privacy and security of the
+reporter of any incident.
+
+## Enforcement Guidelines
+
+Community leaders will follow these Community Impact Guidelines in determining
+the consequences for any action they deem in violation of this Code of Conduct:
+
+### 1. Correction
+
+**Community Impact**: Use of inappropriate language or other behavior deemed
+unprofessional or unwelcome in the community.
+
+**Consequence**: A private, written warning from community leaders, providing
+clarity around the nature of the violation and an explanation of why the
+behavior was inappropriate. A public apology may be requested.
+
+### 2. Warning
+
+**Community Impact**: A violation through a single incident or series of
+actions.
+
+**Consequence**: A warning with consequences for continued behavior. No
+interaction with the people involved, including unsolicited interaction with
+those enforcing the Code of Conduct, for a specified period of time. This
+includes avoiding interactions in community spaces as well as external channels
+like social media. Violating these terms may lead to a temporary or permanent
+ban.
+
+### 3. Temporary Ban
+
+**Community Impact**: A serious violation of community standards, including
+sustained inappropriate behavior.
+
+**Consequence**: A temporary ban from any sort of interaction or public
+communication with the community for a specified period of time. No public or
+private interaction with the people involved, including unsolicited interaction
+with those enforcing the Code of Conduct, is allowed during this period.
+Violating these terms may lead to a permanent ban.
+
+### 4. Permanent Ban
+
+**Community Impact**: Demonstrating a pattern of violation of community
+standards, including sustained inappropriate behavior, harassment of an
+individual, or aggression toward or disparagement of classes of individuals.
+
+**Consequence**: A permanent ban from any sort of public interaction within the
+community.
+
+## Attribution
+
+This Code of Conduct is adapted from the [Contributor Covenant][homepage],
+version 2.1, available at
+[https://www.contributor-covenant.org/version/2/1/code_of_conduct.html][v2.1].
+
+Community Impact Guidelines were inspired by
+[Mozilla's code of conduct enforcement ladder][Mozilla CoC].
+
+For answers to common questions about this code of conduct, see the FAQ at
+[https://www.contributor-covenant.org/faq][FAQ]. Translations are available at
+[https://www.contributor-covenant.org/translations][translations].
+
+[homepage]: https://www.contributor-covenant.org
+[v2.1]: https://www.contributor-covenant.org/version/2/1/code_of_conduct.html
+[Mozilla CoC]: https://github.com/mozilla/diversity
+[FAQ]: https://www.contributor-covenant.org/faq
+[translations]: https://www.contributor-covenant.org/translations
diff --git a/lerobot/CONTRIBUTING.md b/lerobot/CONTRIBUTING.md
new file mode 100644
index 0000000000000000000000000000000000000000..60df93b27965a741448ce842a1d092d6fcdf9312
--- /dev/null
+++ b/lerobot/CONTRIBUTING.md
@@ -0,0 +1,83 @@
+# How to contribute to 🤗 LeRobot
+
+Everyone is welcome to contribute, and we value everybody's contribution. Code is not the only way to help the community. Answering questions, helping others, reaching out, and improving the documentation are immensely valuable.
+
+Whichever way you choose to contribute, please be mindful to respect our [code of conduct](https://github.com/huggingface/lerobot/blob/main/CODE_OF_CONDUCT.md) and our [AI policy](https://github.com/huggingface/lerobot/blob/main/AI_POLICY.md).
+
+## Ways to Contribute
+
+You can contribute in many ways:
+
+- **Fixing issues:** Resolve bugs or improve existing code.
+- **New features:** Develop new features.
+- **Extend:** Implement new models/policies, robots, or simulation environments and upload datasets to the Hugging Face Hub.
+- **Documentation:** Improve examples, guides, and docstrings.
+- **Feedback:** Submit tickets related to bugs or desired new features.
+
+If you are unsure where to start, join our [Discord Channel](https://discord.gg/q8Dzzpym3f).
+
+## Development Setup
+
+To contribute code, you need to set up a development environment.
+
+### 1. Fork and Clone
+
+Fork the repository on GitHub, then clone your fork:
+
+```bash
+git clone https://github.com/<your-handle>/lerobot.git
+cd lerobot
+git remote add upstream https://github.com/huggingface/lerobot.git
+```
+
+### 2. Environment Installation
+
+Please follow our [Installation Guide](https://huggingface.co/docs/lerobot/installation) for the environment setup & installation from source.
+
+## Running Tests & Quality Checks
+
+### Code Style (Pre-commit)
+
+Install `pre-commit` hooks to run checks automatically before you commit:
+
+```bash
+pre-commit install
+```
+
+To run checks manually on all files:
+
+```bash
+pre-commit run --all-files
+```
+
+### Running Tests
+
+We use `pytest`. First, ensure you have test artifacts by installing **git-lfs**:
+
+```bash
+git lfs install
+git lfs pull
+```
+
+Run the full suite (this may require extras installed):
+
+```bash
+pytest -sv ./tests
+```
+
+Or run a specific test file during development:
+
+```bash
+pytest -sv tests/test_specific_feature.py
+```
+
+## Submitting Issues & Pull Requests
+
+Use the templates for required fields and examples.
+
+- **Issues:** Follow the [ticket template](https://github.com/huggingface/lerobot/blob/main/.github/ISSUE_TEMPLATE/bug-report.yml).
+- **Pull requests:** Rebase on `upstream/main`, use a descriptive branch (don't work on `main`), run `pre-commit` and tests locally, and follow the [PR template](https://github.com/huggingface/lerobot/blob/main/.github/PULL_REQUEST_TEMPLATE.md).
+
+One member of the LeRobot team will then review your contribution.
+
+Thank you for contributing to LeRobot!
diff --git a/lerobot/LICENSE b/lerobot/LICENSE
new file mode 100644
index 0000000000000000000000000000000000000000..a603343cdda0228e53d8d9cdeb355f342f1e8f2c
--- /dev/null
+++ b/lerobot/LICENSE
@@ -0,0 +1,507 @@
+Copyright 2024 The Hugging Face team. All rights reserved.
+
+                                 Apache License
+                           Version 2.0, January 2004
+                        http://www.apache.org/licenses/
+
+   TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+
+   1. Definitions.
+
+      "License" shall mean the terms and conditions for use, reproduction,
+      and distribution as defined by Sections 1 through 9 of this document.
+
+      "Licensor" shall mean the copyright owner or entity authorized by
+      the copyright owner that is granting the License.
+
+      "Legal Entity" shall mean the union of the acting entity and all
+      other entities that control, are controlled by, or are under common
+      control with that entity. For the purposes of this definition,
+      "control" means (i) the power, direct or indirect, to cause the
+      direction or management of such entity, whether by contract or
+      otherwise, or (ii) ownership of fifty percent (50%) or more of the
+      outstanding shares, or (iii) beneficial ownership of such entity.
+
+      "You" (or "Your") shall mean an individual or Legal Entity
+      exercising permissions granted by this License.
+
+      "Source" form shall mean the preferred form for making modifications,
+      including but not limited to software source code, documentation
+      source, and configuration files.
+
+      "Object" form shall mean any form resulting from mechanical
+      transformation or translation of a Source form, including but
+      not limited to compiled object code, generated documentation,
+      and conversions to other media types.
+
+      "Work" shall mean the work of authorship, whether in Source or
+      Object form, made available under the License, as indicated by a
+      copyright notice that is included in or attached to the work
+      (an example is provided in the Appendix below).
+
+      "Derivative Works" shall mean any work, whether in Source or Object
+      form, that is based on (or derived from) the Work and for which the
+      editorial revisions, annotations, elaborations, or other modifications
+      represent, as a whole, an original work of authorship. For the purposes
+      of this License, Derivative Works shall not include works that remain
+      separable from, or merely link (or bind by name) to the interfaces of,
+      the Work and Derivative Works thereof.
+
+      "Contribution" shall mean any work of authorship, including
+      the original version of the Work and any modifications or additions
+      to that Work or Derivative Works thereof, that is intentionally
+      submitted to Licensor for inclusion in the Work by the copyright owner
+      or by an individual or Legal Entity authorized to submit on behalf of
+      the copyright owner. For the purposes of this definition, "submitted"
+      means any form of electronic, verbal, or written communication sent
+      to the Licensor or its representatives, including but not limited to
+      communication on electronic mailing lists, source code control systems,
+      and issue tracking systems that are managed by, or on behalf of, the
+      Licensor for the purpose of discussing and improving the Work, but
+      excluding communication that is conspicuously marked or otherwise
+      designated in writing by the copyright owner as "Not a Contribution."
+
+      "Contributor" shall mean Licensor and any individual or Legal Entity
+      on behalf of whom a Contribution has been received by Licensor and
+      subsequently incorporated within the Work.
+
+   2. Grant of Copyright License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      copyright license to reproduce, prepare Derivative Works of,
+      publicly display, publicly perform, sublicense, and distribute the
+      Work and such Derivative Works in Source or Object form.
+
+   3. Grant of Patent License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      (except as stated in this section) patent license to make, have made,
+      use, offer to sell, sell, import, and otherwise transfer the Work,
+      where such license applies only to those patent claims licensable
+      by such Contributor that are necessarily infringed by their
+      Contribution(s) alone or by combination of their Contribution(s)
+      with the Work to which such Contribution(s) was submitted. If You
+      institute patent litigation against any entity (including a
+      cross-claim or counterclaim in a lawsuit) alleging that the Work
+      or a Contribution incorporated within the Work constitutes direct
+      or contributory patent infringement, then any patent licenses
+      granted to You under this License for that Work shall terminate
+      as of the date such litigation is filed.
+
+   4. Redistribution. You may reproduce and distribute copies of the
+      Work or Derivative Works thereof in any medium, with or without
+      modifications, and in Source or Object form, provided that You
+      meet the following conditions:
+
+      (a) You must give any other recipients of the Work or
+          Derivative Works a copy of this License; and
+
+      (b) You must cause any modified files to carry prominent notices
+          stating that You changed the files; and
+
+      (c) You must retain, in the Source form of any Derivative Works
+          that You distribute, all copyright, patent, trademark, and
+          attribution notices from the Source form of the Work,
+          excluding those notices that do not pertain to any part of
+          the Derivative Works; and
+
+      (d) If the Work includes a "NOTICE" text file as part of its
+          distribution, then any Derivative Works that You distribute must
+          include a readable copy of the attribution notices contained
+          within such NOTICE file, excluding those notices that do not
+          pertain to any part of the Derivative Works, in at least one
+          of the following places: within a NOTICE text file distributed
+          as part of the Derivative Works; within the Source form or
+          documentation, if provided along with the Derivative Works; or,
+          within a display generated by the Derivative Works, if and
+          wherever such third-party notices normally appear. The contents
+          of the NOTICE file are for informational purposes only and
+          do not modify the License. You may add Your own attribution
+          notices within Derivative Works that You distribute, alongside
+          or as an addendum to the NOTICE text from the Work, provided
+          that such additional attribution notices cannot be construed
+          as modifying the License.
+
+      You may add Your own copyright statement to Your modifications and
+      may provide additional or different license terms and conditions
+      for use, reproduction, or distribution of Your modifications, or
+      for any such Derivative Works as a whole, provided Your use,
+      reproduction, and distribution of the Work otherwise complies with
+      the conditions stated in this License.
+
+   5. Submission of Contributions. Unless You explicitly state otherwise,
+      any Contribution intentionally submitted for inclusion in the Work
+      by You to the Licensor shall be under the terms and conditions of
+      this License, without any additional terms or conditions.
+      Notwithstanding the above, nothing herein shall supersede or modify
+      the terms of any separate license agreement you may have executed
+      with Licensor regarding such Contributions.
+
+   6. Trademarks. This License does not grant permission to use the trade
+      names, trademarks, service marks, or product names of the Licensor,
+      except as required for reasonable and customary use in describing the
+      origin of the Work and reproducing the content of the NOTICE file.
+
+   7. Disclaimer of Warranty. Unless required by applicable law or
+      agreed to in writing, Licensor provides the Work (and each
+      Contributor provides its Contributions) on an "AS IS" BASIS,
+      WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+      implied, including, without limitation, any warranties or conditions
+      of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+      PARTICULAR PURPOSE. You are solely responsible for determining the
+      appropriateness of using or redistributing the Work and assume any
+      risks associated with Your exercise of permissions under this License.
+
+   8. Limitation of Liability. In no event and under no legal theory,
+      whether in tort (including negligence), contract, or otherwise,
+      unless required by applicable law (such as deliberate and grossly
+      negligent acts) or agreed to in writing, shall any Contributor be
+      liable to You for damages, including any direct, indirect, special,
+      incidental, or consequential damages of any character arising as a
+      result of this License or out of the use or inability to use the
+      Work (including but not limited to damages for loss of goodwill,
+      work stoppage, computer failure or malfunction, or any and all
+      other commercial damages or losses), even if such Contributor
+      has been advised of the possibility of such damages.
+
+   9. Accepting Warranty or Additional Liability. While redistributing
+      the Work or Derivative Works thereof, You may choose to offer,
+      and charge a fee for, acceptance of support, warranty, indemnity,
+      or other liability obligations and/or rights consistent with this
+      License. However, in accepting such obligations, You may act only
+      on Your own behalf and on Your sole responsibility, not on behalf
+      of any other Contributor, and only if You agree to indemnify,
+      defend, and hold each Contributor harmless for any liability
+      incurred by, or claims asserted against, such Contributor by reason
+      of your accepting any such warranty or additional liability.
+
+   END OF TERMS AND CONDITIONS
+
+   APPENDIX: How to apply the Apache License to your work.
+
+      To apply the Apache License to your work, attach the following
+      boilerplate notice, with the fields enclosed by brackets "[]"
+      replaced with your own identifying information. (Don't include
+      the brackets!)  The text should be enclosed in the appropriate
+      comment syntax for the file format. We also recommend that a
+      file or class name and description of purpose be included on the
+      same "printed page" as the copyright notice for easier
+      identification within third-party archives.
+
+   Copyright [yyyy] [name of copyright owner]
+
+   Licensed under the Apache License, Version 2.0 (the "License");
+   you may not use this file except in compliance with the License.
+   You may obtain a copy of the License at
+
+       http://www.apache.org/licenses/LICENSE-2.0
+
+   Unless required by applicable law or agreed to in writing, software
+   distributed under the License is distributed on an "AS IS" BASIS,
+   WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+   See the License for the specific language governing permissions and
+   limitations under the License.
+
+
+## Some of lerobot's code is derived from Diffusion Policy, which is subject to the following copyright notice:
+
+MIT License
+
+Copyright (c) 2023 Columbia Artificial Intelligence and Robotics Lab
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.
+
+
+## Some of lerobot's code is derived from FOWM, which is subject to the following copyright notice:
+
+MIT License
+
+Copyright (c) 2023 Yunhai Feng
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.
+
+
+## Some of lerobot's code is derived from simxarm, which is subject to the following copyright notice:
+
+MIT License
+
+Copyright (c) 2023 Nicklas Hansen & Yanjie Ze
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.
+
+
+## Some of lerobot's code is derived from ALOHA, which is subject to the following copyright notice:
+
+MIT License
+
+Copyright (c) 2023 Tony Z. Zhao
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.
+
+## Some of lerobot's code is derived from DETR, which is subject to the following copyright notice:
+
+                                 Apache License
+                           Version 2.0, January 2004
+                        http://www.apache.org/licenses/
+
+   TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+
+   1. Definitions.
+
+      "License" shall mean the terms and conditions for use, reproduction,
+      and distribution as defined by Sections 1 through 9 of this document.
+
+      "Licensor" shall mean the copyright owner or entity authorized by
+      the copyright owner that is granting the License.
+
+      "Legal Entity" shall mean the union of the acting entity and all
+      other entities that control, are controlled by, or are under common
+      control with that entity. For the purposes of this definition,
+      "control" means (i) the power, direct or indirect, to cause the
+      direction or management of such entity, whether by contract or
+      otherwise, or (ii) ownership of fifty percent (50%) or more of the
+      outstanding shares, or (iii) beneficial ownership of such entity.
+
+      "You" (or "Your") shall mean an individual or Legal Entity
+      exercising permissions granted by this License.
+
+      "Source" form shall mean the preferred form for making modifications,
+      including but not limited to software source code, documentation
+      source, and configuration files.
+
+      "Object" form shall mean any form resulting from mechanical
+      transformation or translation of a Source form, including but
+      not limited to compiled object code, generated documentation,
+      and conversions to other media types.
+
+      "Work" shall mean the work of authorship, whether in Source or
+      Object form, made available under the License, as indicated by a
+      copyright notice that is included in or attached to the work
+      (an example is provided in the Appendix below).
+
+      "Derivative Works" shall mean any work, whether in Source or Object
+      form, that is based on (or derived from) the Work and for which the
+      editorial revisions, annotations, elaborations, or other modifications
+      represent, as a whole, an original work of authorship. For the purposes
+      of this License, Derivative Works shall not include works that remain
+      separable from, or merely link (or bind by name) to the interfaces of,
+      the Work and Derivative Works thereof.
+
+      "Contribution" shall mean any work of authorship, including
+      the original version of the Work and any modifications or additions
+      to that Work or Derivative Works thereof, that is intentionally
+      submitted to Licensor for inclusion in the Work by the copyright owner
+      or by an individual or Legal Entity authorized to submit on behalf of
+      the copyright owner. For the purposes of this definition, "submitted"
+      means any form of electronic, verbal, or written communication sent
+      to the Licensor or its representatives, including but not limited to
+      communication on electronic mailing lists, source code control systems,
+      and issue tracking systems that are managed by, or on behalf of, the
+      Licensor for the purpose of discussing and improving the Work, but
+      excluding communication that is conspicuously marked or otherwise
+      designated in writing by the copyright owner as "Not a Contribution."
+
+      "Contributor" shall mean Licensor and any individual or Legal Entity
+      on behalf of whom a Contribution has been received by Licensor and
+      subsequently incorporated within the Work.
+
+   2. Grant of Copyright License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      copyright license to reproduce, prepare Derivative Works of,
+      publicly display, publicly perform, sublicense, and distribute the
+      Work and such Derivative Works in Source or Object form.
+
+   3. Grant of Patent License. Subject to the terms and conditions of
+      this License, each Contributor hereby grants to You a perpetual,
+      worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+      (except as stated in this section) patent license to make, have made,
+      use, offer to sell, sell, import, and otherwise transfer the Work,
+      where such license applies only to those patent claims licensable
+      by such Contributor that are necessarily infringed by their
+      Contribution(s) alone or by combination of their Contribution(s)
+      with the Work to which such Contribution(s) was submitted. If You
+      institute patent litigation against any entity (including a
+      cross-claim or counterclaim in a lawsuit) alleging that the Work
+      or a Contribution incorporated within the Work constitutes direct
+      or contributory patent infringement, then any patent licenses
+      granted to You under this License for that Work shall terminate
+      as of the date such litigation is filed.
+
+   4. Redistribution. You may reproduce and distribute copies of the
+      Work or Derivative Works thereof in any medium, with or without
+      modifications, and in Source or Object form, provided that You
+      meet the following conditions:
+
+      (a) You must give any other recipients of the Work or
+          Derivative Works a copy of this License; and
+
+      (b) You must cause any modified files to carry prominent notices
+          stating that You changed the files; and
+
+      (c) You must retain, in the Source form of any Derivative Works
+          that You distribute, all copyright, patent, trademark, and
+          attribution notices from the Source form of the Work,
+          excluding those notices that do not pertain to any part of
+          the Derivative Works; and
+
+      (d) If the Work includes a "NOTICE" text file as part of its
+          distribution, then any Derivative Works that You distribute must
+          include a readable copy of the attribution notices contained
+          within such NOTICE file, excluding those notices that do not
+          pertain to any part of the Derivative Works, in at least one
+          of the following places: within a NOTICE text file distributed
+          as part of the Derivative Works; within the Source form or
+          documentation, if provided along with the Derivative Works; or,
+          within a display generated by the Derivative Works, if and
+          wherever such third-party notices normally appear. The contents
+          of the NOTICE file are for informational purposes only and
+          do not modify the License. You may add Your own attribution
+          notices within Derivative Works that You distribute, alongside
+          or as an addendum to the NOTICE text from the Work, provided
+          that such additional attribution notices cannot be construed
+          as modifying the License.
+
+      You may add Your own copyright statement to Your modifications and
+      may provide additional or different license terms and conditions
+      for use, reproduction, or distribution of Your modifications, or
+      for any such Derivative Works as a whole, provided Your use,
+      reproduction, and distribution of the Work otherwise complies with
+      the conditions stated in this License.
+
+   5. Submission of Contributions. Unless You explicitly state otherwise,
+      any Contribution intentionally submitted for inclusion in the Work
+      by You to the Licensor shall be under the terms and conditions of
+      this License, without any additional terms or conditions.
+      Notwithstanding the above, nothing herein shall supersede or modify
+      the terms of any separate license agreement you may have executed
+      with Licensor regarding such Contributions.
+
+   6. Trademarks. This License does not grant permission to use the trade
+      names, trademarks, service marks, or product names of the Licensor,
+      except as required for reasonable and customary use in describing the
+      origin of the Work and reproducing the content of the NOTICE file.
+
+   7. Disclaimer of Warranty. Unless required by applicable law or
+      agreed to in writing, Licensor provides the Work (and each
+      Contributor provides its Contributions) on an "AS IS" BASIS,
+      WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+      implied, including, without limitation, any warranties or conditions
+      of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+      PARTICULAR PURPOSE. You are solely responsible for determining the
+      appropriateness of using or redistributing the Work and assume any
+      risks associated with Your exercise of permissions under this License.
+
+   8. Limitation of Liability. In no event and under no legal theory,
+      whether in tort (including negligence), contract, or otherwise,
+      unless required by applicable law (such as deliberate and grossly
+      negligent acts) or agreed to in writing, shall any Contributor be
+      liable to You for damages, including any direct, indirect, special,
+      incidental, or consequential damages of any character arising as a
+      result of this License or out of the use or inability to use the
+      Work (including but not limited to damages for loss of goodwill,
+      work stoppage, computer failure or malfunction, or any and all
+      other commercial damages or losses), even if such Contributor
+      has been advised of the possibility of such damages.
+
+   9. Accepting Warranty or Additional Liability. While redistributing
+      the Work or Derivative Works thereof, You may choose to offer,
+      and charge a fee for, acceptance of support, warranty, indemnity,
+      or other liability obligations and/or rights consistent with this
+      License. However, in accepting such obligations, You may act only
+      on Your own behalf and on Your sole responsibility, not on behalf
+      of any other Contributor, and only if You agree to indemnify,
+      defend, and hold each Contributor harmless for any liability
+      incurred by, or claims asserted against, such Contributor by reason
+      of your accepting any such warranty or additional liability.
+
+   END OF TERMS AND CONDITIONS
+
+   APPENDIX: How to apply the Apache License to your work.
+
+      To apply the Apache License to your work, attach the following
+      boilerplate notice, with the fields enclosed by brackets "[]"
+      replaced with your own identifying information. (Don't include
+      the brackets!)  The text should be enclosed in the appropriate
+      comment syntax for the file format. We also recommend that a
+      file or class name and description of purpose be included on the
+      same "printed page" as the copyright notice for easier
+      identification within third-party archives.
+
+   Copyright 2020 - present, Facebook, Inc
+
+   Licensed under the Apache License, Version 2.0 (the "License");
+   you may not use this file except in compliance with the License.
+   You may obtain a copy of the License at
+
+       http://www.apache.org/licenses/LICENSE-2.0
+
+   Unless required by applicable law or agreed to in writing, software
+   distributed under the License is distributed on an "AS IS" BASIS,
+   WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+   See the License for the specific language governing permissions and
+   limitations under the License.
diff --git a/lerobot/MANIFEST.in b/lerobot/MANIFEST.in
new file mode 100644
index 0000000000000000000000000000000000000000..c1fce3b5a3b8a0848dce7a8ffe7c5d21a9e5d05f
--- /dev/null
+++ b/lerobot/MANIFEST.in
@@ -0,0 +1,3 @@
+include src/lerobot/templates/lerobot_modelcard_template.md
+include src/lerobot/datasets/card_template.md
+include src/lerobot/envs/metaworld_config.json
diff --git a/lerobot/Makefile b/lerobot/Makefile
new file mode 100644
index 0000000000000000000000000000000000000000..e02f024031b506b275ba58ae41b4080e05fd32e6
--- /dev/null
+++ b/lerobot/Makefile
@@ -0,0 +1,180 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+.PHONY: tests
+
+PYTHON_PATH := $(shell which python)
+
+# If uv is installed and a virtual environment exists, use it
+UV_CHECK := $(shell command -v uv)
+ifneq ($(UV_CHECK),)
+	PYTHON_PATH := $(shell .venv/bin/python)
+endif
+
+export PATH := $(dir $(PYTHON_PATH)):$(PATH)
+
+DEVICE ?= cpu
+
+build-user:
+	docker build -f docker/Dockerfile.user -t lerobot-user .
+
+build-internal:
+	docker build -f docker/Dockerfile.internal -t lerobot-internal .
+
+test-end-to-end:
+	${MAKE} DEVICE=$(DEVICE) test-act-ete-train
+	${MAKE} DEVICE=$(DEVICE) test-act-ete-train-resume
+	${MAKE} DEVICE=$(DEVICE) test-act-ete-eval
+	${MAKE} DEVICE=$(DEVICE) test-diffusion-ete-train
+	${MAKE} DEVICE=$(DEVICE) test-diffusion-ete-eval
+	${MAKE} DEVICE=$(DEVICE) test-tdmpc-ete-train
+	${MAKE} DEVICE=$(DEVICE) test-tdmpc-ete-eval
+	${MAKE} DEVICE=$(DEVICE) test-smolvla-ete-train
+	${MAKE} DEVICE=$(DEVICE) test-smolvla-ete-eval
+
+test-act-ete-train:
+	lerobot-train \
+		--policy.type=act \
+		--policy.dim_model=64 \
+		--policy.n_action_steps=20 \
+		--policy.chunk_size=20 \
+		--policy.device=$(DEVICE) \
+		--policy.push_to_hub=false \
+		--env.type=aloha \
+		--env.episode_length=5 \
+		--dataset.repo_id=lerobot/aloha_sim_transfer_cube_human \
+		--dataset.image_transforms.enable=true \
+		--dataset.episodes="[0]" \
+		--batch_size=2 \
+		--steps=4 \
+		--eval_freq=2 \
+		--eval.n_episodes=1 \
+		--eval.batch_size=1 \
+		--save_freq=2 \
+		--save_checkpoint=true \
+		--log_freq=1 \
+		--wandb.enable=false \
+		--output_dir=tests/outputs/act/
+
+test-act-ete-train-resume:
+	lerobot-train \
+		--config_path=tests/outputs/act/checkpoints/000002/pretrained_model/train_config.json \
+		--resume=true
+
+test-act-ete-eval:
+	lerobot-eval \
+		--policy.path=tests/outputs/act/checkpoints/000004/pretrained_model \
+		--policy.device=$(DEVICE) \
+		--env.type=aloha \
+		--env.episode_length=5 \
+		--eval.n_episodes=1 \
+		--eval.batch_size=1
+
+test-diffusion-ete-train:
+	lerobot-train \
+		--policy.type=diffusion \
+		--policy.down_dims='[64,128,256]' \
+		--policy.diffusion_step_embed_dim=32 \
+		--policy.num_inference_steps=10 \
+		--policy.device=$(DEVICE) \
+		--policy.push_to_hub=false \
+		--env.type=pusht \
+		--env.episode_length=5 \
+		--dataset.repo_id=lerobot/pusht \
+		--dataset.image_transforms.enable=true \
+		--dataset.episodes="[0]" \
+		--batch_size=2 \
+		--steps=2 \
+		--eval_freq=2 \
+		--eval.n_episodes=1 \
+		--eval.batch_size=1 \
+		--save_checkpoint=true \
+		--save_freq=2 \
+		--log_freq=1 \
+		--wandb.enable=false \
+		--output_dir=tests/outputs/diffusion/
+
+test-diffusion-ete-eval:
+	lerobot-eval \
+		--policy.path=tests/outputs/diffusion/checkpoints/000002/pretrained_model \
+		--policy.device=$(DEVICE) \
+		--env.type=pusht \
+		--env.episode_length=5 \
+		--eval.n_episodes=1 \
+		--eval.batch_size=1
+
+test-tdmpc-ete-train:
+	lerobot-train \
+		--policy.type=tdmpc \
+		--policy.device=$(DEVICE) \
+		--policy.push_to_hub=false \
+		--env.type=pusht \
+		--env.episode_length=5 \
+		--dataset.repo_id=lerobot/pusht_image \
+		--dataset.image_transforms.enable=true \
+		--dataset.episodes="[0]" \
+		--batch_size=2 \
+		--steps=2 \
+		--eval_freq=2 \
+		--eval.n_episodes=1 \
+		--eval.batch_size=1 \
+		--save_checkpoint=true \
+		--save_freq=2 \
+		--log_freq=1 \
+		--wandb.enable=false \
+		--output_dir=tests/outputs/tdmpc/
+
+test-tdmpc-ete-eval:
+	lerobot-eval \
+		--policy.path=tests/outputs/tdmpc/checkpoints/000002/pretrained_model \
+		--policy.device=$(DEVICE) \
+		--env.type=pusht \
+		--env.episode_length=5 \
+		--env.observation_height=96 \
+        --env.observation_width=96 \
+		--eval.n_episodes=1 \
+		--eval.batch_size=1
+
+
+test-smolvla-ete-train:
+	lerobot-train \
+		--policy.type=smolvla \
+		--policy.n_action_steps=20 \
+		--policy.chunk_size=20 \
+		--policy.device=$(DEVICE) \
+		--policy.push_to_hub=false \
+		--env.type=aloha \
+		--env.episode_length=5 \
+		--dataset.repo_id=lerobot/aloha_sim_transfer_cube_human \
+		--dataset.image_transforms.enable=true \
+		--dataset.episodes="[0]" \
+		--batch_size=2 \
+		--steps=4 \
+		--eval_freq=2 \
+		--eval.n_episodes=1 \
+		--eval.batch_size=1 \
+		--save_freq=2 \
+		--save_checkpoint=true \
+		--log_freq=1 \
+		--wandb.enable=false \
+		--output_dir=tests/outputs/smolvla/
+
+test-smolvla-ete-eval:
+	lerobot-eval \
+		--policy.path=tests/outputs/smolvla/checkpoints/000004/pretrained_model \
+		--policy.device=$(DEVICE) \
+		--env.type=aloha \
+		--env.episode_length=5 \
+		--eval.n_episodes=1 \
+		--eval.batch_size=1
diff --git a/lerobot/README.md b/lerobot/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..f58b337b30286ca5865e727659c0ba2114674a92
--- /dev/null
+++ b/lerobot/README.md
@@ -0,0 +1,176 @@
+<p align="center">
+  <img alt="LeRobot, Hugging Face Robotics Library" src="./media/readme/lerobot-logo-thumbnail.png" width="100%">
+</p>
+
+<div align="center">
+
+[![Tests](https://github.com/huggingface/lerobot/actions/workflows/nightly.yml/badge.svg?branch=main)](https://github.com/huggingface/lerobot/actions/workflows/nightly.yml?query=branch%3Amain)
+[![Python versions](https://img.shields.io/pypi/pyversions/lerobot)](https://www.python.org/downloads/)
+[![License](https://img.shields.io/badge/License-Apache%202.0-blue.svg)](https://github.com/huggingface/lerobot/blob/main/LICENSE)
+[![Status](https://img.shields.io/pypi/status/lerobot)](https://pypi.org/project/lerobot/)
+[![Version](https://img.shields.io/pypi/v/lerobot)](https://pypi.org/project/lerobot/)
+[![Contributor Covenant](https://img.shields.io/badge/Contributor%20Covenant-v2.1-ff69b4.svg)](https://github.com/huggingface/lerobot/blob/main/CODE_OF_CONDUCT.md)
+[![Discord](https://img.shields.io/badge/Discord-Join_Us-5865F2?style=flat&logo=discord&logoColor=white)](https://discord.gg/q8Dzzpym3f)
+
+</div>
+
+**LeRobot** aims to provide models, datasets, and tools for real-world robotics in PyTorch. The goal is to lower the barrier to entry so that everyone can contribute to and benefit from shared datasets and pretrained models.
+
+🤗 A hardware-agnostic, Python-native interface that standardizes control across diverse platforms, from low-cost arms (SO-100) to humanoids.
+
+🤗 A standardized, scalable LeRobotDataset format (Parquet + MP4 or images) hosted on the Hugging Face Hub, enabling efficient storage, streaming and visualization of massive robotic datasets.
+
+🤗 State-of-the-art policies that have been shown to transfer to the real-world ready for training and deployment.
+
+🤗 Comprehensive support for the open-source ecosystem to democratize physical AI.
+
+## Quick Start
+
+LeRobot can be installed directly from PyPI.
+
+```bash
+pip install lerobot
+lerobot-info
+```
+
+> [!IMPORTANT]
+> For detailed installation guide, please see the [Installation Documentation](https://huggingface.co/docs/lerobot/installation).
+
+## Robots & Control
+
+<div align="center">
+  <img src="./media/readme/robots_control_video.webp" width="640px" alt="Reachy 2 Demo">
+</div>
+
+LeRobot provides a unified `Robot` class interface that decouples control logic from hardware specifics. It supports a wide range of robots and teleoperation devices.
+
+```python
+from lerobot.robots.myrobot import MyRobot
+
+# Connect to a robot
+robot = MyRobot(config=...)
+robot.connect()
+
+# Read observation and send action
+obs = robot.get_observation()
+action = model.select_action(obs)
+robot.send_action(action)
+```
+
+**Supported Hardware:** SO100, LeKiwi, Koch, HopeJR, OMX, EarthRover, Reachy2, Gamepads, Keyboards, Phones, OpenARM, Unitree G1.
+
+While these devices are natively integrated into the LeRobot codebase, the library is designed to be extensible. You can easily implement the Robot interface to utilize LeRobot's data collection, training, and visualization tools for your own custom robot.
+
+For detailed hardware setup guides, see the [Hardware Documentation](https://huggingface.co/docs/lerobot/integrate_hardware).
+
+## LeRobot Dataset
+
+To solve the data fragmentation problem in robotics, we utilize the **LeRobotDataset** format.
+
+- **Structure:** Synchronized MP4 videos (or images) for vision and Parquet files for state/action data.
+- **HF Hub Integration:** Explore thousands of robotics datasets on the [Hugging Face Hub](https://huggingface.co/lerobot).
+- **Tools:** Seamlessly delete episodes, split by indices/fractions, add/remove features, and merge multiple datasets.
+
+```python
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+# Load a dataset from the Hub
+dataset = LeRobotDataset("lerobot/aloha_mobile_cabinet")
+
+# Access data (automatically handles video decoding)
+episode_index=0
+print(f"{dataset[episode_index]['action'].shape=}\n")
+```
+
+Learn more about it in the [LeRobotDataset Documentation](https://huggingface.co/docs/lerobot/lerobot-dataset-v3)
+
+## SoTA Models
+
+LeRobot implements state-of-the-art policies in pure PyTorch, covering Imitation Learning, Reinforcement Learning, and Vision-Language-Action (VLA) models, with more coming soon. It also provides you with the tools to instrument and inspect your training process.
+
+<p align="center">
+  <img alt="Gr00t Architecture" src="./media/readme/VLA_architecture.jpg" width="640px">
+</p>
+
+Training a policy is as simple as running a script configuration:
+
+```bash
+lerobot-train \
+  --policy=act \
+  --dataset.repo_id=lerobot/aloha_mobile_cabinet
+```
+
+| Category                   | Models                                                                                                                                                                                                       |
+| -------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
+| **Imitation Learning**     | [ACT](./docs/source/policy_act_README.md), [Diffusion](./docs/source/policy_diffusion_README.md), [VQ-BeT](./docs/source/policy_vqbet_README.md)                                                             |
+| **Reinforcement Learning** | [HIL-SERL](./docs/source/hilserl.mdx), [TDMPC](./docs/source/policy_tdmpc_README.md) & QC-FQL (coming soon)                                                                                                  |
+| **VLAs Models**            | [Pi0Fast](./docs/source/pi0fast.mdx), [Pi0.5](./docs/source/pi05.mdx), [GR00T N1.5](./docs/source/policy_groot_README.md), [SmolVLA](./docs/source/policy_smolvla_README.md), [XVLA](./docs/source/xvla.mdx) |
+
+Similarly to the hardware, you can easily implement your own policy & leverage LeRobot's data collection, training, and visualization tools, and share your model to the HF Hub
+
+For detailed policy setup guides, see the [Policy Documentation](https://huggingface.co/docs/lerobot/bring_your_own_policies).
+
+## Inference & Evaluation
+
+Evaluate your policies in simulation or on real hardware using the unified evaluation script. LeRobot supports standard benchmarks like **LIBERO**, **MetaWorld** and more to come.
+
+```bash
+# Evaluate a policy on the LIBERO benchmark
+lerobot-eval \
+  --policy.path=lerobot/pi0_libero_finetuned \
+  --env.type=libero \
+  --env.task=libero_object \
+  --eval.n_episodes=10
+```
+
+Learn how to implement your own simulation environment or benchmark and distribute it from the HF Hub by following the [EnvHub Documentation](https://huggingface.co/docs/lerobot/envhub)
+
+## Resources
+
+- **[Documentation](https://huggingface.co/docs/lerobot/index):** The complete guide to tutorials & API.
+- **[Chinese Tutorials: LeRobot+SO-ARM101中文教程-同济子豪兄](https://zihao-ai.feishu.cn/wiki/space/7589642043471924447)** Detailed doc for assembling, teleoperate, dataset, train, deploy. Verified by Seed Studio and 5 global hackathon players.
+- **[Discord](https://discord.gg/q8Dzzpym3f):** Join the `LeRobot` server to discuss with the community.
+- **[X](https://x.com/LeRobotHF):** Follow us on X to stay up-to-date with the latest developments.
+- **[Robot Learning Tutorial](https://huggingface.co/spaces/lerobot/robot-learning-tutorial):** A free, hands-on course to learn robot learning using LeRobot.
+
+## Citation
+
+If you use LeRobot in your project, please cite the GitHub repository to acknowledge the ongoing development and contributors:
+
+```bibtex
+@misc{cadene2024lerobot,
+    author = {Cadene, Remi and Alibert, Simon and Soare, Alexander and Gallouedec, Quentin and Zouitine, Adil and Palma, Steven and Kooijmans, Pepijn and Aractingi, Michel and Shukor, Mustafa and Aubakirova, Dana and Russi, Martino and Capuano, Francesco and Pascal, Caroline and Choghari, Jade and Moss, Jess and Wolf, Thomas},
+    title = {LeRobot: State-of-the-art Machine Learning for Real-World Robotics in Pytorch},
+    howpublished = "\url{https://github.com/huggingface/lerobot}",
+    year = {2024}
+}
+```
+
+If you are referencing our research or the academic paper, please also cite our ICLR publication:
+
+<details>
+<summary><b>ICLR 2026 Paper</b></summary>
+
+```bibtex
+@inproceedings{cadenelerobot,
+  title={LeRobot: An Open-Source Library for End-to-End Robot Learning},
+  author={Cadene, Remi and Alibert, Simon and Capuano, Francesco and Aractingi, Michel and Zouitine, Adil and Kooijmans, Pepijn and Choghari, Jade and Russi, Martino and Pascal, Caroline and Palma, Steven and Shukor, Mustafa and Moss, Jess and Soare, Alexander and Aubakirova, Dana and Lhoest, Quentin and Gallou\'edec, Quentin and Wolf, Thomas},
+  booktitle={The Fourteenth International Conference on Learning Representations},
+  year={2026},
+  url={https://arxiv.org/abs/2602.22818}
+}
+```
+
+</details>
+
+## Contribute
+
+We welcome contributions from everyone in the community! To get started, please read our [CONTRIBUTING.md](https://github.com/huggingface/lerobot/blob/main/CONTRIBUTING.md) guide. Whether you're adding a new feature, improving documentation, or fixing a bug, your help and feedback are invaluable. We're incredibly excited about the future of open-source robotics and can't wait to work with you on what's next—thank you for your support!
+
+<p align="center">
+  <img alt="SO101 Video" src="./media/readme/so100_video.webp" width="640px">
+</p>
+
+<div align="center">
+<sub>Built by the <a href="https://huggingface.co/lerobot">LeRobot</a> team at <a href="https://huggingface.co">Hugging Face</a> with ❤️</sub>
+</div>
diff --git a/lerobot/SECURITY.md b/lerobot/SECURITY.md
new file mode 100644
index 0000000000000000000000000000000000000000..cf58f6cdb6bcd220fae063409fba9b4d1a25d8d1
--- /dev/null
+++ b/lerobot/SECURITY.md
@@ -0,0 +1,48 @@
+# Security Policy
+
+## Project Status & Philosophy
+
+`lerobot` has so far been primarily a research and prototyping tool, which is why deployment security hasn’t been a strong focus until now. As `lerobot` continues to be adopted and deployed in production, we are paying much closer attention to these kinds of issues.
+
+Fortunately, being an open-source project, the community can also help by reporting and fixing vulnerabilities. We appreciate your efforts to responsibly disclose your findings and will make every effort to acknowledge your contributions.
+
+## Reporting a Vulnerability
+
+To report a security issue, please use the GitHub Security Advisory ["Report a Vulnerability"](https://github.com/huggingface/lerobot/security/advisories/new) tab.
+
+The `lerobot` team will send a response indicating the next steps in handling your report. After the initial reply to your report, the security team will keep you informed of the progress towards a fix and full announcement, and may ask for additional information or guidance.
+
+#### Hugging Face Security Team
+
+Since this project is part of the Hugging Face ecosystem, feel free to submit vulnerability reports directly to: **[security@huggingface.co](mailto:security@huggingface.co)**. Someone from the HF security team will review the report and recommend next steps.
+
+#### Open Source Disclosures
+
+If reporting a vulnerability specific to the open-source codebase (and not the underlying Hub infrastructure), you may also use [Huntr](https://huntr.com), a vulnerability disclosure program for open source software.
+
+## Supported Versions
+
+Currently, we treat `lerobot` as a rolling release. We prioritize security updates for the latest available version (`main` branch).
+
+| Version  | Supported |
+| -------- | --------- |
+| Latest   | ✅        |
+| < Latest | ❌        |
+
+## Secure Usage Guidelines
+
+`lerobot` is tightly coupled to the Hugging Face Hub for sharing data and pretrained policies. When downloading artifacts uploaded by others, you expose yourself to risks. Please read below for recommendations to keep your runtime and robot environment safe.
+
+### Remote Artefacts (Weights & Policies)
+
+Models and policies uploaded to the Hugging Face Hub come in different formats. We heavily recommend uploading and downloading models in the [`safetensors`](https://github.com/huggingface/safetensors) format.
+
+`safetensors` was developed specifically to prevent arbitrary code execution on your system, which is critical when running software on physical hardware/robots.
+
+To avoid loading models from unsafe formats (e.g., `pickle`), you should ensure you are prioritizing `safetensors` files.
+
+### Remote Code
+
+Some models or environments on the Hub may require `trust_remote_code=True` to run custom architecture code.
+
+Please **always** verify the content of the modeling files when using this argument. We recommend setting a specific `revision` (commit hash) when loading remote code to ensure you protect yourself from unverified updates to the repository.
diff --git a/lerobot/benchmarks/video/README.md b/lerobot/benchmarks/video/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..1feee69c46e66b3d156e8e92b344c9ccd9893dc0
--- /dev/null
+++ b/lerobot/benchmarks/video/README.md
@@ -0,0 +1,288 @@
+# Video benchmark
+
+## Questions
+
+What is the optimal trade-off between:
+
+- maximizing loading time with random access,
+- minimizing memory space on disk,
+- maximizing success rate of policies,
+- compatibility across devices/platforms for decoding videos (e.g. video players, web browsers).
+
+How to encode videos?
+
+- Which video codec (`-vcodec`) to use? h264, h265, AV1?
+- What pixel format to use (`-pix_fmt`)? `yuv444p` or `yuv420p`?
+- How much compression (`-crf`)? No compression with `0`, intermediate compression with `25` or extreme with `50+`?
+- Which frequency to chose for key frames (`-g`)? A key frame every `10` frames?
+
+How to decode videos?
+
+- Which `decoder`? `torchvision`, `torchaudio`, `ffmpegio`, `decord`, or `nvc`?
+- What scenarios to use for the requesting timestamps during benchmark? (`timestamps_mode`)
+
+## Variables
+
+**Image content & size**
+We don't expect the same optimal settings for a dataset of images from a simulation, or from real-world in an apartment, or in a factory, or outdoor, or with lots of moving objects in the scene, etc. Similarly, loading times might not vary linearly with the image size (resolution).
+For these reasons, we run this benchmark on four representative datasets:
+
+- `lerobot/pusht_image`: (96 x 96 pixels) simulation with simple geometric shapes, fixed camera.
+- `lerobot/aloha_mobile_shrimp_image`: (480 x 640 pixels) real-world indoor, moving camera.
+- `lerobot/paris_street`: (720 x 1280 pixels) real-world outdoor, moving camera.
+- `lerobot/kitchen`: (1080 x 1920 pixels) real-world indoor, fixed camera.
+
+Note: The datasets used for this benchmark need to be image datasets, not video datasets.
+
+**Data augmentations**
+We might revisit this benchmark and find better settings if we train our policies with various data augmentations to make them more robust (e.g. robust to color changes, compression, etc.).
+
+### Encoding parameters
+
+| parameter   | values                                                       |
+| ----------- | ------------------------------------------------------------ |
+| **vcodec**  | `libx264`, `libx265`, `libsvtav1`                            |
+| **pix_fmt** | `yuv444p`, `yuv420p`                                         |
+| **g**       | `1`, `2`, `3`, `4`, `5`, `6`, `10`, `15`, `20`, `40`, `None` |
+| **crf**     | `0`, `5`, `10`, `15`, `20`, `25`, `30`, `40`, `50`, `None`   |
+
+Note that `crf` value might be interpreted differently by various video codecs. In other words, the same value used with one codec doesn't necessarily translate into the same compression level with another codec. In fact, the default value (`None`) isn't the same amongst the different video codecs. Importantly, it is also the case for many other ffmpeg arguments like `g` which specifies the frequency of the key frames.
+
+For a comprehensive list and documentation of these parameters, see the ffmpeg documentation depending on the video codec used:
+
+- h264: https://trac.ffmpeg.org/wiki/Encode/H.264
+- h265: https://trac.ffmpeg.org/wiki/Encode/H.265
+- AV1: https://trac.ffmpeg.org/wiki/Encode/AV1
+
+### Decoding parameters
+
+**Decoder**
+We tested two video decoding backends from torchvision:
+
+- `pyav`
+- `video_reader` (requires to build torchvision from source)
+
+**Requested timestamps**
+Given the way video decoding works, once a keyframe has been loaded, the decoding of subsequent frames is fast.
+This of course is affected by the `-g` parameter during encoding, which specifies the frequency of the keyframes. Given our typical use cases in robotics policies which might request a few timestamps in different random places, we want to replicate these use cases with the following scenarios:
+
+- `1_frame`: 1 frame,
+- `2_frames`: 2 consecutive frames (e.g. `[t, t + 1 / fps]`),
+- `6_frames`: 6 consecutive frames (e.g. `[t + i / fps for i in range(6)]`)
+
+Note that this differs significantly from a typical use case like watching a movie, in which every frame is loaded sequentially from the beginning to the end and it's acceptable to have big values for `-g`.
+
+Additionally, because some policies might request single timestamps that are a few frames apart, we also have the following scenario:
+
+- `2_frames_4_space`: 2 frames with 4 consecutive frames of spacing in between (e.g `[t, t + 5 / fps]`),
+
+However, due to how video decoding is implemented with `pyav`, we don't have access to an accurate seek so in practice this scenario is essentially the same as `6_frames` since all 6 frames between `t` and `t + 5 / fps` will be decoded.
+
+## Metrics
+
+**Data compression ratio (lower is better)**
+`video_images_size_ratio` is the ratio of the memory space on disk taken by the encoded video over the memory space taken by the original images. For instance, `video_images_size_ratio=25%` means that the video takes 4 times less memory space on disk compared to the original images.
+
+**Loading time ratio (lower is better)**
+`video_images_load_time_ratio` is the ratio of the time it takes to decode frames from the video at a given timestamps over the time it takes to load the exact same original images. Lower is better. For instance, `video_images_load_time_ratio=200%` means that decoding from video is 2 times slower than loading the original images.
+
+**Average Mean Square Error (lower is better)**
+`avg_mse` is the average mean square error between each decoded frame and its corresponding original image over all requested timestamps, and also divided by the number of pixels in the image to be comparable when switching to different image sizes.
+
+**Average Peak Signal to Noise Ratio (higher is better)**
+`avg_psnr` measures the ratio between the maximum possible power of a signal and the power of corrupting noise that affects the fidelity of its representation. Higher PSNR indicates better quality.
+
+**Average Structural Similarity Index Measure (higher is better)**
+`avg_ssim` evaluates the perceived quality of images by comparing luminance, contrast, and structure. SSIM values range from -1 to 1, where 1 indicates perfect similarity.
+
+One aspect that can't be measured here with those metrics is the compatibility of the encoding across platforms, in particular on web browser, for visualization purposes.
+h264, h265 and AV1 are all commonly used codecs and should not pose an issue. However, the chroma subsampling (`pix_fmt`) format might affect compatibility:
+
+- `yuv420p` is more widely supported across various platforms, including web browsers.
+- `yuv444p` offers higher color fidelity but might not be supported as broadly.
+
+<!-- **Loss of a pretrained policy (higher is better)** (not available)
+`loss_pretrained` is the result of evaluating with the selected encoding/decoding settings a policy pretrained on original images. It is easier to understand than `avg_l2_error`.
+
+**Success rate after retraining (higher is better)** (not available)
+`success_rate` is the result of training and evaluating a policy with the selected encoding/decoding settings. It is the most difficult metric to get but also the very best. -->
+
+## How the benchmark works
+
+The benchmark evaluates both encoding and decoding of video frames on the first episode of each dataset.
+
+**Encoding:** for each `vcodec` and `pix_fmt` pair, we use a default value for `g` and `crf` upon which we change a single value (either `g` or `crf`) to one of the specified values (we don't test every combination of those as this would be computationally too heavy).
+This gives a unique set of encoding parameters which is used to encode the episode.
+
+**Decoding:** Then, for each of those unique encodings, we iterate through every combination of the decoding parameters `backend` and `timestamps_mode`. For each of them, we record the metrics of a number of samples (given by `--num-samples`). This is parallelized for efficiency and the number of processes can be controlled with `--num-workers`. Ideally, it's best to have a `--num-samples` that is divisible by `--num-workers`.
+
+Intermediate results saved for each `vcodec` and `pix_fmt` combination in csv tables.
+These are then all concatenated to a single table ready for analysis.
+
+## Caveats
+
+We tried to measure the most impactful parameters for both encoding and decoding. However, for computational reasons we can't test out every combination.
+
+Additional encoding parameters exist that are not included in this benchmark. In particular:
+
+- `-preset` which allows for selecting encoding presets. This represents a collection of options that will provide a certain encoding speed to compression ratio. By leaving this parameter unspecified, it is considered to be `medium` for libx264 and libx265 and `8` for libsvtav1.
+- `-tune` which allows to optimize the encoding for certain aspects (e.g. film quality, fast decoding, etc.).
+
+See the documentation mentioned above for more detailed info on these settings and for a more comprehensive list of other parameters.
+
+Similarly on the decoding side, other decoders exist but are not implemented in our current benchmark. To name a few:
+
+- `torchaudio`
+- `ffmpegio`
+- `decord`
+- `nvc`
+
+Note as well that since we are mostly interested in the performance at decoding time (also because encoding is done only once before uploading a dataset), we did not measure encoding times nor have any metrics regarding encoding.
+However, besides the necessity to build ffmpeg from source, encoding did not pose any issue and it didn't take a significant amount of time during this benchmark.
+
+## Install
+
+Building ffmpeg from source is required to include libx265 and libaom/libsvtav1 (av1) video codecs ([compilation guide](https://trac.ffmpeg.org/wiki/CompilationGuide/Ubuntu)).
+
+**Note:** While you still need to build torchvision with a conda-installed `ffmpeg<4.3` to use the `video_reader` decoder (as described in [#220](https://github.com/huggingface/lerobot/pull/220)), you also need another version which is custom-built with all the video codecs for encoding. For the script to then use that version, you can prepend the command above with `PATH="$HOME/bin:$PATH"`, which is where ffmpeg should be built.
+
+## Adding a video decoder
+
+Right now, we're only benchmarking the two video decoder available with torchvision: `pyav` and `video_reader`.
+You can easily add a new decoder to benchmark by adding it to this function in the script:
+
+```diff
+def decode_video_frames(
+    video_path: str,
+    timestamps: list[float],
+    tolerance_s: float,
+    backend: str,
+) -> torch.Tensor:
+    if backend in ["pyav", "video_reader"]:
+        return decode_video_frames_torchvision(
+            video_path, timestamps, tolerance_s, backend
+        )
++    elif backend == ["your_decoder"]:
++        return your_decoder_function(
++            video_path, timestamps, tolerance_s, backend
++        )
+    else:
+        raise NotImplementedError(backend)
+```
+
+## Example
+
+For a quick run, you can try these parameters:
+
+```bash
+python benchmark/video/run_video_benchmark.py \
+    --output-dir outputs/video_benchmark \
+    --repo-ids \
+        lerobot/pusht_image \
+        lerobot/aloha_mobile_shrimp_image \
+    --vcodec libx264 libx265 \
+    --pix-fmt yuv444p yuv420p \
+    --g 2 20 None \
+    --crf 10 40 None \
+    --timestamps-modes 1_frame 2_frames \
+    --backends pyav video_reader \
+    --num-samples 5 \
+    --num-workers 5 \
+    --save-frames 0
+```
+
+## Results
+
+### Reproduce
+
+We ran the benchmark with the following parameters:
+
+```bash
+# h264 and h265 encodings
+python benchmark/video/run_video_benchmark.py \
+    --output-dir outputs/video_benchmark \
+    --repo-ids \
+        lerobot/pusht_image \
+        lerobot/aloha_mobile_shrimp_image \
+        lerobot/paris_street \
+        lerobot/kitchen \
+    --vcodec libx264 libx265 \
+    --pix-fmt yuv444p yuv420p \
+    --g 1 2 3 4 5 6 10 15 20 40 None \
+    --crf 0 5 10 15 20 25 30 40 50 None \
+    --timestamps-modes 1_frame 2_frames 6_frames \
+    --backends pyav video_reader \
+    --num-samples 50 \
+    --num-workers 5 \
+    --save-frames 1
+
+# av1 encoding (only compatible with yuv420p and pyav decoder)
+python benchmark/video/run_video_benchmark.py \
+    --output-dir outputs/video_benchmark \
+    --repo-ids \
+        lerobot/pusht_image \
+        lerobot/aloha_mobile_shrimp_image \
+        lerobot/paris_street \
+        lerobot/kitchen \
+    --vcodec libsvtav1 \
+    --pix-fmt yuv420p \
+    --g 1 2 3 4 5 6 10 15 20 40 None \
+    --crf 0 5 10 15 20 25 30 40 50 None \
+    --timestamps-modes 1_frame 2_frames 6_frames \
+    --backends pyav \
+    --num-samples 50 \
+    --num-workers 5 \
+    --save-frames 1
+```
+
+The full results are available [here](https://docs.google.com/spreadsheets/d/1OYJB43Qu8fC26k_OyoMFgGBBKfQRCi4BIuYitQnq3sw/edit?usp=sharing)
+
+### Parameters selected for LeRobotDataset
+
+Considering these results, we chose what we think is the best set of encoding parameter:
+
+- vcodec: `libsvtav1`
+- pix-fmt: `yuv420p`
+- g: `2`
+- crf: `30`
+
+Since we're using av1 encoding, we're choosing the `pyav` decoder as `video_reader` does not support it (and `pyav` doesn't require a custom build of `torchvision`).
+
+### Summary
+
+These tables show the results for `g=2` and `crf=30`, using `timestamps-modes=6_frames` and `backend=pyav`
+
+| video_images_size_ratio           | vcodec     | pix_fmt |           |           |           |
+| --------------------------------- | ---------- | ------- | --------- | --------- | --------- |
+|                                   | libx264    |         | libx265   |           | libsvtav1 |
+| repo_id                           | yuv420p    | yuv444p | yuv420p   | yuv444p   | yuv420p   |
+| lerobot/pusht_image               | **16.97%** | 17.58%  | 18.57%    | 18.86%    | 22.06%    |
+| lerobot/aloha_mobile_shrimp_image | 2.14%      | 2.11%   | 1.38%     | **1.37%** | 5.59%     |
+| lerobot/paris_street              | 2.12%      | 2.13%   | **1.54%** | **1.54%** | 4.43%     |
+| lerobot/kitchen                   | 1.40%      | 1.39%   | **1.00%** | **1.00%** | 2.52%     |
+
+| video_images_load_time_ratio      | vcodec  | pix_fmt |          |         |           |
+| --------------------------------- | ------- | ------- | -------- | ------- | --------- |
+|                                   | libx264 |         | libx265  |         | libsvtav1 |
+| repo_id                           | yuv420p | yuv444p | yuv420p  | yuv444p | yuv420p   |
+| lerobot/pusht_image               | 6.45    | 5.19    | **1.90** | 2.12    | 2.47      |
+| lerobot/aloha_mobile_shrimp_image | 11.80   | 7.92    | 0.71     | 0.85    | **0.48**  |
+| lerobot/paris_street              | 2.21    | 2.05    | 0.36     | 0.49    | **0.30**  |
+| lerobot/kitchen                   | 1.46    | 1.46    | 0.28     | 0.51    | **0.26**  |
+
+|                                   |          | vcodec   | pix_fmt      |          |           |              |
+| --------------------------------- | -------- | -------- | ------------ | -------- | --------- | ------------ |
+|                                   |          | libx264  |              | libx265  |           | libsvtav1    |
+| repo_id                           | metric   | yuv420p  | yuv444p      | yuv420p  | yuv444p   | yuv420p      |
+| lerobot/pusht_image               | avg_mse  | 2.90E-04 | **2.03E-04** | 3.13E-04 | 2.29E-04  | 2.19E-04     |
+|                                   | avg_psnr | 35.44    | 37.07        | 35.49    | **37.30** | 37.20        |
+|                                   | avg_ssim | 98.28%   | **98.85%**   | 98.31%   | 98.84%    | 98.72%       |
+| lerobot/aloha_mobile_shrimp_image | avg_mse  | 2.76E-04 | 2.59E-04     | 3.17E-04 | 3.06E-04  | **1.30E-04** |
+|                                   | avg_psnr | 35.91    | 36.21        | 35.88    | 36.09     | **40.17**    |
+|                                   | avg_ssim | 95.19%   | 95.18%       | 95.00%   | 95.05%    | **97.73%**   |
+| lerobot/paris_street              | avg_mse  | 6.89E-04 | 6.70E-04     | 4.03E-03 | 4.02E-03  | **3.09E-04** |
+|                                   | avg_psnr | 33.48    | 33.68        | 32.05    | 32.15     | **35.40**    |
+|                                   | avg_ssim | 93.76%   | 93.75%       | 89.46%   | 89.46%    | **95.46%**   |
+| lerobot/kitchen                   | avg_mse  | 2.50E-04 | 2.24E-04     | 4.28E-04 | 4.18E-04  | **1.53E-04** |
+|                                   | avg_psnr | 36.73    | 37.33        | 36.56    | 36.75     | **39.12**    |
+|                                   | avg_ssim | 95.47%   | 95.58%       | 95.52%   | 95.53%    | **96.82%**   |
diff --git a/lerobot/benchmarks/video/run_video_benchmark.py b/lerobot/benchmarks/video/run_video_benchmark.py
new file mode 100644
index 0000000000000000000000000000000000000000..064a84b48e4055955ceae0caa948d0eef38fa690
--- /dev/null
+++ b/lerobot/benchmarks/video/run_video_benchmark.py
@@ -0,0 +1,488 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Assess the performance of video decoding in various configurations.
+
+This script will benchmark different video encoding and decoding parameters.
+See the provided README.md or run `python benchmark/video/run_video_benchmark.py --help` for usage info.
+"""
+
+import argparse
+import datetime as dt
+import itertools
+import random
+import shutil
+from collections import OrderedDict
+from concurrent.futures import ThreadPoolExecutor, as_completed
+from pathlib import Path
+from threading import Lock
+
+import einops
+import numpy as np
+import pandas as pd
+import PIL
+import torch
+from skimage.metrics import mean_squared_error, peak_signal_noise_ratio, structural_similarity
+from tqdm import tqdm
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.video_utils import (
+    decode_video_frames,
+    encode_video_frames,
+)
+from lerobot.utils.constants import OBS_IMAGE
+from lerobot.utils.utils import TimerManager
+
+BASE_ENCODING = OrderedDict(
+    [
+        ("vcodec", "libx264"),
+        ("pix_fmt", "yuv444p"),
+        ("g", 2),
+        ("crf", None),
+        # TODO(aliberts): Add fastdecode
+        # ("fastdecode", 0),
+    ]
+)
+
+
+# TODO(rcadene, aliberts): move to `utils.py` folder when we want to refactor
+def parse_int_or_none(value) -> int | None:
+    if value.lower() == "none":
+        return None
+    try:
+        return int(value)
+    except ValueError as e:
+        raise argparse.ArgumentTypeError(f"Invalid int or None: {value}") from e
+
+
+def check_datasets_formats(repo_ids: list) -> None:
+    for repo_id in repo_ids:
+        dataset = LeRobotDataset(repo_id)
+        if len(dataset.meta.video_keys) > 0:
+            raise ValueError(
+                f"Use only image dataset for running this benchmark. Video dataset provided: {repo_id}"
+            )
+
+
+def get_directory_size(directory: Path) -> int:
+    total_size = 0
+    for item in directory.rglob("*"):
+        if item.is_file():
+            total_size += item.stat().st_size
+    return total_size
+
+
+def load_original_frames(imgs_dir: Path, timestamps: list[float], fps: int) -> torch.Tensor:
+    frames = []
+    for ts in timestamps:
+        idx = int(ts * fps)
+        frame = PIL.Image.open(imgs_dir / f"frame-{idx:06d}.png")
+        frame = torch.from_numpy(np.array(frame))
+        frame = frame.type(torch.float32) / 255
+        frame = einops.rearrange(frame, "h w c -> c h w")
+        frames.append(frame)
+    return torch.stack(frames)
+
+
+def save_decoded_frames(
+    imgs_dir: Path, save_dir: Path, frames: torch.Tensor, timestamps: list[float], fps: int
+) -> None:
+    if save_dir.exists() and len(list(save_dir.glob("frame-*.png"))) == len(timestamps):
+        return
+
+    save_dir.mkdir(parents=True, exist_ok=True)
+    for i, ts in enumerate(timestamps):
+        idx = int(ts * fps)
+        frame_hwc = (frames[i].permute((1, 2, 0)) * 255).type(torch.uint8).cpu().numpy()
+        PIL.Image.fromarray(frame_hwc).save(save_dir / f"frame-{idx:06d}_decoded.png")
+        shutil.copyfile(imgs_dir / f"frame-{idx:06d}.png", save_dir / f"frame-{idx:06d}_original.png")
+
+
+def save_first_episode(imgs_dir: Path, dataset: LeRobotDataset) -> None:
+    episode_index = 0
+    ep_num_images = dataset.meta.episodes["length"][episode_index]
+    if imgs_dir.exists() and len(list(imgs_dir.glob("frame-*.png"))) == ep_num_images:
+        return
+
+    imgs_dir.mkdir(parents=True, exist_ok=True)
+    hf_dataset = dataset.hf_dataset.with_format(None)
+
+    # We only save images from the first camera
+    img_keys = [key for key in hf_dataset.features if key.startswith(OBS_IMAGE)]
+    imgs_dataset = hf_dataset.select_columns(img_keys[0])
+
+    for i, item in enumerate(
+        tqdm(imgs_dataset, desc=f"saving {dataset.repo_id} first episode images", leave=False)
+    ):
+        img = item[img_keys[0]]
+        img.save(str(imgs_dir / f"frame-{i:06d}.png"), quality=100)
+
+        if i >= ep_num_images - 1:
+            break
+
+
+def sample_timestamps(timestamps_mode: str, ep_num_images: int, fps: int) -> list[float]:
+    # Start at 5 to allow for 2_frames_4_space and 6_frames
+    idx = random.randint(5, ep_num_images - 1)
+    match timestamps_mode:
+        case "1_frame":
+            frame_indexes = [idx]
+        case "2_frames":
+            frame_indexes = [idx - 1, idx]
+        case "2_frames_4_space":
+            frame_indexes = [idx - 5, idx]
+        case "6_frames":
+            frame_indexes = [idx - i for i in range(6)][::-1]
+        case _:
+            raise ValueError(timestamps_mode)
+
+    return [idx / fps for idx in frame_indexes]
+
+
+def benchmark_decoding(
+    imgs_dir: Path,
+    video_path: Path,
+    timestamps_mode: str,
+    backend: str,
+    ep_num_images: int,
+    fps: int,
+    num_samples: int = 50,
+    num_workers: int = 4,
+    save_frames: bool = False,
+) -> dict:
+    def process_sample(sample: int, lock: Lock):
+        time_benchmark = TimerManager(log=False)
+        timestamps = sample_timestamps(timestamps_mode, ep_num_images, fps)
+        num_frames = len(timestamps)
+        result = {
+            "psnr_values": [],
+            "ssim_values": [],
+            "mse_values": [],
+        }
+
+        with time_benchmark, lock:
+            frames = decode_video_frames(video_path, timestamps=timestamps, tolerance_s=5e-1, backend=backend)
+        result["load_time_video_ms"] = (time_benchmark.last * 1000) / num_frames
+
+        with time_benchmark:
+            original_frames = load_original_frames(imgs_dir, timestamps, fps)
+        result["load_time_images_ms"] = (time_benchmark.last * 1000) / num_frames
+
+        frames_np, original_frames_np = frames.numpy(), original_frames.numpy()
+        for i in range(num_frames):
+            result["mse_values"].append(mean_squared_error(original_frames_np[i], frames_np[i]))
+            result["psnr_values"].append(
+                peak_signal_noise_ratio(original_frames_np[i], frames_np[i], data_range=1.0)
+            )
+            result["ssim_values"].append(
+                structural_similarity(original_frames_np[i], frames_np[i], data_range=1.0, channel_axis=0)
+            )
+
+        if save_frames and sample == 0:
+            save_dir = video_path.with_suffix("") / f"{timestamps_mode}_{backend}"
+            save_decoded_frames(imgs_dir, save_dir, frames, timestamps, fps)
+
+        return result
+
+    load_times_video_ms = []
+    load_times_images_ms = []
+    mse_values = []
+    psnr_values = []
+    ssim_values = []
+
+    # A sample is a single set of decoded frames specified by timestamps_mode (e.g. a single frame, 2 frames, etc.).
+    # For each sample, we record metrics (loading time and quality metrics) which are then averaged over all samples.
+    # As these samples are independent, we run them in parallel threads to speed up the benchmark.
+    # Use a single shared lock for all worker threads
+    shared_lock = Lock()
+    with ThreadPoolExecutor(max_workers=num_workers) as executor:
+        futures = [executor.submit(process_sample, i, shared_lock) for i in range(num_samples)]
+        for future in tqdm(as_completed(futures), total=num_samples, desc="samples", leave=False):
+            result = future.result()
+            load_times_video_ms.append(result["load_time_video_ms"])
+            load_times_images_ms.append(result["load_time_images_ms"])
+            psnr_values.extend(result["psnr_values"])
+            ssim_values.extend(result["ssim_values"])
+            mse_values.extend(result["mse_values"])
+
+    avg_load_time_video_ms = float(np.array(load_times_video_ms).mean())
+    avg_load_time_images_ms = float(np.array(load_times_images_ms).mean())
+    video_images_load_time_ratio = avg_load_time_video_ms / avg_load_time_images_ms
+
+    return {
+        "avg_load_time_video_ms": avg_load_time_video_ms,
+        "avg_load_time_images_ms": avg_load_time_images_ms,
+        "video_images_load_time_ratio": video_images_load_time_ratio,
+        "avg_mse": float(np.mean(mse_values)),
+        "avg_psnr": float(np.mean(psnr_values)),
+        "avg_ssim": float(np.mean(ssim_values)),
+    }
+
+
+def benchmark_encoding_decoding(
+    dataset: LeRobotDataset,
+    video_path: Path,
+    imgs_dir: Path,
+    encoding_cfg: dict,
+    decoding_cfg: dict,
+    num_samples: int,
+    num_workers: int,
+    save_frames: bool,
+    overwrite: bool = False,
+    seed: int = 1337,
+) -> list[dict]:
+    fps = dataset.fps
+
+    if overwrite or not video_path.is_file():
+        tqdm.write(f"encoding {video_path}")
+        encode_video_frames(
+            imgs_dir=imgs_dir,
+            video_path=video_path,
+            fps=fps,
+            vcodec=encoding_cfg["vcodec"],
+            pix_fmt=encoding_cfg["pix_fmt"],
+            g=encoding_cfg.get("g"),
+            crf=encoding_cfg.get("crf"),
+            # fast_decode=encoding_cfg.get("fastdecode"),
+            overwrite=True,
+        )
+
+    episode_index = 0
+    ep_num_images = dataset.meta.episodes["length"][episode_index]
+    width, height = tuple(dataset[0][dataset.meta.camera_keys[0]].shape[-2:])
+    num_pixels = width * height
+    video_size_bytes = video_path.stat().st_size
+    images_size_bytes = get_directory_size(imgs_dir)
+    video_images_size_ratio = video_size_bytes / images_size_bytes
+
+    random.seed(seed)
+    benchmark_table = []
+    for timestamps_mode in tqdm(
+        decoding_cfg["timestamps_modes"], desc="decodings (timestamps_modes)", leave=False
+    ):
+        for backend in tqdm(decoding_cfg["backends"], desc="decodings (backends)", leave=False):
+            benchmark_row = benchmark_decoding(
+                imgs_dir,
+                video_path,
+                timestamps_mode,
+                backend,
+                ep_num_images,
+                fps,
+                num_samples,
+                num_workers,
+                save_frames,
+            )
+            benchmark_row.update(
+                **{
+                    "repo_id": dataset.repo_id,
+                    "resolution": f"{width} x {height}",
+                    "num_pixels": num_pixels,
+                    "video_size_bytes": video_size_bytes,
+                    "images_size_bytes": images_size_bytes,
+                    "video_images_size_ratio": video_images_size_ratio,
+                    "timestamps_mode": timestamps_mode,
+                    "backend": backend,
+                },
+                **encoding_cfg,
+            )
+            benchmark_table.append(benchmark_row)
+
+    return benchmark_table
+
+
+def main(
+    output_dir: Path,
+    repo_ids: list[str],
+    vcodec: list[str],
+    pix_fmt: list[str],
+    g: list[int],
+    crf: list[int],
+    # fastdecode: list[int],
+    timestamps_modes: list[str],
+    backends: list[str],
+    num_samples: int,
+    num_workers: int,
+    save_frames: bool,
+):
+    check_datasets_formats(repo_ids)
+    encoding_benchmarks = {
+        "g": g,
+        "crf": crf,
+        # "fastdecode": fastdecode,
+    }
+    decoding_benchmarks = {
+        "timestamps_modes": timestamps_modes,
+        "backends": backends,
+    }
+    headers = ["repo_id", "resolution", "num_pixels"]
+    headers += list(BASE_ENCODING.keys())
+    headers += [
+        "timestamps_mode",
+        "backend",
+        "video_size_bytes",
+        "images_size_bytes",
+        "video_images_size_ratio",
+        "avg_load_time_video_ms",
+        "avg_load_time_images_ms",
+        "video_images_load_time_ratio",
+        "avg_mse",
+        "avg_psnr",
+        "avg_ssim",
+    ]
+    file_paths = []
+    for video_codec in tqdm(vcodec, desc="encodings (vcodec)"):
+        for pixel_format in tqdm(pix_fmt, desc="encodings (pix_fmt)", leave=False):
+            benchmark_table = []
+            for repo_id in tqdm(repo_ids, desc="encodings (datasets)", leave=False):
+                dataset = LeRobotDataset(repo_id)
+                imgs_dir = output_dir / "images" / dataset.repo_id.replace("/", "_")
+                # We only use the first episode
+                save_first_episode(imgs_dir, dataset)
+                for duet in [
+                    dict(zip(encoding_benchmarks.keys(), unique_combination, strict=False))
+                    for unique_combination in itertools.product(*encoding_benchmarks.values())
+                ]:
+                    encoding_cfg = BASE_ENCODING.copy()
+                    encoding_cfg["vcodec"] = video_codec
+                    encoding_cfg["pix_fmt"] = pixel_format
+                    for key, value in duet.items():
+                        encoding_cfg[key] = value
+                    args_path = Path("_".join(str(value) for value in encoding_cfg.values()))
+                    video_path = output_dir / "videos" / args_path / f"{repo_id.replace('/', '_')}.mp4"
+                    benchmark_table += benchmark_encoding_decoding(
+                        dataset,
+                        video_path,
+                        imgs_dir,
+                        encoding_cfg,
+                        decoding_benchmarks,
+                        num_samples,
+                        num_workers,
+                        save_frames,
+                    )
+
+            # Save intermediate results
+            benchmark_df = pd.DataFrame(benchmark_table, columns=headers)
+            now = dt.datetime.now()
+            csv_path = (
+                output_dir
+                / f"{now:%Y-%m-%d}_{now:%H-%M-%S}_{video_codec}_{pixel_format}_{num_samples}-samples.csv"
+            )
+            benchmark_df.to_csv(csv_path, header=True, index=False)
+            file_paths.append(csv_path)
+            del benchmark_df
+
+    # Concatenate all results
+    df_list = [pd.read_csv(csv_path) for csv_path in file_paths]
+    concatenated_df = pd.concat(df_list, ignore_index=True)
+    concatenated_path = output_dir / f"{now:%Y-%m-%d}_{now:%H-%M-%S}_all_{num_samples}-samples.csv"
+    concatenated_df.to_csv(concatenated_path, header=True, index=False)
+
+
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser()
+    parser.add_argument(
+        "--output-dir",
+        type=Path,
+        default=Path("outputs/video_benchmark"),
+        help="Directory where the video benchmark outputs are written.",
+    )
+    parser.add_argument(
+        "--repo-ids",
+        type=str,
+        nargs="*",
+        default=[
+            "lerobot/pusht_image",
+            "lerobot/aloha_mobile_shrimp_image",
+            "lerobot/paris_street",
+            "lerobot/kitchen",
+        ],
+        help="Datasets repo-ids to test against. First episodes only are used. Must be images.",
+    )
+    parser.add_argument(
+        "--vcodec",
+        type=str,
+        nargs="*",
+        default=["h264", "hevc", "libsvtav1"],
+        help="Video codecs to be tested",
+    )
+    parser.add_argument(
+        "--pix-fmt",
+        type=str,
+        nargs="*",
+        default=["yuv444p", "yuv420p"],
+        help="Pixel formats (chroma subsampling) to be tested",
+    )
+    parser.add_argument(
+        "--g",
+        type=parse_int_or_none,
+        nargs="*",
+        default=[1, 2, 3, 4, 5, 6, 10, 15, 20, 40, 100, None],
+        help="Group of pictures sizes to be tested.",
+    )
+    parser.add_argument(
+        "--crf",
+        type=parse_int_or_none,
+        nargs="*",
+        default=[0, 5, 10, 15, 20, 25, 30, 40, 50, None],
+        help="Constant rate factors to be tested.",
+    )
+    # parser.add_argument(
+    #     "--fastdecode",
+    #     type=int,
+    #     nargs="*",
+    #     default=[0, 1],
+    #     help="Use the fastdecode tuning option. 0 disables it. "
+    #         "For libx264 and libx265/hevc, only 1 is possible. "
+    #         "For libsvtav1, 1, 2 or 3 are possible values with a higher number meaning a faster decoding optimization",
+    # )
+    parser.add_argument(
+        "--timestamps-modes",
+        type=str,
+        nargs="*",
+        default=[
+            "1_frame",
+            "2_frames",
+            "2_frames_4_space",
+            "6_frames",
+        ],
+        help="Timestamps scenarios to be tested.",
+    )
+    parser.add_argument(
+        "--backends",
+        type=str,
+        nargs="*",
+        default=["torchcodec", "pyav"],
+        help="Torchvision decoding backend to be tested.",
+    )
+    parser.add_argument(
+        "--num-samples",
+        type=int,
+        default=50,
+        help="Number of samples for each encoding x decoding config.",
+    )
+    parser.add_argument(
+        "--num-workers",
+        type=int,
+        default=10,
+        help="Number of processes for parallelized sample processing.",
+    )
+    parser.add_argument(
+        "--save-frames",
+        type=int,
+        default=0,
+        help="Whether to save decoded frames or not. Enter a non-zero number for true.",
+    )
+    args = parser.parse_args()
+    main(**vars(args))
diff --git a/lerobot/docker/Dockerfile.internal b/lerobot/docker/Dockerfile.internal
new file mode 100644
index 0000000000000000000000000000000000000000..b385fc51c8d3e3e81181709dfd3b053260c2fd31
--- /dev/null
+++ b/lerobot/docker/Dockerfile.internal
@@ -0,0 +1,95 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This Dockerfile is designed for HuggingFace internal CI environments
+# that require GPU access. It starts from an NVIDIA CUDA base image.
+
+# docker build -f docker/Dockerfile.internal -t lerobot-internal .
+
+# Configure the base image for CI with GPU access
+# TODO(Steven): Bump these versions
+ARG CUDA_VERSION=12.4.1
+ARG OS_VERSION=22.04
+FROM nvidia/cuda:${CUDA_VERSION}-base-ubuntu${OS_VERSION}
+
+# Define Python version argument
+ARG PYTHON_VERSION=3.12
+
+# Configure environment variables
+ENV DEBIAN_FRONTEND=noninteractive \
+    MUJOCO_GL=egl \
+    PATH=/lerobot/.venv/bin:$PATH \
+    CUDA_VISIBLE_DEVICES=0 \
+    TEST_TYPE=single_gpu \
+    DEVICE=cuda
+
+# Install Python, system dependencies, and uv (as root)
+RUN apt-get update && apt-get install -y --no-install-recommends \
+    software-properties-common build-essential git curl \
+    libglib2.0-0 libgl1-mesa-glx libegl1-mesa ffmpeg \
+    libusb-1.0-0-dev speech-dispatcher libgeos-dev portaudio19-dev \
+    cmake pkg-config ninja-build \
+    && add-apt-repository -y ppa:deadsnakes/ppa \
+    && apt-get update \
+    && apt-get install -y --no-install-recommends \
+       python${PYTHON_VERSION} \
+       python${PYTHON_VERSION}-venv \
+       python${PYTHON_VERSION}-dev \
+    && curl -LsSf https://astral.sh/uv/install.sh | sh \
+    && mv /root/.local/bin/uv /usr/local/bin/uv \
+    && useradd --create-home --shell /bin/bash user_lerobot \
+    && usermod -aG sudo user_lerobot \
+    && apt-get clean && rm -rf /var/lib/apt/lists/*
+
+# Create application directory and set permissions
+WORKDIR /lerobot
+RUN chown -R user_lerobot:user_lerobot /lerobot
+
+# Switch to the non-root user
+USER user_lerobot
+
+# Environment variables for the testing
+ENV HOME=/home/user_lerobot \
+    HF_HOME=/home/user_lerobot/.cache/huggingface \
+    HF_LEROBOT_HOME=/home/user_lerobot/.cache/huggingface/lerobot \
+    TORCH_HOME=/home/user_lerobot/.cache/torch \
+    TRITON_CACHE_DIR=/home/user_lerobot/.cache/triton
+
+# Create the virtual environment
+# We use a virtual environment inside the container—even though the container itself \
+# provides isolation—to ensure compatibility with the cluster and to prevent \
+# issues with MuJoCo and OpenGL drivers.
+RUN uv venv --python python${PYTHON_VERSION}
+
+# Install Python dependencies for caching
+COPY --chown=user_lerobot:user_lerobot setup.py pyproject.toml README.md MANIFEST.in ./
+COPY --chown=user_lerobot:user_lerobot src/ src/
+
+ARG UNBOUND_DEPS=false
+
+RUN if [ "$UNBOUND_DEPS" = "true" ]; then \
+    sed -i 's/,[[:space:]]*<[0-9\.]*//g' pyproject.toml; \
+    echo "Dependencies unbound:" && cat pyproject.toml; \
+    fi
+
+RUN uv pip install --no-cache ".[all]"
+
+RUN chmod +x /lerobot/.venv/lib/python${PYTHON_VERSION}/site-packages/triton/backends/nvidia/bin/ptxas
+
+# Copy the rest of the application source code
+# Make sure to have the git-LFS files for testing
+COPY --chown=user_lerobot:user_lerobot . .
+
+# Set the default command
+CMD ["/bin/bash"]
diff --git a/lerobot/docker/Dockerfile.user b/lerobot/docker/Dockerfile.user
new file mode 100644
index 0000000000000000000000000000000000000000..f267be7f210a73f6e04365085702ba1f5e54f8a9
--- /dev/null
+++ b/lerobot/docker/Dockerfile.user
@@ -0,0 +1,81 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# This Dockerfile is designed for a lerobot user who wants to
+# experiment with the project. It starts from an Python Slim base image.
+
+# docker build -f docker/Dockerfile.user -t lerobot-user .
+# docker run -it --rm lerobot-user
+
+# With USB physical access : docker run -it --device=/dev/ -v /dev/:/dev/ --rm lerobot-user
+
+# Configure the base image
+ARG PYTHON_VERSION=3.12
+FROM python:${PYTHON_VERSION}-slim
+
+# Configure environment variables
+ENV DEBIAN_FRONTEND=noninteractive \
+    MUJOCO_GL=egl \
+    PATH=/lerobot/.venv/bin:$PATH
+
+# Install system dependencies and uv (as root)
+RUN apt-get update && apt-get install -y --no-install-recommends \
+    build-essential git curl libglib2.0-0 libegl1-mesa-dev ffmpeg \
+    libusb-1.0-0-dev speech-dispatcher libgeos-dev portaudio19-dev \
+    cmake pkg-config ninja-build \
+    && curl -LsSf https://astral.sh/uv/install.sh | sh \
+    && mv /root/.local/bin/uv /usr/local/bin/uv \
+    && useradd --create-home --shell /bin/bash user_lerobot \
+    && usermod -aG sudo user_lerobot \
+    && apt-get clean && rm -rf /var/lib/apt/lists/*
+
+# Create application directory and set permissions
+WORKDIR /lerobot
+RUN chown -R user_lerobot:user_lerobot /lerobot
+
+# Switch to the non-root user
+USER user_lerobot
+
+# Environment variables for the testing
+ENV HOME=/home/user_lerobot \
+    HF_HOME=/home/user_lerobot/.cache/huggingface \
+    HF_LEROBOT_HOME=/home/user_lerobot/.cache/huggingface/lerobot \
+    TORCH_HOME=/home/user_lerobot/.cache/torch \
+    TRITON_CACHE_DIR=/home/user_lerobot/.cache/triton
+
+# Create the virtual environment
+# We use a virtual environment inside the container—even though the container itself \
+# provides isolation—to closely resemble local development and allow users to \
+# run other Python projects in the same container without dependency conflicts.
+RUN uv venv
+
+# Install Python dependencies for caching
+COPY --chown=user_lerobot:user_lerobot setup.py pyproject.toml README.md MANIFEST.in ./
+COPY --chown=user_lerobot:user_lerobot src/ src/
+
+ARG UNBOUND_DEPS=false
+
+RUN if [ "$UNBOUND_DEPS" = "true" ]; then \
+    sed -i 's/,[[:space:]]*<[0-9\.]*//g' pyproject.toml; \
+    echo "Dependencies unbound:" && cat pyproject.toml; \
+    fi
+
+RUN uv pip install --no-cache ".[all]"
+
+# Copy the rest of the application code
+# Make sure to have the git-LFS files for testing
+COPY --chown=user_lerobot:user_lerobot . .
+
+# Set the default command
+CMD ["/bin/bash"]
diff --git a/lerobot/docs-requirements.txt b/lerobot/docs-requirements.txt
new file mode 100644
index 0000000000000000000000000000000000000000..e286ad2bb46d14ba2b77d5a3e364669406089965
--- /dev/null
+++ b/lerobot/docs-requirements.txt
@@ -0,0 +1,3 @@
+# docs-requirements.txt
+hf-doc-builder @ git+https://github.com/huggingface/doc-builder.git@main
+watchdog>=6.0.0
diff --git a/lerobot/docs/README.md b/lerobot/docs/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..476eb8dce871b56577ff92a86fb8c699699f51e1
--- /dev/null
+++ b/lerobot/docs/README.md
@@ -0,0 +1,139 @@
+<!---
+Copyright 2020 The HuggingFace Team. All rights reserved.
+
+Licensed under the Apache License, Version 2.0 (the "License");
+you may not use this file except in compliance with the License.
+You may obtain a copy of the License at
+
+    http://www.apache.org/licenses/LICENSE-2.0
+
+Unless required by applicable law or agreed to in writing, software
+distributed under the License is distributed on an "AS IS" BASIS,
+WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+See the License for the specific language governing permissions and
+limitations under the License.
+-->
+
+# Generating the documentation
+
+To generate the documentation, you first have to build it. Several packages are necessary to build the doc,
+you can install them with the following command, at the root of the code repository:
+
+```bash
+pip install -e . -r docs-requirements.txt
+```
+
+You will also need `nodejs`. Please refer to their [installation page](https://nodejs.org/en/download)
+
+---
+
+**NOTE**
+
+You only need to generate the documentation to inspect it locally (if you're planning changes and want to
+check how they look before committing for instance). You don't have to `git commit` the built documentation.
+
+---
+
+## Building the documentation
+
+Once you have setup the `doc-builder` and additional packages, you can generate the documentation by
+typing the following command:
+
+```bash
+doc-builder build lerobot docs/source/ --build_dir ~/tmp/test-build
+```
+
+You can adapt the `--build_dir` to set any temporary folder that you prefer. This command will create it and generate
+the MDX files that will be rendered as the documentation on the main website. You can inspect them in your favorite
+Markdown editor.
+
+## Previewing the documentation
+
+To preview the docs, first install the `watchdog` module with:
+
+```bash
+pip install watchdog
+```
+
+Then run the following command:
+
+```bash
+doc-builder preview lerobot docs/source/
+```
+
+The docs will be viewable at [http://localhost:3000](http://localhost:3000). You can also preview the docs once you have opened a PR. You will see a bot add a comment to a link where the documentation with your changes lives.
+
+---
+
+**NOTE**
+
+The `preview` command only works with existing doc files. When you add a completely new file, you need to update `_toctree.yml` & restart `preview` command (`ctrl-c` to stop it & call `doc-builder preview ...` again).
+
+---
+
+## Adding a new element to the navigation bar
+
+Accepted files are Markdown (.md).
+
+Create a file with its extension and put it in the source directory. You can then link it to the toc-tree by putting
+the filename without the extension in the [`_toctree.yml`](https://github.com/huggingface/lerobot/blob/main/docs/source/_toctree.yml) file.
+
+## Renaming section headers and moving sections
+
+It helps to keep the old links working when renaming the section header and/or moving sections from one document to another. This is because the old links are likely to be used in Issues, Forums, and Social media and it'd make for a much more superior user experience if users reading those months later could still easily navigate to the originally intended information.
+
+Therefore, we simply keep a little map of moved sections at the end of the document where the original section was. The key is to preserve the original anchor.
+
+So if you renamed a section from: "Section A" to "Section B", then you can add at the end of the file:
+
+```
+Sections that were moved:
+
+[ <a href="#section-b">Section A</a><a id="section-a"></a> ]
+```
+
+and of course, if you moved it to another file, then:
+
+```
+Sections that were moved:
+
+[ <a href="../new-file#section-b">Section A</a><a id="section-a"></a> ]
+```
+
+Use the relative style to link to the new file so that the versioned docs continue to work.
+
+For an example of a rich moved sections set please see the very end of [the transformers Trainer doc](https://github.com/huggingface/transformers/blob/main/docs/source/en/main_classes/trainer.md).
+
+### Adding a new tutorial
+
+Adding a new tutorial or section is done in two steps:
+
+- Add a new file under `./source`. This file can either be ReStructuredText (.rst) or Markdown (.md).
+- Link that file in `./source/_toctree.yml` on the correct toc-tree.
+
+Make sure to put your new file under the proper section. If you have a doubt, feel free to ask in a Github Issue or PR.
+
+### Writing source documentation
+
+Values that should be put in `code` should either be surrounded by backticks: \`like so\`. Note that argument names
+and objects like True, None or any strings should usually be put in `code`.
+
+#### Writing a multi-line code block
+
+Multi-line code blocks can be useful for displaying examples. They are done between two lines of three backticks as usual in Markdown:
+
+````
+```
+# first line of code
+# second line
+# etc
+```
+````
+
+#### Adding an image
+
+Due to the rapidly growing repository, it is important to make sure that no files that would significantly weigh down the repository are added. This includes images, videos, and other non-text files. We prefer to leverage a hf.co hosted `dataset` like
+the ones hosted on [`hf-internal-testing`](https://huggingface.co/hf-internal-testing) in which to place these files and reference
+them by URL. We recommend putting them in the following dataset: [huggingface/documentation-images](https://huggingface.co/datasets/huggingface/documentation-images).
+If an external contribution, feel free to add the images to your PR and ask a Hugging Face member to migrate your images
+to this dataset.
diff --git a/lerobot/docs/source/_toctree.yml b/lerobot/docs/source/_toctree.yml
new file mode 100644
index 0000000000000000000000000000000000000000..09d94d28cca95dc4007e219bca77e7f7d1acfa60
--- /dev/null
+++ b/lerobot/docs/source/_toctree.yml
@@ -0,0 +1,136 @@
+- sections:
+  - local: index
+    title: LeRobot
+  - local: installation
+    title: Installation
+  title: Get started
+- sections:
+  - local: il_robots
+    title: Imitation Learning for Robots
+  - local: bring_your_own_policies
+    title: Bring Your Own Policies
+  - local: integrate_hardware
+    title: Bring Your Own Hardware
+  - local: hilserl
+    title: Train a Robot with RL
+  - local: hilserl_sim
+    title: Train RL in Simulation
+  - local: multi_gpu_training
+    title: Multi GPU training
+  - local: peft_training
+    title: Training with PEFT (e.g., LoRA)
+  - local: rename_map
+    title: Using Rename Map and Empty Cameras
+  title: "Tutorials"
+- sections:
+  - local: lerobot-dataset-v3
+    title: Using LeRobotDataset
+  - local: porting_datasets_v3
+    title: Porting Large Datasets
+  - local: using_dataset_tools
+    title: Using the Dataset Tools
+  - local: dataset_subtask
+    title: Using Subtasks in the Dataset
+  - local: streaming_video_encoding
+    title: Streaming Video Encoding
+  title: "Datasets"
+- sections:
+  - local: act
+    title: ACT
+  - local: smolvla
+    title: SmolVLA
+  - local: pi0
+    title: π₀ (Pi0)
+  - local: pi0fast
+    title: π₀-FAST (Pi0Fast)
+  - local: pi05
+    title: π₀.₅ (Pi05)
+  - local: groot
+    title: NVIDIA GR00T N1.5
+  - local: xvla
+    title: X-VLA
+  - local: walloss
+    title: WALL-OSS
+  title: "Policies"
+- sections:
+  - local: sarm
+    title: SARM
+  title: "Reward Models"
+- sections:
+  - local: async
+    title: Use Async Inference
+  - local: rtc
+    title: Real-Time Chunking (RTC)
+  title: "Inference"
+- sections:
+  - local: envhub
+    title: Environments from the Hub
+  - local: envhub_leisaac
+    title: Control & Train Robots in Sim (LeIsaac)
+  - local: envhub_isaaclab_arena
+    title: NVIDIA IsaacLab Arena Environments
+  - local: libero
+    title: Using Libero
+  - local: metaworld
+    title: Using MetaWorld
+  title: "Simulation"
+- sections:
+  - local: introduction_processors
+    title: Introduction to Robot Processors
+  - local: debug_processor_pipeline
+    title: Debug your processor pipeline
+  - local: implement_your_own_processor
+    title: Implement your own processor
+  - local: processors_robots_teleop
+    title: Processors for Robots and Teleoperators
+  - local: env_processor
+    title: Environment Processors
+  title: "Robot Processors"
+- sections:
+  - local: so101
+    title: SO-101
+  - local: so100
+    title: SO-100
+  - local: koch
+    title: Koch v1.1
+  - local: lekiwi
+    title: LeKiwi
+  - local: hope_jr
+    title: Hope Jr
+  - local: reachy2
+    title: Reachy 2
+  - local: unitree_g1
+    title: Unitree G1
+  - local: earthrover_mini_plus
+    title: Earth Rover Mini
+  - local: omx
+    title: OMX
+  - local: openarm
+    title: OpenArm
+  title: "Robots"
+- sections:
+  - local: phone_teleop
+    title: Phone
+  title: "Teleoperators"
+- sections:
+  - local: cameras
+    title: Cameras
+  title: "Sensors"
+- sections:
+  - local: torch_accelerators
+    title: PyTorch accelerators
+  title: "Supported Hardware"
+- sections:
+  - local: notebooks
+    title: Notebooks
+  - local: feetech
+    title: Updating Feetech Firmware
+  - local: damiao
+    title: Damiao Motors and CAN Bus
+  title: "Resources"
+- sections:
+  - local: contributing
+    title: Contribute to LeRobot
+  - local: backwardcomp
+    title: Backward compatibility
+  title: "About"
diff --git a/lerobot/docs/source/act.mdx b/lerobot/docs/source/act.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..453bcbba89aabf6c6c0e195f3f2599cc56e27bb7
--- /dev/null
+++ b/lerobot/docs/source/act.mdx
@@ -0,0 +1,95 @@
+# ACT (Action Chunking with Transformers)
+
+ACT is a **lightweight and efficient policy for imitation learning**, especially well-suited for fine-grained manipulation tasks. It's the **first model we recommend when you're starting out** with LeRobot due to its fast training time, low computational requirements, and strong performance.
+
+<div class="video-container">
+  <iframe
+    width="100%"
+    height="415"
+    src="https://www.youtube.com/embed/ft73x0LfGpM"
+    title="LeRobot ACT Tutorial"
+    frameborder="0"
+    allow="accelerometer; autoplay; clipboard-write; encrypted-media; gyroscope; picture-in-picture"
+    allowfullscreen
+  ></iframe>
+</div>
+
+_Watch this tutorial from the LeRobot team to learn how ACT works: [LeRobot ACT Tutorial](https://www.youtube.com/watch?v=ft73x0LfGpM)_
+
+## Model Overview
+
+Action Chunking with Transformers (ACT) was introduced in the paper [Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware](https://arxiv.org/abs/2304.13705) by Zhao et al. The policy was designed to enable precise, contact-rich manipulation tasks using affordable hardware and minimal demonstration data.
+
+### Why ACT is Great for Beginners
+
+ACT stands out as an excellent starting point for several reasons:
+
+- **Fast Training**: Trains in a few hours on a single GPU
+- **Lightweight**: Only ~80M parameters, making it efficient and easy to work with
+- **Data Efficient**: Often achieves high success rates with just 50 demonstrations
+
+### Architecture
+
+ACT uses a transformer-based architecture with three main components:
+
+1. **Vision Backbone**: ResNet-18 processes images from multiple camera viewpoints
+2. **Transformer Encoder**: Synthesizes information from camera features, joint positions, and a learned latent variable
+3. **Transformer Decoder**: Generates coherent action sequences using cross-attention
+
+The policy takes as input:
+
+- Multiple RGB images (e.g., from wrist cameras, front/top cameras)
+- Current robot joint positions
+- A latent style variable `z` (learned during training, set to zero during inference)
+
+And outputs a chunk of `k` future action sequences.
+
+## Installation Requirements
+
+1. Install LeRobot by following our [Installation Guide](./installation).
+2. ACT is included in the base LeRobot installation, so no additional dependencies are needed!
+
+## Training ACT
+
+ACT works seamlessly with the standard LeRobot training pipeline. Here's a complete example for training ACT on your dataset:
+
+```bash
+lerobot-train \
+  --dataset.repo_id=${HF_USER}/your_dataset \
+  --policy.type=act \
+  --output_dir=outputs/train/act_your_dataset \
+  --job_name=act_your_dataset \
+  --policy.device=cuda \
+  --wandb.enable=true \
+  --policy.repo_id=${HF_USER}/act_policy
+```
+
+### Training Tips
+
+1. **Start with defaults**: ACT's default hyperparameters work well for most tasks
+2. **Training duration**: Expect a few hours for 100k training steps on a single GPU
+3. **Batch size**: Start with batch size 8 and adjust based on your GPU memory
+
+### Train using Google Colab
+
+If your local computer doesn't have a powerful GPU, you can utilize Google Colab to train your model by following the [ACT training notebook](./notebooks#training-act).
+
+## Evaluating ACT
+
+Once training is complete, you can evaluate your ACT policy using the `lerobot-record` command with your trained policy. This will run inference and record evaluation episodes:
+
+```bash
+lerobot-record \
+  --robot.type=so100_follower \
+  --robot.port=/dev/ttyACM0 \
+  --robot.id=my_robot \
+  --robot.cameras="{ front: {type: opencv, index_or_path: 0, width: 640, height: 480, fps: 30}}" \
+  --display_data=true \
+  --dataset.repo_id=${HF_USER}/eval_act_your_dataset \
+  --dataset.num_episodes=10 \
+  --dataset.single_task="Your task description" \
+  --dataset.streaming_encoding=true \
+  --dataset.encoder_threads=2 \
+  # --dataset.vcodec=auto \
+  --policy.path=${HF_USER}/act_policy
+```
diff --git a/lerobot/docs/source/async.mdx b/lerobot/docs/source/async.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..a46408a0d05ee817be678032d2f893b2fff61400
--- /dev/null
+++ b/lerobot/docs/source/async.mdx
@@ -0,0 +1,313 @@
+# Asynchronous Inference
+
+With our [SmolVLA](https://huggingface.co/papers/2506.01844) we introduced a new way to run inference on real-world robots, **decoupling action prediction from action execution**.
+In this tutorial, we'll show how to use asynchronous inference (_async inference_) using a finetuned version of SmolVLA, and all the policies supported by LeRobot.
+**Try async inference with all the policies** supported by LeRobot!
+
+**What you'll learn:**
+
+1. Why asynchronous inference matters and how it compares to, more traditional, sequential inference.
+2. How to spin-up a `PolicyServer` and connect a `RobotClient` from the same machine, and even over the network.
+3. How to tune key parameters (`actions_per_chunk`, `chunk_size_threshold`) for your robot and policy.
+
+If you get stuck, hop into our [Discord community](https://discord.gg/s3KuuzsPFb)!
+
+In a nutshell: with _async inference_, your robot keeps acting while the policy server is already busy computing the next chunk of actions---eliminating "wait-for-inference" lags and unlocking smoother, more reactive behaviours.
+This is fundamentally different from synchronous inference (sync), where the robot stays idle while the policy computes the next chunk of actions.
+
+---
+
+## Getting started with async inference
+
+You can read more information on asynchronous inference in our [blogpost](https://huggingface.co/blog/async-robot-inference). This guide is designed to help you quickly set up and run asynchronous inference in your environment.
+
+First, install `lerobot` with the `async` tag, to install the extra dependencies required to run async inference.
+
+```shell
+pip install -e ".[async]"
+```
+
+Then, spin up a policy server (in one terminal, or in a separate machine) specifying the host address and port for the client to connect to.
+You can spin up a policy server running:
+
+```shell
+python -m lerobot.async_inference.policy_server \
+     --host=127.0.0.1 \
+     --port=8080
+```
+
+This will start a policy server listening on `127.0.0.1:8080` (`localhost`, port 8080). At this stage, the policy server is empty, as all information related to which policy to run and with which parameters are specified during the first handshake with the client. Spin up a client with:
+
+```shell
+python -m lerobot.async_inference.robot_client \
+    --server_address=127.0.0.1:8080 \ # SERVER: the host address and port of the policy server
+    --robot.type=so100_follower \ # ROBOT: your robot type
+    --robot.port=/dev/tty.usbmodem585A0076841 \ # ROBOT: your robot port
+    --robot.id=follower_so100 \ # ROBOT: your robot id, to load calibration file
+    --robot.cameras="{ laptop: {type: opencv, index_or_path: 0, width: 1920, height: 1080, fps: 30}, phone: {type: opencv, index_or_path: 0, width: 1920, height: 1080, fps: 30}}" \ # POLICY: the cameras used to acquire frames, with keys matching the keys expected by the policy
+    --task="dummy" \ # POLICY: The task to run the policy on (`Fold my t-shirt`). Not necessarily defined for all policies, such as `act`
+    --policy_type=your_policy_type \ # POLICY: the type of policy to run (smolvla, act, etc)
+    --pretrained_name_or_path=user/model \ # POLICY: the model name/path on server to the checkpoint to run (e.g., lerobot/smolvla_base)
+    --policy_device=mps \ # POLICY: the device to run the policy on, on the server (cuda, mps, xpu, cpu)
+    --actions_per_chunk=50 \ # POLICY: the number of actions to output at once
+    --chunk_size_threshold=0.5 \ # CLIENT: the threshold for the chunk size before sending a new observation to the server
+    --aggregate_fn_name=weighted_average \ # CLIENT: the function to aggregate actions on overlapping portions
+    --debug_visualize_queue_size=True # CLIENT: whether to visualize the queue size at runtime
+```
+
+In summary, you need to specify instructions for:
+
+- `SERVER`: the address and port of the policy server
+- `ROBOT`: the type of robot to connect to, the port to connect to, and the local `id` of the robot
+- `POLICY`: the type of policy to run, and the model name/path on server to the checkpoint to run. You also need to specify which device should the sever be using, and how many actions to output at once (capped at the policy max actions value).
+- `CLIENT`: the threshold for the chunk size before sending a new observation to the server, and the function to aggregate actions on overlapping portions. Optionally, you can also visualize the queue size at runtime, to help you tune the `CLIENT` parameters.
+
+Importantly,
+
+- `actions_per_chunk` and `chunk_size_threshold` are key parameters to tune for your setup.
+- `aggregate_fn_name` is the function to aggregate actions on overlapping portions. You can either add a new one to a registry of functions, or add your own in `robot_client.py` (see [here](NOTE:addlinktoLOC))
+- `debug_visualize_queue_size` is a useful tool to tune the `CLIENT` parameters.
+
+## Done! You should see your robot moving around by now 😉
+
+## Async vs. synchronous inference
+
+Synchronous inference relies on interleaving action chunk prediction and action execution. This inherently results in _idle frames_, frames where the robot awaits idle the policy's output: a new action chunk.
+In turn, inference is plagued by evident real-time lags, where the robot simply stops acting due to the lack of available actions.
+With robotics models increasing in size, this problem risks becoming only more severe.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/async-inference/sync.png"
+    width="80%"
+  ></img>
+</p>
+<p align="center">
+  <i>Synchronous inference</i> makes the robot idle while the policy is
+  computing the next chunk of actions.
+</p>
+
+To overcome this, we design async inference, a paradigm where action planning and execution are decoupled, resulting in (1) higher adaptability and, most importantly, (2) no idle frames.
+Crucially, with async inference, the next action chunk is computed _before_ the current one is exhausted, resulting in no idleness.
+Higher adaptability is ensured by aggregating the different action chunks on overlapping portions, obtaining an up-to-date plan and a tighter control loop.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/async-inference/async.png"
+    width="80%"
+  ></img>
+</p>
+<p align="center">
+  <i>Asynchronous inference</i> results in no idleness because the next chunk is
+  computed before the current chunk is exhausted.
+</p>
+
+---
+
+## Start the Policy Server
+
+Policy servers are wrappers around a `PreTrainedPolicy` interfacing them with observations coming from a robot client.
+Policy servers are initialized as empty containers which are populated with the requested policy specified in the initial handshake between the robot client and the policy server.
+As such, spinning up a policy server is as easy as specifying the host address and port. If you're running the policy server on the same machine as the robot client, you can use `localhost` as the host address.
+
+<hfoptions id="start_policy_server">
+<hfoption id="Command">
+```bash
+python -m lerobot.async_inference.policy_server \
+     --host=127.0.0.1 \
+     --port=8080
+```
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.async_inference.configs import PolicyServerConfig
+from lerobot.async_inference.policy_server import serve
+
+config = PolicyServerConfig(
+    host="localhost",
+    port=8080,
+)
+serve(config)
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+This listens on `localhost:8080` for an incoming connection from the associated`RobotClient`, which will communicate which policy to run during the first client-server handshake.
+
+---
+
+## Launch the Robot Client
+
+`RobotClient` is a wrapper around a `Robot` instance, which `RobotClient` connects to the (possibly remote) `PolicyServer`.
+The `RobotClient` streams observations to the `PolicyServer`, and receives action chunks obtained running inference on the server (which we assume to have better computational resources than the robot controller).
+
+<hfoptions id="start_robot_client">
+<hfoption id="Command">
+```bash
+python -m lerobot.async_inference.robot_client \
+    --server_address=127.0.0.1:8080 \ # SERVER: the host address and port of the policy server
+    --robot.type=so100_follower \ # ROBOT: your robot type
+    --robot.port=/dev/tty.usbmodem585A0076841 \ # ROBOT: your robot port
+    --robot.id=follower_so100 \ # ROBOT: your robot id, to load calibration file
+    --robot.cameras="{ laptop: {type: opencv, index_or_path: 0, width: 1920, height: 1080, fps: 30}, phone: {type: opencv, index_or_path: 0, width: 1920, height: 1080, fps: 30}}" \ # POLICY: the cameras used to acquire frames, with keys matching the keys expected by the policy
+    --task="dummy" \ # POLICY: The task to run the policy on (`Fold my t-shirt`). Not necessarily defined for all policies, such as `act`
+    --policy_type=your_policy_type \ # POLICY: the type of policy to run (smolvla, act, etc)
+    --pretrained_name_or_path=user/model \ # POLICY: the model name/path on server to the checkpoint to run (e.g., lerobot/smolvla_base)
+    --policy_device=mps \ # POLICY: the device to run the policy on, on the server
+    --actions_per_chunk=50 \ # POLICY: the number of actions to output at once
+    --chunk_size_threshold=0.5 \ # CLIENT: the threshold for the chunk size before sending a new observation to the server
+    --aggregate_fn_name=weighted_average \ # CLIENT: the function to aggregate actions on overlapping portions
+    --debug_visualize_queue_size=True # CLIENT: whether to visualize the queue size at runtime
+```
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+import threading
+from lerobot.robots.so_follower import SO100FollowerConfig
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.async_inference.configs import RobotClientConfig
+from lerobot.async_inference.robot_client import RobotClient
+from lerobot.async_inference.helpers import visualize_action_queue_size
+
+# 1. Create the robot instance
+"""Check out the cameras available in your setup by running `python lerobot/find_cameras.py`"""
+# these cameras must match the ones expected by the policy
+# check the config.json on the Hub for the policy you are using
+camera_cfg = {
+    "top": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=30),
+    "side": OpenCVCameraConfig(index_or_path=1, width=640, height=480, fps=30)
+}
+
+robot_cfg = SO100FollowerConfig(
+  port="/dev/tty.usbmodem585A0076841",
+  id="follower_so100",
+  cameras=camera_cfg
+)
+
+# 3. Create client configuration
+client_cfg = RobotClientConfig(
+    robot=robot_cfg,
+    server_address="localhost:8080",
+    policy_device="mps",
+    client_device="cpu",
+    policy_type="smolvla",
+    pretrained_name_or_path="<user>/smolvla_async",
+    chunk_size_threshold=0.5,
+    actions_per_chunk=50,  # make sure this is less than the max actions of the policy
+)
+
+# 4. Create and start client
+client = RobotClient(client_cfg)
+
+# 5. Specify the task
+task = "Don't do anything, stay still"
+
+if client.start():
+    # Start action receiver thread
+    action_receiver_thread = threading.Thread(target=client.receive_actions, daemon=True)
+    action_receiver_thread.start()
+
+    try:
+        # Run the control loop
+        client.control_loop(task)
+    except KeyboardInterrupt:
+        client.stop()
+        action_receiver_thread.join()
+        # (Optionally) plot the action queue size
+        visualize_action_queue_size(client.action_queue_size)
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+The following two parameters are key in every setup:
+
+<table>
+  <thead>
+    <tr>
+      <th>Hyperparameter</th>
+      <th>Default</th>
+      <th>What it does</th>
+    </tr>
+  </thead>
+  <tbody>
+    <tr>
+      <td>
+        <code>actions_per_chunk</code>
+      </td>
+      <td>50</td>
+      <td>
+        How many actions the policy outputs at once. Typical values: 10-50.
+      </td>
+    </tr>
+    <tr>
+      <td>
+        <code>chunk_size_threshold</code>
+      </td>
+      <td>0.7</td>
+      <td>
+        When the queue is ≤ 50% full, the client sends a fresh observation.
+        Value in [0, 1].
+      </td>
+    </tr>
+  </tbody>
+</table>
+
+<Tip>
+  Different values of `actions_per_chunk` and `chunk_size_threshold` do result
+  in different behaviours.
+</Tip>
+
+On the one hand, increasing the value of `actions_per_chunk` will result in reducing the likelihood of ending up with no actions to execute, as more actions will be available when the new chunk is computed.
+However, larger values of `actions_per_chunk` might also result in less precise actions, due to the compounding errors consequent to predicting actions over longer timespans.
+
+On the other hand, increasing the value of `chunk_size_threshold` will result in sending out to the `PolicyServer` observations for inference more often, resulting in a larger number of updates action chunks, overlapping on significant portions. This results in high adaptability, in the limit predicting one action chunk for each observation, which is in turn only marginally consumed while a new one is produced.
+This option does also put more pressure on the inference pipeline, as a consequence of the many requests. Conversely, values of `chunk_size_threshold` close to 0.0 collapse to the synchronous edge case, whereby new observations are only sent out whenever the current chunk is exhausted.
+
+We found the default values of `actions_per_chunk` and `chunk_size_threshold` to work well in the experiments we developed for the [SmolVLA paper](https://huggingface.co/papers/2506.01844), but recommend experimenting with different values to find the best fit for your setup.
+
+### Tuning async inference for your setup
+
+1. **Choose your computational resources carefully.** [PI0](https://huggingface.co/lerobot/pi0) occupies 14GB of memory at inference time, while [SmolVLA](https://huggingface.co/lerobot/smolvla_base) requires only ~2GB. You should identify the best computational resource for your use case keeping in mind smaller policies require less computational resources. The combination of policy and device used (CPU-intensive, using MPS, or the number of CUDA cores on a given NVIDIA GPU) directly impacts the average inference latency you should expect.
+2. **Adjust your `fps` based on inference latency.** While the server generates a new action chunk, the client is not idle and is stepping through its current action queue. If the two processes happen at fundamentally different speeds, the client might end up with an empty queue. As such, you should reduce your fps if you consistently run out of actions in queue.
+3. **Adjust `chunk_size_threshold`**.
+   - Values closer to `0.0` result in almost sequential behavior. Values closer to `1.0` → send observation every step (more bandwidth, relies on good world-model).
+   - We found values around 0.5-0.6 to work well. If you want to tweak this, spin up a `RobotClient` setting the `--debug_visualize_queue_size` to `True`. This will plot the action queue size evolution at runtime, and you can use it to find the value of `chunk_size_threshold` that works best for your setup.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/async-inference/queues.png"
+    width="80%"
+  ></img>
+</p>
+<p align="center">
+  <i>
+    The action queue size is plotted at runtime when the
+    `--debug_visualize_queue_size` flag is passed, for various levels of
+    `chunk_size_threshold` (`g` in the SmolVLA paper).
+  </i>
+</p>
+
+---
+
+## Conclusion
+
+Asynchronous inference represents a significant advancement in real-time robotics control, addressing the fundamental challenge of inference latency that has long plagued robotics applications. Through this tutorial, you've learned how to implement a complete async inference pipeline that eliminates idle frames and enables smoother, more reactive robot behaviors.
+
+**Key Takeaways:**
+
+- **Paradigm Shift**: Async inference decouples action prediction from execution, allowing robots to continue acting while new action chunks are computed in parallel
+- **Performance Benefits**: Eliminates "wait-for-inference" lags that are inherent in synchronous approaches, becoming increasingly important as policy models grow larger
+- **Flexible Architecture**: The server-client design enables distributed computing, where inference can run on powerful remote hardware while maintaining real-time robot control
+- **Tunable Parameters**: Success depends on properly configuring `actions_per_chunk` and `chunk_size_threshold` for your specific hardware, policy, and task requirements
+- **Universal Compatibility**: Works with all LeRobot-supported policies, from lightweight ACT models to vision-language models like SmolVLA
+
+Start experimenting with the default parameters, monitor your action queue sizes, and iteratively refine your setup to achieve optimal performance for your specific use case.
+If you want to discuss this further, hop into our [Discord community](https://discord.gg/s3KuuzsPFb), or open an issue on our [GitHub repository](https://github.com/huggingface/lerobot/issues).
diff --git a/lerobot/docs/source/backwardcomp.mdx b/lerobot/docs/source/backwardcomp.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..3366c8ab916cbb98063ea89c0824d5a15a60e294
--- /dev/null
+++ b/lerobot/docs/source/backwardcomp.mdx
@@ -0,0 +1,151 @@
+# Backward compatibility
+
+## Policy Normalization Migration (PR #1452)
+
+**Breaking Change**: LeRobot policies no longer have built-in normalization layers embedded in their weights. Normalization is now handled by external `PolicyProcessorPipeline` components.
+
+### What changed?
+
+|                            | Before PR #1452                                  | After PR #1452                                               |
+| -------------------------- | ------------------------------------------------ | ------------------------------------------------------------ |
+| **Normalization Location** | Embedded in model weights (`normalize_inputs.*`) | External `PolicyProcessorPipeline` components                |
+| **Model State Dict**       | Contains normalization statistics                | **Clean weights only** - no normalization parameters         |
+| **Usage**                  | `policy(batch)` handles everything               | `preprocessor(batch)` → `policy(...)` → `postprocessor(...)` |
+
+### Impact on existing models
+
+- Models trained **before** PR #1452 have normalization embedded in their weights
+- These models need migration to work with the new `PolicyProcessorPipeline` system
+- The migration extracts normalization statistics and creates separate processor pipelines
+
+### Migrating old models
+
+Use the migration script to convert models with embedded normalization:
+
+```shell
+python src/lerobot/processor/migrate_policy_normalization.py \
+    --pretrained-path lerobot/act_aloha_sim_transfer_cube_human \
+    --push-to-hub \
+    --branch migrated
+```
+
+The script:
+
+1. **Extracts** normalization statistics from model weights
+2. **Creates** external preprocessor and postprocessor pipelines
+3. **Removes** normalization layers from model weights
+4. **Saves** clean model + processor pipelines
+5. **Pushes** to Hub with automatic PR creation
+
+### Using migrated models
+
+```python
+# New usage pattern (after migration)
+from lerobot.policies.factory import make_policy, make_pre_post_processors
+
+# Load model and processors separately
+policy = make_policy(config, ds_meta=dataset.meta)
+preprocessor, postprocessor = make_pre_post_processors(
+    policy_cfg=config,
+    dataset_stats=dataset.meta.stats
+)
+
+# Process data through pipeline
+processed_batch = preprocessor(raw_batch)
+action = policy.select_action(processed_batch)
+final_action = postprocessor(action)
+```
+
+## Hardware API redesign
+
+PR [#777](https://github.com/huggingface/lerobot/pull/777) improves the LeRobot calibration but is **not backward-compatible**. Below is a overview of what changed and how you can continue to work with datasets created before this pull request.
+
+### What changed?
+
+|                                   | Before PR #777                                    | After PR #777                                                |
+| --------------------------------- | ------------------------------------------------- | ------------------------------------------------------------ |
+| **Joint range**                   | Degrees `-180...180°`                             | **Normalised range** Joints: `–100...100` Gripper: `0...100` |
+| **Zero position (SO100 / SO101)** | Arm fully extended horizontally                   | **In middle of the range for each joint**                    |
+| **Boundary handling**             | Software safeguards to detect ±180 ° wrap-arounds | No wrap-around logic needed due to mid-range zero            |
+
+---
+
+### Impact on existing datasets
+
+- Recorded trajectories created **before** PR #777 will replay incorrectly if loaded directly:
+  - Joint angles are offset and incorrectly normalized.
+- Any models directly finetuned or trained on the old data will need their inputs and outputs converted.
+
+### Using datasets made with the previous calibration system
+
+We provide a migration example script for replaying an episode recorded with the previous calibration here: `examples/backward_compatibility/replay.py`.
+Below we take you through the modifications that are done in the example script to make the previous calibration datasets work.
+
+```diff
++   key = f"{name.removeprefix('main_')}.pos"
+    action[key] = action_array[i].item()
++   action["shoulder_lift.pos"] = -(action["shoulder_lift.pos"] - 90)
++   action["elbow_flex.pos"] -= 90
+```
+
+Let's break this down.
+New codebase uses `.pos` suffix for the position observations and we have removed `main_` prefix:
+
+<!-- prettier-ignore-start -->
+```python
+key = f"{name.removeprefix('main_')}.pos"
+```
+<!-- prettier-ignore-end -->
+
+For `"shoulder_lift"` (id = 2), the 0 position is changed by -90 degrees and the direction is reversed compared to old calibration/code.
+
+<!-- prettier-ignore-start -->
+```python
+action["shoulder_lift.pos"] = -(action["shoulder_lift.pos"] - 90)
+```
+<!-- prettier-ignore-end -->
+
+For `"elbow_flex"` (id = 3), the 0 position is changed by -90 degrees compared to old calibration/code.
+
+<!-- prettier-ignore-start -->
+```python
+action["elbow_flex.pos"] -= 90
+```
+<!-- prettier-ignore-end -->
+
+To use degrees normalization we then set the `--robot.use_degrees` option to `true`.
+
+```diff
+python examples/backward_compatibility/replay.py \
+    --robot.type=so101_follower \
+    --robot.port=/dev/tty.usbmodem5A460814411 \
+    --robot.id=blue \
++   --robot.use_degrees=true \
+    --dataset.repo_id=my_dataset_id \
+    --dataset.episode=0
+```
+
+### Using policies trained with the previous calibration system
+
+Policies output actions in the same format as the datasets (`torch.Tensors`). Therefore, the same transformations should be applied.
+
+To find these transformations, we recommend to first try and and replay an episode of the dataset your policy was trained on using the section above.
+Then, add these same transformations on your inference script (shown here in the `record.py` script):
+
+```diff
+action_values = predict_action(
+    observation_frame,
+    policy,
+    get_safe_torch_device(policy.config.device),
+    policy.config.use_amp,
+    task=single_task,
+    robot_type=robot.robot_type,
+    )
+    action = {key: action_values[i].item() for i, key in enumerate(robot.action_features)}
+
++   action["shoulder_lift.pos"] = -(action["shoulder_lift.pos"] - 90)
++   action["elbow_flex.pos"] -= 90
+    robot.send_action(action)
+```
+
+If you have questions or run into migration issues, feel free to ask them on [Discord](https://discord.gg/s3KuuzsPFb)
diff --git a/lerobot/docs/source/bring_your_own_policies.mdx b/lerobot/docs/source/bring_your_own_policies.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..9266c9e5bb667cfa2b822904122167db9c950286
--- /dev/null
+++ b/lerobot/docs/source/bring_your_own_policies.mdx
@@ -0,0 +1,175 @@
+# Bring Your Own Policies
+
+This tutorial explains how to integrate your own custom policy implementations into the LeRobot ecosystem, allowing you to leverage all LeRobot tools for training, evaluation, and deployment while using your own algorithms.
+
+## Step 1: Create a Policy Package
+
+Your custom policy should be organized as an installable Python package following LeRobot's plugin conventions.
+
+### Package Structure
+
+Create a package with the prefix `lerobot_policy_` (IMPORTANT!) followed by your policy name:
+
+```bash
+lerobot_policy_my_custom_policy/
+├── pyproject.toml
+└── src/
+    └── lerobot_policy_my_custom_policy/
+        ├── __init__.py
+        ├── configuration_my_custom_policy.py
+        ├── modeling_my_custom_policy.py
+        └── processor_my_custom_policy.py
+```
+
+### Package Configuration
+
+Set up your `pyproject.toml`:
+
+```toml
+[project]
+name = "lerobot_policy_my_custom_policy"
+version = "0.1.0"
+dependencies = [
+    # your policy-specific dependencies
+]
+requires-python = ">= 3.12"
+
+[build-system]
+build-backend = # your-build-backend
+requires = # your-build-system
+```
+
+## Step 2: Define the Policy Configuration
+
+Create a configuration class that inherits from `PreTrainedConfig` and registers your policy type:
+
+```python
+# configuration_my_custom_policy.py
+from dataclasses import dataclass, field
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import NormalizationMode
+
+@PreTrainedConfig.register_subclass("my_custom_policy")
+@dataclass
+class MyCustomPolicyConfig(PreTrainedConfig):
+    """Configuration class for MyCustomPolicy.
+
+    Args:
+        n_obs_steps: Number of observation steps to use as input
+        horizon: Action prediction horizon
+        n_action_steps: Number of action steps to execute
+        hidden_dim: Hidden dimension for the policy network
+        # Add your policy-specific parameters here
+    """
+    # ...PreTrainedConfig fields...
+    pass
+
+    def __post_init__(self):
+        super().__post_init__()
+        # Add any validation logic here
+
+    def validate_features(self) -> None:
+        """Validate input/output feature compatibility."""
+        # Implement validation logic for your policy's requirements
+        pass
+```
+
+## Step 3: Implement the Policy Class
+
+Create your policy implementation by inheriting from LeRobot's base `PreTrainedPolicy` class:
+
+```python
+# modeling_my_custom_policy.py
+import torch
+import torch.nn as nn
+from typing import Any
+
+from lerobot.policies.pretrained import PreTrainedPolicy
+from .configuration_my_custom_policy import MyCustomPolicyConfig
+
+class MyCustomPolicy(PreTrainedPolicy):
+    config_class = MyCustomPolicyConfig
+    name = "my_custom_policy"
+
+    def __init__(self, config: MyCustomPolicyConfig, dataset_stats: dict[str, Any] = None):
+        super().__init__(config, dataset_stats)
+        ...
+```
+
+## Step 4: Add Data Processors
+
+Create processor functions:
+
+```python
+# processor_my_custom_policy.py
+from typing import Any
+import torch
+
+
+def make_my_custom_policy_pre_post_processors(
+    config,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """Create preprocessing and postprocessing functions for your policy."""
+    pass  # Define your preprocessing and postprocessing logic here
+
+```
+
+## Step 5: Package Initialization
+
+Expose your classes in the package's `__init__.py`:
+
+```python
+# __init__.py
+"""Custom policy package for LeRobot."""
+
+try:
+    import lerobot  # noqa: F401
+except ImportError:
+    raise ImportError(
+        "lerobot is not installed. Please install lerobot to use this policy package."
+    )
+
+from .configuration_my_custom_policy import MyCustomPolicyConfig
+from .modeling_my_custom_policy import MyCustomPolicy
+from .processor_my_custom_policy import make_my_custom_policy_pre_post_processors
+
+__all__ = [
+    "MyCustomPolicyConfig",
+    "MyCustomPolicy",
+    "make_my_custom_policy_pre_post_processors",
+]
+```
+
+## Step 6: Installation and Usage
+
+### Install Your Policy Package
+
+```bash
+cd lerobot_policy_my_custom_policy
+pip install -e .
+
+# Or install from PyPI if published
+pip install lerobot_policy_my_custom_policy
+```
+
+### Use Your Policy
+
+Once installed, your policy automatically integrates with LeRobot's training and evaluation tools:
+
+```bash
+lerobot-train \
+    --policy.type my_custom_policy \
+    --env.type pusht \
+    --steps 200000
+```
+
+## Examples and Community Contributions
+
+Check out these example policy implementations:
+
+- [DiTFlow Policy](https://github.com/danielsanjosepro/lerobot_policy_ditflow) - Diffusion Transformer policy with flow-matching objective. Try it out in this example: [DiTFlow Example](https://github.com/danielsanjosepro/test_lerobot_policy_ditflow)
+
+Share your policy implementations with the community! 🤗
diff --git a/lerobot/docs/source/cameras.mdx b/lerobot/docs/source/cameras.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..8af0f5ae52c2b78fc853b3ec8c9d858766e0b046
--- /dev/null
+++ b/lerobot/docs/source/cameras.mdx
@@ -0,0 +1,220 @@
+# Cameras
+
+LeRobot offers multiple options for video capture:
+
+| Class             | Supported Cameras                   |
+| ----------------- | ----------------------------------- |
+| `OpenCVCamera`    | Phone, built-in laptop, USB webcams |
+| `ZMQCamera`       | Network-connected cameras           |
+| `RealSenseCamera` | Intel RealSense (with depth)        |
+| `Reachy2Camera`   | Reachy 2 robot cameras              |
+
+> [!TIP]
+> For `OpenCVCamera` compatibility details, see the [Video I/O with OpenCV Overview](https://docs.opencv.org/4.x/d0/da7/videoio_overview.html).
+
+### Find your camera
+
+Every camera requires a unique identifier to be instantiated, allowing you to distinguish between multiple connected devices.
+
+`OpenCVCamera` and `RealSenseCamera` support auto-discovery. Run the command below to list available devices and their identifiers. Note that these identifiers may change after rebooting your computer or re-plugging the camera, depending on your operating system.
+
+```bash
+lerobot-find-cameras opencv # or realsense for Intel Realsense cameras
+```
+
+The output will look something like this if you have two cameras connected:
+
+```bash
+--- Detected Cameras ---
+Camera #0:
+  Name: OpenCV Camera @ 0
+  Type: OpenCV
+  Id: 0
+  Backend api: AVFOUNDATION
+  Default stream profile:
+    Format: 16.0
+    Width: 1920
+    Height: 1080
+    Fps: 15.0
+--------------------
+(more cameras ...)
+```
+
+> [!WARNING]
+> When using Intel RealSense cameras in `macOS`, you could get this [error](https://github.com/IntelRealSense/librealsense/issues/12307): `Error finding RealSense cameras: failed to set power state`, this can be solved by running the same command with `sudo` permissions. Note that using RealSense cameras in `macOS` is unstable.
+
+`ZMQCamera` and `Reachy2Camera` do not support auto-discovery. They must be configured manually by providing their network address and port or robot SDK settings.
+
+## Use cameras
+
+### Frame access modes
+
+All camera classes implement three access modes for capturing frames:
+
+| Method                    | Behavior                                                                                                                                                   | Blocks?        | Best For                                 |
+| ------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------- | ---------------------------------------- |
+| `read()`                  | Waits for the camera hardware to return a frame. May block for a long time depending on the camera and SDK.                                                | Yes            | Simple scripts, sequential capture       |
+| `async_read(timeout_ms)`  | Returns the latest unconsumed frame from background thread. Blocks only if buffer is empty, up to `timeout_ms`. Raises `TimeoutError` if no frame arrives. | With a timeout | Control loops synchronized to camera FPS |
+| `read_latest(max_age_ms)` | Peeks at the most recent frame in buffer (may be stale). Raises `TimeoutError` if frame is older than `max_age_ms`.                                        | No             | UI visualization, logging, monitoring    |
+
+### Usage examples
+
+The following examples show how to use the camera API to configure and capture frames from different camera types.
+
+- **Blocking and non-blocking frame capture** using an OpenCV-based camera
+- **Color and depth capture** using an Intel RealSense camera
+
+> [!WARNING]
+> Failing to cleanly disconnect cameras can cause resource leaks. Use the context manager protocol to ensure automatic cleanup:
+>
+> ```python
+> with OpenCVCamera(config) as camera:
+>     ...
+> ```
+>
+> You can also call `connect()` and `disconnect()` manually, but always use a `finally` block for the latter.
+
+<hfoptions id="shell_restart">
+<hfoption id="Open CV Camera">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.cameras.opencv.camera_opencv import OpenCVCamera
+from lerobot.cameras.configs import ColorMode, Cv2Rotation
+
+# Construct an `OpenCVCameraConfig` with your desired FPS, resolution, color mode, and rotation.
+config = OpenCVCameraConfig(
+    index_or_path=0,
+    fps=15,
+    width=1920,
+    height=1080,
+    color_mode=ColorMode.RGB,
+    rotation=Cv2Rotation.NO_ROTATION
+)
+
+# Instantiate and connect an `OpenCVCamera`, performing a warm-up read (default).
+with OpenCVCamera(config) as camera:
+
+    # Read a frame synchronously — blocks until hardware delivers a new frame
+    frame = camera.read()
+    print(f"read() call returned frame with shape:", frame.shape)
+
+    # Read a frame asynchronously with a timeout — returns the latest unconsumed frame or waits up to timeout_ms for a new one
+    try:
+        for i in range(10):
+            frame = camera.async_read(timeout_ms=200)
+            print(f"async_read call returned frame {i} with shape:", frame.shape)
+    except TimeoutError as e:
+        print(f"No frame received within timeout: {e}")
+
+    # Instantly return a frame - returns the most recent frame captured by the camera
+    try:
+        initial_frame = camera.read_latest(max_age_ms=1000)
+        for i in range(10):
+            frame = camera.read_latest(max_age_ms=1000)
+            print(f"read_latest call returned frame {i} with shape:", frame.shape)
+            print(f"Was a new frame received by the camera? {not (initial_frame == frame).any()}")
+    except TimeoutError as e:
+        print(f"Frame too old: {e}")
+
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+<hfoption id="Intel Realsense Camera">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.cameras.realsense.configuration_realsense import RealSenseCameraConfig
+from lerobot.cameras.realsense.camera_realsense import RealSenseCamera
+from lerobot.cameras.configs import ColorMode, Cv2Rotation
+
+# Create a `RealSenseCameraConfig` specifying your camera’s serial number and enabling depth.
+config = RealSenseCameraConfig(
+    serial_number_or_name="233522074606",
+    fps=15,
+    width=640,
+    height=480,
+    color_mode=ColorMode.RGB,
+    use_depth=True,
+    rotation=Cv2Rotation.NO_ROTATION
+)
+
+# Instantiate and connect a `RealSenseCamera` with warm-up read (default).
+camera = RealSenseCamera(config)
+camera.connect()
+
+# Capture a color frame via `read()` and a depth map via `read_depth()`.
+try:
+    color_frame = camera.read()
+    depth_map = camera.read_depth()
+    print("Color frame shape:", color_frame.shape)
+    print("Depth map shape:", depth_map.shape)
+finally:
+    camera.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+## Use your phone's camera
+
+<hfoptions id="use phone">
+<hfoption id="iPhone & macOS">
+
+To use your iPhone as a camera on macOS, enable the Continuity Camera feature:
+
+- Ensure your Mac is running macOS 13 or later, and your iPhone is on iOS 16 or later.
+- Sign in both devices with the same Apple ID.
+- Connect your devices with a USB cable or turn on Wi-Fi and Bluetooth for a wireless connection.
+
+For more details, visit [Apple support](https://support.apple.com/en-gb/guide/mac-help/mchl77879b8a/mac).
+
+</hfoption>
+<hfoption id="OBS virtual camera">
+
+If you want to use your phone as a camera using OBS, follow these steps to set up a virtual camera.
+
+1. _(Linux only) Install `v4l2loopback-dkms` and `v4l-utils`_. These packages create virtual camera devices and verify their settings. Install with:
+
+```bash
+sudo apt install v4l2loopback-dkms v4l-utils
+```
+
+2. _Install the [DroidCam app](https://droidcam.app) on your phone_. This app is available for both iOS and Android.
+3. _Download and install [OBS Studio](https://obsproject.com)_.
+4. _Download and install the [DroidCam OBS plugin](https://droidcam.app/obs)_.
+5. _Start OBS Studio_.
+
+6. _Add your phone as a source_. Follow the instructions [here](https://droidcam.app/obs/usage). Be sure to set the resolution to `640x480` to avoid the watermarks.
+7. _Adjust resolution settings_. In OBS Studio, go to `File > Settings > Video` or `OBS > Preferences... > Video`. Change the `Base(Canvas) Resolution` and the `Output(Scaled) Resolution` to `640x480` by manually typing it.
+8. _Start virtual camera_. In OBS Studio, follow the instructions [here](https://obsproject.com/kb/virtual-camera-guide).
+9. _Verify the virtual camera setup and resolution_.
+   - **Linux**: Use `v4l2-ctl` to list devices and check resolution:
+     ```bash
+     v4l2-ctl --list-devices  # find VirtualCam and note its /dev/videoX path
+     v4l2-ctl -d /dev/videoX --get-fmt-video  # replace with your VirtualCam path
+     ```
+     You should see `VirtualCam` listed and resolution `640x480`.
+   - **macOS**: Open Photo Booth or FaceTime and select "OBS Virtual Camera" as the input.
+   - **Windows**: The native Camera app doesn't support virtual cameras. Use a video conferencing app (Zoom, Teams) or run `lerobot-find-cameras opencv` directly to verify.
+
+<details>
+<summary><strong>Troubleshooting</strong></summary>
+
+> The virtual camera resolution is incorrect.
+
+Delete the virtual camera source and recreate it. The resolution cannot be changed after creation.
+
+> Error reading frame in background thread for OpenCVCamera(X): OpenCVCamera(X) frame width=640 or height=480 do not match configured width=1920 or height=1080.
+
+This error is caused by OBS Virtual Camera advertising a `1920x1080` resolution despite rescaling. The only fix for now is to comment out the width and height check in `_postprocess_image()`.
+
+</details>
+
+</hfoption>
+</hfoptions>
+
+If everything is set up correctly, your phone will appear as a standard OpenCV camera and can be used with `OpenCVCamera`.
diff --git a/lerobot/docs/source/contributing.md b/lerobot/docs/source/contributing.md
new file mode 120000
index 0000000000000000000000000000000000000000..f939e75f21a8badb5c40f527abd0e098fe9bc472
--- /dev/null
+++ b/lerobot/docs/source/contributing.md
@@ -0,0 +1 @@
+../../CONTRIBUTING.md
\ No newline at end of file
diff --git a/lerobot/docs/source/damiao.mdx b/lerobot/docs/source/damiao.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..45388ab9b815b1b0821b9e25b45b63d20091bb6d
--- /dev/null
+++ b/lerobot/docs/source/damiao.mdx
@@ -0,0 +1,165 @@
+# Damiao Motors and CAN Bus
+
+This guide covers setup and usage of Damiao motors with LeRobot via CAN bus communication.
+
+Currently, only Linux is supported, as the OpenArms CAN adapter only has drivers for Linux.
+
+## Linux CAN Setup
+
+Before using Damiao motors, you need to set up the CAN interface on your Linux system.
+
+### Install CAN Utilities
+
+```bash
+sudo apt-get install can-utils
+```
+
+### Configure CAN Interface (Manual)
+
+For standard CAN FD (recommended for OpenArms):
+
+```bash
+sudo ip link set can0 down
+sudo ip link set can0 type can bitrate 1000000 dbitrate 5000000 fd on
+sudo ip link set can0 up
+```
+
+For standard CAN (without FD):
+
+```bash
+sudo ip link set can0 down
+sudo ip link set can0 type can bitrate 1000000
+sudo ip link set can0 up
+```
+
+### Configure CAN Interface (Using LeRobot)
+
+LeRobot provides a utility script to setup and test CAN interfaces:
+
+```bash
+# Setup multiple interfaces (e.g., OpenArms Followers with 2 CAN buses)
+lerobot-setup-can --mode=setup --interfaces=can0,can1
+```
+
+## Debugging CAN Communication
+
+Use the built-in debug tools to test motor communication:
+
+```bash
+# Test motors on all interfaces
+lerobot-setup-can --mode=test --interfaces=can0,can1
+
+# Run speed/latency test
+lerobot-setup-can --mode=speed --interfaces=can0
+```
+
+The test mode will scan for motors (IDs 0x01-0x08) and report which ones respond. Example output:
+
+```
+can0: UP (CAN FD)
+  Motor 0x01 (joint_1): ✓ FOUND
+    → Response 0x11 [FD]: 00112233...
+  Motor 0x02 (joint_2): ✓ FOUND
+  Motor 0x03 (joint_3): ✗ No response
+  ...
+  Summary: 2/8 motors found
+```
+
+## Usage
+
+### Basic Setup
+
+```python
+from lerobot.motors import Motor
+from lerobot.motors.damiao import DamiaoMotorsBus
+
+# Define your motors with send/receive CAN IDs
+motors = {
+    "joint_1": Motor(id=0x01, motor_type_str="dm8009", recv_id=0x11),
+    "joint_2": Motor(id=0x02, motor_type_str="dm4340", recv_id=0x12),
+    "joint_3": Motor(id=0x03, motor_type_str="dm4310", recv_id=0x13),
+}
+
+# Create the bus
+bus = DamiaoMotorsBus(
+    port="can0",  # Linux socketcan interface
+    motors=motors,
+)
+
+# Connect
+bus.connect()
+```
+
+### Reading Motor States
+
+```python
+# Read single motor position (degrees)
+position = bus.read("Present_Position", "joint_1")
+
+# Read from multiple motors
+positions = bus.sync_read("Present_Position")  # All motors
+positions = bus.sync_read("Present_Position", ["joint_1", "joint_2"])
+
+# Read all states at once (position, velocity, torque)
+states = bus.sync_read_all_states()
+# Returns: {'joint_1': {'position': 45.2, 'velocity': 1.3, 'torque': 0.5}, ...}
+```
+
+### Writing Motor Commands
+
+```python
+# Enable torque
+bus.enable_torque()
+
+# Set goal position (degrees)
+bus.write("Goal_Position", "joint_1", 45.0)
+
+# Set positions for multiple motors
+bus.sync_write("Goal_Position", {
+    "joint_1": 45.0,
+    "joint_2": -30.0,
+    "joint_3": 90.0,
+})
+
+# Disable torque
+bus.disable_torque()
+```
+
+## Configuration Options
+
+| Parameter      | Default   | Description                                                 |
+| -------------- | --------- | ----------------------------------------------------------- |
+| `port`         | -         | CAN interface (`can0`) or serial port (`/dev/cu.usbmodem*`) |
+| `use_can_fd`   | `True`    | Enable CAN FD for higher data rates                         |
+| `bitrate`      | `1000000` | Nominal bitrate (1 Mbps)                                    |
+| `data_bitrate` | `5000000` | CAN FD data bitrate (5 Mbps)                                |
+
+## Motor Configuration
+
+Each motor requires:
+
+- `id`: CAN ID for sending commands
+- `motor_type`: One of the supported motor types (e.g., `"dm8009"`, `"dm4340"`)
+- `recv_id`: CAN ID for receiving responses
+
+OpenArms default IDs follow the pattern: send ID `0x0N`, receive ID `0x1N` where N is the joint number.
+
+## Troubleshooting
+
+### No Response from Motors
+
+1. **Check power**
+2. **Verify CAN wiring**: Check CAN-H, CAN-L, and GND connections
+3. **Check motor IDs**: Use Damiao Debugging Tools to verify/configure IDs
+4. **Test CAN interface**: Run `candump can0` to see if messages are being received
+5. **Run diagnostics**: `lerobot-setup-can --mode=test --interfaces=can0`
+
+### Motor Timeout Parameter
+
+If motors were configured with timeout=0, they won't respond to commands. Use Damiao Debugging Tools to set a non-zero timeout value.
+
+### Verify CAN FD Status
+
+```bash
+ip -d link show can0 | grep fd
+```
diff --git a/lerobot/docs/source/dataset_subtask.mdx b/lerobot/docs/source/dataset_subtask.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..beb5d80bd81c67ecefe8b0e659b3687a2fc31957
--- /dev/null
+++ b/lerobot/docs/source/dataset_subtask.mdx
@@ -0,0 +1,278 @@
+# Using Subtasks in LeRobot Datasets
+
+Subtask support in robotics datasets has proven effective in improving robot reasoning and understanding. Subtasks are particularly useful for:
+
+- **Hierarchical policies**: Building policies that include subtask predictions to visualize robot reasoning in real time
+- **Reward modeling**: Helping reward models understand task progression (e.g., SARM-style stage-aware reward models)
+- **Task decomposition**: Breaking down complex manipulation tasks into atomic, interpretable steps
+
+LeRobotDataset now supports subtasks as part of its dataset structure, alongside tasks.
+
+## What are Subtasks?
+
+While a **task** describes the overall goal (e.g., "Pick up the apple and place it in the basket"), **subtasks** break down the execution into finer-grained steps:
+
+1. "Approach the apple"
+2. "Grasp the apple"
+3. "Lift the apple"
+4. "Move to basket"
+5. "Release the apple"
+
+Each frame in the dataset can be annotated with its corresponding subtask, enabling models to learn and predict these intermediate stages.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/subtask-asset.png"
+  alt="An overview of subtask annotation showing how frames are labeled with intermediate subtask stages"
+  width="80%"
+/>
+
+<p>
+  <em>Figure: Overview of subtask annotation.</em>
+</p>
+
+**Reference:** _Subtask-learning based for robot self-assembly in flexible collaborative assembly in manufacturing_, Original Article, Published: 19 April 2022.
+
+## Dataset Structure
+
+Subtask information is stored in the dataset metadata:
+
+```
+my-dataset/
+├── data/
+│   └── ...
+├── meta/
+│   ├── info.json
+│   ├── stats.json
+│   ├── tasks.parquet
+│   ├── subtasks.parquet      # Subtask index → subtask string mapping
+│   └── episodes/
+│       └── ...
+└── videos/
+    └── ...
+```
+
+### Subtasks Parquet File
+
+The `meta/subtasks.parquet` file maps subtask indices to their natural language descriptions:
+
+| subtask_index | subtask (index column) |
+| ------------- | ---------------------- |
+| 0             | "Approach the apple"   |
+| 1             | "Grasp the apple"      |
+| 2             | "Lift the apple"       |
+| ...           | ...                    |
+
+### Frame-Level Annotations
+
+Each frame in the dataset can include a `subtask_index` field that references the subtasks parquet file:
+
+```python
+# Example frame data in the parquet file
+{
+    "index": 42,
+    "timestamp": 1.4,
+    "episode_index": 0,
+    "task_index": 0,
+    "subtask_index": 2,  # References "Lift the apple"
+    "observation.state": [...],
+    "action": [...],
+}
+```
+
+## Annotating Datasets with Subtasks
+
+We provide a HuggingFace Space for easily annotating any LeRobotDataset with subtasks:
+
+**[https://huggingface.co/spaces/lerobot/annotate](https://huggingface.co/spaces/lerobot/annotate)**
+
+After completing your annotation:
+
+1. Click "Push to Hub" to upload your annotated dataset
+2. You can also run the annotation space locally by following the instructions at [github.com/huggingface/lerobot-annotate](https://github.com/huggingface/lerobot-annotate)
+
+## Loading Datasets with Subtasks
+
+When you load a dataset with subtask annotations, the subtask information is automatically available:
+
+```python
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+# Load a dataset with subtask annotations
+dataset = LeRobotDataset("jadechoghari/collect-fruit-annotated")
+
+# Access a sample
+sample = dataset[100]
+
+# The sample includes both task and subtask information
+print(sample["task"])        # "Collect the fruit"
+print(sample["subtask"])     # "Grasp the apple"
+print(sample["task_index"])  # tensor(0)
+print(sample["subtask_index"])  # tensor(2)
+```
+
+### Checking for Subtask Support
+
+You can check if a dataset has subtask annotations:
+
+```python
+# Check if subtasks are available
+has_subtasks = (
+    "subtask_index" in dataset.features
+    and dataset.meta.subtasks is not None
+)
+
+if has_subtasks:
+    print(f"Dataset has {len(dataset.meta.subtasks)} unique subtasks")
+    print("Subtasks:", list(dataset.meta.subtasks.index))
+```
+
+## Using Subtasks for Training
+
+### With the Tokenizer Processor
+
+The `TokenizerProcessor` automatically handles subtask tokenization for Vision-Language Action (VLA) models:
+
+```python
+from lerobot.processor.tokenizer_processor import TokenizerProcessor
+from lerobot.processor.pipeline import ProcessorPipeline
+
+# Create a tokenizer processor
+tokenizer_processor = TokenizerProcessor(
+    tokenizer_name_or_path="google/paligemma-3b-pt-224",
+    padding="max_length",
+    max_length=64,
+)
+
+# The processor will automatically tokenize subtasks if present in the batch
+# and add them to the observation under:
+# - "observation.subtask.tokens"
+# - "observation.subtask.attention_mask"
+```
+
+When subtasks are available in the batch, the tokenizer processor adds:
+
+- `observation.subtask.tokens`: Tokenized subtask text
+- `observation.subtask.attention_mask`: Attention mask for the subtask tokens
+
+### DataLoader with Subtasks
+
+```python
+import torch
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+dataset = LeRobotDataset("jadechoghari/collect-fruit-annotated")
+
+dataloader = torch.utils.data.DataLoader(
+    dataset,
+    batch_size=16,
+    shuffle=True,
+)
+
+for batch in dataloader:
+    # Access subtask information in the batch
+    subtasks = batch["subtask"]  # List of subtask strings
+    subtask_indices = batch["subtask_index"]  # Tensor of subtask indices
+
+    # Use for training hierarchical policies or reward models
+    print(f"Batch subtasks: {set(subtasks)}")
+```
+
+## Example Datasets with Subtask Annotations
+
+Try loading a dataset with subtask annotations:
+
+```python
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+# Example dataset with subtask annotations
+dataset = LeRobotDataset("jadechoghari/collect-fruit-annotated")
+
+# Explore the subtasks
+print("Available subtasks:")
+for subtask_name in dataset.meta.subtasks.index:
+    print(f"  - {subtask_name}")
+
+# Get subtask distribution
+subtask_counts = {}
+for i in range(len(dataset)):
+    sample = dataset[i]
+    subtask = sample["subtask"]
+    subtask_counts[subtask] = subtask_counts.get(subtask, 0) + 1
+
+print("\nSubtask distribution:")
+for subtask, count in sorted(subtask_counts.items(), key=lambda x: -x[1]):
+    print(f"  {subtask}: {count} frames")
+```
+
+## Use Cases
+
+### 1. Hierarchical Policy Training
+
+Train policies that predict both actions and current subtask:
+
+```python
+class HierarchicalPolicy(nn.Module):
+    def __init__(self, num_subtasks):
+        super().__init__()
+        self.action_head = nn.Linear(hidden_dim, action_dim)
+        self.subtask_head = nn.Linear(hidden_dim, num_subtasks)
+
+    def forward(self, observations):
+        features = self.encoder(observations)
+        actions = self.action_head(features)
+        subtask_logits = self.subtask_head(features)
+        return actions, subtask_logits
+```
+
+### 2. Stage-Aware Reward Modeling (SARM)
+
+Build reward models that understand task progression:
+
+```python
+# SARM predicts:
+# - Stage: Which subtask is being executed (discrete)
+# - Progress: How far along the subtask (continuous 0-1)
+
+class SARMRewardModel(nn.Module):
+    def forward(self, observations):
+        features = self.encoder(observations)
+        stage_logits = self.stage_classifier(features)
+        progress = self.progress_regressor(features)
+        return stage_logits, progress
+```
+
+### 3. Progress Visualization
+
+Monitor robot execution by tracking subtask progression:
+
+```python
+def visualize_execution(model, observations):
+    for t, obs in enumerate(observations):
+        action, subtask_logits = model(obs)
+        predicted_subtask = subtask_names[subtask_logits.argmax()]
+        print(f"t={t}: Executing '{predicted_subtask}'")
+```
+
+## API Reference
+
+### LeRobotDataset Properties
+
+| Property                    | Type                   | Description                                |
+| --------------------------- | ---------------------- | ------------------------------------------ |
+| `meta.subtasks`             | `pd.DataFrame \| None` | DataFrame mapping subtask names to indices |
+| `features["subtask_index"]` | `dict`                 | Feature spec for subtask_index if present  |
+
+### Sample Keys
+
+When subtasks are available, each sample includes:
+
+| Key             | Type           | Description                          |
+| --------------- | -------------- | ------------------------------------ |
+| `subtask_index` | `torch.Tensor` | Integer index of the current subtask |
+| `subtask`       | `str`          | Natural language subtask description |
+
+## Related Resources
+
+- [SARM Paper](https://arxiv.org/pdf/2509.25358) - Stage-Aware Reward Modeling for Long Horizon Robot Manipulation
+- [LeRobot Annotate Space](https://huggingface.co/spaces/lerobot/annotate) - Interactive annotation tool
+- [LeRobotDataset v3.0](./lerobot-dataset-v3) - Dataset format documentation
diff --git a/lerobot/docs/source/debug_processor_pipeline.mdx b/lerobot/docs/source/debug_processor_pipeline.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..4826c947ecc2df9a7dfc5184d5847cf1029ec63c
--- /dev/null
+++ b/lerobot/docs/source/debug_processor_pipeline.mdx
@@ -0,0 +1,299 @@
+# Debug Your Processor Pipeline
+
+Processor pipelines can be complex, especially when chaining multiple transformation steps.
+Unlike simple function calls, pipelines lack natural observability, you can't easily see what happens
+between each step or where things go wrong.
+This guide provides debugging tools and techniques specifically designed to address these challenges
+and help you understand data flow through your pipelines.
+
+We'll explore three complementary debugging approaches: **hooks** for runtime monitoring, **step-through debugging** for detailed inspection, and **feature validation** for catching structural mismatches. Each serves a different purpose and together they provide complete visibility into your pipeline's behavior.
+
+## Understanding Hooks
+
+Hooks are functions that get called at specific points during pipeline execution.
+They provide a way to inspect, monitor, or modify data without changing your pipeline code.
+Think of them as "event listeners" for your pipeline.
+
+### What is a Hook?
+
+A hook is a callback function that gets automatically invoked at specific moments during pipeline execution.
+The concept comes from event-driven programming, imagine you could "hook into" the pipeline's execution flow to observe or react to what's happening.
+
+Think of hooks like inserting checkpoints into your pipeline. Every time the pipeline reaches one of these checkpoints, it pauses briefly to call your hook function, giving you a chance to inspect the current state, log information, and validate data.
+
+A hook is simply a function that accepts two parameters:
+
+- `step_idx: int` - The index of the current processing step (0, 1, 2, etc.)
+- `transition: EnvTransition` - The data transition at that point in the pipeline
+
+The beauty of hooks is their non-invasive nature: you can add monitoring, validation, or debugging logic without changing a single line of your pipeline code. The pipeline remains clean and focused on its core logic, while hooks handle the cross-cutting concerns like logging, monitoring, and debugging.
+
+### Before vs After Hooks
+
+The pipeline supports two types of hooks:
+
+- **Before hooks** (`register_before_step_hook`) - Called before each step executes
+- **After hooks** (`register_after_step_hook`) - Called after each step completes
+
+```python
+def before_hook(step_idx: int, transition: EnvTransition):
+    """Called before step processes the transition."""
+    print(f"About to execute step {step_idx}")
+    # Useful for: logging, validation, setup
+
+def after_hook(step_idx: int, transition: EnvTransition):
+    """Called after step has processed the transition."""
+    print(f"Completed step {step_idx}")
+    # Useful for: monitoring results, cleanup, debugging
+
+processor.register_before_step_hook(before_hook)
+processor.register_after_step_hook(after_hook)
+```
+
+### Implementing a NaN Detection Hook
+
+Here's a practical example of a hook that detects NaN values:
+
+```python
+def check_nans(step_idx: int, transition: EnvTransition):
+    """Check for NaN values in observations."""
+    obs = transition.get(TransitionKey.OBSERVATION)
+    if obs:
+        for key, value in obs.items():
+            if isinstance(value, torch.Tensor) and torch.isnan(value).any():
+                print(f"NaN detected in {key} at step {step_idx}")
+
+# Register the hook to run after each step
+processor.register_after_step_hook(check_nans)
+
+# Process your data - the hook will be called automatically
+output = processor(input_data)
+
+# Remove the hook when done debugging
+processor.unregister_after_step_hook(check_nans)
+```
+
+### How Hooks Work Internally
+
+Understanding the internal mechanism helps you use hooks more effectively. The pipeline maintains two separate lists: one for before-step hooks and another for after-step hooks. When you register a hook, it's simply appended to the appropriate list.
+
+During execution, the pipeline follows a strict sequence: for each processing step, it first calls all before-hooks in registration order, then executes the actual step transformation, and finally calls all after-hooks in registration order. This creates a predictable, sandwich-like structure around each step.
+
+The key insight is that hooks don't change the core pipeline logic—they're purely additive. The pipeline's `_forward` method orchestrates this dance between hooks and processing steps, ensuring that your debugging or monitoring code runs at exactly the right moments without interfering with the main data flow.
+
+Here's a simplified view of how the pipeline executes hooks:
+
+```python
+class DataProcessorPipeline:
+    def __init__(self):
+        self.steps = [...]
+        self.before_step_hooks = []  # List of before hooks
+        self.after_step_hooks = []   # List of after hooks
+
+    def _forward(self, transition):
+        """Internal method that processes the transition through all steps."""
+        for step_idx, processor_step in enumerate(self.steps):
+            # 1. Call all BEFORE hooks
+            for hook in self.before_step_hooks:
+                hook(step_idx, transition)
+
+            # 2. Execute the actual processing step
+            transition = processor_step(transition)
+
+            # 3. Call all AFTER hooks
+            for hook in self.after_step_hooks:
+                hook(step_idx, transition)
+
+        return transition
+
+    def register_before_step_hook(self, hook_fn):
+        self.before_step_hooks.append(hook_fn)
+
+    def register_after_step_hook(self, hook_fn):
+        self.after_step_hooks.append(hook_fn)
+```
+
+### Execution Flow
+
+The execution flow looks like this:
+
+```
+Input → Before Hook → Step 0 → After Hook → Before Hook → Step 1 → After Hook → ... → Output
+```
+
+For example, with 3 steps and both hook types:
+
+```python
+def timing_before(step_idx, transition):
+    print(f"⏱️  Starting step {step_idx}")
+
+def validation_after(step_idx, transition):
+    print(f"✅ Completed step {step_idx}")
+
+processor.register_before_step_hook(timing_before)
+processor.register_after_step_hook(validation_after)
+
+# This will output:
+# ⏱️  Starting step 0
+# ✅ Completed step 0
+# ⏱️  Starting step 1
+# ✅ Completed step 1
+# ⏱️  Starting step 2
+# ✅ Completed step 2
+```
+
+### Multiple Hooks
+
+You can register multiple hooks of the same type - they execute in the order registered:
+
+```python
+def log_shapes(step_idx: int, transition: EnvTransition):
+    obs = transition.get(TransitionKey.OBSERVATION)
+    if obs:
+        print(f"Step {step_idx} observation shapes:")
+        for key, value in obs.items():
+            if isinstance(value, torch.Tensor):
+                print(f"  {key}: {value.shape}")
+
+processor.register_after_step_hook(check_nans)      # Executes first
+processor.register_after_step_hook(log_shapes)     # Executes second
+
+# Both hooks will be called after each step in registration order
+output = processor(input_data)
+```
+
+While hooks are excellent for monitoring specific issues (like NaN detection) or gathering metrics during normal pipeline execution, sometimes you need to dive deeper. When you want to understand exactly what happens at each step or debug complex transformation logic, step-through debugging provides the detailed inspection you need.
+
+## Step-Through Debugging
+
+Step-through debugging is like having a slow-motion replay for your pipeline. Instead of watching your data get transformed in one quick blur from input to output, you can pause and examine what happens after each individual step.
+
+This approach is particularly valuable when you're trying to understand a complex pipeline, debug unexpected behavior, or verify that each transformation is working as expected. Unlike hooks, which are great for automated monitoring, step-through debugging gives you manual, interactive control over the inspection process.
+
+The `step_through()` method is a generator that yields the transition state after each processing step, allowing you to inspect intermediate results. Think of it as creating a series of snapshots of your data as it flows through the pipeline—each snapshot shows you exactly what your data looks like after one more transformation has been applied.
+
+### How Step-Through Works
+
+The `step_through()` method fundamentally changes how the pipeline executes. Instead of running all steps in sequence and only returning the final result, it transforms the pipeline into an iterator that yields intermediate results.
+
+Here's what happens internally: the method starts by converting your input data into the pipeline's internal transition format, then yields this initial state. Next, it applies the first processing step and yields the result. Then it applies the second step to that result and yields again, and so on. Each `yield` gives you a complete snapshot of the transition at that point.
+
+This generator pattern is powerful because it's lazy—the pipeline only computes the next step when you ask for it. This means you can stop at any point, inspect the current state thoroughly, and decide whether to continue. You're not forced to run the entire pipeline just to debug one problematic step.
+
+Instead of running the entire pipeline and only seeing the final result, `step_through()` pauses after each step and gives you the intermediate transition:
+
+```python
+# This creates a generator that yields intermediate states
+for i, intermediate_result in enumerate(processor.step_through(input_data)):
+    print(f"=== After step {i} ===")
+
+    # Inspect the observation at this stage
+    obs = intermediate_result.get(TransitionKey.OBSERVATION)
+    if obs:
+        for key, value in obs.items():
+            if isinstance(value, torch.Tensor):
+                print(f"{key}: shape={value.shape}, dtype={value.dtype}")
+```
+
+### Interactive Debugging with Breakpoints
+
+You can add breakpoints in the step-through loop to interactively debug:
+
+```python
+# Step through the pipeline with debugging
+for i, intermediate in enumerate(processor.step_through(data)):
+    print(f"Step {i}: {processor.steps[i].__class__.__name__}")
+
+    # Set a breakpoint to inspect the current state
+    breakpoint()  # Debugger will pause here
+
+    # You can now inspect 'intermediate' in the debugger:
+    # - Check tensor shapes and values
+    # - Verify expected transformations
+    # - Look for unexpected changes
+```
+
+During the debugger session, you can:
+
+- Examine `intermediate[TransitionKey.OBSERVATION]` to see observation data
+- Check `intermediate[TransitionKey.ACTION]` for action transformations
+- Inspect any part of the transition to understand what each step does
+
+Step-through debugging is perfect for understanding the _data_ transformations, but what about the _structure_ of that data? While hooks and step-through help you debug runtime behavior, you also need to ensure your pipeline produces data in the format expected by downstream components. This is where feature contract validation comes in.
+
+## Validating Feature Contracts
+
+Feature contracts define what data structure your pipeline expects as input and produces as output.
+Validating these contracts helps catch mismatches early.
+
+### Understanding Feature Contracts
+
+Each processor step has a `transform_features()` method that describes how it changes the data structure:
+
+```python
+# Get the expected output features from your pipeline
+initial_features = {
+    PipelineFeatureType.OBSERVATION: {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(7,)),
+        "observation.image": PolicyFeature(type=FeatureType.IMAGE, shape=(3, 224, 224))
+    },
+    PipelineFeatureType.ACTION: {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(4,))
+    }
+}
+
+# Check what your pipeline will output
+output_features = processor.transform_features(initial_features)
+
+print("Input features:")
+for feature_type, features in initial_features.items():
+    print(f"  {feature_type}:")
+    for key, feature in features.items():
+        print(f"    {key}: {feature.type.value}, shape={feature.shape}")
+
+print("\nOutput features:")
+for feature_type, features in output_features.items():
+    print(f"  {feature_type}:")
+    for key, feature in features.items():
+        print(f"    {key}: {feature.type.value}, shape={feature.shape}")
+```
+
+### Verifying Expected Features
+
+Check that your pipeline produces the features you expect:
+
+```python
+# Define what features you expect the pipeline to produce
+expected_keys = ["observation.state", "observation.image", "action"]
+
+print("Validating feature contract...")
+for expected_key in expected_keys:
+    found = False
+    for feature_type, features in output_features.items():
+        if expected_key in features:
+            feature = features[expected_key]
+            print(f"✅ {expected_key}: {feature.type.value}, shape={feature.shape}")
+            found = True
+            break
+
+    if not found:
+        print(f"❌ Missing expected feature: {expected_key}")
+```
+
+This validation helps ensure your pipeline will work correctly with downstream components that expect specific data structures.
+
+## Summary
+
+Now that you understand the three debugging approaches, you can tackle any pipeline issue systematically:
+
+1. **Hooks** - For runtime monitoring and validation without modifying pipeline code
+2. **Step-through** - For inspecting intermediate states and understanding transformations
+3. **Feature validation** - For ensuring data structure contracts are met
+
+**When to use each approach:**
+
+- Start with **step-through debugging** when you need to understand what your pipeline does or when something unexpected happens
+- Add **hooks** for continuous monitoring during development and production to catch issues automatically
+- Use **feature validation** before deployment to ensure your pipeline works with downstream components
+
+These three tools work together to give you the complete observability that complex pipelines naturally lack. With hooks watching for issues, step-through helping you understand behavior, and feature validation ensuring compatibility, you'll be able to debug any pipeline confidently and efficiently.
diff --git a/lerobot/docs/source/earthrover_mini_plus.mdx b/lerobot/docs/source/earthrover_mini_plus.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..884e84d8c05e94cc5d5acc952114ee63285dd8f8
--- /dev/null
+++ b/lerobot/docs/source/earthrover_mini_plus.mdx
@@ -0,0 +1,238 @@
+# EarthRover Mini Plus
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/Earth_Rover_Mini_5_240c9adc-4f9e-44b7-982f-5d1dc24af1d8.png.webp"
+  alt="EarthRover Mini Plus"
+  width="70%"
+/>
+
+The EarthRover Mini Plus is a fully open source mobile robot that connects through the cloud using the Frodobots SDK. This lets you control the robot and record datasets for training AI models.
+
+## What You Need
+
+### Hardware
+
+- EarthRover Mini robot
+- Computer with Python 3.12 or newer
+- Internet connection
+
+### Setting Up the Frodobots SDK
+
+The robot needs the [Frodobots SDK](https://github.com/frodobots-org/earth-rovers-sdk) running on your computer. Here's how:
+
+1. Download and install the SDK:
+
+```bash
+git clone https://github.com/frodobots-org/earth-rovers-sdk.git
+cd earth-rovers-sdk
+pip install -r requirements.txt
+```
+
+2. Save Credentials:
+
+Write your .env variables with the SDK API key and bot name provided by the Frodobots team.
+
+```bash
+SDK_API_TOKEN=your_sdk_api_token_here
+BOT_SLUG=your_bot_slug_here
+CHROME_EXECUTABLE_PATH=/path/to/chrome_or_chromium
+# Default value is MAP_ZOOM_LEVEL=18 https://wiki.openstreetmap.org/wiki/Zoom_levels
+MAP_ZOOM_LEVEL=18
+MISSION_SLUG=your_mission_slug_here
+# Image quality between 0.1 and 1.0 (default: 0.8)
+# Recommended: 0.8 for better performance
+IMAGE_QUALITY=0.8
+# Image format: jpeg, png or webp (default: png)
+# Recommended: jpeg for better performance and lower bandwidth usage
+IMAGE_FORMAT=jpeg
+```
+
+3. Start the SDK:
+
+```bash
+hypercorn main:app --reload
+```
+
+4. Open your web browser and go to `http://localhost:8000`, then click "Join"
+
+The SDK gives you:
+
+- Live video from front and rear cameras
+
+> [!IMPORTANT]
+> The SDK must be running before you can use the robot.
+
+## Install LeRobot
+
+Follow our [Installation Guide](./installation) to install LeRobot.
+
+In addition to the base installation, install the EarthRover Mini dependencies:
+
+```bash
+pip install -e .
+```
+
+## How It Works
+
+The robot uses the internet to communicate:
+
+- **Movement commands**: Sent through the SDK
+- **Camera video**: Received from the SDK
+- **Robot info**: Battery, location, speed from the SDK
+
+You don't need to plug anything in - it all works through the SDK.
+
+## Calibration
+
+No calibration needed! The robot is ready to use as soon as the SDK is running.
+
+## Controlling the Robot
+
+You control the robot using your keyboard - just like playing a video game with WASD keys.
+
+### Keyboard Controls
+
+| Key | Action                           |
+| --- | -------------------------------- |
+| W   | Move forward                     |
+| S   | Move backward                    |
+| A   | Turn left (with forward motion)  |
+| D   | Turn right (with forward motion) |
+| Q   | Rotate left in place             |
+| E   | Rotate right in place            |
+| X   | Stop all movement                |
+| +/= | Increase speed                   |
+| -   | Decrease speed                   |
+| ESC | Disconnect                       |
+
+### Speed Settings
+
+You can adjust how fast the robot moves:
+
+- **Forward/backward speed**: Default is full speed (1.0)
+- **Turning speed**: Default is full speed (1.0)
+- **Speed changes**: Use +/- keys to adjust by 0.1 each time
+
+### Try It Out
+
+Test driving the robot before recording data:
+
+```python
+from lerobot.robots.earthrover_mini_plus import EarthRoverMiniPlus, EarthRoverMiniPlusConfig
+from lerobot.teleoperators.keyboard import KeyboardRoverTeleop, KeyboardRoverTeleopConfig
+
+# Initialize robot
+robot_config = EarthRoverMiniPlusConfig()
+robot = EarthRoverMiniPlus(robot_config)
+
+# Initialize teleoperator
+teleop_config = KeyboardRoverTeleopConfig(
+    linear_speed=1.0,
+    angular_speed=1.0,
+    speed_increment=0.1
+)
+teleop = KeyboardRoverTeleop(teleop_config)
+
+# Connect
+robot.connect()
+teleop.connect()
+
+# Teleoperate (use keyboard controls)
+try:
+    while True:
+        action = teleop.get_action()
+        robot.send_action(action)
+except KeyboardInterrupt:
+    pass
+finally:
+    robot.disconnect()
+    teleop.disconnect()
+```
+
+> [!TIP]
+> If you're using a Mac, you might need to give Terminal permission to access your keyboard for teleoperation. Go to System Preferences > Security & Privacy > Input Monitoring and check the box for Terminal.
+
+## Recording Data
+
+Once you can drive the robot well, you can start recording data to train AI models. The system records:
+
+- **What you do**: How you move the robot (forward, backward, turning)
+- **What the robot sees**:
+  - Videos from both cameras
+  - Robot speed and direction
+  - Battery level and location
+  - GPS position and signal
+  - Other sensor data
+- **When it happened**: Timestamps for everything
+
+### Setting Up Hugging Face
+
+We use Hugging Face to store your data online. First, log in with your token from [Hugging Face settings](https://huggingface.co/settings/tokens):
+
+```bash
+hf auth login --token ${HUGGINGFACE_TOKEN} --add-to-git-credential
+```
+
+Store your Hugging Face username:
+
+```bash
+HF_USER=$(hf auth whoami | awk -F': *' 'NR==1 {print $2}')
+echo $HF_USER
+```
+
+### Start Recording
+
+Use the standard recording command:
+
+```bash
+lerobot-record \
+    --robot.type=earthrover_mini_plus \
+    --teleop.type=keyboard_rover \
+    --dataset.repo_id=your_username/dataset_name \
+    --dataset.num_episodes=2 \
+    --dataset.fps=10 \
+    --dataset.single_task="Navigate around obstacles" \
+    --dataset.streaming_encoding=true \
+    --dataset.encoder_threads=2 \
+    # --dataset.vcodec=auto \
+    --display_data=true
+```
+
+Replace `your_username/dataset_name` with your Hugging Face username and a name for your dataset.
+
+### What Gets Saved
+
+Your dataset includes:
+
+**Your Actions (2 features)**:
+
+- `linear_velocity`: How much you moved forward/backward
+- `angular_velocity`: How much you turned left/right
+
+**Robot Observations (24 features)**:
+
+- Front camera video
+- Rear camera video
+- Current speed
+- Battery level
+- Orientation
+- GPS (latitude, longitude, signal strength)
+- Network signal strength
+- Vibration level
+- Lamp state (on/off)
+- Accelerometer (x, y, z)
+- Gyroscope (x, y, z)
+- Magnetometer (x, y, z)
+- Wheel RPMs (4 wheels)
+
+### Where Your Data Goes
+
+On your computer: `~/.cache/huggingface/lerobot/{repo-id}`
+
+After recording, your data automatically uploads to your Hugging Face page:
+
+```bash
+echo https://huggingface.co/datasets/${HF_USER}/earthrover-navigation
+```
+
+Your dataset will be tagged with `LeRobot` for community discovery.
diff --git a/lerobot/docs/source/env_processor.mdx b/lerobot/docs/source/env_processor.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..8dbf315c7a1ab4294a95990176275be647ed4a00
--- /dev/null
+++ b/lerobot/docs/source/env_processor.mdx
@@ -0,0 +1,418 @@
+# Environment Processors
+
+Environment processors are a critical layer in LeRobot's data processing architecture that handle **environment-specific** transformations, separate from policy-specific processing. This separation of concerns enables cleaner code, better modularity, and easier experimentation with different environments and policies.
+
+## Why Environment Processors?
+
+When working with different robot environments (LIBERO, MetaWorld, Aloha, etc.), each environment often has unique data formats, coordinate systems, and conventions that need standardization **before** policy processing. Without environment processors, these transformations would be:
+
+1. **Hardcoded in environment code** - Making it difficult to experiment with different state representations
+2. **Duplicated across policies** - Each policy would need to handle environment-specific quirks
+3. **Mixed with policy logic** - Violating separation of concerns and making debugging harder
+
+Environment processors solve this by providing a **dedicated processing layer** between raw environment observations and policy inputs.
+
+## The Processing Pipeline
+
+Here's how data flows through the complete processing pipeline during evaluation:
+
+```python
+# In lerobot_eval.py rollout() function:
+
+# 1. Raw environment observation (numpy arrays, various formats)
+raw_observation = env.step(action)
+
+# 2. Convert numpy to torch, normalize images [0,1]
+observation = preprocess_observation(raw_observation)
+
+# 3. Add task metadata (for multi-task environments)
+observation = add_envs_task(env, observation)
+
+# 4. ENVIRONMENT-SPECIFIC preprocessing (NEW!)
+#    - Flatten robot states
+#    - Rotate images to match dataset conventions
+#    - Handle environment-specific coordinate systems
+observation = env_preprocessor(observation)
+
+# 5. POLICY-SPECIFIC preprocessing
+#    - Normalize with dataset statistics
+#    - Add batch dimensions
+#    - Move to GPU
+#    - Tokenize language instructions
+observation = preprocessor(observation)
+
+# 6. Policy inference
+action = policy.select_action(observation)
+
+# 7. POLICY-SPECIFIC postprocessing
+#    - Unnormalize actions
+#    - Remove batch dimensions
+action = postprocessor(action)
+
+# 8. ENVIRONMENT-SPECIFIC postprocessing (NEW!)
+#    - Convert action formats if needed
+#    - Apply environment-specific constraints
+action_transition = {"action": action}
+action_transition = env_postprocessor(action_transition)
+action = action_transition["action"]
+
+# 9. Execute in environment
+env.step(action)
+```
+
+## The Benefits
+
+### 1. **Separation of Concerns**
+
+Environment processors handle transformations specific to the **environment's data format**, while policy processors handle transformations specific to the **model's requirements**.
+
+```python
+# ❌ Before: Mixed concerns
+class LiberoVLAPolicy:
+    def preprocess(self, obs):
+        # Environment-specific: Flatten robot state (shouldn't be in policy!)
+        state = self._flatten_robot_state(obs["robot_state"])
+        # Policy-specific: Normalize with dataset stats
+        state = self.normalizer(state)
+        return state
+
+# ✅ After: Clear separation
+# Environment processor: Handles LIBERO's nested robot state
+env_preprocessor = LiberoProcessorStep()  # Flattens robot_state
+
+# Policy processor: Handles model requirements
+policy_preprocessor = NormalizerProcessorStep(stats=dataset_stats)
+```
+
+### 2. **Flexibility and Reusability**
+
+The same policy can work with different environment processors, and the same environment processor can work with different policies:
+
+```python
+# Use SmolVLA policy with LIBERO environment
+libero_preprocessor, libero_postprocessor = make_env_pre_post_processors(libero_cfg)
+smolvla_preprocessor, smolvla_postprocessor = make_pre_post_processors(smolvla_cfg)
+
+# Or use ACT policy with the same LIBERO environment
+libero_preprocessor, libero_postprocessor = make_env_pre_post_processors(libero_cfg)
+act_preprocessor, act_postprocessor = make_pre_post_processors(act_cfg)
+```
+
+### 3. **Easier Experimentation**
+
+Want to try different state representations for LIBERO? Just create a new processor:
+
+```python
+# Original: 8D state (pos + quat→axisangle + gripper)
+@ProcessorStepRegistry.register("libero_processor")
+class LiberoProcessorStep(ObservationProcessorStep):
+    def _process_observation(self, obs):
+        eef_pos = robot_state["eef"]["pos"]          # 3D
+        eef_axisangle = quat2axisangle(quat)         # 3D
+        gripper = robot_state["gripper"]["qpos"]     # 2D
+        state = torch.cat([eef_pos, eef_axisangle, gripper], dim=-1)  # 8D
+        return state
+
+# Experiment: Add velocity for better control
+@ProcessorStepRegistry.register("libero_velocity_processor")
+class LiberoVelocityProcessorStep(ObservationProcessorStep):
+    def _process_observation(self, obs):
+        # Include velocities for 14D state
+        eef_pos = robot_state["eef"]["pos"]          # 3D
+        eef_axisangle = quat2axisangle(quat)         # 3D
+        eef_vel = robot_state["eef"]["vel"]          # 3D  (NEW)
+        gripper_pos = robot_state["gripper"]["qpos"] # 2D
+        gripper_vel = robot_state["gripper"]["qvel"] # 3D  (NEW)
+        state = torch.cat([eef_pos, eef_axisangle, eef_vel,
+                          gripper_pos, gripper_vel], dim=-1)  # 14D
+        return state
+```
+
+### 4. **Cleaner Environment Code**
+
+Environments expose **all available data** without needing to know what downstream models will use:
+
+```python
+# LIBERO environment exposes full robot state
+observation = {
+    "pixels": {"image": img, "image2": img2},
+    "robot_state": {
+        "eef": {"pos": ..., "quat": ..., "vel": ..., "mat": ..., "axisangle": ...},
+        "gripper": {"qpos": ..., "qvel": ...},
+        "joints": {"pos": ..., "vel": ...}
+    }
+}
+
+# Environment processor decides what to use
+# Policy processor handles model-specific transformations
+```
+
+## Using Environment Processors
+
+### Factory Function
+
+The `make_env_pre_post_processors` function follows the same pattern as `make_pre_post_processors` for policies:
+
+```python
+from lerobot.envs.factory import make_env_pre_post_processors
+from lerobot.envs.configs import LiberoEnv, PushtEnv
+
+# For LIBERO: Returns LiberoProcessorStep in preprocessor
+libero_cfg = LiberoEnv(task="libero_spatial", camera_name=["agentview"])
+env_preprocessor, env_postprocessor = make_env_pre_post_processors(libero_cfg)
+
+# For other environments: Returns identity processors (no-op)
+pusht_cfg = PushtEnv()
+env_preprocessor, env_postprocessor = make_env_pre_post_processors(pusht_cfg)
+```
+
+### Implementation in `envs/factory.py`
+
+```python
+def make_env_pre_post_processors(
+    env_cfg: EnvConfig,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+]:
+    """
+    Create preprocessor and postprocessor pipelines for environment observations.
+
+    Args:
+        env_cfg: The configuration of the environment.
+
+    Returns:
+        A tuple containing:
+            - preprocessor: Pipeline that processes environment observations
+            - postprocessor: Pipeline that processes environment outputs
+    """
+    # For LIBERO environments, add the LiberoProcessorStep to preprocessor
+    if isinstance(env_cfg, LiberoEnv) or "libero" in env_cfg.type:
+        preprocessor = PolicyProcessorPipeline(steps=[LiberoProcessorStep()])
+    else:
+        # For all other environments, return an identity preprocessor
+        preprocessor = PolicyProcessorPipeline(steps=[])
+
+    # Postprocessor is currently identity for all environments
+    # Future: Could add environment-specific action transformations
+    postprocessor = PolicyProcessorPipeline(steps=[])
+
+    return preprocessor, postprocessor
+```
+
+### Integration in Evaluation
+
+In `lerobot_eval.py`, the environment processors are created once and used throughout:
+
+```python
+def eval_main(cfg: EvalPipelineConfig):
+    # Create environment
+    envs = make_env(cfg.env, n_envs=cfg.eval.batch_size)
+
+    # Create policy
+    policy = make_policy(cfg=cfg.policy, env_cfg=cfg.env)
+
+    # Create policy processors
+    preprocessor, postprocessor = make_pre_post_processors(
+        policy_cfg=cfg.policy,
+        pretrained_path=cfg.policy.pretrained_path,
+    )
+
+    # Create environment processors (NEW!)
+    env_preprocessor, env_postprocessor = make_env_pre_post_processors(env_cfg=cfg.env)
+
+    # Run evaluation with both processor types
+    eval_policy_all(
+        envs=envs,
+        policy=policy,
+        env_preprocessor=env_preprocessor,      # Environment-specific
+        env_postprocessor=env_postprocessor,    # Environment-specific
+        preprocessor=preprocessor,              # Policy-specific
+        postprocessor=postprocessor,            # Policy-specific
+        n_episodes=cfg.eval.n_episodes,
+    )
+```
+
+## Example: LIBERO Environment Processor
+
+The `LiberoProcessorStep` demonstrates a real-world environment processor:
+
+```python
+from lerobot.processor.pipeline import ObservationProcessorStep
+
+@dataclass
+@ProcessorStepRegistry.register(name="libero_processor")
+class LiberoProcessorStep(ObservationProcessorStep):
+    """
+    Processes LIBERO observations into the LeRobot format.
+
+    **State Processing:**
+    - Extracts end-effector position (3D)
+    - Converts quaternion to axis-angle representation (3D)
+    - Extracts gripper joint positions (2D)
+    - Concatenates into 8D state vector
+
+    **Image Processing:**
+    - Rotates images 180° to match HuggingFaceVLA/libero convention
+    """
+
+    def _process_observation(self, observation):
+        processed_obs = observation.copy()
+
+        # Process images: Flip 180° for camera convention
+        for key in list(processed_obs.keys()):
+            if key.startswith("observation.images."):
+                img = processed_obs[key]
+                img = torch.flip(img, dims=[2, 3])  # Flip H and W
+                processed_obs[key] = img
+
+        # Process robot_state: Flatten to 8D vector
+        if "observation.robot_state" in processed_obs:
+            robot_state = processed_obs.pop("observation.robot_state")
+
+            eef_pos = robot_state["eef"]["pos"]           # (B, 3)
+            eef_quat = robot_state["eef"]["quat"]         # (B, 4)
+            gripper_qpos = robot_state["gripper"]["qpos"] # (B, 2)
+
+            # Convert quaternion to axis-angle
+            eef_axisangle = self._quat2axisangle(eef_quat)  # (B, 3)
+
+            # Concatenate into single state vector
+            state = torch.cat((eef_pos, eef_axisangle, gripper_qpos), dim=-1)
+            state = state.float()
+
+            processed_obs["observation.state"] = state
+
+        return processed_obs
+```
+
+### Why These Transformations?
+
+1. **Image Rotation**: The HuggingFaceVLA/libero dataset has images rotated 180° from the raw LIBERO simulator. The processor handles this convention mismatch so policies trained on the dataset work seamlessly.
+
+2. **State Flattening**: The raw LIBERO environment exposes nested dictionaries with all available state information (position, quaternion, velocity, matrix representation, etc.). The processor:
+   - Selects the relevant components (pos, quat, gripper)
+   - Converts quaternion to axis-angle (more suitable for learning)
+   - Flattens to a single 8D vector that policies expect
+
+3. **Flexibility**: The environment still exposes **all** raw data. If you want to try different state representations (e.g., including velocities, using matrix representation instead of axis-angle), you can create a new processor without modifying the environment code.
+
+## Adding Environment Processors for New Environments
+
+To add environment processors for a new environment:
+
+### 1. Create the Processor Step
+
+```python
+# In src/lerobot/processor/env_processor.py
+
+@dataclass
+@ProcessorStepRegistry.register(name="myenv_processor")
+class MyEnvProcessorStep(ObservationProcessorStep):
+    """Process observations from MyEnv."""
+
+    def _process_observation(self, observation):
+        processed = observation.copy()
+
+        # Your environment-specific transformations
+        if "myenv.specific.state" in processed:
+            state = processed.pop("myenv.specific.state")
+            # Transform to standard format
+            processed["observation.state"] = self._transform_state(state)
+
+        return processed
+```
+
+### 2. Update the Factory
+
+```python
+# In src/lerobot/envs/factory.py
+
+def make_env_pre_post_processors(env_cfg: EnvConfig):
+    if isinstance(env_cfg, LiberoEnv) or "libero" in env_cfg.type:
+        preprocessor = PolicyProcessorPipeline(steps=[LiberoProcessorStep()])
+    elif isinstance(env_cfg, MyEnvConfig) or "myenv" in env_cfg.type:
+        preprocessor = PolicyProcessorPipeline(steps=[MyEnvProcessorStep()])
+    else:
+        preprocessor = PolicyProcessorPipeline(steps=[])
+
+    postprocessor = PolicyProcessorPipeline(steps=[])
+    return preprocessor, postprocessor
+```
+
+### 3. Use in Evaluation
+
+No changes needed! The evaluation script automatically uses the appropriate processor:
+
+```bash
+lerobot-eval \
+    --policy.path=lerobot/my_policy \
+    --env.type=myenv \  # Automatically uses MyEnvProcessorStep
+    --eval.n_episodes=10
+```
+
+## Future: Environment Postprocessors
+
+Currently, postprocessors are identity (no-op) for all environments. Future use cases include:
+
+### Action Space Transformations
+
+```python
+@dataclass
+class MyEnvActionPostprocessor(ProcessorStep):
+    """Convert policy actions to environment-specific format."""
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        action = transition["action"]
+
+        # Example: Convert from Cartesian to joint space
+        if self.action_space == "joint":
+            action = self.ik_solver(action)
+
+        # Example: Apply environment-specific safety limits
+        action = torch.clamp(action, self.min_action, self.max_action)
+
+        transition["action"] = action
+        return transition
+```
+
+### Coordinate System Conversions
+
+```python
+@dataclass
+class CoordinateTransformPostprocessor(ProcessorStep):
+    """Transform actions between coordinate systems."""
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        action = transition["action"]
+
+        # Example: Policy outputs in world frame, env expects base frame
+        action = self.world_to_base_transform(action)
+
+        transition["action"] = action
+        return transition
+```
+
+## Best Practices
+
+1. **Keep environment processors simple**: They should only handle environment-specific data format issues, not complex learning-related transformations.
+
+2. **Use policy processors for model requirements**: Normalization, batching, device placement, and tokenization belong in policy processors.
+
+3. **Expose all data from environments**: Let processors decide what to use rather than hardcoding choices in the environment.
+
+4. **Document conventions**: Clearly document any coordinate system conventions, camera orientations, or data formats that your processor handles.
+
+5. **Test independently**: Environment processors should be testable without loading full policies or environments.
+
+## Summary
+
+Environment processors provide a **clean separation** between environment-specific data transformations and policy-specific model requirements. This architecture:
+
+- ✅ Enables easy experimentation with different state representations
+- ✅ Allows policies to work seamlessly across different environments
+- ✅ Keeps environment code focused on simulation/hardware interface
+- ✅ Makes processor pipelines more maintainable and debuggable
+- ✅ Follows the single responsibility principle
+
+The key insight: **Environments define data formats, processors standardize them, policies consume standardized data.** Each layer has a clear, focused responsibility.
diff --git a/lerobot/docs/source/envhub.mdx b/lerobot/docs/source/envhub.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..36c08a8b32ae87e92f987b7976594a37cd0220a1
--- /dev/null
+++ b/lerobot/docs/source/envhub.mdx
@@ -0,0 +1,431 @@
+# Loading Environments from the Hub
+
+The **EnvHub** feature allows you to load simulation environments directly from the Hugging Face Hub with a single line of code. This unlocks a powerful new model for collaboration: instead of environments being locked away inside monolithic libraries, anyone can publish custom environments and share them with the community.
+
+## What is EnvHub?
+
+EnvHub lets you create custom robotics simulation environments with your own robot models and scenarios, and make them easily usable by anyone through the LeRobot framework.
+
+EnvHub packages are stored on the Hugging Face Hub, and can be seamlessly pulled and used in your AI robotics projects through LeRobot with a single line of code.
+
+Thanks to EnvHub, you can:
+
+1. **Create and publish environments** to the Hugging Face Hub as Git repositories, and distribute complex physics simulations without packaging hassles
+2. **Load environments** dynamically, without installing them as packages
+3. **Version and track** environment changes using Git semantics
+4. **Discover** new simulation tasks shared by the community
+
+This design means you can go from discovering an interesting environment on the Hub to running experiments in seconds, or create your own custom robot and environment without worrying about dependency conflicts or complex installation procedures.
+
+When you create an EnvHub package, you can build anything you want inside it and use any simulation tool you like: this is your own space to play with. The only requirement is that the package contains an `env.py` file that defines the environment and allows LeRobot to load and use your EnvHub package.
+
+This `env.py` file needs to expose a small API so LeRobot can load and run it. In particular, you must provide a `make_env(n_envs: int = 1, use_async_envs: bool = False)` or `make_env(n_envs: int = 1, use_async_envs: bool = False, cfg: EnvConfig)` function, which is the main entry point for LeRobot. It should return one of:
+
+- A `gym.vector.VectorEnv` (most common)
+- A single `gym.Env` (will be automatically wrapped)
+- A dict mapping `{suite_name: {task_id: VectorEnv}}` (for multi-task benchmarks)
+
+You can also pass an `EnvConfig` object to `make_env` to configure the environment (e.g. the number of environments, task, camera name, initial states, control mode, episode length, etc.).
+
+Finally, your environment must implement the standard `gym.vector.VectorEnv` interface so it works with LeRobot, including methods like `reset` and `step`.
+
+## Quick Start
+
+Loading an environment from the Hub is as simple as:
+
+```python
+from lerobot.envs.factory import make_env
+
+# Load a hub environment (requires explicit consent to run remote code)
+env = make_env("lerobot/cartpole-env", trust_remote_code=True)
+```
+
+<Tip warning={true}>
+  **Security Notice**: Loading environments from the Hub executes Python code
+  from third-party repositories. Only use `trust_remote_code=True` with
+  repositories you trust. We strongly recommend pinning to a specific commit
+  hash for reproducibility and security.
+</Tip>
+
+## Repository Structure
+
+To make your environment loadable from the Hub, your repository must contain at minimum:
+
+### Required Files
+
+**`env.py`** (or custom Python file)
+
+- Must expose a `make_env(n_envs: int, use_async_envs: bool)` function
+- This function should return one of:
+  - A `gym.vector.VectorEnv` (most common)
+  - A single `gym.Env` (will be automatically wrapped)
+  - A dict mapping `{suite_name: {task_id: VectorEnv}}` (for multi-task benchmarks)
+
+### Optional Files
+
+**`requirements.txt`**
+
+- List any additional dependencies your environment needs
+- Users will need to install these manually before loading your environment
+
+**`README.md`**
+
+- Document your environment: what task it implements, observation/action spaces, rewards, etc.
+- Include usage examples and any special setup instructions
+
+**`.gitignore`**
+
+- Exclude unnecessary files from your repository
+
+### Example Repository Structure
+
+```
+my-environment-repo/
+├── env.py                 # Main environment definition (required)
+├── requirements.txt       # Dependencies (optional)
+├── README.md             # Documentation (recommended)
+├── assets/               # Images, videos, etc. (optional)
+│   └── demo.gif
+└── configs/              # Config files if needed (optional)
+    └── task_config.yaml
+```
+
+## Creating Your Environment Repository
+
+### Step 1: Define Your Environment
+
+Create an `env.py` file with a `make_env` function:
+
+```python
+# env.py
+import gymnasium as gym
+
+def make_env(n_envs: int = 1, use_async_envs: bool = False):
+    """
+    Create vectorized environments for your custom task.
+
+    Args:
+        n_envs: Number of parallel environments
+        use_async_envs: Whether to use AsyncVectorEnv or SyncVectorEnv
+
+    Returns:
+        gym.vector.VectorEnv or dict mapping suite names to vectorized envs
+    """
+    def _make_single_env():
+        # Create your custom environment
+        return gym.make("CartPole-v1")
+
+    # Choose vector environment type
+    env_cls = gym.vector.AsyncVectorEnv if use_async_envs else gym.vector.SyncVectorEnv
+
+    # Create vectorized environment
+    vec_env = env_cls([_make_single_env for _ in range(n_envs)])
+
+    return vec_env
+```
+
+### Step 2: Test Locally
+
+Before uploading, test your environment locally:
+
+```python
+from lerobot.envs.utils import _load_module_from_path, _call_make_env, _normalize_hub_result
+
+# Load your module
+module = _load_module_from_path("./env.py")
+
+# Test the make_env function
+result = _call_make_env(module, n_envs=2, use_async_envs=False)
+normalized = _normalize_hub_result(result)
+
+# Verify it works
+suite_name = next(iter(normalized))
+env = normalized[suite_name][0]
+obs, info = env.reset()
+print(f"Observation shape: {obs.shape if hasattr(obs, 'shape') else type(obs)}")
+env.close()
+```
+
+### Step 3: Upload to the Hub
+
+Upload your repository to Hugging Face:
+
+```bash
+# Install huggingface_hub if needed
+pip install huggingface_hub
+
+# Login to Hugging Face
+hf auth login
+
+# Create a new repository
+hf repo create my-org/my-custom-env
+
+# Initialize git and push
+git init
+git add .
+git commit -m "Initial environment implementation"
+git remote add origin https://huggingface.co/my-org/my-custom-env
+git push -u origin main
+```
+
+Alternatively, use the `huggingface_hub` Python API:
+
+```python
+from huggingface_hub import HfApi
+
+api = HfApi()
+
+# Create repository
+api.create_repo("my-custom-env", repo_type="space")
+
+# Upload files
+api.upload_folder(
+    folder_path="./my-env-folder",
+    repo_id="username/my-custom-env",
+    repo_type="space",
+)
+```
+
+## Loading Environments from the Hub
+
+### Basic Usage
+
+```python
+from lerobot.envs.factory import make_env
+
+# Load from the hub
+envs_dict = make_env(
+    "username/my-custom-env",
+    n_envs=4,
+    trust_remote_code=True
+)
+
+# Access the environment
+suite_name = next(iter(envs_dict))
+env = envs_dict[suite_name][0]
+
+# Use it like any gym environment
+obs, info = env.reset()
+action = env.action_space.sample()
+obs, reward, terminated, truncated, info = env.step(action)
+```
+
+### Advanced: Pinning to Specific Versions
+
+For reproducibility and security, pin to a specific Git revision:
+
+```python
+# Pin to a specific branch
+env = make_env("username/my-env@main", trust_remote_code=True)
+
+# Pin to a specific commit (recommended for papers/experiments)
+env = make_env("username/my-env@abc123def456", trust_remote_code=True)
+
+# Pin to a tag
+env = make_env("username/my-env@v1.0.0", trust_remote_code=True)
+```
+
+### Custom File Paths
+
+If your environment definition is not in `env.py`:
+
+```python
+# Load from a custom file
+env = make_env("username/my-env:custom_env.py", trust_remote_code=True)
+
+# Combine with version pinning
+env = make_env("username/my-env@v1.0:envs/task_a.py", trust_remote_code=True)
+```
+
+### Async Environments
+
+For better performance with multiple environments:
+
+```python
+envs_dict = make_env(
+    "username/my-env",
+    n_envs=8,
+    use_async_envs=True,  # Use AsyncVectorEnv for parallel execution
+    trust_remote_code=True
+)
+```
+
+## URL Format Reference
+
+The hub URL format supports several patterns:
+
+| Pattern              | Description                    | Example                                |
+| -------------------- | ------------------------------ | -------------------------------------- |
+| `user/repo`          | Load `env.py` from main branch | `make_env("lerobot/pusht-env")`        |
+| `user/repo@revision` | Load from specific revision    | `make_env("lerobot/pusht-env@main")`   |
+| `user/repo:path`     | Load custom file               | `make_env("lerobot/envs:pusht.py")`    |
+| `user/repo@rev:path` | Revision + custom file         | `make_env("lerobot/envs@v1:pusht.py")` |
+
+## Multi-Task Environments
+
+For benchmarks with multiple tasks (like LIBERO), return a nested dictionary:
+
+```python
+def make_env(n_envs: int = 1, use_async_envs: bool = False):
+    env_cls = gym.vector.AsyncVectorEnv if use_async_envs else gym.vector.SyncVectorEnv
+
+    # Return dict: {suite_name: {task_id: VectorEnv}}
+    return {
+        "suite_1": {
+            0: env_cls([lambda: gym.make("Task1-v0") for _ in range(n_envs)]),
+            1: env_cls([lambda: gym.make("Task2-v0") for _ in range(n_envs)]),
+        },
+        "suite_2": {
+            0: env_cls([lambda: gym.make("Task3-v0") for _ in range(n_envs)]),
+        }
+    }
+```
+
+## Security Considerations
+
+<Tip warning={true}>
+  **Important**: The `trust_remote_code=True` flag is required to execute
+  environment code from the Hub. This is by design for security.
+</Tip>
+
+When loading environments from the Hub:
+
+1. **Review the code first**: Visit the repository and inspect `env.py` before loading
+2. **Pin to commits**: Use specific commit hashes for reproducibility
+3. **Check dependencies**: Review `requirements.txt` for suspicious packages
+4. **Use trusted sources**: Prefer official organizations or well-known researchers
+5. **Sandbox if needed**: Run untrusted code in isolated environments (containers, VMs)
+
+Example of safe usage:
+
+```python
+# ❌ BAD: Loading without inspection
+env = make_env("random-user/untrusted-env", trust_remote_code=True)
+
+# ✅ GOOD: Review code, then pin to specific commit
+# 1. Visit https://huggingface.co/trusted-org/verified-env
+# 2. Review the env.py file
+# 3. Copy the commit hash
+env = make_env("trusted-org/verified-env@a1b2c3d4", trust_remote_code=True)
+```
+
+## Example: CartPole from the Hub
+
+Here's a complete example using the reference CartPole environment:
+
+```python
+from lerobot.envs.factory import make_env
+import numpy as np
+
+# Load the environment
+envs_dict = make_env("lerobot/cartpole-env", n_envs=4, trust_remote_code=True)
+
+# Get the vectorized environment
+suite_name = next(iter(envs_dict))
+env = envs_dict[suite_name][0]
+
+# Run a simple episode
+obs, info = env.reset()
+done = np.zeros(env.num_envs, dtype=bool)
+total_reward = np.zeros(env.num_envs)
+
+while not done.all():
+    # Random policy
+    action = env.action_space.sample()
+    obs, reward, terminated, truncated, info = env.step(action)
+    total_reward += reward
+    done = terminated | truncated
+
+print(f"Average reward: {total_reward.mean():.2f}")
+env.close()
+```
+
+## Benefits of EnvHub
+
+### For Environment Authors
+
+- **Easy distribution**: No PyPI packaging required
+- **Version control**: Use Git for environment versioning
+- **Rapid iteration**: Push updates instantly
+- **Documentation**: Hub README renders beautifully
+- **Community**: Reach LeRobot users directly
+
+### For Researchers
+
+- **Quick experiments**: Load any environment in one line
+- **Reproducibility**: Pin to specific commits
+- **Discovery**: Browse environments on the Hub
+- **No conflicts**: No need to install conflicting packages
+
+### For the Community
+
+- **Growing ecosystem**: More diverse simulation tasks
+- **Standardization**: Common `make_env` API
+- **Collaboration**: Fork and improve existing environments
+- **Accessibility**: Lower barrier to sharing research
+
+## Troubleshooting
+
+### "Refusing to execute remote code"
+
+You must explicitly pass `trust_remote_code=True`:
+
+```python
+env = make_env("user/repo", trust_remote_code=True)
+```
+
+### "Module X not found"
+
+The hub environment has dependencies you need to install:
+
+```bash
+# Check the repo's requirements.txt and install dependencies
+pip install gymnasium numpy
+```
+
+### "make_env not found in module"
+
+Your `env.py` must expose a `make_env` function:
+
+```python
+def make_env(n_envs: int, use_async_envs: bool):
+    # Your implementation
+    pass
+```
+
+### Environment returns wrong type
+
+The `make_env` function must return:
+
+- A `gym.vector.VectorEnv`, or
+- A single `gym.Env`, or
+- A dict `{suite_name: {task_id: VectorEnv}}`
+
+## Best Practices
+
+1. **Document your environment**: Include observation/action space descriptions, reward structure, and termination conditions in your README
+2. **Add requirements.txt**: List all dependencies with versions
+3. **Test thoroughly**: Verify your environment works locally before pushing
+4. **Use semantic versioning**: Tag releases with version numbers
+5. **Add examples**: Include usage examples in your README
+6. **Keep it simple**: Minimize dependencies when possible
+7. **License your work**: Add a LICENSE file to clarify usage terms
+
+## Future Directions
+
+The EnvHub ecosystem enables exciting possibilities:
+
+- **GPU-accelerated physics**: Share Isaac Gym or Brax environments
+- **Photorealistic rendering**: Distribute environments with advanced graphics
+- **Multi-agent scenarios**: Complex interaction tasks
+- **Real-world simulators**: Digital twins of physical setups
+- **Procedural generation**: Infinite task variations
+- **Domain randomization**: Pre-configured DR pipelines
+
+As more researchers and developers contribute, the diversity and quality of available environments will grow, benefiting the entire robotics learning community.
+
+## See Also
+
+- [Hugging Face Hub Documentation](https://huggingface.co/docs/hub/en/index)
+- [Gymnasium Documentation](https://gymnasium.farama.org/index.html)
+- [Example Hub Environment](https://huggingface.co/lerobot/cartpole-env)
diff --git a/lerobot/docs/source/envhub_isaaclab_arena.mdx b/lerobot/docs/source/envhub_isaaclab_arena.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..828d51bad30db3a263e6ca43098bcba64044cf20
--- /dev/null
+++ b/lerobot/docs/source/envhub_isaaclab_arena.mdx
@@ -0,0 +1,510 @@
+# NVIDIA IsaacLab Arena & LeRobot
+
+LeRobot EnvHub now supports **GPU-accelerated simulation** with IsaacLab Arena for policy evaluation at scale.
+Train and evaluate imitation learning policies with high-fidelity simulation — all integrated into the LeRobot ecosystem.
+
+<img
+  src="https://huggingface.co/nvidia/isaaclab-arena-envs/resolve/main/assets/Gr1OpenMicrowaveEnvironment.png"
+  alt="IsaacLab Arena - GR1 Microwave Environment"
+  style={{ maxWidth: "100%", borderRadius: "8px", marginBottom: "1rem" }}
+/>
+
+[IsaacLab Arena](https://github.com/isaac-sim/IsaacLab-Arena) integrates with NVIDIA IsaacLab to provide:
+
+- 🤖 **Humanoid embodiments**: GR1, G1, Galileo with various configurations
+- 🎯 **Manipulation & loco-manipulation tasks**: Door opening, pick-and-place, button pressing, and more
+- ⚡ **GPU-accelerated rollouts**: Parallel environment execution on NVIDIA GPUs
+- 🖼️ **RTX Rendering**: Evaluate vision-based policies with realistic rendering, reflections and refractions
+- 📦 **LeRobot-compatible datasets**: Ready for training with GR00T N1x, PI0, SmolVLA, ACT, and Diffusion policies
+- 🔄 **EnvHub integration**: Load environments from HuggingFace EnvHub with one line
+
+## Installation
+
+### Prerequisites
+
+Hardware requirements are shared with Isaac Sim, and are detailed in [Isaac Sim Requirements](https://docs.isaacsim.omniverse.nvidia.com/5.1.0/installation/requirements.html).
+
+- NVIDIA GPU with CUDA support
+- NVIDIA driver compatible with IsaacSim 5.1.0
+- Linux (Ubuntu 22.04 / 24.04)
+
+### Setup
+
+```bash
+# 1. Create conda environment
+conda create -y -n lerobot-arena python=3.11
+conda activate lerobot-arena
+conda install -y -c conda-forge ffmpeg=7.1.1
+
+# 2. Install Isaac Sim 5.1.0
+pip install "isaacsim[all,extscache]==5.1.0" --extra-index-url https://pypi.nvidia.com
+
+# Accept NVIDIA EULA (required)
+export ACCEPT_EULA=Y
+export PRIVACY_CONSENT=Y
+
+# 3. Install IsaacLab 2.3.0
+git clone https://github.com/isaac-sim/IsaacLab.git
+cd IsaacLab
+git checkout v2.3.0
+./isaaclab.sh -i
+cd ..
+
+# 4. Install IsaacLab Arena
+git clone https://github.com/isaac-sim/IsaacLab-Arena.git
+cd IsaacLab-Arena
+git checkout release/0.1.1
+pip install -e .
+cd ..
+
+
+# 5. Install LeRobot
+git clone https://github.com/huggingface/lerobot.git
+cd lerobot
+pip install -e .
+cd ..
+
+
+# 6. Install additional dependencies
+pip install onnxruntime==1.23.2 lightwheel-sdk==1.0.1 vuer[all]==0.0.70 qpsolvers==4.8.1
+pip install numpy==1.26.0 # Isaac Sim 5.1 depends on numpy==1.26.0, this will be fixed in next release
+```
+
+## Evaluating Policies
+
+### Pre-trained Policies
+
+The following trained policies are available:
+
+| Policy                      | Architecture | Task          | Link                                                                     |
+| :-------------------------- | :----------- | :------------ | :----------------------------------------------------------------------- |
+| pi05-arena-gr1-microwave    | PI0.5        | GR1 Microwave | [HuggingFace](https://huggingface.co/nvidia/pi05-arena-gr1-microwave)    |
+| smolvla-arena-gr1-microwave | SmolVLA      | GR1 Microwave | [HuggingFace](https://huggingface.co/nvidia/smolvla-arena-gr1-microwave) |
+
+### Evaluate SmolVLA
+
+```bash
+pip install -e ".[smolvla]"
+pip install numpy==1.26.0 # revert numpy to version 1.26
+```
+
+```bash
+lerobot-eval \
+    --policy.path=nvidia/smolvla-arena-gr1-microwave \
+    --env.type=isaaclab_arena \
+    --env.hub_path=nvidia/isaaclab-arena-envs \
+    --rename_map='{"observation.images.robot_pov_cam_rgb": "observation.images.robot_pov_cam"}' \
+    --policy.device=cuda \
+    --env.environment=gr1_microwave \
+    --env.embodiment=gr1_pink \
+    --env.object=mustard_bottle \
+    --env.headless=false \
+    --env.enable_cameras=true \
+    --env.video=true \
+    --env.video_length=10 \
+    --env.video_interval=15 \
+    --env.state_keys=robot_joint_pos \
+    --env.camera_keys=robot_pov_cam_rgb \
+    --trust_remote_code=True \
+    --eval.batch_size=1
+```
+
+### Evaluate PI0.5
+
+```bash
+pip install -e ".[pi]"
+pip install numpy==1.26.0 # revert numpy to version 1.26
+```
+
+<Tip>PI0.5 requires disabling torch compile for evaluation:</Tip>
+
+```bash
+TORCH_COMPILE_DISABLE=1 TORCHINDUCTOR_DISABLE=1 lerobot-eval \
+    --policy.path=nvidia/pi05-arena-gr1-microwave \
+    --env.type=isaaclab_arena \
+    --env.hub_path=nvidia/isaaclab-arena-envs \
+    --rename_map='{"observation.images.robot_pov_cam_rgb": "observation.images.robot_pov_cam"}' \
+    --policy.device=cuda \
+    --env.environment=gr1_microwave \
+    --env.embodiment=gr1_pink \
+    --env.object=mustard_bottle \
+    --env.headless=false \
+    --env.enable_cameras=true \
+    --env.video=true \
+    --env.video_length=15 \
+    --env.video_interval=15 \
+    --env.state_keys=robot_joint_pos \
+    --env.camera_keys=robot_pov_cam_rgb \
+    --trust_remote_code=True \
+    --eval.batch_size=1
+```
+
+<Tip>
+  To change the number of parallel environments, use the ```--eval.batch_size```
+  flag.
+</Tip>
+
+### What to Expect
+
+During evaluation, you will see a progress bar showing the running success rate:
+
+```
+Stepping through eval batches:   8%|██████▍    | 4/50 [00:45<08:06, 10.58s/it, running_success_rate=25.0%]
+```
+
+### Video Recording
+
+To enable video recording during evaluation, add the following flags to your command:
+
+```bash
+--env.video=true \
+--env.video_length=15 \
+--env.video_interval=15
+```
+
+For more details on video recording, see the [IsaacLab Recording Documentation](https://isaac-sim.github.io/IsaacLab/main/source/how-to/record_video.html).
+
+<Tip>
+When running headless with `--env.headless=true`, you must also enable cameras explicitly for camera enabled environments:
+
+```bash
+--env.headless=true --env.enable_cameras=true
+```
+
+</Tip>
+
+### Output Directory
+
+Evaluation videos are saved to the output directory with the following structure:
+
+```
+outputs/eval/<date>/<timestamp>_<env>_<policy>/videos/<task>_<env_id>/eval_episode_<n>.mp4
+```
+
+For example:
+
+```
+outputs/eval/2026-01-02/14-38-01_isaaclab_arena_smolvla/videos/gr1_microwave_0/eval_episode_0.mp4
+```
+
+## Training Policies
+
+To learn more about training policies with LeRobot, please refer to the training documentation:
+
+- [SmolVLA](./smolvla)
+- [Pi0.5](./pi05)
+- [GR00T N1.5](./groot)
+
+Sample IsaacLab Arena datasets are available on HuggingFace Hub for experimentation:
+
+| Dataset                                                                                                   | Description                | Frames |
+| :-------------------------------------------------------------------------------------------------------- | :------------------------- | :----- |
+| [Arena-GR1-Manipulation-Task](https://huggingface.co/datasets/nvidia/Arena-GR1-Manipulation-Task-v3)      | GR1 microwave manipulation | ~4K    |
+| [Arena-G1-Loco-Manipulation-Task](https://huggingface.co/datasets/nvidia/Arena-G1-Loco-Manipulation-Task) | G1 loco-manipulation       | ~4K    |
+
+## Environment Configuration
+
+### Full Configuration Options
+
+```python
+from lerobot.envs.configs import IsaaclabArenaEnv
+
+config = IsaaclabArenaEnv(
+    # Environment selection
+    environment="gr1_microwave",      # Task environment
+    embodiment="gr1_pink",            # Robot embodiment
+    object="power_drill",             # Object to manipulate
+
+    # Simulation settings
+    episode_length=300,               # Max steps per episode
+    headless=True,                    # Run without GUI
+    device="cuda:0",                  # GPU device
+    seed=42,                          # Random seed
+
+    # Observation configuration
+    state_keys="robot_joint_pos",     # State observation keys (comma-separated)
+    camera_keys="robot_pov_cam_rgb",  # Camera observation keys (comma-separated)
+    state_dim=54,                     # Expected state dimension
+    action_dim=36,                    # Expected action dimension
+    camera_height=512,                # Camera image height
+    camera_width=512,                 # Camera image width
+    enable_cameras=True,              # Enable camera observations
+
+    # Video recording
+    video=False,                      # Enable video recording
+    video_length=100,                 # Frames per video
+    video_interval=200,               # Steps between recordings
+
+    # Advanced
+    mimic=False,                      # Enable mimic mode
+    teleop_device=None,               # Teleoperation device
+    disable_fabric=False,             # Disable fabric optimization
+    enable_pinocchio=True,            # Enable Pinocchio for IK
+)
+```
+
+### Using Environment Hub directly for advanced usage
+
+Create a file called `test_env_load_arena.py` or [download from the EnvHub](https://huggingface.co/nvidia/isaaclab-arena-envs/blob/main/tests/test_env_load_arena.py):
+
+```python
+import logging
+from dataclasses import asdict
+from pprint import pformat
+import torch
+import tqdm
+from lerobot.configs import parser
+from lerobot.configs.eval import EvalPipelineConfig
+
+
+@parser.wrap()
+def main(cfg: EvalPipelineConfig):
+    """Run random action rollout for IsaacLab Arena environment."""
+    logging.info(pformat(asdict(cfg)))
+
+    from lerobot.envs.factory import make_env
+
+    env_dict = make_env(
+        cfg.env,
+        n_envs=cfg.env.num_envs,
+        trust_remote_code=True,
+    )
+    env = next(iter(env_dict.values()))[0]
+    env.reset()
+    for _ in tqdm.tqdm(range(cfg.env.episode_length)):
+        with torch.inference_mode():
+            actions = env.action_space.sample()
+            obs, rewards, terminated, truncated, info = env.step(actions)
+            if terminated.any() or truncated.any():
+                obs, info = env.reset()
+    env.close()
+
+
+if __name__ == "__main__":
+    main()
+```
+
+Run with:
+
+```bash
+python test_env_load_arena.py \
+    --env.environment=g1_locomanip_pnp \
+    --env.embodiment=gr1_pink \
+    --env.object=cracker_box \
+    --env.num_envs=4 \
+    --env.enable_cameras=true \
+    --env.seed=1000 \
+    --env.video=true \
+    --env.video_length=10 \
+    --env.video_interval=15 \
+    --env.headless=false \
+    --env.hub_path=nvidia/isaaclab-arena-envs \
+    --env.type=isaaclab_arena
+```
+
+## Creating New Environments
+
+First create a new IsaacLab Arena environment by following the [IsaacLab Arena Documentation](https://isaac-sim.github.io/IsaacLab-Arena/release/0.1.1/index.html).
+
+Clone our EnvHub repo:
+
+```bash
+git clone https://huggingface.co/nvidia/isaaclab-arena-envs
+```
+
+Modify the `example_envs.yaml` file based on your new environment.
+[Upload](./envhub#step-3-upload-to-the-hub) your modified repo to HuggingFace EnvHub.
+
+<Tip>
+  Your IsaacLab Arena environment code must be locally available during
+  evaluation. Users can clone your environment repository separately, or you can
+  bundle the environment code and assets directly in your EnvHub repo.
+</Tip>
+
+Then, when evaluating, use your new environment:
+
+```bash
+lerobot-eval \
+    --env.hub_path=<your-env-hub-path>/isaaclab-arena-envs \
+    --env.environment=<your new environment> \
+    ...other flags...
+```
+
+We look forward to your contributions!
+
+## Troubleshooting
+
+### CUDA out of memory
+
+Reduce `batch_size` or use a GPU with more VRAM:
+
+```bash
+--eval.batch_size=1
+```
+
+### EULA not accepted
+
+Set environment variables before running:
+
+```bash
+export ACCEPT_EULA=Y
+export PRIVACY_CONSENT=Y
+```
+
+### Video recording not working
+
+Enable cameras when running headless:
+
+```bash
+--env.video=true --env.enable_cameras=true --env.headless=true
+```
+
+### Policy output dimension mismatch
+
+Ensure `action_dim` matches your policy:
+
+```bash
+--env.action_dim=36
+```
+
+### libGLU.so.1 Errors during Isaac Sim initialization
+
+Ensure you have the following dependencies installed, this is likely to happen on headless machines.
+
+```bash
+sudo apt update && sudo apt install -y libglu1-mesa libxt6
+```
+
+## See Also
+
+- [EnvHub Documentation](./envhub.mdx) - General EnvHub usage
+- [IsaacLab Arena GitHub](https://github.com/isaac-sim/IsaacLab-Arena)
+- [IsaacLab Documentation](https://isaac-sim.github.io/IsaacLab/)
+
+## Lightwheel LW-BenchHub
+
+[Lightwheel](https://www.lightwheel.ai) is bringing `Lightwheel-Libero-Tasks` and `Lightwheel-RoboCasa-Tasks` with 268 tasks to the LeRobot ecosystem.
+LW-BenchHub collects and generates large-scale datasets via teleoperation that comply with the LeRobot specification, enabling out-of-the-box training and evaluation workflows.
+With the unified interface provided by EnvHub, developers can quickly build end-to-end experimental pipelines.
+
+### Install
+
+Assuming you followed the [Installation](#installation) steps, you can install LW-BenchHub with:
+
+```bash
+conda install pinocchio -c conda-forge -y
+pip install numpy==1.26.0 # revert numpy to version 1.26
+
+sudo apt-get install git-lfs && git lfs install
+
+git clone https://github.com/LightwheelAI/lw_benchhub
+git lfs pull # Ensure LFS files (e.g., .usd assets) are downloaded
+
+cd lw_benchhub
+pip install -e .
+```
+
+For more detailed instructions, please refer to the [LW-BenchHub Documentation](https://docs.lightwheel.net/lw_benchhub/usage/Installation).
+
+### Lightwheel Tasks Dataset
+
+LW-BenchHub datasets are available on HuggingFace Hub:
+
+| Dataset                                                                                                       | Description             | Tasks | Frames |
+| :------------------------------------------------------------------------------------------------------------ | :---------------------- | :---- | :----- |
+| [Lightwheel-Tasks-X7S](https://huggingface.co/datasets/LightwheelAI/Lightwheel-Tasks-X7S)                     | X7S LIBERO and RoboCasa | 117   | ~10.3M |
+| [Lightwheel-Tasks-Double-Piper](https://huggingface.co/datasets/LightwheelAI/Lightwheel-Tasks-Double-Piper)   | Double-Piper LIBERO     | 130   | ~6.0M  |
+| [Lightwheel-Tasks-G1-Controller](https://huggingface.co/datasets/LightwheelAI/Lightwheel-Tasks-G1-Controller) | G1-Controller LIBERO    | 62    | ~2.7M  |
+| [Lightwheel-Tasks-G1-WBC](https://huggingface.co/datasets/LightwheelAI/Lightwheel-Tasks-G1-WBC)               | G1-WBC RoboCasa         | 32    | ~1.5M  |
+
+For training policies, refer to the [Training Policies](#training-policies) section.
+
+### Evaluating Policies
+
+#### Pre-trained Policies
+
+The following trained policies are available:
+
+| Policy                   | Architecture | Task                           | Layout     | Robot           | Link                                                                                  |
+| :----------------------- | :----------- | :----------------------------- | :--------- | :-------------- | :------------------------------------------------------------------------------------ |
+| smolvla-double-piper-pnp | SmolVLA      | L90K1PutTheBlackBowlOnThePlate | libero-1-1 | DoublePiper-Abs | [HuggingFace](https://huggingface.co/LightwheelAI/smolvla-double-piper-pnp/tree/main) |
+
+#### Evaluate SmolVLA
+
+```bash
+lerobot-eval \
+  --policy.path=LightwheelAI/smolvla-double-piper-pnp \
+  --env.type=isaaclab_arena \
+  --rename_map='{"observation.images.left_hand_camera_rgb": "observation.images.left_hand", "observation.images.right_hand_camera_rgb": "observation.images.right_hand", "observation.images.first_person_camera_rgb": "observation.images.first_person"}' \
+  --env.hub_path=LightwheelAI/lw_benchhub_env \
+  --env.kwargs='{"config_path": "configs/envhub/example.yml"}' \
+  --trust_remote_code=true \
+  --env.state_keys=joint_pos \
+  --env.action_dim=12 \
+  --env.camera_keys=left_hand_camera_rgb,right_hand_camera_rgb,first_person_camera_rgb \
+  --policy.device=cuda \
+  --eval.batch_size=10 \
+  --eval.n_episodes=100
+```
+
+### Environment Configuration
+
+Evaluation can be quickly launched by modifying the `robot`, `task`, and `layout` settings in the configuration file.
+
+#### Full Configuration Options
+
+```yml
+# =========================
+# Basic Settings
+# =========================
+disable_fabric: false
+device: cuda:0
+sensitivity: 1.0
+step_hz: 50
+enable_cameras: true
+execute_mode: eval
+episode_length_s: 20.0 # Episode length in seconds, increase if episodes timeout during eval
+
+# =========================
+# Robot Settings
+# =========================
+robot: DoublePiper-Abs # Robot type, DoublePiper-Abs, X7S-Abs, G1-Controller or G1-Controller-DecoupledWBC
+robot_scale: 1.0
+
+# =========================
+# Task & Scene Settings
+# =========================
+task: L90K1PutTheBlackBowlOnThePlate # Task name
+scene_backend: robocasa
+task_backend: robocasa
+debug_assets: null
+layout: libero-1-1 # Layout and style ID
+sources:
+  - objaverse
+  - lightwheel
+  - aigen_objs
+object_projects: []
+usd_simplify: false
+seed: 42
+
+# =========================
+# Object Placement Retry Settings
+# =========================
+max_scene_retry: 4
+max_object_placement_retry: 3
+
+resample_objects_placement_on_reset: true
+resample_robot_placement_on_reset: true
+
+# =========================
+# Replay Configuration Settings
+# =========================
+replay_cfgs:
+  add_camera_to_observation: true
+  render_resolution: [640, 480]
+```
+
+### See Also
+
+- [LW-BenchHub GitHub](https://github.com/LightwheelAI/LW-BenchHub)
+- [LW-BenchHub Documentation](https://docs.lightwheel.net/lw_benchhub/)
diff --git a/lerobot/docs/source/envhub_leisaac.mdx b/lerobot/docs/source/envhub_leisaac.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..2537700a55a24fa365311ec51626783a375fa839
--- /dev/null
+++ b/lerobot/docs/source/envhub_leisaac.mdx
@@ -0,0 +1,302 @@
+# LeIsaac × LeRobot EnvHub
+
+LeRobot EnvHub now supports **imitation learning in simulation** with LeIsaac.
+Spin up everyday manipulation tasks, teleoperate the robot, collect demos, push them to the Hub, and train policies in LeRobot — all in one loop.
+
+[LeIsaac](https://github.com/LightwheelAI/leisaac) integrates with IsaacLab and the SO101 Leader/Follower setup to provide:
+
+- 🕹️ **Teleoperation-first workflows** for data collection
+- 📦 **Built-in data conversion** ready for LeRobot training
+- 🤖 **Everyday skills** like picking oranges, lifting cubes, cleaning tables, and folding cloth
+- ☁️ **Ongoing upgrades** from [LightWheel](https://lightwheel.ai/): cloud simulation, EnvHub support, Sim2Real tooling, and more
+
+Below you’ll find the currently supported LeIsaac tasks exposed through LeRobot EnvHub.
+
+# Available Environments
+
+The following table lists all available tasks and environments in LeIsaac x LeRobot Envhub. You can also get the latest list of environments by running the following command:
+
+```bash
+python scripts/environments/list_envs.py
+```
+
+| Task                                                                                                                                                            | Environment ID                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                | Task Description                                                                                                           | Related Robot                                              |
+| :-------------------------------------------------------------------------------------------------------------------------------------------------------------- | :-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :------------------------------------------------------------------------------------------------------------------------- | :--------------------------------------------------------- |
+| <video src="https://github.com/user-attachments/assets/466eddff-f720-4f99-94d5-5e123e4c302c" autoplay loop muted playsinline style="max-width: 300px;"></video> | [LeIsaac-SO101-PickOrange-v0](https://github.com/LightwheelAI/leisaac/blob/main/source/leisaac/leisaac/tasks/pick_orange/pick_orange_env_cfg.py)<br /><br />[LeIsaac-SO101-PickOrange-Direct-v0](https://github.com/LightwheelAI/leisaac/blob/main/source/leisaac/leisaac/tasks/pick_orange/direct/pick_orange_env.py)                                                                                                                                                                                                                        | Pick three oranges and put them into the plate, then reset the arm to rest state.                                          | Single-Arm SO101 Follower                                  |
+| <video src="https://github.com/user-attachments/assets/1e4eb83a-0b38-40fb-a0b2-ddb0fe201e6d" autoplay loop muted playsinline style="max-width: 300px;"></video> | [LeIsaac-SO101-LiftCube-v0](https://github.com/LightwheelAI/leisaac/blob/main/source/leisaac/leisaac/tasks/lift_cube/lift_cube_env_cfg.py)<br /><br />[LeIsaac-SO101-LiftCube-Direct-v0](https://github.com/LightwheelAI/leisaac/blob/main/source/leisaac/leisaac/tasks/lift_cube/direct/lift_cube_env.py)                                                                                                                                                                                                                                    | Lift the red cube up.                                                                                                      | Single-Arm SO101 Follower                                  |
+| <video src="https://github.com/user-attachments/assets/e49d8f1c-dcc9-412b-a88f-100680d8a45b" autoplay loop muted playsinline style="max-width: 300px;"></video> | [LeIsaac-SO101-CleanToyTable-v0](https://github.com/LightwheelAI/leisaac/blob/main/source/leisaac/leisaac/tasks/clean_toy_table/clean_toy_table_env_cfg.py)<br /><br />[LeIsaac-SO101-CleanToyTable-BiArm-v0](https://github.com/LightwheelAI/leisaac/blob/main/source/leisaac/leisaac/tasks/clean_toy_table/clean_toy_table_bi_arm_env_cfg.py)<br /><br />[LeIsaac-SO101-CleanToyTable-BiArm-Direct-v0](https://github.com/LightwheelAI/leisaac/blob/main/source/leisaac/leisaac/tasks/clean_toy_table/direct/clean_toy_table_bi_arm_env.py) | Pick two letter e objects into the box, and reset the arm to rest state.                                                   | Single-Arm SO101 Follower<br /><br />Bi-Arm SO101 Follower |
+| <video src="https://github.com/user-attachments/assets/e29a0f8a-9286-4ce6-b45d-342c3d3ba754" autoplay loop muted playsinline style="max-width: 300px;"></video> | [LeIsaac-SO101-FoldCloth-BiArm-v0](https://github.com/LightwheelAI/leisaac/blob/main/source/leisaac/leisaac/tasks/fold_cloth/fold_cloth_bi_arm_env_cfg.py)<br /><br />[LeIsaac-SO101-FoldCloth-BiArm-Direct-v0](https://github.com/LightwheelAI/leisaac/blob/main/source/leisaac/leisaac/tasks/fold_cloth/direct/fold_cloth_bi_arm_env.py)                                                                                                                                                                                                    | Fold the cloth, and reset the arm to rest state.<br /><br />_Note: Only the DirectEnv support check_success in this task._ | Bi-Arm SO101 Follower                                      |
+
+# Load LeIsaac directly in LeRobot with one line of code
+
+> EnvHub: Share LeIsaac environments through HuggingFace
+
+[EnvHub](https://huggingface.co/docs/lerobot/envhub) is our reproducible environment hub, spin up a packaged simulation with one line, experiment immediately, and publish your own tasks for the community.
+
+LeIsaac offers EnvHub support so you can consume or share tasks with only a few commands.
+
+<video
+  controls
+  src="https://github.com/user-attachments/assets/687666f5-ebe0-421d-84a0-eb86116ac5f8"
+  style={{ width: "100%", maxWidth: "960px", borderRadius: "8px" }}
+/>
+
+## How to get started, environment Setup
+
+Run the following commands to setup your code environments:
+
+```bash
+# Refer to Getting Started/Installation to install leisaac firstly
+conda create -n leisaac_envhub python=3.11
+conda activate leisaac_envhub
+
+conda install -c "nvidia/label/cuda-12.8.1" cuda-toolkit
+pip install -U torch==2.7.0 torchvision==0.22.0 --index-url https://download.pytorch.org/whl/cu128
+pip install 'leisaac[isaaclab] @ git+https://github.com/LightwheelAI/leisaac.git#subdirectory=source/leisaac' --extra-index-url https://pypi.nvidia.com
+
+# Install lerobot
+pip install lerobot==0.4.1
+
+# Fix numpy version
+pip install numpy==1.26.0
+```
+
+## Usage Example
+
+EnvHub exposes every LeIsaac-supported task in a uniform interface. The examples below load `so101_pick_orange` and demonstrate a random-action rollout and an interactive teleoperation.
+
+### Random Action
+
+<details>
+<summary>Click to expand code example</summary>
+
+```python
+# envhub_random_action.py
+
+import torch
+from lerobot.envs.factory import make_env
+
+# Load from the hub
+envs_dict = make_env("LightwheelAI/leisaac_env:envs/so101_pick_orange.py", n_envs=1, trust_remote_code=True)
+
+# Access the environment
+suite_name = next(iter(envs_dict))
+sync_vector_env = envs_dict[suite_name][0]
+# retrieve the isaac environment from the sync vector env
+env = sync_vector_env.envs[0].unwrapped
+
+# Use it like any gym environment
+obs, info = env.reset()
+
+while True:
+    action = torch.tensor(env.action_space.sample())
+    obs, reward, terminated, truncated, info = env.step(action)
+    if terminated or truncated:
+        obs, info = env.reset()
+
+env.close()
+```
+
+</details>
+
+```bash
+python envhub_random_action.py
+```
+
+You should see the SO101 arm swinging under purely random commands.
+
+### Teleoperation
+
+LeRobot’s teleoperation stack can drive the simulated arm.
+
+Connect the SO101 Leader controller, run the calibration command below.
+
+```bash
+lerobot-calibrate \
+    --teleop.type=so101_leader \
+    --teleop.port=/dev/ttyACM0 \
+    --teleop.id=leader
+```
+
+And then launch the teleop script.
+
+<details>
+<summary>Click to expand code example</summary>
+
+```python
+# envhub_teleop_example.py
+
+import logging
+import time
+import gymnasium as gym
+
+from dataclasses import asdict, dataclass
+from pprint import pformat
+
+from lerobot.teleoperators import (  # noqa: F401
+    Teleoperator,
+    TeleoperatorConfig,
+    make_teleoperator_from_config,
+    so_leader,
+    bi_so_leader,
+)
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import init_logging
+from lerobot.envs.factory import make_env
+
+
+@dataclass
+class TeleoperateConfig:
+    teleop: TeleoperatorConfig
+    env_name: str = "so101_pick_orange"
+    fps: int = 60
+
+
+@dataclass
+class EnvWrap:
+    env: gym.Env
+
+
+def make_env_from_leisaac(env_name: str = "so101_pick_orange"):
+    envs_dict = make_env(
+        f'LightwheelAI/leisaac_env:envs/{env_name}.py',
+        n_envs=1,
+        trust_remote_code=True
+    )
+    suite_name = next(iter(envs_dict))
+    sync_vector_env = envs_dict[suite_name][0]
+    env = sync_vector_env.envs[0].unwrapped
+
+    return env
+
+
+def teleop_loop(teleop: Teleoperator, env: gym.Env, fps: int):
+    from leisaac.devices.action_process import preprocess_device_action
+    from leisaac.assets.robots.lerobot import SO101_FOLLOWER_MOTOR_LIMITS
+    from leisaac.utils.env_utils import dynamic_reset_gripper_effort_limit_sim
+
+    env_wrap = EnvWrap(env=env)
+
+    obs, info = env.reset()
+    while True:
+        loop_start = time.perf_counter()
+        if env.cfg.dynamic_reset_gripper_effort_limit:
+            dynamic_reset_gripper_effort_limit_sim(env, 'so101leader')
+
+        raw_action = teleop.get_action()
+        processed_action = preprocess_device_action(
+            dict(
+                so101_leader=True,
+                joint_state={
+                    k.removesuffix(".pos"): v for k, v in raw_action.items()},
+                motor_limits=SO101_FOLLOWER_MOTOR_LIMITS),
+            env_wrap
+        )
+        obs, reward, terminated, truncated, info = env.step(processed_action)
+        if terminated or truncated:
+            obs, info = env.reset()
+
+        dt_s = time.perf_counter() - loop_start
+        precise_sleep(max(1 / fps - dt_s, 0.0))
+        loop_s = time.perf_counter() - loop_start
+        print(f"\ntime: {loop_s * 1e3:.2f}ms ({1 / loop_s:.0f} Hz)")
+
+
+def teleoperate(cfg: TeleoperateConfig):
+    init_logging()
+    logging.info(pformat(asdict(cfg)))
+
+    teleop = make_teleoperator_from_config(cfg.teleop)
+    env = make_env_from_leisaac(cfg.env_name)
+
+    teleop.connect()
+    if hasattr(env, 'initialize'):
+        env.initialize()
+    try:
+        teleop_loop(teleop=teleop, env=env, fps=cfg.fps)
+    except KeyboardInterrupt:
+        pass
+    finally:
+        teleop.disconnect()
+        env.close()
+
+
+def main():
+    teleoperate(TeleoperateConfig(
+        teleop=so_leader.SO101LeaderConfig(
+            port="/dev/ttyACM0",
+            id='leader',
+            use_degrees=False,
+        ),
+        env_name="so101_pick_orange",
+        fps=60,
+    ))
+
+
+if __name__ == "__main__":
+    main()
+
+```
+
+</details>
+
+```bash
+python envhub_teleop_example.py
+```
+
+Running the script lets you operate the simulated arm using the physical Leader device.
+
+## ☁️ Cloud Simulation (No GPU Required)
+
+Don’t have a local GPU or the right drivers? No problem! You can run LeIsaac entirely in the cloud with zero setup.
+LeIsaac works out-of-the-box on **NVIDIA Brev**, giving you a fully configured environment directly in your browser.
+
+👉 **Start here:** [https://lightwheelai.github.io/leisaac/docs/cloud_simulation/nvidia_brev](https://lightwheelai.github.io/leisaac/docs/cloud_simulation/nvidia_brev)
+
+Once your instance is deployed, simply open the link for **port 80 (HTTP)** to launch **Visual Studio Code Server** (default password: `password`). From there, you can run simulations, edit code, and visualize IsaacLab environments — all from your web browser.
+
+**No GPU, no drivers, no local installation. Just click and run.**
+
+## Additional Notes
+
+We keep EnvHub coverage aligned with the LeIsaac task. Currently supported:
+
+- `so101_pick_orange`
+- `so101_lift_cube`
+- `so101_clean_toytable`
+- `bi_so101_fold_cloth`
+
+Switch tasks by targeting a different script when calling `make_env`, for example:
+
+```python
+envs_dict_pick_orange = make_env("LightwheelAI/leisaac_env:envs/so101_pick_orange.py", n_envs=1, trust_remote_code=True)
+envs_dict_lift_cube = make_env("LightwheelAI/leisaac_env:envs/so101_lift_cube.py", n_envs=1, trust_remote_code=True)
+envs_dict_clean_toytable = make_env("LightwheelAI/leisaac_env:envs/so101_clean_toytable.py", n_envs=1, trust_remote_code=True)
+envs_dict_fold_cloth = make_env("LightwheelAI/leisaac_env:envs/bi_so101_fold_cloth.py", n_envs=1, trust_remote_code=True)
+```
+
+Note: when working with `bi_so101_fold_cloth`, call `initialize()` immediately after retrieving the env before performing any other operations:
+
+<details>
+<summary>Click to expand code example</summary>
+
+```python
+import torch
+from lerobot.envs.factory import make_env
+
+# Load from the hub
+envs_dict = make_env("LightwheelAI/leisaac_env:envs/bi_so101_fold_cloth.py", n_envs=1, trust_remote_code=True)
+
+# Access the environment
+suite_name = next(iter(envs_dict))
+sync_vector_env = envs_dict[suite_name][0]
+# retrieve the isaac environment from the sync vector env
+env = sync_vector_env.envs[0].unwrapped
+
+# NOTE: initialize() first
+env.initialize()
+
+# other operation with env...
+```
+
+</details>
diff --git a/lerobot/docs/source/feetech.mdx b/lerobot/docs/source/feetech.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..bba60e4cc4bba26425a986bae694ba3e92e1dae8
--- /dev/null
+++ b/lerobot/docs/source/feetech.mdx
@@ -0,0 +1,71 @@
+# Feetech Motor Firmware Update
+
+This tutorial guides you through updating the firmware of Feetech motors using the official Feetech software.
+
+## Prerequisites
+
+- Windows computer (Feetech software is only available for Windows)
+- Feetech motor control board
+- USB cable to connect the control board to your computer
+- Feetech motors connected to the control board
+
+## Step 1: Download Feetech Software
+
+1. Visit the official Feetech software download page: [https://www.feetechrc.com/software.html](https://www.feetechrc.com/software.html)
+2. Download the latest version of the Feetech debugging software (FD)
+3. Install the software on your Windows computer
+
+## Step 2: Hardware Setup
+
+1. Connect your Feetech motors to the motor control board
+2. Connect the motor control board to your Windows computer via USB cable
+3. Ensure power is supplied to the motors
+
+## Step 3: Configure Connection
+
+1. Launch the Feetech debugging software
+2. Select the correct COM port from the port dropdown menu
+   - If unsure which port to use, check Windows Device Manager under "Ports (COM & LPT)"
+3. Set the appropriate baud rate (typically 1000000 for most Feetech motors)
+4. Click "Open" to establish communication with the control board
+
+## Step 4: Scan for Motors
+
+1. Once connected, click the "Search" button to detect all connected motors
+2. The software will automatically discover and list all motors on the bus
+3. Each motor will appear with its ID number
+
+## Step 5: Update Firmware
+
+For each motor you want to update:
+
+1. **Select the motor** from the list by clicking on it
+2. **Click on Upgrade tab**:
+3. **Click on Online button**:
+   - If an potential firmware update is found, it will be displayed in the box
+4. **Click on Upgrade button**:
+   - The update progress will be displayed
+
+## Step 6: Verify Update
+
+1. After the update completes, the software should automatically refresh the motor information
+2. Verify that the firmware version has been updated to the expected version
+
+## Important Notes
+
+⚠️ **Warning**: Do not disconnect power or USB during firmware updates, it will potentially brick the motor.
+
+## Bonus: Motor Debugging on Linux/macOS
+
+For debugging purposes only, you can use the open-source Feetech Debug Tool:
+
+- **Repository**: [FT_SCServo_Debug_Qt](https://github.com/CarolinePascal/FT_SCServo_Debug_Qt/tree/fix/port-search-timer)
+
+### Installation Instructions
+
+Follow the instructions in the repository to install the tool, for Ubuntu you can directly install it, for MacOS you need to build it from source.
+
+**Limitations:**
+
+- This tool is for debugging and parameter adjustment only
+- Firmware updates must still be done on Windows with official Feetech software
diff --git a/lerobot/docs/source/groot.mdx b/lerobot/docs/source/groot.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..0ef591466f5f93446bc4aef810136d54b0c6c917
--- /dev/null
+++ b/lerobot/docs/source/groot.mdx
@@ -0,0 +1,134 @@
+# GR00T N1.5 Policy
+
+GR00T N1.5 is an open foundation model from NVIDIA designed for generalized humanoid robot reasoning and skills. It is a cross-embodiment model that accepts multimodal input, including language and images, to perform manipulation tasks in diverse environments.
+
+This document outlines the specifics of its integration and usage within the LeRobot framework.
+
+## Model Overview
+
+NVIDIA Isaac GR00T N1.5 is an upgraded version of the GR00T N1 foundation model. It is built to improve generalization and language-following abilities for humanoid robots.
+
+Developers and researchers can post-train GR00T N1.5 with their own real or synthetic data to adapt it for specific humanoid robots or tasks.
+
+GR00T N1.5 (specifically the GR00T-N1.5-3B model) is built using pre-trained vision and language encoders. It utilizes a flow matching action transformer to model a chunk of actions, conditioned on vision, language, and proprioception.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/lerobot-groot-paper1%20(1).png"
+  alt="An overview of GR00T"
+  width="80%"
+/>
+
+Its strong performance comes from being trained on an expansive and diverse humanoid dataset, which includes:
+
+- Real captured data from robots.
+- Synthetic data generated using NVIDIA Isaac GR00T Blueprint.
+- Internet-scale video data.
+
+This approach allows the model to be highly adaptable through post-training for specific embodiments, tasks, and environments.
+
+## Installation Requirements
+
+As of today, GR00T N1.5 requires flash attention for it's internal working.
+
+We are working on making this optional, but in the meantime that means that we require an extra installation step and it can only be used in CUDA enabled devices.
+
+1. Following the Environment Setup of our [Installation Guide](./installation). **Attention** don't install `lerobot` in this step.
+2. Install [Flash Attention](https://github.com/Dao-AILab/flash-attention) by running:
+
+```bash
+# Check https://pytorch.org/get-started/locally/ for your system
+pip install "torch>=2.2.1,<2.8.0" "torchvision>=0.21.0,<0.23.0" # --index-url https://download.pytorch.org/whl/cu1XX
+pip install ninja "packaging>=24.2,<26.0" # flash attention dependencies
+pip install "flash-attn>=2.5.9,<3.0.0" --no-build-isolation
+python -c "import flash_attn; print(f'Flash Attention {flash_attn.__version__} imported successfully')"
+```
+
+3. Install LeRobot by running:
+
+```bash
+pip install lerobot[groot]
+```
+
+## Usage
+
+To use GR00T in your LeRobot configuration, specify the policy type as:
+
+```python
+policy.type=groot
+```
+
+## Training
+
+### Training Command Example
+
+Here's a complete training command for finetuning the base GR00T model on your own dataset:
+
+```bash
+# Using a multi-GPU setup
+accelerate launch \
+  --multi_gpu \
+  --num_processes=$NUM_GPUS \
+  $(which lerobot-train) \
+  --output_dir=$OUTPUT_DIR \
+  --save_checkpoint=true \
+  --batch_size=$BATCH_SIZE \
+  --steps=$NUM_STEPS \
+  --save_freq=$SAVE_FREQ \
+  --log_freq=$LOG_FREQ \
+  --policy.push_to_hub=true \
+  --policy.type=groot \
+  --policy.repo_id=$REPO_ID \
+  --policy.tune_diffusion_model=false \
+  --dataset.repo_id=$DATASET_ID \
+  --wandb.enable=true \
+  --wandb.disable_artifact=true \
+  --job_name=$JOB_NAME
+```
+
+## Performance Results
+
+### Libero Benchmark Results
+
+> [!NOTE]
+> Follow our instructions for Libero usage: [Libero](./libero)
+
+GR00T has demonstrated strong performance on the Libero benchmark suite. To compare and test its LeRobot implementation, we finetuned the GR00T N1.5 model for 30k steps on the Libero dataset and compared the results to the GR00T reference results.
+
+| Benchmark          | LeRobot Implementation | GR00T Reference |
+| ------------------ | ---------------------- | --------------- |
+| **Libero Spatial** | 82.0%                  | 92.0%           |
+| **Libero Object**  | 99.0%                  | 92.0%           |
+| **Libero Long**    | 82.0%                  | 76.0%           |
+| **Average**        | 87.0%                  | 87.0%           |
+
+These results demonstrate GR00T's strong generalization capabilities across diverse robotic manipulation tasks. To reproduce these results, you can follow the instructions in the [Libero](https://huggingface.co/docs/lerobot/libero) section.
+
+### Evaluate in your hardware setup
+
+Once you have trained your model using your parameters you can run inference in your downstream task. Follow the instructions in [Imitation Learning for Robots](./il_robots). For example:
+
+```bash
+lerobot-record \
+  --robot.type=bi_so_follower \
+  --robot.left_arm_port=/dev/ttyACM1 \
+  --robot.right_arm_port=/dev/ttyACM0 \
+  --robot.id=bimanual_follower \
+  --robot.cameras='{ right: {"type": "opencv", "index_or_path": 0, "width": 640, "height": 480, "fps": 30},
+    left: {"type": "opencv", "index_or_path": 2, "width": 640, "height": 480, "fps": 30},
+    top: {"type": "opencv", "index_or_path": 4, "width": 640, "height": 480, "fps": 30},
+  }' \
+  --display_data=true \
+  --dataset.repo_id=<user>/eval_groot-bimanual  \
+  --dataset.num_episodes=10 \
+  --dataset.single_task="Grab and handover the red cube to the other arm" \
+  --dataset.streaming_encoding=true \
+  --dataset.encoder_threads=2 \
+  # --dataset.vcodec=auto \
+  --policy.path=<user>/groot-bimanual \ # your trained model
+  --dataset.episode_time_s=30 \
+  --dataset.reset_time_s=10
+```
+
+## License
+
+This model follows the **Apache 2.0 License**, consistent with the original [GR00T repository](https://github.com/NVIDIA/Isaac-GR00T).
diff --git a/lerobot/docs/source/hilserl.mdx b/lerobot/docs/source/hilserl.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..ad1c74f9a79faf951ac656dbdb8dae348824f7a5
--- /dev/null
+++ b/lerobot/docs/source/hilserl.mdx
@@ -0,0 +1,923 @@
+# HIL-SERL Real Robot Training Workflow Guide
+
+In this tutorial you will go through the full Human-in-the-Loop Sample-Efficient Reinforcement Learning (HIL-SERL) workflow using LeRobot. You will master training a policy with RL on a real robot in just a few hours.
+
+HIL-SERL is a sample-efficient reinforcement learning algorithm that combines human demonstrations with online learning and human interventions. The approach starts from a small set of human demonstrations, uses them to train a reward classifier, and then employs an actor-learner architecture where humans can intervene during policy execution to guide exploration and correct unsafe behaviors. In this tutorial, you'll use a gamepad to provide interventions and control the robot during the learning process.
+
+It combines three key ingredients:
+
+1. **Offline demonstrations & reward classifier:** a handful of human-teleop episodes plus a vision-based success detector give the policy a shaped starting point.
+
+2. **On-robot actor / learner loop with human interventions:** a distributed Soft Actor Critic (SAC) learner updates the policy while an actor explores on the physical robot; the human can jump in at any time to correct dangerous or unproductive behaviour.
+
+3. **Safety & efficiency tools:** joint/end-effector (EE) bounds, crop region of interest (ROI) preprocessing and WandB monitoring keep the data useful and the hardware safe.
+
+Together these elements let HIL-SERL reach near-perfect task success and faster cycle times than imitation-only baselines.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/hilserl-main-figure.png"
+    alt="HIL-SERL workflow"
+    title="HIL-SERL workflow"
+    width="100%"
+  ></img>
+</p>
+
+<p align="center">
+  <i>HIL-SERL workflow, Luo et al. 2024</i>
+</p>
+
+This guide provides step-by-step instructions for training a robot policy using LeRobot's HilSerl implementation to train on a real robot.
+
+## What do I need?
+
+- A gamepad (recommended) or keyboard to control the robot
+- A Nvidia GPU
+- A real robot with a follower and leader arm (optional if you use the keyboard or the gamepad)
+- A URDF file for the robot for the kinematics package (check `lerobot/model/kinematics.py`)
+
+## What kind of tasks can I train?
+
+One can use HIL-SERL to train on a variety of manipulation tasks. Some recommendations:
+
+- Start with a simple task to understand how the system works.
+  - Push cube to a goal region
+  - Pick and lift cube with the gripper
+- Avoid extremely long horizon tasks. Focus on tasks that can be completed in 5-10 seconds.
+- Once you have a good idea of how the system works, you can try more complex tasks and longer horizons.
+  - Pick and place cube
+  - Bimanual tasks to pick objects with two arms
+  - Hand-over tasks to transfer objects from one arm to another
+  - Go crazy!
+
+## Install LeRobot with HIL-SERL
+
+To install LeRobot with HIL-SERL, you need to install the `hilserl` extra.
+
+```bash
+pip install -e ".[hilserl]"
+```
+
+## Real Robot Training Workflow
+
+### Understanding Configuration
+
+The training process begins with proper configuration for the HILSerl environment. The main configuration class is `GymManipulatorConfig` in `lerobot/rl/gym_manipulator.py`, which contains nested `HILSerlRobotEnvConfig` and `DatasetConfig`. The configuration is organized into focused, nested sub-configs:
+
+<!-- prettier-ignore-start -->
+```python
+class GymManipulatorConfig:
+    env: HILSerlRobotEnvConfig    # Environment configuration (nested)
+    dataset: DatasetConfig    # Dataset recording/replay configuration (nested)
+    mode: str | None = None    # "record", "replay", or None (for training)
+    device: str = "cpu"    # Compute device
+
+class HILSerlRobotEnvConfig(EnvConfig):
+    robot: RobotConfig | None = None    # Main robot agent (defined in `lerobot/robots`)
+    teleop: TeleoperatorConfig | None = None    # Teleoperator agent, e.g., gamepad or leader arm
+    processor: HILSerlProcessorConfig    # Processing pipeline configuration (nested)
+    name: str = "real_robot"    # Environment name
+    task: str | None = None    # Task identifier
+    fps: int = 10    # Control frequency
+
+# Nested processor configuration
+class HILSerlProcessorConfig:
+    control_mode: str = "gamepad"    # Control mode
+    observation: ObservationConfig | None = None    # Observation processing settings
+    image_preprocessing: ImagePreprocessingConfig | None = None    # Image crop/resize settings
+    gripper: GripperConfig | None = None    # Gripper control and penalty settings
+    reset: ResetConfig | None = None    # Environment reset and timing settings
+    inverse_kinematics: InverseKinematicsConfig | None = None    # IK processing settings
+    reward_classifier: RewardClassifierConfig | None = None    # Reward classifier settings
+    max_gripper_pos: float | None = 100.0    # Maximum gripper position
+
+# Sub-configuration classes
+class ObservationConfig:
+    add_joint_velocity_to_observation: bool = False    # Add joint velocities to state
+    add_current_to_observation: bool = False    # Add motor currents to state
+    display_cameras: bool = False    # Display camera feeds during execution
+
+class ImagePreprocessingConfig:
+    crop_params_dict: dict[str, tuple[int, int, int, int]] | None = None    # Image cropping parameters
+    resize_size: tuple[int, int] | None = None    # Target image size
+
+class GripperConfig:
+    use_gripper: bool = True    # Enable gripper control
+    gripper_penalty: float = 0.0    # Penalty for inappropriate gripper usage
+
+class ResetConfig:
+    fixed_reset_joint_positions: Any | None = None    # Joint positions for reset
+    reset_time_s: float = 5.0    # Time to wait during reset
+    control_time_s: float = 20.0    # Maximum episode duration
+    terminate_on_success: bool = True    # Whether to terminate episodes on success detection
+
+class InverseKinematicsConfig:
+    urdf_path: str | None = None    # Path to robot URDF file
+    target_frame_name: str | None = None    # End-effector frame name
+    end_effector_bounds: dict[str, list[float]] | None = None    # EE workspace bounds
+    end_effector_step_sizes: dict[str, float] | None = None    # EE step sizes per axis
+
+class RewardClassifierConfig:
+    pretrained_path: str | None = None    # Path to pretrained reward classifier
+    success_threshold: float = 0.5    # Success detection threshold
+    success_reward: float = 1.0    # Reward value for successful episodes
+
+# Dataset configuration
+class DatasetConfig:
+    repo_id: str    # LeRobot dataset repository ID
+    task: str    # Task identifier
+    root: str | None = None    # Local dataset root directory
+    num_episodes_to_record: int = 5    # Number of episodes for recording
+    replay_episode: int | None = None    # Episode index for replay
+    push_to_hub: bool = False    # Whether to push datasets to Hub
+```
+<!-- prettier-ignore-end -->
+
+### Processor Pipeline Architecture
+
+HIL-SERL uses a modular processor pipeline architecture that processes robot observations and actions through a series of composable steps. The pipeline is divided into two main components:
+
+#### Environment Processor Pipeline
+
+The environment processor (`env_processor`) handles incoming observations and environment state:
+
+1. **VanillaObservationProcessorStep**: Converts raw robot observations into standardized format
+2. **JointVelocityProcessorStep** (optional): Adds joint velocity information to observations
+3. **MotorCurrentProcessorStep** (optional): Adds motor current readings to observations
+4. **ForwardKinematicsJointsToEE** (optional): Computes end-effector pose from joint positions
+5. **ImageCropResizeProcessorStep** (optional): Crops and resizes camera images
+6. **TimeLimitProcessorStep** (optional): Enforces episode time limits
+7. **GripperPenaltyProcessorStep** (optional): Applies penalties for inappropriate gripper usage
+8. **RewardClassifierProcessorStep** (optional): Automated reward detection using vision models
+9. **AddBatchDimensionProcessorStep**: Converts data to batch format for neural network processing
+10. **DeviceProcessorStep**: Moves data to the specified compute device (CPU/GPU)
+
+#### Action Processor Pipeline
+
+The action processor (`action_processor`) handles outgoing actions and human interventions:
+
+1. **AddTeleopActionAsComplimentaryDataStep**: Captures teleoperator actions for logging
+2. **AddTeleopEventsAsInfoStep**: Records intervention events and episode control signals
+3. **InterventionActionProcessorStep**: Handles human interventions and episode termination
+4. **Inverse Kinematics Pipeline** (when enabled):
+   - **MapDeltaActionToRobotActionStep**: Converts delta actions to robot action format
+   - **EEReferenceAndDelta**: Computes end-effector reference and delta movements
+   - **EEBoundsAndSafety**: Enforces workspace safety bounds
+   - **InverseKinematicsEEToJoints**: Converts end-effector actions to joint targets
+   - **GripperVelocityToJoint**: Handles gripper control commands
+
+#### Configuration Examples
+
+**Basic Observation Processing**:
+
+```json
+{
+  "env": {
+    "processor": {
+      "observation": {
+        "add_joint_velocity_to_observation": true,
+        "add_current_to_observation": false,
+        "display_cameras": false
+      }
+    }
+  }
+}
+```
+
+**Image Processing**:
+
+```json
+{
+  "env": {
+    "processor": {
+      "image_preprocessing": {
+        "crop_params_dict": {
+          "observation.images.front": [180, 250, 120, 150],
+          "observation.images.side": [180, 207, 180, 200]
+        },
+        "resize_size": [128, 128]
+      }
+    }
+  }
+}
+```
+
+**Inverse Kinematics Setup**:
+
+```json
+{
+  "env": {
+    "processor": {
+      "inverse_kinematics": {
+        "urdf_path": "path/to/robot.urdf",
+        "target_frame_name": "end_effector",
+        "end_effector_bounds": {
+          "min": [0.16, -0.08, 0.03],
+          "max": [0.24, 0.2, 0.1]
+        },
+        "end_effector_step_sizes": {
+          "x": 0.02,
+          "y": 0.02,
+          "z": 0.02
+        }
+      }
+    }
+  }
+}
+```
+
+### Advanced Observation Processing
+
+The HIL-SERL framework supports additional observation processing features that can improve policy learning:
+
+#### Joint Velocity Processing
+
+Enable joint velocity estimation to provide the policy with motion information:
+
+```json
+{
+  "env": {
+    "processor": {
+      "observation": {
+        "add_joint_velocity_to_observation": true
+      }
+    }
+  }
+}
+```
+
+This processor:
+
+- Estimates joint velocities using finite differences between consecutive joint position readings
+- Adds velocity information to the observation state vector
+- Useful for policies that need motion awareness for dynamic tasks
+
+#### Motor Current Processing
+
+Monitor motor currents to detect contact forces and load conditions:
+
+```json
+{
+  "env": {
+    "processor": {
+      "observation": {
+        "add_current_to_observation": true
+      }
+    }
+  }
+}
+```
+
+This processor:
+
+- Reads motor current values from the robot's control system
+- Adds current measurements to the observation state vector
+- Helps detect contact events, object weights, and mechanical resistance
+- Useful for contact-rich manipulation tasks
+
+#### Combined Observation Processing
+
+You can enable multiple observation processing features simultaneously:
+
+```json
+{
+  "env": {
+    "processor": {
+      "observation": {
+        "add_joint_velocity_to_observation": true,
+        "add_current_to_observation": true,
+        "display_cameras": false
+      }
+    }
+  }
+}
+```
+
+**Note**: Enabling additional observation features increases the state space dimensionality, which may require adjusting your policy network architecture and potentially collecting more training data.
+
+### Finding Robot Workspace Bounds
+
+Before collecting demonstrations, you need to determine the appropriate operational bounds for your robot.
+
+This helps simplify the problem of learning on the real robot in two ways: 1) by limiting the robot's operational space to a specific region that solves the task and avoids unnecessary or unsafe exploration, and 2) by allowing training in end-effector space rather than joint space. Empirically, learning in joint space for reinforcement learning in manipulation is often a harder problem - some tasks are nearly impossible to learn in joint space but become learnable when the action space is transformed to end-effector coordinates.
+
+**Using lerobot-find-joint-limits**
+
+This script helps you find the safe operational bounds for your robot's end-effector. Given that you have a follower and leader arm, you can use the script to find the bounds for the follower arm that will be applied during training.
+Bounding the action space will reduce the redundant exploration of the agent and guarantees safety.
+
+```bash
+lerobot-find-joint-limits \
+  --robot.type=so100_follower \
+  --robot.port=/dev/tty.usbmodem58760431541 \
+  --robot.id=black \
+  --teleop.type=so100_leader \
+  --teleop.port=/dev/tty.usbmodem58760431551 \
+  --teleop.id=blue
+```
+
+**Workflow**
+
+1. Run the script and move the robot through the space that solves the task
+2. The script will record the minimum and maximum end-effector positions and the joint angles and prints them to the console, for example:
+   ```
+   Max ee position [0.2417 0.2012 0.1027]
+   Min ee position [0.1663 -0.0823 0.0336]
+   Max joint positions [-20.0, -20.0, -20.0, -20.0, -20.0, -20.0]
+   Min joint positions [50.0, 50.0, 50.0, 50.0, 50.0, 50.0]
+   ```
+3. Use these values in the configuration of your teleoperation device (TeleoperatorConfig) under the `end_effector_bounds` field
+
+**Example Configuration**
+
+```json
+"end_effector_bounds": {
+    "max": [0.24, 0.20, 0.10],
+    "min": [0.16, -0.08, 0.03]
+}
+```
+
+### Collecting Demonstrations
+
+With the bounds defined, you can safely collect demonstrations for training. Training RL with off-policy algorithm allows us to use offline datasets collected in order to improve the efficiency of the learning process.
+
+**Setting Up Record Mode**
+
+Create a configuration file for recording demonstrations (or edit an existing one like [env_config.json](https://huggingface.co/datasets/lerobot/config_examples/resolve/main/rl/env_config.json)):
+
+1. Set `mode` to `"record"` at the root level
+2. Specify a unique `repo_id` for your dataset in the `dataset` section (e.g., "username/task_name")
+3. Set `num_episodes_to_record` in the `dataset` section to the number of demonstrations you want to collect
+4. Set `env.processor.image_preprocessing.crop_params_dict` to `{}` initially (we'll determine crops later)
+5. Configure `env.robot`, `env.teleop`, and other hardware settings in the `env` section
+
+Example configuration section:
+
+```json
+{
+  "env": {
+    "type": "gym_manipulator",
+    "name": "real_robot",
+    "fps": 10,
+    "processor": {
+      "control_mode": "gamepad",
+      "observation": {
+        "display_cameras": false
+      },
+      "image_preprocessing": {
+        "crop_params_dict": {},
+        "resize_size": [128, 128]
+      },
+      "gripper": {
+        "use_gripper": true,
+        "gripper_penalty": 0.0
+      },
+      "reset": {
+        "reset_time_s": 5.0,
+        "control_time_s": 20.0
+      }
+    },
+    "robot": {
+      // ... robot configuration ...
+    },
+    "teleop": {
+      // ... teleoperator configuration ...
+    }
+  },
+  "dataset": {
+    "repo_id": "username/pick_lift_cube",
+    "root": null,
+    "task": "pick_and_lift",
+    "num_episodes_to_record": 15,
+    "replay_episode": 0,
+    "push_to_hub": true
+  },
+  "mode": "record",
+  "device": "cpu"
+}
+```
+
+### Using a Teleoperation Device
+
+Along with your robot, you will need a teleoperation device to control it in order to collect datasets of your task and perform interventions during the online training.
+We support using a gamepad or a keyboard or the leader arm of the robot.
+
+HIL-Serl learns actions in the end-effector space of the robot. Therefore, the teleoperation will control the end-effector's x,y,z displacements.
+
+For that we need to define a version of the robot that takes actions in the end-effector space. Check the robot class `SO100FollowerEndEffector` and its configuration `SO100FollowerEndEffectorConfig` for the default parameters related to the end-effector space.
+
+<!-- prettier-ignore-start -->
+```python
+class SO100FollowerEndEffectorConfig(SO100FollowerConfig):
+    """Configuration for the SO100FollowerEndEffector robot."""
+
+    # Default bounds for the end-effector position (in meters)
+    end_effector_bounds: dict[str, list[float]] = field( # bounds for the end-effector in x,y,z direction
+        default_factory=lambda: {
+            "min": [-1.0, -1.0, -1.0],  # min x, y, z
+            "max": [1.0, 1.0, 1.0],  # max x, y, z
+        }
+    )
+
+    max_gripper_pos: float = 50 # maximum gripper position that the gripper will be open at
+
+    end_effector_step_sizes: dict[str, float] = field( # maximum step size for the end-effector in x,y,z direction
+        default_factory=lambda: {
+            "x": 0.02,
+            "y": 0.02,
+            "z": 0.02,
+        }
+    )
+```
+<!-- prettier-ignore-end -->
+
+The `Teleoperator` defines the teleoperation device. You can check the list of available teleoperators in `lerobot/teleoperators`.
+
+**Setting up the Gamepad**
+
+The gamepad provides a very convenient way to control the robot and the episode state.
+
+To setup the gamepad, you need to set the `control_mode` to `"gamepad"` and define the `teleop` section in the configuration file.
+
+```json
+{
+  "env": {
+    "teleop": {
+      "type": "gamepad",
+      "use_gripper": true
+    },
+    "processor": {
+      "control_mode": "gamepad",
+      "gripper": {
+        "use_gripper": true
+      }
+    }
+  }
+}
+```
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/gamepad_guide.jpg?raw=true"
+    alt="Figure shows the control mappings on a Logitech gamepad."
+    title="Gamepad Control Mapping"
+    width="100%"
+  ></img>
+</p>
+<p align="center">
+  <i>Gamepad button mapping for robot control and episode management</i>
+</p>
+
+**Setting up the SO101 leader**
+
+The SO101 leader arm has reduced gears that allows it to move and track the follower arm during exploration. Therefore, taking over is much smoother than the gearless SO100.
+
+To setup the SO101 leader, you need to set the `control_mode` to `"leader"` and define the `teleop` section in the configuration file.
+
+```json
+{
+  "env": {
+    "teleop": {
+      "type": "so101_leader",
+      "port": "/dev/tty.usbmodem585A0077921",
+      "use_degrees": true
+    },
+    "processor": {
+      "control_mode": "leader",
+      "gripper": {
+        "use_gripper": true
+      }
+    }
+  }
+}
+```
+
+In order to annotate the success/failure of the episode, **you will need** to use a keyboard to press `s` for success, `esc` for failure.
+During the online training, press `space` to take over the policy and `space` again to give the control back to the policy.
+
+<details>
+<summary><strong>Video: SO101 leader teleoperation</strong></summary>
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so101_leader_tutorial.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+<p align="center"><i>SO101 leader teleoperation example, the leader tracks the follower, press `space` to intervene</i></p>
+</details>
+
+**Recording Demonstrations**
+
+Start the recording process, an example of the config file can be found [here](https://huggingface.co/datasets/aractingi/lerobot-example-config-files/blob/main/env_config_so100.json):
+
+```bash
+python -m lerobot.rl.gym_manipulator --config_path src/lerobot/configs/env_config_so100.json
+```
+
+During recording:
+
+1. The robot will reset to the initial position defined in the configuration file `env.processor.reset.fixed_reset_joint_positions`
+2. Complete the task successfully
+3. The episode ends with a reward of 1 when you press the "success" button
+4. If the time limit is reached, or the fail button is pressed, the episode ends with a reward of 0
+5. You can rerecord an episode by pressing the "rerecord" button
+6. The process automatically continues to the next episode
+7. After recording all episodes, the dataset is pushed to the Hugging Face Hub (optional) and saved locally
+
+### Processing the Dataset
+
+After collecting demonstrations, process them to determine optimal camera crops.
+Reinforcement learning is sensitive to background distractions, so it is important to crop the images to the relevant workspace area.
+
+Visual RL algorithms learn directly from pixel inputs, making them vulnerable to irrelevant visual information. Background elements like changing lighting, shadows, people moving, or objects outside the workspace can confuse the learning process. Good ROI selection should:
+
+- Include only the essential workspace where the task happens
+- Capture the robot's end-effector and all objects involved in the task
+- Exclude unnecessary background elements and distractions
+
+Note: If you already know the crop parameters, you can skip this step and just set the `crop_params_dict` in the configuration file during recording.
+
+**Determining Crop Parameters**
+
+Use the `crop_dataset_roi.py` script to interactively select regions of interest in your camera images:
+
+```bash
+python -m lerobot.rl.crop_dataset_roi --repo-id username/pick_lift_cube
+```
+
+1. For each camera view, the script will display the first frame
+2. Draw a rectangle around the relevant workspace area
+3. Press 'c' to confirm the selection
+4. Repeat for all camera views
+5. The script outputs cropping parameters and creates a new cropped dataset
+
+Example output:
+
+```
+Selected Rectangular Regions of Interest (top, left, height, width):
+observation.images.side: [180, 207, 180, 200]
+observation.images.front: [180, 250, 120, 150]
+```
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/crop_dataset.gif"
+    width="600"
+  />
+</p>
+
+<p align="center">
+  <i>Interactive cropping tool for selecting regions of interest</i>
+</p>
+
+**Updating Configuration**
+
+Add these crop parameters to your training configuration:
+
+```json
+{
+  "env": {
+    "processor": {
+      "image_preprocessing": {
+        "crop_params_dict": {
+          "observation.images.side": [180, 207, 180, 200],
+          "observation.images.front": [180, 250, 120, 150]
+        },
+        "resize_size": [128, 128]
+      }
+    }
+  }
+}
+```
+
+**Recommended image resolution**
+
+Most vision-based policies have been validated on square inputs of either **128×128** (default) or **64×64** pixels. We therefore advise setting the resize_size parameter to [128, 128] – or [64, 64] if you need to save GPU memory and bandwidth. Other resolutions are possible but have not been extensively tested.
+
+### Training a Reward Classifier
+
+The reward classifier plays an important role in the HIL-SERL workflow by automating reward assignment and automatically detecting episode success. Instead of manually defining reward functions or relying on human feedback for every timestep, the reward classifier learns to predict success/failure from visual observations. This enables the RL algorithm to learn efficiently by providing consistent and automated reward signals based on the robot's camera inputs.
+
+This guide explains how to train a reward classifier for human-in-the-loop reinforcement learning implementation of LeRobot. Reward classifiers learn to predict the reward value given a state which can be used in an RL setup to train a policy.
+
+**Note**: Training a reward classifier is optional. You can start the first round of RL experiments by annotating the success manually with your gamepad or keyboard device.
+
+The reward classifier implementation in `modeling_classifier.py` uses a pretrained vision model to process the images. It can output either a single value for binary rewards to predict success/fail cases or multiple values for multi-class settings.
+
+**Collecting a Dataset for the reward classifier**
+
+Before training, you need to collect a dataset with labeled examples. The `record_dataset` function in `gym_manipulator.py` enables the process of collecting a dataset of observations, actions, and rewards.
+
+To collect a dataset, you need to modify some parameters in the environment configuration based on HILSerlRobotEnvConfig.
+
+```bash
+python -m lerobot.rl.gym_manipulator --config_path src/lerobot/configs/reward_classifier_train_config.json
+```
+
+**Key Parameters for Data Collection**
+
+- **mode**: set it to `"record"` to collect a dataset (at root level)
+- **dataset.repo_id**: `"hf_username/dataset_name"`, name of the dataset and repo on the hub
+- **dataset.num_episodes_to_record**: Number of episodes to record
+- **env.processor.reset.terminate_on_success**: Whether to automatically terminate episodes when success is detected (default: `true`)
+- **env.fps**: Number of frames per second to record
+- **dataset.push_to_hub**: Whether to push the dataset to the hub
+
+The `env.processor.reset.terminate_on_success` parameter allows you to control episode termination behavior. When set to `false`, episodes will continue even after success is detected, allowing you to collect more positive examples with the reward=1 label. This is crucial for training reward classifiers as it provides more success state examples in your dataset. When set to `true` (default), episodes terminate immediately upon success detection.
+
+**Important**: For reward classifier training, set `terminate_on_success: false` to collect sufficient positive examples. For regular HIL-SERL training, keep it as `true` to enable automatic episode termination when the task is completed successfully.
+
+Example configuration section for data collection:
+
+```json
+{
+  "env": {
+    "type": "gym_manipulator",
+    "name": "real_robot",
+    "fps": 10,
+    "processor": {
+      "reset": {
+        "reset_time_s": 5.0,
+        "control_time_s": 20.0,
+        "terminate_on_success": false
+      },
+      "gripper": {
+        "use_gripper": true
+      }
+    },
+    "robot": {
+      // ... robot configuration ...
+    },
+    "teleop": {
+      // ... teleoperator configuration ...
+    }
+  },
+  "dataset": {
+    "repo_id": "hf_username/dataset_name",
+    "dataset_root": "data/your_dataset",
+    "task": "reward_classifier_task",
+    "num_episodes_to_record": 20,
+    "replay_episode": null,
+    "push_to_hub": true
+  },
+  "mode": "record",
+  "device": "cpu"
+}
+```
+
+**Reward Classifier Configuration**
+
+The reward classifier is configured using `configuration_classifier.py`. Here are the key parameters:
+
+- **model_name**: Base model architecture (e.g., we mainly use `"helper2424/resnet10"`)
+- **model_type**: `"cnn"` or `"transformer"`
+- **num_cameras**: Number of camera inputs
+- **num_classes**: Number of output classes (typically 2 for binary success/failure)
+- **hidden_dim**: Size of hidden representation
+- **dropout_rate**: Regularization parameter
+- **learning_rate**: Learning rate for optimizer
+
+Example configuration for training the [reward classifier](https://huggingface.co/datasets/aractingi/lerobot-example-config-files/blob/main/reward_classifier_train_config.json):
+
+```json
+{
+  "policy": {
+    "type": "reward_classifier",
+    "model_name": "helper2424/resnet10",
+    "model_type": "cnn",
+    "num_cameras": 2,
+    "num_classes": 2,
+    "hidden_dim": 256,
+    "dropout_rate": 0.1,
+    "learning_rate": 1e-4,
+    "device": "cuda",
+    "use_amp": true,
+    "input_features": {
+      "observation.images.front": {
+        "type": "VISUAL",
+        "shape": [3, 128, 128]
+      },
+      "observation.images.side": {
+        "type": "VISUAL",
+        "shape": [3, 128, 128]
+      }
+    }
+  }
+}
+```
+
+**Training the Classifier**
+
+To train the classifier, use the `train.py` script with your configuration:
+
+```bash
+lerobot-train --config_path path/to/reward_classifier_train_config.json
+```
+
+**Deploying and Testing the Model**
+
+To use your trained reward classifier, configure the `HILSerlRobotEnvConfig` to use your model:
+
+<!-- prettier-ignore-start -->
+```python
+config = GymManipulatorConfig(
+    env=HILSerlRobotEnvConfig(
+        processor=HILSerlProcessorConfig(
+            reward_classifier=RewardClassifierConfig(
+                pretrained_path="path_to_your_pretrained_trained_model"
+            )
+        ),
+        # Other environment parameters
+    ),
+    dataset=DatasetConfig(...),
+    mode=None  # For training
+)
+```
+<!-- prettier-ignore-end -->
+
+or set the argument in the json config file.
+
+```json
+{
+  "env": {
+    "processor": {
+      "reward_classifier": {
+        "pretrained_path": "path_to_your_pretrained_model",
+        "success_threshold": 0.7,
+        "success_reward": 1.0
+      },
+      "reset": {
+        "terminate_on_success": true
+      }
+    }
+  }
+}
+```
+
+Run `gym_manipulator.py` to test the model.
+
+```bash
+python -m lerobot.rl.gym_manipulator --config_path path/to/env_config.json
+```
+
+The reward classifier will automatically provide rewards based on the visual input from the robot's cameras.
+
+**Example Workflow for training the reward classifier**
+
+1. **Create the configuration files**:
+   Create the necessary json configuration files for the reward classifier and the environment. Check the examples [here](https://huggingface.co/datasets/lerobot/config_examples/resolve/main/reward_classifier/config.json).
+
+2. **Collect a dataset**:
+
+   ```bash
+   python -m lerobot.rl.gym_manipulator --config_path src/lerobot/configs/env_config.json
+   ```
+
+3. **Train the classifier**:
+
+   ```bash
+   lerobot-train --config_path src/lerobot/configs/reward_classifier_train_config.json
+   ```
+
+4. **Test the classifier**:
+   ```bash
+   python -m lerobot.rl.gym_manipulator --config_path src/lerobot/configs/env_config.json
+   ```
+
+### Training with Actor-Learner
+
+The LeRobot system uses a distributed actor-learner architecture for training. This architecture decouples robot interactions from the learning process, allowing them to run concurrently without blocking each other. The actor server handles robot observations and actions, sending interaction data to the learner server. The learner server performs gradient descent and periodically updates the actor's policy weights. You will need to start two processes: a learner and an actor.
+
+**Configuration Setup**
+
+Create a training configuration file (example available [here](https://huggingface.co/datasets/lerobot/config_examples/resolve/main/rl/train_config.json)). The training config is based on the main `TrainRLServerPipelineConfig` class in `lerobot/configs/train.py`.
+
+1. Configure the policy settings (`type="sac"`, `device`, etc.)
+2. Set `dataset` to your cropped dataset
+3. Configure environment settings with crop parameters
+4. Check the other parameters related to SAC in [configuration_sac.py](https://github.com/huggingface/lerobot/blob/main/src/lerobot/policies/sac/configuration_sac.py#L79).
+5. Verify that the `policy` config is correct with the right `input_features` and `output_features` for your task.
+
+**Starting the Learner**
+
+First, start the learner server process:
+
+```bash
+python -m lerobot.rl.learner --config_path src/lerobot/configs/train_config_hilserl_so100.json
+```
+
+The learner:
+
+- Initializes the policy network
+- Prepares replay buffers
+- Opens a `gRPC` server to communicate with actors
+- Processes transitions and updates the policy
+
+**Starting the Actor**
+
+In a separate terminal, start the actor process with the same configuration:
+
+```bash
+python -m lerobot.rl.actor --config_path src/lerobot/configs/train_config_hilserl_so100.json
+```
+
+The actor:
+
+- Connects to the learner via `gRPC`
+- Initializes the environment
+- Execute rollouts of the policy to collect experience
+- Sends transitions to the learner
+- Receives updated policy parameters
+
+**Training Flow**
+
+The training proceeds automatically:
+
+1. The actor executes the policy in the environment
+2. Transitions are collected and sent to the learner
+3. The learner updates the policy based on these transitions
+4. Updated policy parameters are sent back to the actor
+5. The process continues until the specified step limit is reached
+
+**Human in the Loop**
+
+- The key to learning efficiently is to have human interventions to provide corrective feedback and completing the task to aide the policy learning and exploration.
+- To perform human interventions, you can press the upper right trigger button on the gamepad (or the `space` key on the keyboard). This will pause the policy actions and allow you to take over.
+- A successful experiment is one where the human has to intervene at the start but then reduces the amount of interventions as the policy improves. You can monitor the intervention rate in the `wandb` dashboard.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/hil_effect.png?raw=true"
+    alt="Figure shows the control mappings on a Logitech gamepad."
+    title="Gamepad Control Mapping"
+    width="100%"
+  ></img>
+</p>
+
+<p align="center">
+  <i>
+    Example showing how human interventions help guide policy learning over time
+  </i>
+</p>
+
+- The figure shows the plot of the episodic reward over interaction step. The figure shows the effect of human interventions on the policy learning.
+- The orange curve is an experiment without any human interventions. While the pink and blue curves are experiments with human interventions.
+- We can observe that the number of steps where the policy starts achieving the maximum reward is cut by a quarter when human interventions are present.
+
+**Monitoring and Debugging**
+
+If you have `wandb.enable` set to `true` in your configuration, you can monitor training progress in real-time through the [Weights & Biases](https://wandb.ai/site/) dashboard.
+
+### Guide to Human Interventions
+
+The learning process is very sensitive to the intervention strategy. It will takes a few runs to understand how to intervene effectively. Some tips and hints:
+
+- Allow the policy to explore for a few episodes at the start of training.
+- Avoid intervening for long periods of time. Try to intervene in situation to correct the robot's behaviour when it goes off track.
+- Once the policy starts achieving the task, even if its not perfect, you can limit your interventions to simple quick actions like a simple grasping commands.
+
+The ideal behaviour is that your intervention rate should drop gradually during training as shown in the figure below.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/intervention_rate_tutorial_rl.png?raw=true"
+    alt="Intervention rate"
+    title="Intervention rate during training"
+    width="100%"
+  ></img>
+</p>
+
+<p align="center">
+  <i>
+    Plot of the intervention rate during a training run on a pick and lift cube
+    task
+  </i>
+</p>
+
+### Key hyperparameters to tune
+
+Some configuration values have a disproportionate impact on training stability and speed:
+
+- **`temperature_init`** (`policy.temperature_init`) – initial entropy temperature in SAC. Higher values encourage more exploration; lower values make the policy more deterministic early on. A good starting point is `1e-2`. We observed that setting it too high can make human interventions ineffective and slow down learning.
+- **`policy_parameters_push_frequency`** (`policy.actor_learner_config.policy_parameters_push_frequency`) – interval in _seconds_ between two weight pushes from the learner to the actor. The default is `4 s`. Decrease to **1-2 s** to provide fresher weights (at the cost of more network traffic); increase only if your connection is slow, as this will reduce sample efficiency.
+- **`storage_device`** (`policy.storage_device`) – device on which the learner keeps the policy parameters. If you have spare GPU memory, set this to `"cuda"` (instead of the default `"cpu"`). Keeping the weights on-GPU removes CPU→GPU transfer overhead and can significantly increase the number of learner updates per second.
+
+Congrats 🎉, you have finished this tutorial!
+
+> [!TIP]
+> If you have any questions or need help, please reach out on [Discord](https://discord.com/invite/s3KuuzsPFb).
+
+Paper citation:
+
+```
+@article{luo2024precise,
+  title={Precise and Dexterous Robotic Manipulation via Human-in-the-Loop Reinforcement Learning},
+  author={Luo, Jianlan and Xu, Charles and Wu, Jeffrey and Levine, Sergey},
+  journal={arXiv preprint arXiv:2410.21845},
+  year={2024}
+}
+```
diff --git a/lerobot/docs/source/hilserl_sim.mdx b/lerobot/docs/source/hilserl_sim.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..e2dddd9edb116cdcd215eaec4db323c801d46e7c
--- /dev/null
+++ b/lerobot/docs/source/hilserl_sim.mdx
@@ -0,0 +1,154 @@
+# Train RL in Simulation
+
+This guide explains how to use the `gym_hil` simulation environments as an alternative to real robots when working with the LeRobot framework for Human-In-the-Loop (HIL) reinforcement learning.
+
+`gym_hil` is a package that provides Gymnasium-compatible simulation environments specifically designed for Human-In-the-Loop reinforcement learning. These environments allow you to:
+
+- Train policies in simulation to test the RL stack before training on real robots
+
+- Collect demonstrations in sim using external devices like gamepads or keyboards
+- Perform human interventions during policy learning
+
+Currently, the main environment is a Franka Panda robot simulation based on MuJoCo, with tasks like picking up a cube.
+
+## Installation
+
+First, install the `gym_hil` package within the LeRobot environment:
+
+```bash
+pip install -e ".[hilserl]"
+```
+
+## What do I need?
+
+- A gamepad or keyboard to control the robot
+- A Nvidia GPU
+
+## Configuration
+
+To use `gym_hil` with LeRobot, you need to create a configuration file. An example is provided [here](https://huggingface.co/datasets/lerobot/config_examples/resolve/main/rl/gym_hil/env_config.json). Key configuration sections include:
+
+### Environment Type and Task
+
+```json
+{
+  "env": {
+    "type": "gym_manipulator",
+    "name": "gym_hil",
+    "task": "PandaPickCubeGamepad-v0",
+    "fps": 10
+  },
+  "device": "cuda"
+}
+```
+
+Available tasks:
+
+- `PandaPickCubeBase-v0`: Basic environment
+- `PandaPickCubeGamepad-v0`: With gamepad control
+- `PandaPickCubeKeyboard-v0`: With keyboard control
+
+### Processor Configuration
+
+```json
+{
+  "env": {
+    "processor": {
+      "control_mode": "gamepad",
+      "gripper": {
+        "use_gripper": true,
+        "gripper_penalty": -0.02
+      },
+      "reset": {
+        "control_time_s": 15.0,
+        "fixed_reset_joint_positions": [
+          0.0, 0.195, 0.0, -2.43, 0.0, 2.62, 0.785
+        ]
+      },
+      "inverse_kinematics": {
+        "end_effector_step_sizes": {
+          "x": 0.025,
+          "y": 0.025,
+          "z": 0.025
+        }
+      }
+    }
+  }
+}
+```
+
+Important parameters:
+
+- `gripper.gripper_penalty`: Penalty for excessive gripper movement
+- `gripper.use_gripper`: Whether to enable gripper control
+- `inverse_kinematics.end_effector_step_sizes`: Size of the steps in the x,y,z axes of the end-effector
+- `control_mode`: Set to `"gamepad"` to use a gamepad controller
+
+## Running with HIL RL of LeRobot
+
+### Basic Usage
+
+To run the environment, set mode to null:
+
+```bash
+python -m lerobot.rl.gym_manipulator --config_path path/to/gym_hil_env.json
+```
+
+### Recording a Dataset
+
+To collect a dataset, set the mode to `record` whilst defining the repo_id and number of episodes to record:
+
+```json
+{
+  "env": {
+    "type": "gym_manipulator",
+    "name": "gym_hil",
+    "task": "PandaPickCubeGamepad-v0"
+  },
+  "dataset": {
+    "repo_id": "username/sim_dataset",
+    "root": null,
+    "task": "pick_cube",
+    "num_episodes_to_record": 10,
+    "replay_episode": null,
+    "push_to_hub": true
+  },
+  "mode": "record"
+}
+```
+
+```bash
+python -m lerobot.rl.gym_manipulator --config_path path/to/gym_hil_env.json
+```
+
+### Training a Policy
+
+To train a policy, checkout the configuration example available [here](https://huggingface.co/datasets/lerobot/config_examples/resolve/main/rl/gym_hil/train_config.json) and run the actor and learner servers:
+
+```bash
+python -m lerobot.rl.actor --config_path path/to/train_gym_hil_env.json
+```
+
+In a different terminal, run the learner server:
+
+```bash
+python -m lerobot.rl.learner --config_path path/to/train_gym_hil_env.json
+```
+
+The simulation environment provides a safe and repeatable way to develop and test your Human-In-the-Loop reinforcement learning components before deploying to real robots.
+
+Congrats 🎉, you have finished this tutorial!
+
+> [!TIP]
+> If you have any questions or need help, please reach out on [Discord](https://discord.com/invite/s3KuuzsPFb).
+
+Paper citation:
+
+```
+@article{luo2024precise,
+  title={Precise and Dexterous Robotic Manipulation via Human-in-the-Loop Reinforcement Learning},
+  author={Luo, Jianlan and Xu, Charles and Wu, Jeffrey and Levine, Sergey},
+  journal={arXiv preprint arXiv:2410.21845},
+  year={2024}
+}
+```
diff --git a/lerobot/docs/source/hope_jr.mdx b/lerobot/docs/source/hope_jr.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..8826d975803cf43dc17b1e44f8c641fad336feb8
--- /dev/null
+++ b/lerobot/docs/source/hope_jr.mdx
@@ -0,0 +1,283 @@
+# HopeJR
+
+## Prerequisites
+
+- [Hardware Setup](https://github.com/TheRobotStudio/HOPEJr)
+
+## Install LeRobot
+
+Follow the [installation instructions](https://github.com/huggingface/lerobot#installation) to install LeRobot.
+
+Install LeRobot with HopeJR dependencies:
+
+```bash
+pip install -e ".[hopejr]"
+```
+
+## Device Configuration
+
+Before starting calibration and operation, you need to identify the USB ports for each HopeJR component. Run this script to find the USB ports for the arm, hand, glove, and exoskeleton:
+
+```bash
+lerobot-find-port
+```
+
+This will display the available USB ports and their associated devices. Make note of the port paths (e.g., `/dev/tty.usbmodem58760433331`, `/dev/tty.usbmodem11301`) as you'll need to specify them in the `--robot.port` and `--teleop.port` parameters when recording data, replaying episodes, or running teleoperation scripts.
+
+## Step 1: Calibration
+
+Before performing teleoperation, HopeJR's limbs need to be calibrated. Calibration files will be saved in `~/.cache/huggingface/lerobot/calibration`
+
+### 1.1 Calibrate Robot Hand
+
+```bash
+lerobot-calibrate \
+    --robot.type=hope_jr_hand \
+    --robot.port=/dev/tty.usbmodem58760432281 \
+    --robot.id=blue \
+    --robot.side=right
+```
+
+When running the calibration script, a calibration GUI will pop up. Finger joints are named as follows:
+
+**Thumb**:
+
+- **CMC**: base joint connecting thumb to hand
+- **MCP**: knuckle joint
+- **PIP**: first finger joint
+- **DIP** : fingertip joint
+
+**Index, Middle, Ring, and Pinky fingers**:
+
+- **Radial flexor**: Moves base of finger towards the thumb
+- **Ulnar flexor**: Moves base of finger towards the pinky
+- **PIP/DIP**: Flexes the distal and proximal phalanx of the finger
+
+Each one of these will need to be calibrated individually via the GUI.
+Note that ulnar and radial flexors should have ranges of the same size (but with different offsets) in order to get symmetric movement.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/calibration_gui_1.png"
+    alt="Setting boundaries in the hand calibration GUI"
+    title="Setting boundaries in the hand calibration GUI"
+    width="100%"
+  ></img>
+</p>
+
+Use the calibration interface to set the range boundaries for each joint as shown above.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/calibration_gui_2.png"
+    alt="Saving calibration values"
+    title="Saving calibration values"
+    width="100%"
+  ></img>
+</p>
+
+Once you have set the appropriate boundaries for all joints, click "Save" to save the calibration values to the motors.
+
+### 1.2 Calibrate Teleoperator Glove
+
+```bash
+lerobot-calibrate \
+    --teleop.type=homunculus_glove \
+    --teleop.port=/dev/tty.usbmodem11201 \
+    --teleop.id=red \
+    --teleop.side=right
+```
+
+Move each finger through its full range of motion, starting from the thumb.
+
+```
+Move thumb through its entire range of motion.
+Recording positions. Press ENTER to stop...
+
+-------------------------------------------
+NAME      |    MIN |    POS |    MAX
+thumb_cmc |   1790 |   1831 |   1853
+thumb_mcp |   1497 |   1514 |   1528
+thumb_pip |   1466 |   1496 |   1515
+thumb_dip |   1463 |   1484 |   1514
+```
+
+Continue with each finger:
+
+```
+Move middle through its entire range of motion.
+Recording positions. Press ENTER to stop...
+
+-------------------------------------------
+NAME                 |    MIN |    POS |    MAX
+middle_mcp_abduction |   1598 |   1718 |   1820
+middle_mcp_flexion   |   1512 |   1658 |   2136
+middle_dip           |   1484 |   1500 |   1547
+```
+
+Once calibration is complete, the system will save the calibration to `/Users/your_username/.cache/huggingface/lerobot/calibration/teleoperators/homunculus_glove/red.json`
+
+### 1.3 Calibrate Robot Arm
+
+```bash
+lerobot-calibrate \
+    --robot.type=hope_jr_arm \
+    --robot.port=/dev/tty.usbserial-1110 \
+    --robot.id=white
+```
+
+This will open a calibration GUI where you can set the range limits for each motor. The arm motions are organized as follows:
+
+- **Shoulder**: pitch, yaw, and roll
+- **Elbow**: flex
+- **Wrist**: pitch, yaw, and roll
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/calibration_gui_2.png"
+    alt="Setting boundaries in the arm calibration GUI"
+    title="Setting boundaries in the arm calibration GUI"
+    width="100%"
+  ></img>
+</p>
+
+Use the calibration interface to set the range boundaries for each joint. Move each joint through its full range of motion and adjust the minimum and maximum values accordingly. Once you have set the appropriate boundaries for all joints, save the calibration.
+
+### 1.4 Calibrate Teleoperator Exoskeleton
+
+```bash
+lerobot-calibrate \
+    --teleop.type=homunculus_arm \
+    --teleop.port=/dev/tty.usbmodem11201 \
+    --teleop.id=black
+```
+
+The exoskeleton allows one to control the robot arm. During calibration, you'll be prompted to move all joints through their full range of motion:
+
+```
+Move all joints through their entire range of motion.
+Recording positions. Press ENTER to stop...
+
+-------------------------------------------
+-------------------------------------------
+NAME            |    MIN |    POS |    MAX
+shoulder_pitch  |    586 |    736 |    895
+shoulder_yaw    |   1257 |   1374 |   1390
+shoulder_roll   |    449 |   1034 |   2564
+elbow_flex      |   3023 |   3117 |   3134
+wrist_roll      |   3073 |   3096 |   3147
+wrist_yaw       |   2143 |   2171 |   2185
+wrist_pitch     |   1975 |   1993 |   2074
+Calibration saved to /Users/your_username/.cache/huggingface/lerobot/calibration/teleoperators/homunculus_arm/black.json
+```
+
+## Step 2: Teleoperation
+
+Due to global variable conflicts in the Feetech middleware, teleoperation for arm and hand must run in separate shell sessions:
+
+### Hand
+
+```bash
+lerobot-teleoperate \
+    --robot.type=hope_jr_hand \
+    --robot.port=/dev/tty.usbmodem58760432281 \
+    --robot.id=blue \
+    --robot.side=right \
+    --teleop.type=homunculus_glove \
+    --teleop.port=/dev/tty.usbmodem11201 \
+    --teleop.id=red \
+    --teleop.side=right \
+    --display_data=true \
+    --fps=30
+```
+
+### Arm
+
+```bash
+lerobot-teleoperate \
+    --robot.type=hope_jr_arm \
+    --robot.port=/dev/tty.usbserial-1110 \
+    --robot.id=white \
+    --teleop.type=homunculus_arm \
+    --teleop.port=/dev/tty.usbmodem11201 \
+    --teleop.id=black \
+    --display_data=true \
+    --fps=30
+```
+
+## Step 3: Record, Replay, Train
+
+Record, Replay and Train with Hope-JR is still experimental.
+
+### Record
+
+This step records the dataset, which can be seen as an example [here](https://huggingface.co/datasets/nepyope/hand_record_test_with_video_data/settings).
+
+```bash
+lerobot-record \
+    --robot.type=hope_jr_hand \
+    --robot.port=/dev/tty.usbmodem58760432281 \
+    --robot.id=right \
+    --robot.side=right \
+    --robot.cameras='{"main": {"type": "opencv", "index_or_path": 0, "width": 640, "height": 480, "fps": 30}}' \
+    --teleop.type=homunculus_glove \
+    --teleop.port=/dev/tty.usbmodem1201 \
+    --teleop.id=right \
+    --teleop.side=right \
+    --dataset.repo_id=<USER>/hand_record_test_with_video_data \
+    --dataset.single_task="Hand recording test with video data" \
+    --dataset.num_episodes=1 \
+    --dataset.episode_time_s=5 \
+    --dataset.push_to_hub=true \
+    --dataset.private=true \
+    --dataset.streaming_encoding=true \
+    --dataset.encoder_threads=2 \
+    # --dataset.vcodec=auto \
+    --display_data=true
+```
+
+### Replay
+
+```bash
+lerobot-replay \
+    --robot.type=hope_jr_hand \
+    --robot.port=/dev/tty.usbmodem58760432281 \
+    --robot.id=right \
+    --robot.side=right \
+    --dataset.repo_id=<USER>/hand_record_test_with_camera \
+    --dataset.episode=0
+```
+
+### Train
+
+```bash
+lerobot-train \
+  --dataset.repo_id=<USER>/hand_record_test_with_video_data \
+  --policy.type=act \
+  --output_dir=outputs/train/hopejr_hand \
+  --job_name=hopejr \
+  --policy.device=mps \
+  --wandb.enable=true \
+  --policy.repo_id=<USER>/hand_test_policy
+```
+
+### Evaluate
+
+This training run can be viewed as an example [here](https://wandb.ai/tino/lerobot/runs/rp0k8zvw?nw=nwusertino).
+
+```bash
+lerobot-record \
+  --robot.type=hope_jr_hand \
+  --robot.port=/dev/tty.usbmodem58760432281 \
+  --robot.id=right \
+  --robot.side=right \
+  --robot.cameras='{"main": {"type": "opencv", "index_or_path": 0, "width": 640, "height": 480, "fps": 30}}' \
+  --display_data=false \
+  --dataset.repo_id=<USER>/eval_hopejr \
+  --dataset.single_task="Evaluate hopejr hand policy" \
+  --dataset.num_episodes=10 \
+  --dataset.streaming_encoding=true \
+  --dataset.encoder_threads=2 \
+  # --dataset.vcodec=auto \
+  --policy.path=outputs/train/hopejr_hand/checkpoints/last/pretrained_model
+```
diff --git a/lerobot/docs/source/il_robots.mdx b/lerobot/docs/source/il_robots.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..245634382e8ee712cb77e024834379dc7fa0a7a9
--- /dev/null
+++ b/lerobot/docs/source/il_robots.mdx
@@ -0,0 +1,626 @@
+# Imitation Learning on Real-World Robots
+
+This tutorial will explain how to train a neural network to control a real robot autonomously.
+
+**You'll learn:**
+
+1. How to record and visualize your dataset.
+2. How to train a policy using your data and prepare it for evaluation.
+3. How to evaluate your policy and visualize the results.
+
+By following these steps, you'll be able to replicate tasks, such as picking up a Lego block and placing it in a bin with a high success rate, as shown in the video below.
+
+<details>
+<summary><strong>Video: pickup lego block task</strong></summary>
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/lerobot_task.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+</details>
+
+This tutorial isn’t tied to a specific robot: we walk you through the commands and API snippets you can adapt for any supported platform.
+
+During data collection, you’ll use a “teloperation” device, such as a leader arm or keyboard to teleoperate the robot and record its motion trajectories.
+
+Once you’ve gathered enough trajectories, you’ll train a neural network to imitate these trajectories and deploy the trained model so your robot can perform the task autonomously.
+
+If you run into any issues at any point, jump into our [Discord community](https://discord.com/invite/s3KuuzsPFb) for support.
+
+## Set up and Calibrate
+
+If you haven't yet set up and calibrated your robot and teleop device, please do so by following the robot-specific tutorial.
+
+## Teleoperate
+
+In this example, we’ll demonstrate how to teleoperate the SO101 robot. For each command, we also provide a corresponding API example.
+
+Note that the `id` associated with a robot is used to store the calibration file. It's important to use the same `id` when teleoperating, recording, and evaluating when using the same setup.
+
+<hfoptions id="teleoperate_so101">
+<hfoption id="Command">
+```bash
+lerobot-teleoperate \
+    --robot.type=so101_follower \
+    --robot.port=/dev/tty.usbmodem58760431541 \
+    --robot.id=my_awesome_follower_arm \
+    --teleop.type=so101_leader \
+    --teleop.port=/dev/tty.usbmodem58760431551 \
+    --teleop.id=my_awesome_leader_arm
+```
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.teleoperators.so_leader import SO101LeaderConfig, SO101Leader
+from lerobot.robots.so_follower import SO101FollowerConfig, SO101Follower
+
+robot_config = SO101FollowerConfig(
+    port="/dev/tty.usbmodem58760431541",
+    id="my_red_robot_arm",
+)
+
+teleop_config = SO101LeaderConfig(
+    port="/dev/tty.usbmodem58760431551",
+    id="my_blue_leader_arm",
+)
+
+robot = SO101Follower(robot_config)
+teleop_device = SO101Leader(teleop_config)
+robot.connect()
+teleop_device.connect()
+
+while True:
+    action = teleop_device.get_action()
+    robot.send_action(action)
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+The teleoperate command will automatically:
+
+1. Identify any missing calibrations and initiate the calibration procedure.
+2. Connect the robot and teleop device and start teleoperation.
+
+## Cameras
+
+To add cameras to your setup, follow this [Guide](./cameras#setup-cameras).
+
+## Teleoperate with cameras
+
+With `rerun`, you can teleoperate again while simultaneously visualizing the camera feeds and joint positions. In this example, we’re using the Koch arm.
+
+<hfoptions id="teleoperate_koch_camera">
+<hfoption id="Command">
+```bash
+lerobot-teleoperate \
+    --robot.type=koch_follower \
+    --robot.port=/dev/tty.usbmodem58760431541 \
+    --robot.id=my_awesome_follower_arm \
+    --robot.cameras="{ front: {type: opencv, index_or_path: 0, width: 1920, height: 1080, fps: 30}}" \
+    --teleop.type=koch_leader \
+    --teleop.port=/dev/tty.usbmodem58760431551 \
+    --teleop.id=my_awesome_leader_arm \
+    --display_data=true
+```
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.teleoperators.koch_leader import KochLeaderConfig, KochLeader
+from lerobot.robots.koch_follower import KochFollowerConfig, KochFollower
+
+camera_config = {
+    "front": OpenCVCameraConfig(index_or_path=0, width=1920, height=1080, fps=30)
+}
+
+robot_config = KochFollowerConfig(
+    port="/dev/tty.usbmodem585A0076841",
+    id="my_red_robot_arm",
+    cameras=camera_config
+)
+
+teleop_config = KochLeaderConfig(
+    port="/dev/tty.usbmodem58760431551",
+    id="my_blue_leader_arm",
+)
+
+robot = KochFollower(robot_config)
+teleop_device = KochLeader(teleop_config)
+robot.connect()
+teleop_device.connect()
+
+while True:
+    observation = robot.get_observation()
+    action = teleop_device.get_action()
+    robot.send_action(action)
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+## Record a dataset
+
+Once you're familiar with teleoperation, you can record your first dataset.
+
+We use the Hugging Face hub features for uploading your dataset. If you haven't previously used the Hub, make sure you can login via the cli using a write-access token, this token can be generated from the [Hugging Face settings](https://huggingface.co/settings/tokens).
+
+Add your token to the CLI by running this command:
+
+```bash
+hf auth login --token ${HUGGINGFACE_TOKEN} --add-to-git-credential
+```
+
+Then store your Hugging Face repository name in a variable:
+
+```bash
+HF_USER=$(NO_COLOR=1 hf auth whoami | awk -F': *' 'NR==1 {print $2}')
+echo $HF_USER
+```
+
+Now you can record a dataset. To record 5 episodes and upload your dataset to the hub, adapt the code below for your robot and execute the command or API example.
+
+<hfoptions id="record">
+<hfoption id="Command">
+```bash
+lerobot-record \
+    --robot.type=so101_follower \
+    --robot.port=/dev/tty.usbmodem585A0076841 \
+    --robot.id=my_awesome_follower_arm \
+    --robot.cameras="{ front: {type: opencv, index_or_path: 0, width: 1920, height: 1080, fps: 30}}" \
+    --teleop.type=so101_leader \
+    --teleop.port=/dev/tty.usbmodem58760431551 \
+    --teleop.id=my_awesome_leader_arm \
+    --display_data=true \
+    --dataset.repo_id=${HF_USER}/record-test \
+    --dataset.num_episodes=5 \
+    --dataset.single_task="Grab the black cube" \
+    --dataset.streaming_encoding=true \
+    # --dataset.vcodec=auto \
+    --dataset.encoder_threads=2
+```
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.utils import hw_to_dataset_features
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+from lerobot.teleoperators.so_leader.config_so100_leader import SO100LeaderConfig
+from lerobot.teleoperators.so_leader.so100_leader import SO100Leader
+from lerobot.utils.control_utils import init_keyboard_listener
+from lerobot.utils.utils import log_say
+from lerobot.utils.visualization_utils import init_rerun
+from lerobot.scripts.lerobot_record import record_loop
+from lerobot.processor import make_default_processors
+
+NUM_EPISODES = 5
+FPS = 30
+EPISODE_TIME_SEC = 60
+RESET_TIME_SEC = 10
+TASK_DESCRIPTION = "My task description"
+
+# Create robot configuration
+robot_config = SO100FollowerConfig(
+    id="my_awesome_follower_arm",
+    cameras={
+        "front": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=FPS) # Optional: fourcc="MJPG" for troubleshooting OpenCV async error.
+    },
+    port="/dev/tty.usbmodem58760434471",
+)
+
+teleop_config = SO100LeaderConfig(
+    id="my_awesome_leader_arm",
+    port="/dev/tty.usbmodem585A0077581",
+)
+
+# Initialize the robot and teleoperator
+robot = SO100Follower(robot_config)
+teleop = SO100Leader(teleop_config)
+
+# Configure the dataset features
+action_features = hw_to_dataset_features(robot.action_features, "action")
+obs_features = hw_to_dataset_features(robot.observation_features, "observation")
+dataset_features = {**action_features, **obs_features}
+
+# Create the dataset
+dataset = LeRobotDataset.create(
+    repo_id="<hf_username>/<dataset_repo_id>",
+    fps=FPS,
+    features=dataset_features,
+    robot_type=robot.name,
+    use_videos=True,
+    image_writer_threads=4,
+)
+
+# Initialize the keyboard listener and rerun visualization
+_, events = init_keyboard_listener()
+init_rerun(session_name="recording")
+
+# Connect the robot and teleoperator
+robot.connect()
+teleop.connect()
+
+# Create the required processors
+teleop_action_processor, robot_action_processor, robot_observation_processor = make_default_processors()
+
+episode_idx = 0
+while episode_idx < NUM_EPISODES and not events["stop_recording"]:
+    log_say(f"Recording episode {episode_idx + 1} of {NUM_EPISODES}")
+
+    record_loop(
+        robot=robot,
+        events=events,
+        fps=FPS,
+        teleop_action_processor=teleop_action_processor,
+        robot_action_processor=robot_action_processor,
+        robot_observation_processor=robot_observation_processor,
+        teleop=teleop,
+        dataset=dataset,
+        control_time_s=EPISODE_TIME_SEC,
+        single_task=TASK_DESCRIPTION,
+        display_data=True,
+    )
+
+    # Reset the environment if not stopping or re-recording
+    if not events["stop_recording"] and (episode_idx < NUM_EPISODES - 1 or events["rerecord_episode"]):
+        log_say("Reset the environment")
+        record_loop(
+            robot=robot,
+            events=events,
+            fps=FPS,
+            teleop_action_processor=teleop_action_processor,
+            robot_action_processor=robot_action_processor,
+            robot_observation_processor=robot_observation_processor,
+            teleop=teleop,
+            control_time_s=RESET_TIME_SEC,
+            single_task=TASK_DESCRIPTION,
+            display_data=True,
+        )
+
+    if events["rerecord_episode"]:
+        log_say("Re-recording episode")
+        events["rerecord_episode"] = False
+        events["exit_early"] = False
+        dataset.clear_episode_buffer()
+        continue
+
+    dataset.save_episode()
+    episode_idx += 1
+
+# Clean up
+log_say("Stop recording")
+robot.disconnect()
+teleop.disconnect()
+dataset.push_to_hub()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+#### Dataset upload
+
+Locally, your dataset is stored in this folder: `~/.cache/huggingface/lerobot/{repo-id}`. At the end of data recording, your dataset will be uploaded on your Hugging Face page (e.g. `https://huggingface.co/datasets/${HF_USER}/so101_test`) that you can obtain by running:
+
+```bash
+echo https://huggingface.co/datasets/${HF_USER}/so101_test
+```
+
+Your dataset will be automatically tagged with `LeRobot` for the community to find it easily, and you can also add custom tags (in this case `tutorial` for example).
+
+You can look for other LeRobot datasets on the hub by searching for `LeRobot` [tags](https://huggingface.co/datasets?other=LeRobot).
+
+You can also push your local dataset to the Hub manually, running:
+
+```bash
+hf upload ${HF_USER}/record-test ~/.cache/huggingface/lerobot/{repo-id} --repo-type dataset
+```
+
+#### Record function
+
+The `record` function provides a suite of tools for capturing and managing data during robot operation:
+
+##### 1. Data Storage
+
+- Data is stored using the `LeRobotDataset` format and is stored on disk during recording.
+- By default, the dataset is pushed to your Hugging Face page after recording.
+  - To disable uploading, use `--dataset.push_to_hub=False`.
+
+##### 2. Checkpointing and Resuming
+
+- Checkpoints are automatically created during recording.
+- If an issue occurs, you can resume by re-running the same command with `--resume=true`. When resuming a recording, `--dataset.num_episodes` must be set to the **number of additional episodes to be recorded**, and not to the targeted total number of episodes in the dataset !
+- To start recording from scratch, **manually delete** the dataset directory.
+
+##### 3. Recording Parameters
+
+Set the flow of data recording using command-line arguments:
+
+- `--dataset.episode_time_s=60`
+  Duration of each data recording episode (default: **60 seconds**).
+- `--dataset.reset_time_s=60`
+  Duration for resetting the environment after each episode (default: **60 seconds**).
+- `--dataset.num_episodes=50`
+  Total number of episodes to record (default: **50**).
+
+##### 4. Keyboard Controls During Recording
+
+Control the data recording flow using keyboard shortcuts:
+
+- Press **Right Arrow (`→`)**: Early stop the current episode or reset time and move to the next.
+- Press **Left Arrow (`←`)**: Cancel the current episode and re-record it.
+- Press **Escape (`ESC`)**: Immediately stop the session, encode videos, and upload the dataset.
+
+#### Tips for gathering data
+
+Once you're comfortable with data recording, you can create a larger dataset for training. A good starting task is grasping an object at different locations and placing it in a bin. We suggest recording at least 50 episodes, with 10 episodes per location. Keep the cameras fixed and maintain consistent grasping behavior throughout the recordings. Also make sure the object you are manipulating is visible on the camera's. A good rule of thumb is you should be able to do the task yourself by only looking at the camera images.
+
+In the following sections, you’ll train your neural network. After achieving reliable grasping performance, you can start introducing more variations during data collection, such as additional grasp locations, different grasping techniques, and altering camera positions.
+
+Avoid adding too much variation too quickly, as it may hinder your results.
+
+If you want to dive deeper into this important topic, you can check out the [blog post](https://huggingface.co/blog/lerobot-datasets#what-makes-a-good-dataset) we wrote on what makes a good dataset.
+
+#### Troubleshooting:
+
+- On Linux, if the left and right arrow keys and escape key don't have any effect during data recording, make sure you've set the `$DISPLAY` environment variable. See [pynput limitations](https://pynput.readthedocs.io/en/latest/limitations.html#linux).
+
+## Visualize a dataset
+
+If you uploaded your dataset to the hub with `--control.push_to_hub=true`, you can [visualize your dataset online](https://huggingface.co/spaces/lerobot/visualize_dataset) by copy pasting your repo id given by:
+
+```bash
+echo ${HF_USER}/so101_test
+```
+
+## Replay an episode
+
+A useful feature is the `replay` function, which allows you to replay any episode that you've recorded or episodes from any dataset out there. This function helps you test the repeatability of your robot's actions and assess transferability across robots of the same model.
+
+You can replay the first episode on your robot with either the command below or with the API example:
+
+<hfoptions id="replay">
+<hfoption id="Command">
+```bash
+lerobot-replay \
+    --robot.type=so101_follower \
+    --robot.port=/dev/tty.usbmodem58760431541 \
+    --robot.id=my_awesome_follower_arm \
+    --dataset.repo_id=${HF_USER}/record-test \
+    --dataset.episode=0 # choose the episode you want to replay
+```
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+import time
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.robots.so_follower.config_so100_follower import SO100FollowerConfig
+from lerobot.robots.so_follower.so100_follower import SO100Follower
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import log_say
+
+episode_idx = 0
+
+robot_config = SO100FollowerConfig(port="/dev/tty.usbmodem58760434471", id="my_awesome_follower_arm")
+
+robot = SO100Follower(robot_config)
+robot.connect()
+
+dataset = LeRobotDataset("<hf_username>/<dataset_repo_id>", episodes=[episode_idx])
+actions = dataset.hf_dataset.select_columns("action")
+
+log_say(f"Replaying episode {episode_idx}")
+for idx in range(dataset.num_frames):
+    t0 = time.perf_counter()
+
+    action = {
+        name: float(actions[idx]["action"][i]) for i, name in enumerate(dataset.features["action"]["names"])
+    }
+    robot.send_action(action)
+
+    precise_sleep(max(1.0 / dataset.fps - (time.perf_counter() - t0), 0.0))
+
+robot.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+Your robot should replicate movements similar to those you recorded. For example, check out [this video](https://x.com/RemiCadene/status/1793654950905680090) where we use `replay` on a Aloha robot from [Trossen Robotics](https://www.trossenrobotics.com).
+
+## Train a policy
+
+To train a policy to control your robot, use the [`lerobot-train`](https://github.com/huggingface/lerobot/blob/main/src/lerobot/scripts/lerobot_train.py) script. A few arguments are required. Here is an example command:
+
+```bash
+lerobot-train \
+  --dataset.repo_id=${HF_USER}/so101_test \
+  --policy.type=act \
+  --output_dir=outputs/train/act_so101_test \
+  --job_name=act_so101_test \
+  --policy.device=cuda \
+  --wandb.enable=true \
+  --policy.repo_id=${HF_USER}/my_policy
+```
+
+Let's explain the command:
+
+1. We provided the dataset as argument with `--dataset.repo_id=${HF_USER}/so101_test`.
+2. We provided the policy with `policy.type=act`. This loads configurations from [`configuration_act.py`](https://github.com/huggingface/lerobot/blob/main/src/lerobot/policies/act/configuration_act.py). Importantly, this policy will automatically adapt to the number of motor states, motor actions and cameras of your robot (e.g. `laptop` and `phone`) which have been saved in your dataset.
+3. We provided `policy.device=cuda` since we are training on a Nvidia GPU, but you could use `policy.device=mps` to train on Apple silicon.
+4. We provided `wandb.enable=true` to use [Weights and Biases](https://docs.wandb.ai/quickstart) for visualizing training plots. This is optional but if you use it, make sure you are logged in by running `wandb login`.
+
+Training should take several hours. You will find checkpoints in `outputs/train/act_so101_test/checkpoints`.
+
+To resume training from a checkpoint, below is an example command to resume from `last` checkpoint of the `act_so101_test` policy:
+
+```bash
+lerobot-train \
+  --config_path=outputs/train/act_so101_test/checkpoints/last/pretrained_model/train_config.json \
+  --resume=true
+```
+
+If you do not want to push your model to the hub after training use `--policy.push_to_hub=false`.
+
+Additionally you can provide extra `tags` or specify a `license` for your model or make the model repo `private` by adding this: `--policy.private=true --policy.tags=\[ppo,rl\] --policy.license=mit`
+
+#### Train using Google Colab
+
+If your local computer doesn't have a powerful GPU you could utilize Google Colab to train your model by following the [ACT training notebook](./notebooks#training-act).
+
+#### Upload policy checkpoints
+
+Once training is done, upload the latest checkpoint with:
+
+```bash
+hf upload ${HF_USER}/act_so101_test \
+  outputs/train/act_so101_test/checkpoints/last/pretrained_model
+```
+
+You can also upload intermediate checkpoints with:
+
+```bash
+CKPT=010000
+hf upload ${HF_USER}/act_so101_test${CKPT} \
+  outputs/train/act_so101_test/checkpoints/${CKPT}/pretrained_model
+```
+
+## Run inference and evaluate your policy
+
+You can use the `record` script from [`lerobot-record`](https://github.com/huggingface/lerobot/blob/main/src/lerobot/scripts/lerobot_record.py) with a policy checkpoint as input, to run inference and evaluate your policy. For instance, run this command or API example to run inference and record 10 evaluation episodes:
+
+<hfoptions id="eval">
+<hfoption id="Command">
+```bash
+lerobot-record  \
+  --robot.type=so100_follower \
+  --robot.port=/dev/ttyACM1 \
+  --robot.cameras="{ up: {type: opencv, index_or_path: /dev/video10, width: 640, height: 480, fps: 30}, side: {type: intelrealsense, serial_number_or_name: 233522074606, width: 640, height: 480, fps: 30}}" \
+  --robot.id=my_awesome_follower_arm \
+  --display_data=false \
+  --dataset.repo_id=${HF_USER}/eval_so100 \
+  --dataset.single_task="Put lego brick into the transparent box" \
+  --dataset.streaming_encoding=true \
+  --dataset.encoder_threads=2 \
+  # --dataset.vcodec=auto \
+  # <- Teleop optional if you want to teleoperate in between episodes \
+  # --teleop.type=so100_leader \
+  # --teleop.port=/dev/ttyACM0 \
+  # --teleop.id=my_awesome_leader_arm \
+  --policy.path=${HF_USER}/my_policy
+```
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.utils import hw_to_dataset_features
+from lerobot.policies.act.modeling_act import ACTPolicy
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.robots.so_follower.config_so100_follower import SO100FollowerConfig
+from lerobot.robots.so_follower.so100_follower import SO100Follower
+from lerobot.scripts.lerobot_record import record_loop
+from lerobot.utils.control_utils import init_keyboard_listener
+from lerobot.utils.utils import log_say
+from lerobot.utils.visualization_utils import init_rerun
+
+
+NUM_EPISODES = 5
+FPS = 30
+EPISODE_TIME_SEC = 60
+TASK_DESCRIPTION = "My task description"
+HF_MODEL_ID = "<hf_username>/<model_repo_id>"
+HF_DATASET_ID = "<hf_username>/<eval_dataset_repo_id>"
+
+# Create the robot configuration
+camera_config = {"front": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=FPS)}
+robot_config = SO100FollowerConfig(
+    port="/dev/tty.usbmodem58760434471", id="my_awesome_follower_arm", cameras=camera_config
+)
+
+# Initialize the robot
+robot = SO100Follower(robot_config)
+
+# Initialize the policy
+policy = ACTPolicy.from_pretrained(HF_MODEL_ID)
+
+# Configure the dataset features
+action_features = hw_to_dataset_features(robot.action_features, "action")
+obs_features = hw_to_dataset_features(robot.observation_features, "observation")
+dataset_features = {**action_features, **obs_features}
+
+# Create the dataset
+dataset = LeRobotDataset.create(
+    repo_id=HF_DATASET_ID,
+    fps=FPS,
+    features=dataset_features,
+    robot_type=robot.name,
+    use_videos=True,
+    image_writer_threads=4,
+)
+
+# Initialize the keyboard listener and rerun visualization
+_, events = init_keyboard_listener()
+init_rerun(session_name="recording")
+
+# Connect the robot
+robot.connect()
+
+preprocessor, postprocessor = make_pre_post_processors(
+    policy_cfg=policy,
+    pretrained_path=HF_MODEL_ID,
+    dataset_stats=dataset.meta.stats,
+)
+
+for episode_idx in range(NUM_EPISODES):
+    log_say(f"Running inference, recording eval episode {episode_idx + 1} of {NUM_EPISODES}")
+
+    # Run the policy inference loop
+    record_loop(
+        robot=robot,
+        events=events,
+        fps=FPS,
+        policy=policy,
+        preprocessor=preprocessor,
+        postprocessor=postprocessor,
+        dataset=dataset,
+        control_time_s=EPISODE_TIME_SEC,
+        single_task=TASK_DESCRIPTION,
+        display_data=True,
+    )
+
+    dataset.save_episode()
+
+# Clean up
+robot.disconnect()
+dataset.push_to_hub()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+As you can see, it's almost the same command as previously used to record your training dataset. Two things changed:
+
+1. There is an additional `--control.policy.path` argument which indicates the path to your policy checkpoint with (e.g. `outputs/train/eval_act_so101_test/checkpoints/last/pretrained_model`). You can also use the model repository if you uploaded a model checkpoint to the hub (e.g. `${HF_USER}/act_so101_test`).
+2. The name of dataset begins by `eval` to reflect that you are running inference (e.g. `${HF_USER}/eval_act_so101_test`).
diff --git a/lerobot/docs/source/implement_your_own_processor.mdx b/lerobot/docs/source/implement_your_own_processor.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..5b7d4f78ac1e2d355dc14ba2a37f3484204b936c
--- /dev/null
+++ b/lerobot/docs/source/implement_your_own_processor.mdx
@@ -0,0 +1,273 @@
+# Implement your own Robot Processor
+
+In this tutorial, you'll learn how to implement your own Robot Processor.
+It begins by exploring the need for a custom processor, then uses the `NormalizerProcessorStep` as the running example to explain how to implement, configure, and serialize a processor. Finally, it lists all helper processors that ship with LeRobot.
+
+## Why would you need a custom processor?
+
+In most cases, when reading raw data from sensors or when models output actions, you need to process this data to make it compatible with your target system. For example, a common need is normalizing data ranges to make them suitable for neural networks.
+
+LeRobot's `NormalizerProcessorStep` handles this crucial task:
+
+```python
+# Input: raw joint positions in [0, 180] degrees
+raw_action = torch.tensor([90.0, 45.0, 135.0])
+
+# After processing: normalized to [-1, 1] range for model training
+normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=dataset_stats)
+normalized_result = normalizer(transition)
+# ...
+```
+
+Other common processing needs include:
+
+- **Device placement**: Moving tensors between CPU/GPU and converting data types
+- **Format conversion**: Transforming between different data structures
+- **Batching**: Adding/removing batch dimensions for model compatibility
+- **Safety constraints**: Applying limits to robot commands
+
+```python
+# Example pipeline combining multiple processors
+pipeline = PolicyProcessorPipeline([
+    RenameObservationsProcessorStep(rename_map={}),
+    AddBatchDimensionProcessorStep(),
+    NormalizerProcessorStep(features=features, stats=stats),
+    DeviceProcessorStep(device="cuda"),
+    # ...
+])
+```
+
+LeRobot provides a pipeline mechanism to implement sequences of processing steps for both input data and output actions, making it easy to compose these transformations in the right order for optimal performance.
+
+## How to implement your own processor?
+
+We'll use the `NormalizerProcessorStep` as our main example because it demonstrates essential processor patterns including state management, configuration serialization, and tensor handling that you'll commonly need.
+
+Prepare the sequence of processing steps necessary for your problem. A processor step is a class that implements the following methods:
+
+- `__call__`: implements the processing step for the input transition.
+- `get_config`: gets the configuration of the processor step.
+- `state_dict`: gets the state of the processor step.
+- `load_state_dict`: loads the state of the processor step.
+- `reset`: resets the state of the processor step.
+- `feature_contract`: displays the modification to the feature space during the processor step.
+
+### Implement the `__call__` method
+
+The `__call__` method is the core of your processor step. It takes an `EnvTransition` and returns a modified `EnvTransition`. Here's how the `NormalizerProcessorStep` works:
+
+```python
+@dataclass
+@ProcessorStepRegistry.register("normalizer_processor")
+class NormalizerProcessorStep(ProcessorStep):
+    """Normalize observations/actions using dataset statistics."""
+
+    features: dict[str, PolicyFeature]
+    norm_map: dict[FeatureType, NormalizationMode]
+    stats: dict[str, dict[str, Any]] | None = None
+    eps: float = 1e-8
+    _tensor_stats: dict = field(default_factory=dict, init=False, repr=False)
+
+    def __post_init__(self):
+        """Convert stats to tensors for efficient computation."""
+        self.stats = self.stats or {}
+        self._tensor_stats = to_tensor(self.stats, device=self.device, dtype=torch.float32)
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        new_transition = transition.copy()
+        # Normalize observations
+        # ...
+        # Normalize action
+        # ...
+        return new_transition
+
+```
+
+See the full implementation in `src/lerobot/processor/normalize_processor.py` for complete details.
+
+**Key principles:**
+
+- **Always use `transition.copy()`** to avoid side effects
+- **Handle both observations and actions** consistently
+- **Separate config from state**: `get_config()` returns JSON-serializable params, `state_dict()` returns tensors
+- **Convert stats to tensors** in `__post_init__()` for efficient computation
+
+### Configuration and State Management
+
+Processors support serialization through three methods that separate configuration from tensor state. The `NormalizerProcessorStep` demonstrates this perfectly - it carries dataset statistics (tensors) in its state, and hyperparameters in its config:
+
+```python
+# Continuing the NormalizerProcessorStep example...
+
+def get_config(self) -> dict[str, Any]:
+    """JSON-serializable configuration (no tensors)."""
+    return {
+        "eps": self.eps,
+        "features": {k: {"type": v.type.value, "shape": v.shape} for k, v in self.features.items()},
+        "norm_map": {ft.value: nm.value for ft, nm in self.norm_map.items()},
+        # ...
+    }
+
+def state_dict(self) -> dict[str, torch.Tensor]:
+    """Tensor state only (e.g., dataset statistics)."""
+    flat: dict[str, torch.Tensor] = {}
+    for key, sub in self._tensor_stats.items():
+        for stat_name, tensor in sub.items():
+            flat[f"{key}.{stat_name}"] = tensor.cpu()  # Always save to CPU
+    return flat
+
+def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+    """Restore tensor state at runtime."""
+    self._tensor_stats.clear()
+    for flat_key, tensor in state.items():
+        key, stat_name = flat_key.rsplit(".", 1)
+        # Load to processor's configured device
+        self._tensor_stats.setdefault(key, {})[stat_name] = tensor.to(
+            dtype=torch.float32, device=self.device
+        )
+        # ...
+```
+
+**Usage:**
+
+```python
+# Save (e.g., inside a policy)
+config = normalizer.get_config()
+tensors = normalizer.state_dict()
+
+# Restore (e.g., loading a pretrained policy)
+new_normalizer = NormalizerProcessorStep(**config)
+new_normalizer.load_state_dict(tensors)
+# Now new_normalizer has the same stats and configuration
+```
+
+### Transform features
+
+The `transform_features` method defines how your processor transforms feature names and shapes. This is crucial for policy configuration and debugging.
+
+For `NormalizerProcessorStep`, features are typically preserved unchanged since normalization doesn't alter keys or shapes:
+
+```python
+def transform_features(self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+    """Normalization preserves all feature definitions."""
+    return features  # No changes to feature structure
+    # ...
+```
+
+When your processor renames or reshapes data, implement this method to reflect the mapping for downstream components. For example, a simple rename processor:
+
+```python
+def transform_features(self, features: dict[str, PolicyFeature]) -> dict[str, PolicyFeature]:
+    # Simple renaming
+    if "pixels" in features:
+        features["observation.image"] = features.pop("pixels")
+
+    # Pattern-based renaming
+    for key in list(features.keys()):
+        if key.startswith("env_state."):
+            suffix = key[len("env_state."):]
+            features[f"observation.{suffix}"] = features.pop(key)
+            # ...
+
+    return features
+```
+
+**Key principles:**
+
+- Use `features.pop(old_key)` to remove and get the old feature
+- Use `features[new_key] = old_feature` to add the renamed feature
+- Always return the modified features dictionary
+- Document transformations clearly in the docstring
+
+### Using overrides
+
+You can override step parameters at load-time using `overrides`. This is handy for non-serializable objects or site-specific settings. It works both in policy factories and with `DataProcessorPipeline.from_pretrained(...)`.
+
+**Foundational model adaptation**: This is particularly useful when working with foundational pretrained policies where you rarely have access to the original training statistics. You can inject your own dataset statistics to adapt the normalizer to your specific robot or environment data.
+
+Example: during policy evaluation on the robot, override the device and rename map.
+Use this to run a policy trained on CUDA on a CPU-only robot, or to remap camera keys when the robot uses different names than the dataset.
+
+Direct usage with `from_pretrained`:
+
+```python
+from lerobot.processor import RobotProcessorPipeline
+
+# Load a foundational policy trained on diverse robot data
+# but adapt normalization to your specific robot/environment
+new_stats = LeRobotDataset(repo_id="username/my-dataset").meta.stats
+processor = RobotProcessorPipeline.from_pretrained(
+    "huggingface/foundational-robot-policy",  # Pretrained foundation model
+    overrides={
+        "normalizer_processor": {"stats": new_stats},     # Inject your robot's statistics
+        "device_processor": {"device": "cuda:0"},         # registry name for registered steps
+        "rename_processor": {"rename_map": robot_key_map}, # Map your robot's observation keys
+        # ...
+    },
+)
+```
+
+## Best Practices
+
+Based on analysis of all LeRobot processor implementations, here are the key patterns and practices:
+
+### 1. **Safe Data Handling**
+
+Always create copies of input data to avoid unintended side effects. Use `transition.copy()` and `observation.copy()` rather than modifying data in-place. This prevents your processor from accidentally affecting other components in the pipeline.
+
+Check for required data before processing and handle missing data gracefully. If your processor expects certain keys (like `"pixels"` for image processing), validate their presence first. For optional data, use safe access patterns like `transition.get()` and handle `None` values appropriately.
+
+When data validation fails, provide clear, actionable error messages that help users understand what went wrong and how to fix it.
+
+### 2. **Choose Appropriate Base Classes**
+
+LeRobot provides specialized base classes that reduce boilerplate code and ensure consistency. Use `ObservationProcessorStep` when you only need to modify observations, `ActionProcessorStep` for action-only processing, and `RobotActionProcessorStep` specifically for dictionary-based robot actions.
+
+Only inherit directly from `ProcessorStep` when you need full control over the entire transition or when processing multiple transition components simultaneously. The specialized base classes handle the transition management for you and provide type safety.
+
+### 3. **Registration and Naming**
+
+Register your processors with descriptive, namespaced names using `@ProcessorStepRegistry.register()`. Use organization prefixes like `"robotics_lab/safety_clipper"` or `"acme_corp/vision_enhancer"` to avoid naming conflicts. Avoid generic names like `"processor"` or `"step"` that could clash with other implementations.
+
+Good registration makes your processors discoverable and enables clean serialization/deserialization when saving and loading pipelines.
+
+### 4. **State Management Patterns**
+
+Distinguish between configuration parameters (JSON-serializable values) and internal state (tensors, buffers). Use dataclass fields with `init=False, repr=False` for internal state that shouldn't appear in the constructor or string representation.
+
+Implement the `reset()` method to clear internal state between episodes. This is crucial for stateful processors that accumulate data over time, like moving averages or temporal filters.
+
+Remember that `get_config()` should only return JSON-serializable configuration, while `state_dict()` handles tensor state separately.
+
+### 5. **Input Validation and Error Handling**
+
+Validate input types and shapes before processing. Check tensor properties like `dtype` and dimensions to ensure compatibility with your algorithms. For robot actions, verify that required pose components or joint values are present and within expected ranges.
+
+Use early returns for edge cases where no processing is needed. Provide clear, descriptive error messages that include the expected vs. actual data types or shapes. This makes debugging much easier for users.
+
+### 6. **Device and Dtype Awareness**
+
+Design your processors to automatically adapt to the device and dtype of input tensors. Internal tensors (like normalization statistics) should match the input tensor's device and dtype to ensure compatibility with multi-GPU training, mixed precision, and distributed setups.
+
+Implement a `to()` method that moves your processor's internal state to the specified device. Check device/dtype compatibility at runtime and automatically migrate internal state when needed. This pattern enables seamless operation across different hardware configurations without manual intervention.
+
+## Conclusion
+
+You now have all the tools to implement custom processors in LeRobot! The key steps are:
+
+1. **Define your processor** as a dataclass with the required methods (`__call__`, `get_config`, `state_dict`, `load_state_dict`, `reset`, `transform_features`)
+2. **Register it** using `@ProcessorStepRegistry.register("name")` for discoverability
+3. **Integrate it** into a `DataProcessorPipeline` with other processing steps
+4. **Use base classes** like `ObservationProcessorStep` when possible to reduce boilerplate
+5. **Implement device/dtype awareness** to support multi-GPU and mixed precision setups
+
+The processor system is designed to be modular and composable, allowing you to build complex data processing pipelines from simple, focused components. Whether you're preprocessing sensor data for training or post-processing model outputs for robot execution, custom processors give you the flexibility to handle any data transformation your robotics application requires.
+
+Key principles for robust processors:
+
+- **Device/dtype adaptation**: Internal tensors should match input tensors
+- **Clear error messages**: Help users understand what went wrong
+- **Base class usage**: Leverage specialized base classes to reduce boilerplate
+- **Feature contracts**: Declare data structure changes with `transform_features()`
+
+Start simple, test thoroughly, and ensure your processors work seamlessly across different hardware configurations!
diff --git a/lerobot/docs/source/index.mdx b/lerobot/docs/source/index.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..a2f919e7d81ddaf3b5bb2b2175f964e17fb41eb9
--- /dev/null
+++ b/lerobot/docs/source/index.mdx
@@ -0,0 +1,23 @@
+<div class="flex justify-center">
+  <a target="_blank" href="https://huggingface.co/lerobot">
+    <img
+      alt="HuggingFace Expert Acceleration Program"
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/lerobot-logo-thumbnail.png"
+      style="width: 100%"
+    ></img>
+  </a>
+</div>
+
+# LeRobot
+
+**State-of-the-art machine learning for real-world robotics**
+
+🤗 LeRobot aims to provide models, datasets, and tools for real-world robotics in PyTorch. The goal is to lower the barrier for entry to robotics so that everyone can contribute and benefit from sharing datasets and pretrained models.
+
+🤗 LeRobot contains state-of-the-art approaches that have been shown to transfer to the real-world with a focus on imitation learning and reinforcement learning.
+
+🤗 LeRobot already provides a set of pretrained models, datasets with human collected demonstrations, and simulated environments so that everyone can get started.
+
+🤗 LeRobot hosts pretrained models and datasets on the LeRobot HuggingFace page.
+
+Join the LeRobot community on [Discord](https://discord.gg/s3KuuzsPFb)
diff --git a/lerobot/docs/source/installation.mdx b/lerobot/docs/source/installation.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..80f705e889d6d4160b0fa5132876d2badf73ef83
--- /dev/null
+++ b/lerobot/docs/source/installation.mdx
@@ -0,0 +1,183 @@
+# Installation
+
+This guide uses `conda` (via miniforge) to manage environments (recommended). If you prefer another environment manager (e.g. `uv`, `venv`), ensure you have Python >=3.12 and `ffmpeg` installed with the `libsvtav1` encoder, then skip ahead to [Environment Setup](#step-2-environment-setup).
+
+## Step 1 (`conda` only): Install [`miniforge`](https://conda-forge.org/download/)
+
+```bash
+wget "https://github.com/conda-forge/miniforge/releases/latest/download/Miniforge3-$(uname)-$(uname -m).sh"
+bash Miniforge3-$(uname)-$(uname -m).sh
+```
+
+## Step 2: Environment Setup
+
+Create a virtual environment with Python 3.12:
+
+<!-- prettier-ignore-start -->
+<hfoptions id="create_venv">
+<hfoption id="conda">
+```bash
+conda create -y -n lerobot python=3.12
+```
+</hfoption>
+<hfoption id="uv">
+```bash
+uv python install 3.12
+uv venv --python 3.12
+```
+</hfoption>
+</hfoptions>
+<!-- prettier-ignore-end -->
+
+Then activate your virtual environment, you have to do this each time you open a shell to use lerobot:
+
+<!-- prettier-ignore-start -->
+<hfoptions id="activate_venv">
+<hfoption id="conda">```bash
+conda activate lerobot
+```</hfoption>
+<hfoption id="uv">
+```bash
+# Linux/macOSsource
+source .venv/bin/activate
+# Windows PowerShell
+source .venv\Scripts\Activate.ps1
+```
+</hfoption>
+</hfoptions>
+<!-- prettier-ignore-end -->
+
+When using `conda`, install `ffmpeg` in your environment:
+
+```bash
+conda install ffmpeg -c conda-forge
+ffmpeg -version  # ffmpeg 8.X is not yet supported !
+```
+
+> [!TIP]
+> This usually installs `ffmpeg 7.X` for your platform compiled with the `libsvtav1` encoder. If `libsvtav1` is not supported (check supported encoders with `ffmpeg -encoders`), you can:
+>
+> - _[On any platform]_ Explicitly install `ffmpeg 7.X` using:
+>
+> ```bash
+> conda install ffmpeg=7.1.1 -c conda-forge
+> ```
+>
+> - _[On Linux only]_ If you want to bring your own ffmpeg: Install [ffmpeg build dependencies](https://trac.ffmpeg.org/wiki/CompilationGuide/Ubuntu#GettheDependencies) and [compile ffmpeg from source with libsvtav1](https://trac.ffmpeg.org/wiki/CompilationGuide/Ubuntu#libsvtav1), and make sure you use the corresponding ffmpeg binary to your install with `which ffmpeg`.
+
+> [!NOTE]
+> When installing LeRobot inside WSL (Windows Subsystem for Linux), make sure to install `evdev` with the following command:
+>
+> ```bash
+> conda install evdev -c conda-forge
+> ```
+
+> [!IMPORTANT]
+> If you are using `uv` you will have to install `ffmpeg` system-wide (outside of the virtual environment). You rely on `uv` and `torchcodec` ability to dynamically link to the system `ffmpeg`.
+
+## Step 3: Install LeRobot 🤗
+
+### From Source
+
+First, clone the repository and navigate into the directory:
+
+```bash
+git clone https://github.com/huggingface/lerobot.git
+cd lerobot
+```
+
+Then, install the library in editable mode. This is useful if you plan to contribute to the code.
+
+<!-- prettier-ignore-start -->
+<hfoptions id="install_lerobot_src">
+<hfoption id="conda">
+```bash
+pip install -e .
+```
+</hfoption>
+<hfoption id="uv">
+```bash
+uv pip install -e .
+```
+</hfoption>
+</hfoptions>
+<!-- prettier-ignore-end -->
+
+### Installation from PyPI
+
+**Core Library:**
+Install the base package with:
+
+<!-- prettier-ignore-start -->
+<hfoptions id="install_lerobot_pypi">
+<hfoption id="conda">
+```bash
+pip install lerobot
+```
+</hfoption>
+<hfoption id="uv">
+```bash
+uv pip install lerobot
+```
+</hfoption>
+</hfoptions>
+<!-- prettier-ignore-end -->
+
+_This installs only the default dependencies._
+
+**Extra Features:**
+To install additional functionality, use one of the following (If you are using `uv`, replace `pip install` with `uv pip install` in the commands below.):
+
+```bash
+pip install 'lerobot[all]'          # All available features
+pip install 'lerobot[aloha,pusht]'  # Specific features (Aloha & Pusht)
+pip install 'lerobot[feetech]'      # Feetech motor support
+```
+
+_Replace `[...]` with your desired features._
+
+**Available Tags:**
+For a full list of optional dependencies, see:
+https://pypi.org/project/lerobot/
+
+### Troubleshooting
+
+If you encounter build errors, you may need to install additional dependencies: `cmake`, `build-essential`, and `ffmpeg libs`.
+To install these for Linux run:
+
+```bash
+sudo apt-get install cmake build-essential python3-dev pkg-config libavformat-dev libavcodec-dev libavdevice-dev libavutil-dev libswscale-dev libswresample-dev libavfilter-dev
+```
+
+For other systems, see: [Compiling PyAV](https://pyav.org/docs/develop/overview/installation.html#bring-your-own-ffmpeg)
+
+## Optional dependencies
+
+LeRobot provides optional extras for specific functionalities. Multiple extras can be combined (e.g., `.[aloha,feetech]`). For all available extras, refer to `pyproject.toml`. If you are using `uv`, replace `pip install` with `uv pip install` in the commands below.
+
+### Simulations
+
+Install environment packages: `aloha` ([gym-aloha](https://github.com/huggingface/gym-aloha)), or `pusht` ([gym-pusht](https://github.com/huggingface/gym-pusht))
+Example:
+
+```bash
+pip install -e ".[aloha]" # or "[pusht]" for example
+```
+
+### Motor Control
+
+For Koch v1.1 install the Dynamixel SDK, for SO100/SO101/Moss install the Feetech SDK.
+
+```bash
+pip install -e ".[feetech]" # or "[dynamixel]" for example
+```
+
+### Experiment Tracking
+
+To use [Weights and Biases](https://docs.wandb.ai/quickstart) for experiment tracking, log in with
+
+```bash
+wandb login
+```
+
+You can now assemble your robot if it's not ready yet, look for your robot type on the left. Then follow the link below to use Lerobot with your robot.
diff --git a/lerobot/docs/source/integrate_hardware.mdx b/lerobot/docs/source/integrate_hardware.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..fa36e717082359f3a8c54a9701660fcdf9fbd0f7
--- /dev/null
+++ b/lerobot/docs/source/integrate_hardware.mdx
@@ -0,0 +1,476 @@
+# Bring Your Own Hardware
+
+This tutorial will explain how to integrate your own robot design into the LeRobot ecosystem and have it access all of our tools (data collection, control pipelines, policy training and inference).
+
+To that end, we provide the [`Robot`](https://github.com/huggingface/lerobot/blob/main/src/lerobot/robots/robot.py) base class in the LeRobot which specifies a standard interface for physical robot integration. Let's see how to implement it.
+
+## Prerequisites
+
+- Your own robot which exposes a communication interface (e.g. serial, CAN, TCP)
+- A way to read sensor data and send motor commands programmatically, e.g. manufacturer's SDK or API, or your own protocol implementation.
+- LeRobot installed in your environment. Follow our [Installation Guide](./installation).
+
+## Choose your motors
+
+If you're using Feetech or Dynamixel motors, LeRobot provides built-in bus interfaces:
+
+- [`FeetechMotorsBus`](https://github.com/huggingface/lerobot/blob/main/src/lerobot/motors/feetech/feetech.py) – for controlling Feetech servos
+- [`DynamixelMotorsBus`](https://github.com/huggingface/lerobot/blob/main/src/lerobot/motors/dynamixel/dynamixel.py) – for controlling Dynamixel servos
+
+Please refer to the [`MotorsBus`](https://github.com/huggingface/lerobot/blob/main/src/lerobot/motors/motors_bus.py) abstract class to learn about its API.
+For a good example of how it can be used, you can have a look at our own [SO101 follower implementation](https://github.com/huggingface/lerobot/blob/main/src/lerobot/robots/so_follower/so101_follower/so101_follower.py)
+
+Use these if compatible. Otherwise, you'll need to find or write a Python interface (not covered in this tutorial):
+
+- Find an existing SDK in Python (or use bindings to C/C++)
+- Or implement a basic communication wrapper (e.g., via pyserial, socket, or CANopen)
+
+You're not alone—many community contributions use custom boards or firmware!
+
+For Feetech and Dynamixel, we currently support these servos: - Feetech: - STS & SMS series (protocol 0): `sts3215`, `sts3250`, `sm8512bl` - SCS series (protocol 1): `scs0009` - Dynamixel (protocol 2.0 only): `xl330-m077`, `xl330-m288`, `xl430-w250`, `xm430-w350`, `xm540-w270`, `xc430-w150`
+
+If you are using Feetech or Dynamixel servos that are not in this list, you can add those in the [Feetech table](https://github.com/huggingface/lerobot/blob/main/src/lerobot/motors/feetech/tables.py) or [Dynamixel table](https://github.com/huggingface/lerobot/blob/main/src/lerobot/motors/dynamixel/tables.py). Depending on the model, this will require you to add model-specific information. In most cases though, there shouldn't be a lot of additions to do.
+
+In the next sections, we'll use a `FeetechMotorsBus` as the motors interface for the examples. Replace it and adapt to your motors if necessary.
+
+## Step 1: Subclass the `Robot` Interface
+
+You’ll first need to specify the config class and a string identifier (`name`) for your robot. If your robot has special needs that you'd like to be able to change easily, it should go here (e.g. port/address, baudrate).
+
+Here, we'll add the port name and one camera by default for our robot:
+
+<!-- prettier-ignore-start -->
+```python
+from dataclasses import dataclass, field
+
+from lerobot.cameras import CameraConfig
+from lerobot.cameras.opencv import OpenCVCameraConfig
+from lerobot.robots import RobotConfig
+
+
+@RobotConfig.register_subclass("my_cool_robot")
+@dataclass
+class MyCoolRobotConfig(RobotConfig):
+    port: str
+    cameras: dict[str, CameraConfig] = field(
+        default_factory={
+            "cam_1": OpenCVCameraConfig(
+                index_or_path=2,
+                fps=30,
+                width=480,
+                height=640,
+            ),
+        }
+    )
+```
+<!-- prettier-ignore-end -->
+
+[Cameras tutorial](./cameras) to understand how to detect and add your camera.
+
+Next, we'll create our actual robot class which inherits from `Robot`. This abstract class defines a contract you must follow for your robot to be usable with the rest of the LeRobot tools.
+
+Here we'll create a simple 5-DoF robot with one camera. It could be a simple arm but notice that the `Robot` abstract class does not assume anything on your robot's form factor. You can let you imagination run wild when designing new robots!
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.cameras import make_cameras_from_configs
+from lerobot.motors import Motor, MotorNormMode
+from lerobot.motors.feetech import FeetechMotorsBus
+from lerobot.robots import Robot
+
+class MyCoolRobot(Robot):
+    config_class = MyCoolRobotConfig
+    name = "my_cool_robot"
+
+    def __init__(self, config: MyCoolRobotConfig):
+        super().__init__(config)
+        self.bus = FeetechMotorsBus(
+            port=self.config.port,
+            motors={
+                "joint_1": Motor(1, "sts3250", MotorNormMode.RANGE_M100_100),
+                "joint_2": Motor(2, "sts3215", MotorNormMode.RANGE_M100_100),
+                "joint_3": Motor(3, "sts3215", MotorNormMode.RANGE_M100_100),
+                "joint_4": Motor(4, "sts3215", MotorNormMode.RANGE_M100_100),
+                "joint_5": Motor(5, "sts3215", MotorNormMode.RANGE_M100_100),
+            },
+            calibration=self.calibration,
+        )
+        self.cameras = make_cameras_from_configs(config.cameras)
+```
+<!-- prettier-ignore-end -->
+
+## Step 2: Define Observation and Action Features
+
+These two properties define the _interface contract_ between your robot and tools that consume it (such as data collection or learning pipelines).
+
+> [!WARNING]
+> Note that these properties must be callable even if the robot is not yet connected, so avoid relying on runtime hardware state to define them.
+
+### `observation_features`
+
+This property should return a dictionary describing the structure of sensor outputs from your robot. The keys match what `get_observation()` returns, and the values describe either the shape (for arrays/images) or the type (for simple values).
+
+Example for our 5-DoF arm with one camera:
+
+<!-- prettier-ignore-start -->
+```python
+@property
+def _motors_ft(self) -> dict[str, type]:
+    return {
+        "joint_1.pos": float,
+        "joint_2.pos": float,
+        "joint_3.pos": float,
+        "joint_4.pos": float,
+        "joint_5.pos": float,
+    }
+
+@property
+def _cameras_ft(self) -> dict[str, tuple]:
+    return {
+        cam: (self.cameras[cam].height, self.cameras[cam].width, 3) for cam in self.cameras
+    }
+
+@property
+def observation_features(self) -> dict:
+    return {**self._motors_ft, **self._cameras_ft}
+```
+<!-- prettier-ignore-end -->
+
+In this case, observations consist of a simple dict storing each motor's position and a camera image.
+
+### `action_features`
+
+This property describes the commands your robot expects via `send_action()`. Again, keys must match the expected input format, and values define the shape/type of each command.
+
+Here, we simply use the same joints proprioceptive features (`self._motors_ft`) as with `observation_features`: the action sent will simply the goal position for each motor.
+
+<!-- prettier-ignore-start -->
+```python
+def action_features(self) -> dict:
+    return self._motors_ft
+```
+<!-- prettier-ignore-end -->
+
+## Step 3: Handle Connection and Disconnection
+
+These methods should handle opening and closing communication with your hardware (e.g. serial ports, CAN interfaces, USB devices, cameras).
+
+### `is_connected`
+
+This property should simply reflect that communication with the robot's hardware is established. When this property is `True`, it should be possible to read and write to the hardware using `get_observation()` and `send_action()`.
+
+<!-- prettier-ignore-start -->
+```python
+@property
+def is_connected(self) -> bool:
+    return self.bus.is_connected and all(cam.is_connected for cam in self.cameras.values())
+```
+<!-- prettier-ignore-end -->
+
+### `connect()`
+
+This method should establish communication with the hardware. Moreover, if your robot needs calibration and is not calibrated, it should start a calibration procedure by default. If your robot needs some specific configuration, this should also be called here.
+
+<!-- prettier-ignore-start -->
+```python
+def connect(self, calibrate: bool = True) -> None:
+    self.bus.connect()
+    if not self.is_calibrated and calibrate:
+        self.calibrate()
+
+    for cam in self.cameras.values():
+        cam.connect()
+
+    self.configure()
+```
+<!-- prettier-ignore-end -->
+
+### `disconnect()`
+
+This method should gracefully terminate communication with the hardware: free any related resources (threads or processes), close ports, etc.
+
+Here, we already handle this in our `MotorsBus` and `Camera` classes so we just need to call their own `disconnect()` methods:
+
+<!-- prettier-ignore-start -->
+```python
+def disconnect(self) -> None:
+    self.bus.disconnect()
+    for cam in self.cameras.values():
+        cam.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+## Step 4: Support Calibration and Configuration
+
+LeRobot supports saving and loading calibration data automatically. This is useful for joint offsets, zero positions, or sensor alignment.
+
+> Note that depending on your hardware, this may not apply. If that's the case, you can simply leave these methods as no-ops:
+
+<!-- prettier-ignore-start -->
+```python
+@property
+def is_calibrated(self) -> bool:
+    return True
+
+def calibrate(self) -> None:
+    pass
+```
+<!-- prettier-ignore-end -->
+
+### `is_calibrated`
+
+This should reflect whether your robot has the required calibration loaded.
+
+<!-- prettier-ignore-start -->
+```python
+@property
+def is_calibrated(self) -> bool:
+    return self.bus.is_calibrated
+```
+<!-- prettier-ignore-end -->
+
+### `calibrate()`
+
+The goal of the calibration is twofold:
+
+- Know the physical range of motion of each motors in order to only send commands within this range.
+- Normalize raw motors positions to sensible continuous values (e.g. percentages, degrees) instead of arbitrary discrete value dependant on the specific motor used that will not replicate elsewhere.
+
+It should implement the logic for calibration (if relevant) and update the `self.calibration` dictionary. If you are using Feetech or Dynamixel motors, our bus interfaces already include methods to help with this.
+
+<!-- prettier-ignore-start -->
+```python
+def calibrate(self) -> None:
+    self.bus.disable_torque()
+    for motor in self.bus.motors:
+        self.bus.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+
+    input(f"Move {self} to the middle of its range of motion and press ENTER....")
+    homing_offsets = self.bus.set_half_turn_homings()
+
+    print(
+        "Move all joints sequentially through their entire ranges "
+        "of motion.\nRecording positions. Press ENTER to stop..."
+    )
+    range_mins, range_maxes = self.bus.record_ranges_of_motion()
+
+    self.calibration = {}
+    for motor, m in self.bus.motors.items():
+        self.calibration[motor] = MotorCalibration(
+            id=m.id,
+            drive_mode=0,
+            homing_offset=homing_offsets[motor],
+            range_min=range_mins[motor],
+            range_max=range_maxes[motor],
+        )
+
+    self.bus.write_calibration(self.calibration)
+    self._save_calibration()
+    print("Calibration saved to", self.calibration_fpath)
+```
+<!-- prettier-ignore-end -->
+
+### `configure()`
+
+Use this to set up any configuration for your hardware (servos control modes, controller gains, etc.). This should usually be run at connection time and be idempotent.
+
+<!-- prettier-ignore-start -->
+```python
+def configure(self) -> None:
+    with self.bus.torque_disabled():
+        self.bus.configure_motors()
+        for motor in self.bus.motors:
+            self.bus.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+            self.bus.write("P_Coefficient", motor, 16)
+            self.bus.write("I_Coefficient", motor, 0)
+            self.bus.write("D_Coefficient", motor, 32)
+```
+<!-- prettier-ignore-end -->
+
+## Step 5: Implement Sensors Reading and Action Sending
+
+These are the most important runtime functions: the core I/O loop.
+
+### `get_observation()`
+
+Returns a dictionary of sensor values from the robot. These typically include motor states, camera frames, various sensors, etc. In the LeRobot framework, these observations are what will be fed to a policy in order to predict the actions to take. The dictionary keys and structure must match `observation_features`.
+
+<!-- prettier-ignore-start -->
+```python
+def get_observation(self) -> dict[str, Any]:
+    if not self.is_connected:
+        raise ConnectionError(f"{self} is not connected.")
+
+    # Read arm position
+    obs_dict = self.bus.sync_read("Present_Position")
+    obs_dict = {f"{motor}.pos": val for motor, val in obs_dict.items()}
+
+    # Capture images from cameras
+    for cam_key, cam in self.cameras.items():
+        obs_dict[cam_key] = cam.async_read()
+
+    return obs_dict
+```
+<!-- prettier-ignore-end -->
+
+### `send_action()`
+
+Takes a dictionary that matches `action_features`, and sends it to your hardware. You can add safety limits (clipping, smoothing) and return what was actually sent.
+
+For simplicity, we won't be adding any modification of the actions in our example here.
+
+<!-- prettier-ignore-start -->
+```python
+def send_action(self, action: dict[str, Any]) -> dict[str, Any]:
+    goal_pos = {key.removesuffix(".pos"): val for key, val in action.items()}
+
+    # Send goal position to the arm
+    self.bus.sync_write("Goal_Position", goal_pos)
+
+    return action
+```
+<!-- prettier-ignore-end -->
+
+## Adding a Teleoperator
+
+For implementing teleoperation devices, we also provide a [`Teleoperator`](https://github.com/huggingface/lerobot/blob/main/src/lerobot/teleoperators/teleoperator.py) base class. This class is very similar to the `Robot` base class and also doesn't assume anything on form factor.
+
+The main differences are in the I/O functions: a teleoperator allows you to produce action via `get_action` and can receive feedback actions via `send_feedback`. Feedback could be anything controllable on the teleoperation device that could help the person controlling it understand the consequences of the actions sent. Think motion/force feedback on a leader arm, vibrations on a gamepad controller for example. To implement a teleoperator, you can follow this same tutorial and adapt it for these two methods.
+
+## Using Your Own `LeRobot` Devices 🔌
+
+You can easily extend `lerobot` with your own custom hardware—be it a camera, robot, or teleoperation device—by creating a separate, installable Python package. If you follow a few simple conventions, the `lerobot` command-line tools (like `lerobot-teleop` and `lerobot-record`) will **automatically discover and integrate your creations** without requiring any changes to the `lerobot` source code.
+
+This guide outlines the conventions your plugin must follow.
+
+### The 4 Core Conventions
+
+To ensure your custom device is discoverable, you must adhere to the following four rules.
+
+#### 1\. Create an Installable Package with a Specific Prefix
+
+Your project must be a standard, installable Python package. Crucially, the name of your package (as defined in `pyproject.toml` or `setup.py`) must begin with one of these prefixes:
+
+- `lerobot_robot_` for a robot.
+- `lerobot_camera_` for a camera.
+- `lerobot_teleoperator_` for a teleoperation device.
+
+This prefix system is how `lerobot` automatically finds your plugin in the Python environment.
+
+#### 2\. Follow the `SomethingConfig`/`Something` Naming Pattern
+
+Your device's implementation class must be named after its configuration class, simply by removing the `Config` suffix.
+
+- **Config Class:** `MyAwesomeTeleopConfig`
+- **Device Class:** `MyAwesomeTeleop`
+
+#### 3\. Place Your Files in a Predictable Structure
+
+The device class (`MyAwesomeTeleop`) must be located in a predictable module relative to its configuration class (`MyAwesomeTeleopConfig`). `lerobot` will automatically search in these locations:
+
+- In the **same module** as the config class.
+- In a **submodule named after the device** (e.g., `my_awesome_teleop.py`).
+
+The recommended and simplest structure is to place them in separate, clearly named files within the same directory.
+
+#### 4\. Expose Classes in `__init__.py`
+
+Your package's `__init__.py` file should import and expose both the configuration and the device classes, making them easily accessible.
+
+### Putting It All Together: A Complete Example
+
+Let's create a new teleoperator called `my_awesome_teleop`.
+
+#### Directory Structure
+
+Here is what the project folder should look like. The package name, `lerobot_teleoperator_my_awesome_teleop`, follows **Convention \#1**.
+
+```
+lerobot_teleoperator_my_awesome_teleop/
+├── pyproject.toml # (or setup.py) lists lerobot as a dependency
+└── lerobot_teleoperator_my_awesome_teleop/
+    ├── __init__.py
+    ├── config_my_awesome_teleop.py
+    └── my_awesome_teleop.py
+```
+
+#### File Contents
+
+- **`config_my_awesome_teleop.py`**: Defines the configuration class. Note the `Config` suffix (**Convention \#2**).
+
+  ```python
+  from dataclasses import dataclass
+
+  from lerobot.teleoperators.config import TeleoperatorConfig
+
+  @TeleoperatorConfig.register_subclass("my_awesome_teleop")
+  @dataclass
+  class MyAwesomeTeleopConfig(TeleoperatorConfig):
+      # Your configuration fields go here
+      port: str = "192.168.1.1"
+  ```
+
+- **`my_awesome_teleop.py`**: Implements the device. The class name `MyAwesomeTeleop` matches its config class name (**Convention \#2**). This file structure adheres to **Convention \#3**.
+
+  ```python
+  from lerobot.teleoperators.teleoperator import Teleoperator
+
+  from .config_my_awesome_teleop import MyAwesomeTeleopConfig
+
+  class MyAwesomeTeleop(Teleoperator):
+      config_class = MyAwesomeTeleopConfig
+      name = "my_awesome_teleop"
+
+      def __init__(self, config: MyAwesomeTeleopConfig):
+          super().__init__(config)
+          self.config = config
+
+      # Your device logic (e.g., connect) goes here
+  ```
+
+- **`__init__.py`**: Exposes the key classes (**Convention \#4**).
+
+  ```python
+  from .config_my_awesome_teleop import MyAwesomeTeleopConfig
+  from .my_awesome_teleop import MyAwesomeTeleop
+  ```
+
+### Installation and Usage
+
+1.  **Install your new plugin in your Python environment.** You can install your local plugin package using `pip`'s editable mode or from PyPi.
+
+    ```bash
+    # Locally
+    # Navigate to your plugin's root directory and install it
+    cd lerobot_teleoperator_my_awesome_teleop
+    pip install -e .
+
+    # From PyPi
+    pip install lerobot_teleoperator_my_awesome_teleop
+    ```
+
+2.  **Use it directly from the command line.** Now, you can use your custom device by referencing its type.
+
+    ```bash
+    lerobot-teleoperate --teleop.type=my_awesome_teleop \
+    # other arguments
+    ```
+
+And that's it\! Your custom device is now fully integrated.
+
+### Looking for an example ?
+
+Check out these two packages from the community:
+
+- https://github.com/SpesRobotics/lerobot-robot-xarm
+- https://github.com/SpesRobotics/lerobot-teleoperator-teleop
+
+## Wrapping Up
+
+Once your robot class is complete, you can leverage the LeRobot ecosystem:
+
+- Control your robot with available teleoperators or integrate directly your teleoperating device
+- Record training data and visualize it
+- Integrate it into RL or imitation learning pipelines
+
+Don't hesitate to reach out to the community for help on our [Discord](https://discord.gg/s3KuuzsPFb) 🤗
diff --git a/lerobot/docs/source/introduction_processors.mdx b/lerobot/docs/source/introduction_processors.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..6f3768615419c29c050c77ac8717df9383b745c5
--- /dev/null
+++ b/lerobot/docs/source/introduction_processors.mdx
@@ -0,0 +1,314 @@
+# Introduction to Processors
+
+In robotics, there's a fundamental mismatch between the data that robots and humans produce and what machine learning models expect.
+Robots output raw sensor data like camera images and joint positions that need normalization, batching, and device placement before models can process them.
+Language instructions from humans must be tokenized into numerical representations, and different robots use different coordinate systems that need standardization.
+
+The challenge extends to model outputs as well.
+Models might output end-effector positions while robots need joint-space commands, or teleoperators produce relative movements while robots expect absolute commands.
+Model predictions are often normalized and need conversion back to real-world scales.
+
+Cross-domain translation adds another layer of complexity.
+Training data from one robot setup needs adaptation for deployment on different hardware, models trained with specific camera configurations must work with new arrangements, and datasets with different naming conventions need harmonization.
+
+**That's where processors come in.** They serve as universal translators that bridge these gaps, ensuring seamless data flow from sensors to models to actuators.
+Processors handle all the preprocessing and postprocessing steps needed to convert raw environment data into model-ready inputs and vice versa.
+
+This means that your favorite policy can be used like this:
+
+```python
+import torch
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.policies.your_policy import YourPolicy
+from lerobot.processor.pipeline import RobotProcessorPipeline, PolicyProcessorPipeline
+dataset = LeRobotDataset("hf_user/dataset", episodes=[0])
+sample = dataset[10]
+
+model = YourPolicy.from_pretrained(
+    "hf_user/model",
+)
+model.eval()
+model.to("cuda")
+preprocessor, postprocessor = make_pre_post_processors(model.config, pretrained_path="hf_user/model", dataset_stats=dataset.meta.stats)
+
+preprocessed_sample = preprocessor(sample)
+action = model.select_action(preprocessed_sample)
+postprocessed_action = postprocessor(action)
+```
+
+## What are Processors?
+
+In robotics, data comes in many forms: images from cameras, joint positions from sensors, text instructions from users, and more. Each type of data requires specific transformations before a model can use it effectively. Models need this data to be:
+
+- **Normalized**: Scaled to appropriate ranges for neural network processing
+- **Batched**: Organized with proper dimensions for batch processing
+- **Tokenized**: Text converted to numerical representations
+- **Device-placed**: Moved to the right hardware (CPU/GPU)
+- **Type-converted**: Cast to appropriate data types
+
+Processors handle these transformations through composable, reusable steps that can be chained together into pipelines. Think of them as a modular assembly line where each station performs a specific transformation on your data.
+
+## Core Concepts
+
+### EnvTransition: The Universal Data Container
+
+The `EnvTransition` is the fundamental data structure that flows through all processors.
+It's a typed dictionary that represents a complete robot-environment interaction:
+
+- **OBSERVATION**: All sensor data (images, states, proprioception)
+- **ACTION**: The action to execute or that was executed
+- **REWARD**: Reinforcement learning signal
+- **DONE/TRUNCATED**: Episode boundary indicators
+- **INFO**: Arbitrary metadata
+- **COMPLEMENTARY_DATA**: Task descriptions, indices, padding flags, inter-step data
+
+### ProcessorStep: The Building Block
+
+A `ProcessorStep` is a single transformation unit that processes transitions. It's an abstract base class with two required methods:
+
+```python
+from lerobot.processor import ProcessorStep, EnvTransition
+
+class MyProcessorStep(ProcessorStep):
+    """Example processor step - inherit and implement abstract methods."""
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Transform the transition - REQUIRED abstract method."""
+        # Your processing logic here
+        return transition
+
+    def transform_features(self, features):
+        """Declare how this step transforms feature shapes/types - REQUIRED abstract method."""
+        return features  # Most processors return features unchanged
+```
+
+`__call__` is the core of your processor step. It takes an `EnvTransition` and returns a modified `EnvTransition`.
+
+`transform_features` is used to declare how this step transforms feature shapes/types.
+
+### DataProcessorPipeline: The Generic Orchestrator
+
+The `DataProcessorPipeline[TInput, TOutput]` chains multiple `ProcessorStep` instances together:
+
+```python
+from lerobot.processor import RobotProcessorPipeline, PolicyProcessorPipeline
+
+# For robot hardware (unbatched data)
+robot_processor = RobotProcessorPipeline[RobotAction, RobotAction](
+    steps=[step1, step2, step3],
+    name="robot_pipeline"
+)
+
+# For model training/inference (batched data)
+policy_processor = PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+    steps=[step1, step2, step3],
+    name="policy_pipeline"
+)
+```
+
+## RobotProcessorPipeline vs PolicyProcessorPipeline
+
+The key distinction is in the data structures they handle:
+
+| Aspect          | RobotProcessorPipeline                       | PolicyProcessorPipeline                  |
+| --------------- | -------------------------------------------- | ---------------------------------------- |
+| **Input**       | `dict[str, Any]` - Individual robot values   | `dict[str, Any]` - Batched tensors       |
+| **Output**      | `dict[str, Any]` - Individual robot commands | `torch.Tensor` - Policy predictions      |
+| **Use Case**    | Real-time robot control                      | Model training/inference                 |
+| **Data Format** | Unbatched, heterogeneous                     | Batched, homogeneous                     |
+| **Examples**    | `{"joint_1": 0.5}`                           | `{"observation.state": tensor([[0.5]])}` |
+
+**Use `RobotProcessorPipeline`** for robot hardware interfaces:
+
+```python
+# Robot data structures: dict[str, Any] for observations and actions
+robot_obs: dict[str, Any] = {
+    "joint_1": 0.5,           # Individual joint values
+    "joint_2": -0.3,
+    "camera_0": image_array   # Raw camera data
+}
+
+robot_action: dict[str, Any] = {
+    "joint_1": 0.2,          # Target joint positions
+    "joint_2": 0.1,
+    "gripper": 0.8
+}
+```
+
+**Use `PolicyProcessorPipeline`** for model training and batch processing:
+
+```python
+# Policy data structures: batch dicts and tensors
+policy_batch: dict[str, Any] = {
+    "observation.state": torch.tensor([[0.5, -0.3]]),      # Batched states
+    "observation.images.camera0": torch.tensor(...),        # Batched images
+    "action": torch.tensor([[0.2, 0.1, 0.8]])              # Batched actions
+}
+
+policy_action: torch.Tensor = torch.tensor([[0.2, 0.1, 0.8]])  # Model output tensor
+```
+
+## Converter Functions
+
+LeRobot provides converter functions to bridge different data formats in `lerobot.processor.converters`. These functions handle the crucial translations between robot hardware data structures, policy model formats, and the internal `EnvTransition` representation that flows through processor pipelines.
+
+| Category                       | Function                      | Description                     |
+| ------------------------------ | ----------------------------- | ------------------------------- |
+| **Robot Hardware Converters**  | `robot_action_to_transition`  | Robot dict → EnvTransition      |
+|                                | `observation_to_transition`   | Robot obs → EnvTransition       |
+|                                | `transition_to_robot_action`  | EnvTransition → Robot dict      |
+| **Policy/Training Converters** | `batch_to_transition`         | Batch dict → EnvTransition      |
+|                                | `transition_to_batch`         | EnvTransition → Batch dict      |
+|                                | `policy_action_to_transition` | Policy tensor → EnvTransition   |
+|                                | `transition_to_policy_action` | EnvTransition → Policy tensor   |
+| **Utilities**                  | `create_transition`           | Build transitions with defaults |
+|                                | `identity_transition`         | Pass-through converter          |
+
+The key insight is that **robot hardware converters** work with individual values and dictionaries, while **policy/training converters** work with batched tensors and model outputs. The converter functions automatically handle the structural differences, so your processor steps can focus on the core transformations without worrying about data format compatibility.
+
+## Processor Examples
+
+The following examples demonstrate real-world processor configurations for policy training and inference.
+
+Here is an example processor for policy training and inference:
+
+```python
+# Training data preprocessing (optimized order for GPU performance)
+training_preprocessor = PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+    steps=[
+        RenameObservationsProcessorStep(rename_map={}),     # Standardize keys
+        AddBatchDimensionProcessorStep(),                   # Add batch dims
+        TokenizerProcessorStep(tokenizer_name="...", ...),  # Tokenize language
+        DeviceProcessorStep(device="cuda"),                 # Move to GPU first
+        NormalizerProcessorStep(features=..., stats=...),   # Normalize on GPU
+    ]
+)
+
+# Model output postprocessing
+training_postprocessor = PolicyProcessorPipeline[torch.Tensor, torch.Tensor](
+    steps=[
+        DeviceProcessorStep(device="cpu"),                  # Move to CPU
+        UnnormalizerProcessorStep(features=..., stats=...), # Denormalize
+    ]
+    to_transition=policy_action_to_transition,
+    to_output=transition_to_policy_action,
+)
+```
+
+### An interaction between a robot and a policy with processors
+
+The most common real-world scenario combines both pipeline types robot hardware generates observations that need policy processing, and policy outputs need robot-compatible postprocessing:
+
+```python
+# Real deployment: Robot sensors → Model → Robot commands
+with torch.no_grad():
+    while not done:
+        raw_obs = robot.get_observation()  # dict[str, Any]
+
+        # Add your robot observation to policy observation processor
+
+        policy_input = policy_preprocessor(raw_obs)  # Batched dict
+
+        policy_output = policy.select_action(policy_input)  # Policy tensor
+
+        policy_action = policy_postprocessor(policy_output)
+
+        # Add your robot action to policy action processor
+
+        robot.send_action(policy_action)
+```
+
+## Feature Contracts: Shape and Type Transformation
+
+Processors don't just transform data - they can also **change the data structure itself**. The `transform_features()` method declares these changes, which is crucial for dataset recording and policy creation.
+
+### Why Feature Contracts Matter
+
+When building datasets or policies, LeRobot needs to know:
+
+- **What data fields will exist** after processing
+- **What shapes and types** each field will have
+- **How to configure models** for the expected data structure
+
+```python
+# Example: A processor that adds velocity to observations
+class VelocityProcessor(ObservationProcessorStep):
+    def observation(self, obs):
+        new_obs = obs.copy()
+        if "observation.state" in obs:
+            # concatenate computed velocity field to the state
+            new_obs["observation.state"] = self._compute_velocity(obs["observation.state"])
+        return new_obs
+
+    def transform_features(self, features):
+        """Declare the new velocity field we're adding."""
+        state_feature = features[PipelineFeatureType.OBSERVATION].get("observation.state")
+        if state_feature:
+            double_shape = (state_feature.shape[0] * 2,) if state_feature.shape else (2,)
+            features[PipelineFeatureType.OBSERVATION]["observation.state"] = PolicyFeature(
+                type=FeatureType.STATE, shape=double_shape
+            )
+        return features
+```
+
+### Feature Specification Functions
+
+`create_initial_features()` and `aggregate_pipeline_dataset_features()` solve a critical dataset creation problem: determining the exact final data structure before any data is processed.
+Since processor pipelines can add new features (like velocity fields), change tensor shapes (like cropping images), or rename keys, datasets need to know the complete output specification upfront to allocate proper storage and define schemas.
+These functions work together by starting with robot hardware specifications (`create_initial_features()`) then simulating the entire pipeline transformation (`aggregate_pipeline_dataset_features()`) to compute the final feature dictionary that gets passed to `LeRobotDataset.create()`, ensuring perfect alignment between what processors output and what datasets expect to store.
+
+```python
+from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features
+
+# Start with robot's raw features
+initial_features = create_initial_features(
+    observation=robot.observation_features,  # {"joint_1.pos": float, "camera_0": (480,640,3)}
+    action=robot.action_features            # {"joint_1.pos": float, "gripper.pos": float}
+)
+
+# Apply processor pipeline to compute final features
+final_features = aggregate_pipeline_dataset_features(
+    pipeline=my_processor_pipeline,
+    initial_features=initial_features,
+    use_videos=True
+)
+
+# Use for dataset creation
+dataset = LeRobotDataset.create(
+    repo_id="my_dataset",
+    features=final_features,  # Knows exactly what data to expect
+    ...
+)
+```
+
+## Common Processor Steps
+
+LeRobot provides many registered processor steps. Here are the most commonly used core processors:
+
+### Essential Processors
+
+- **`normalizer_processor`**: Normalize observations/actions using dataset statistics (mean/std or min/max)
+- **`device_processor`**: Move tensors to CPU/GPU with optional dtype conversion
+- **`to_batch_processor`**: Add batch dimensions to transitions for model compatibility
+- **`rename_observations_processor`**: Rename observation keys using mapping dictionaries
+- **`tokenizer_processor`**: Tokenize natural language task descriptions into tokens and attention masks
+
+### Next Steps
+
+- **[Implement Your Own Processor](./implement_your_own_processor)** - Create custom processor steps
+- **[Debug Your Pipeline](./debug_processor_pipeline)** - Troubleshoot and optimize pipelines
+- **[Processors for Robots and Teleoperators](./processors_robots_teleop)** - Real-world integration patterns
+
+## Summary
+
+Processors solve the data translation problem in robotics by providing:
+
+- **Modular transformations**: Composable, reusable processing steps
+- **Type safety**: Generic pipelines with compile-time checking
+- **Performance optimization**: GPU-accelerated operations
+- **Robot/Policy distinction**: Separate pipelines for different data structures
+- **Comprehensive ecosystem**: 30+ registered processors for common tasks
+
+The key insight: `RobotProcessorPipeline` handles unbatched robot hardware data, while `PolicyProcessorPipeline` handles batched model data. Choose the right tool for your data structure!
diff --git a/lerobot/docs/source/koch.mdx b/lerobot/docs/source/koch.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..813b9bd67d5ede9b6c400daee56c81ae1770f154
--- /dev/null
+++ b/lerobot/docs/source/koch.mdx
@@ -0,0 +1,283 @@
+# Koch v1.1
+
+In the steps below, we explain how to assemble the Koch v1.1 robot.
+
+## Order and assemble the parts
+
+Follow the sourcing and assembling instructions provided in this [README](https://github.com/jess-moss/koch-v1-1). This will guide you through setting up both the follower and leader arms, as shown in the image below.
+
+For a visual walkthrough of the assembly process, you can refer to [this video tutorial](https://youtu.be/8nQIg9BwwTk).
+
+> [!WARNING]
+> Since the production of this video, we simplified the configuration phase. Because of this, two things differ from the instructions in that video:
+>
+> - Don't plug in all the motor cables right away and wait to be instructed to do so in [Configure the motors](#configure-the-motors).
+> - Don't screw in the controller board (PCB) to the base right away and wait for being instructed to do so in [Configure the motors](#configure-the-motors).
+
+## Install LeRobot 🤗
+
+To install LeRobot follow, our [Installation Guide](./installation)
+
+In addition to these instructions, you need to install the Dynamixel SDK:
+
+```bash
+pip install -e ".[dynamixel]"
+```
+
+## Configure the motors
+
+### 1. Find the USB ports associated with each arm
+
+To find the port for each bus servo adapter, run this script:
+
+```bash
+lerobot-find-port
+```
+
+<hfoptions id="example">
+<hfoption id="Mac">
+
+Example output:
+
+```
+Finding all available ports for the MotorBus.
+['/dev/tty.usbmodem575E0032081', '/dev/tty.usbmodem575E0031751']
+Remove the USB cable from your MotorsBus and press Enter when done.
+
+[...Disconnect corresponding leader or follower arm and press Enter...]
+
+The port of this MotorsBus is /dev/tty.usbmodem575E0032081
+Reconnect the USB cable.
+```
+
+Where the found port is: `/dev/tty.usbmodem575E0032081` corresponding to your leader or follower arm.
+
+</hfoption>
+<hfoption id="Linux">
+
+On Linux, you might need to give access to the USB ports by running:
+
+```bash
+sudo chmod 666 /dev/ttyACM0
+sudo chmod 666 /dev/ttyACM1
+```
+
+Example output:
+
+```
+Finding all available ports for the MotorBus.
+['/dev/ttyACM0', '/dev/ttyACM1']
+Remove the usb cable from your MotorsBus and press Enter when done.
+
+[...Disconnect corresponding leader or follower arm and press Enter...]
+
+The port of this MotorsBus is /dev/ttyACM1
+Reconnect the USB cable.
+```
+
+Where the found port is: `/dev/ttyACM1` corresponding to your leader or follower arm.
+
+</hfoption>
+</hfoptions>
+
+### 2. Set the motors ids and baudrates
+
+Each motor is identified by a unique id on the bus. When brand new, motors usually come with a default id of `1`. For the communication to work properly between the motors and the controller, we first need to set a unique, different id to each motor. Additionally, the speed at which data is transmitted on the bus is determined by the baudrate. In order to talk to each other, the controller and all the motors need to be configured with the same baudrate.
+
+To that end, we first need to connect to each motor individually with the controller in order to set these. Since we will write these parameters in the non-volatile section of the motors' internal memory (EEPROM), we'll only need to do this once.
+
+If you are repurposing motors from another robot, you will probably also need to perform this step, as the ids and baudrate likely won't match.
+
+#### Follower
+
+Connect the usb cable from your computer and the 5V power supply to the follower arm's controller board. Then, run the following command or run the API example with the port you got from the previous step. You'll also need to give your leader arm a name with the `id` parameter.
+
+For a visual reference on how to set the motor ids please refer to [this video](https://huggingface.co/docs/lerobot/en/so101#setup-motors-video) where we follow the process for the SO101 arm.
+
+<hfoptions id="setup_motors">
+<hfoption id="Command">
+
+```bash
+lerobot-setup-motors \
+    --robot.type=koch_follower \
+    --robot.port=/dev/tty.usbmodem575E0031751 # <- paste here the port found at previous step
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.robots.koch_follower import KochFollower, KochFollowerConfig
+
+config = KochFollowerConfig(
+    port="/dev/tty.usbmodem575E0031751",
+    id="my_awesome_follower_arm",
+)
+follower = KochFollower(config)
+follower.setup_motors()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+You should see the following instruction.
+
+```
+Connect the controller board to the 'gripper' motor only and press enter.
+```
+
+As instructed, plug the gripper's motor. Make sure it's the only motor connected to the board, and that the motor itself is not yet daisy-chained to any other motor. As you press `[Enter]`, the script will automatically set the id and baudrate for that motor.
+
+<details>
+<summary>Troubleshooting</summary>
+
+If you get an error at that point, check your cables and make sure they are plugged in properly:
+
+<ul>
+  <li>Power supply</li>
+  <li>USB cable between your computer and the controller board</li>
+  <li>The 3-pin cable from the controller board to the motor</li>
+</ul>
+
+If you are using a Waveshare controller board, make sure that the two jumpers are set on the `B` channel (USB).
+
+</details>
+
+You should then see the following message:
+
+```
+'gripper' motor id set to 6
+```
+
+Followed by the next instruction:
+
+```
+Connect the controller board to the 'wrist_roll' motor only and press enter.
+```
+
+You can disconnect the 3-pin cable from the controller board but you can leave it connected to the gripper motor on the other end as it will already be in the right place. Now, plug in another 3-pin cable to the wrist roll motor and connect it to the controller board. As with the previous motor, make sure it is the only motor connected to the board and that the motor itself isn't connected to any other one.
+
+Repeat the operation for each motor as instructed.
+
+> [!TIP]
+> Check your cabling at each step before pressing Enter. For instance, the power supply cable might disconnect as you manipulate the board.
+
+When you are done, the script will simply finish, at which point the motors are ready to be used. You can now plug the 3-pin cable from each motor to the next one, and the cable from the first motor (the 'shoulder pan' with id=1) to the controller board, which can now be attached to the base of the arm.
+
+#### Leader
+
+Do the same steps for the leader arm but modify the command or script accordingly.
+
+<hfoptions id="setup_motors">
+<hfoption id="Command">
+
+```bash
+lerobot-setup-motors \
+    --teleop.type=koch_leader \
+    --teleop.port=/dev/tty.usbmodem575E0031751 \  # <- paste here the port found at previous step
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.teleoperators.koch_leader import KochLeader, KochLeaderConfig
+
+config = KochLeaderConfig(
+    port="/dev/tty.usbmodem575E0031751",
+    id="my_awesome_leader_arm",
+)
+leader = KochLeader(config)
+leader.setup_motors()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+## Calibrate
+
+Next, you'll need to calibrate your robot to ensure that the leader and follower arms have the same position values when they are in the same physical position.
+The calibration process is very important because it allows a neural network trained on one robot to work on another.
+
+#### Follower
+
+Run the following command or API example to calibrate the follower arm:
+
+<hfoptions id="calibrate_follower">
+<hfoption id="Command">
+
+```bash
+lerobot-calibrate \
+    --robot.type=koch_follower \
+    --robot.port=/dev/tty.usbmodem58760431551 \ # <- The port of your robot
+    --robot.id=my_awesome_follower_arm  # <- Give the robot a unique name
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.robots.koch_follower import KochFollowerConfig, KochFollower
+
+config = KochFollowerConfig(
+    port="/dev/tty.usbmodem585A0076891",
+    id="my_awesome_follower_arm",
+)
+
+follower = KochFollower(config)
+follower.connect(calibrate=False)
+follower.calibrate()
+follower.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+We unified the calibration method for most robots. Thus, the calibration steps for this Koch arm are the same as the steps for the SO100 and SO101. First, we have to move the robot to the position where each joint is in the middle of its range, then we press `Enter`. Secondly, we move all joints through their full range of motion. A video of this same process for the SO101 as reference can be found [here](https://huggingface.co/docs/lerobot/en/so101#calibration-video).
+
+#### Leader
+
+Do the same steps to calibrate the leader arm, run the following command or API example:
+
+<hfoptions id="calibrate_leader">
+<hfoption id="Command">
+
+```bash
+lerobot-calibrate \
+    --teleop.type=koch_leader \
+    --teleop.port=/dev/tty.usbmodem58760431551 \ # <- The port of your robot
+    --teleop.id=my_awesome_leader_arm  # <- Give the robot a unique name
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.teleoperators.koch_leader import KochLeaderConfig, KochLeader
+
+config = KochLeaderConfig(
+    port="/dev/tty.usbmodem575E0031751",
+    id="my_awesome_leader_arm",
+)
+
+leader = KochLeader(config)
+leader.connect(calibrate=False)
+leader.calibrate()
+leader.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+Congrats 🎉, your robot is all set to learn a task on its own. Start training it by following this tutorial: [Getting started with real-world robots](./il_robots)
+
+> [!TIP]
+> If you have any questions or need help, please reach out on [Discord](https://discord.com/invite/s3KuuzsPFb).
diff --git a/lerobot/docs/source/lekiwi.mdx b/lerobot/docs/source/lekiwi.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..7e7c1a6804ff2c5b47338385afc85c2aeb928f66
--- /dev/null
+++ b/lerobot/docs/source/lekiwi.mdx
@@ -0,0 +1,343 @@
+# LeKiwi
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/1740517739083.jpeg"
+  alt="LeKiwi"
+  width="70%"
+/>
+
+In the steps below, we explain how to assemble the LeKiwi mobile robot.
+
+## Source the parts
+
+Follow this [README](https://github.com/SIGRobotics-UIUC/LeKiwi). It contains the bill of materials, with a link to source the parts, as well as the instructions to 3D print the parts.
+And advise if it's your first time printing or if you don't own a 3D printer.
+
+### Wired version
+
+If you have the **wired** LeKiwi version, you can skip the installation of the Raspberry Pi and setting up SSH. You can also run all commands directly on your PC for both the LeKiwi scripts and the leader arm scripts for teleoperating.
+
+## Install software on Pi
+
+Now we have to set up the remote PC that will run on the LeKiwi Robot. This is normally a Raspberry Pi, but can be any PC that can run on 5V and has enough usb ports (2 or more) for the cameras and motor control board.
+
+### Install OS
+
+For setting up the Raspberry Pi and its SD-card see: [Setup PI](https://www.raspberrypi.com/documentation/computers/getting-started.html). Here is explained how to download the [Imager](https://www.raspberrypi.com/software/) to install Raspberry Pi OS or Ubuntu.
+
+### Setup SSH
+
+After setting up your Pi, you should enable and set up [SSH](https://www.raspberrypi.com/news/coding-on-raspberry-pi-remotely-with-visual-studio-code/) (Secure Shell Protocol) so you can log in to the Pi from your laptop without requiring a screen, keyboard, and mouse on the Pi. A great tutorial on how to do this can be found [here](https://www.raspberrypi.com/documentation/computers/remote-access.html#ssh). Logging into your Pi can be done in your Command Prompt (cmd) or, if you use VSCode you can use [this](https://marketplace.visualstudio.com/items?itemName=ms-vscode-remote.remote-ssh) extension.
+
+### Install LeRobot on Pi 🤗
+
+On your Raspberry Pi install LeRobot using our [Installation Guide](./installation)
+
+In addition to these instructions, you need to install the Feetech SDK & ZeroMQ on your Pi:
+
+```bash
+pip install -e ".[lekiwi]"
+```
+
+## Install LeRobot locally
+
+If you already have installed LeRobot on your laptop/pc you can skip this step; otherwise, please follow along as we do the same steps we did on the Pi.
+
+Follow our [Installation Guide](./installation)
+
+In addition to these instructions, you need to install the Feetech SDK & ZeroMQ on your laptop/pc:
+
+```bash
+pip install -e ".[lekiwi]"
+```
+
+Great :hugs:! You are now done installing LeRobot, and we can begin assembling the SO100/SO101 arms and the mobile base :robot:.
+Every time you now want to use LeRobot, you can go to the `~/lerobot` folder where we installed LeRobot and run one of the commands.
+
+# Step-by-Step Assembly Instructions
+
+First, we will assemble the two SO100/SO101 arms. One to attach to the mobile base and one for teleoperation. Then we will assemble the mobile base. The instructions for assembling can be found on these two pages:
+
+- [Assemble SO101](./so101#step-by-step-assembly-instructions)
+- [Assemble LeKiwi](https://github.com/SIGRobotics-UIUC/LeKiwi/blob/main/Assembly.md)
+
+### Find the USB ports associated with motor board
+
+To find the port for each bus servo adapter, run this script:
+
+```bash
+lerobot-find-port
+```
+
+<hfoptions id="example">
+<hfoption id="Mac">
+
+Example output:
+
+```
+Finding all available ports for the MotorBus.
+['/dev/tty.usbmodem575E0032081']
+Remove the USB cable from your MotorsBus and press Enter when done.
+
+[...Disconnect corresponding leader or follower arm and press Enter...]
+
+The port of this MotorsBus is /dev/tty.usbmodem575E0032081
+Reconnect the USB cable.
+```
+
+Where the found port is: `/dev/tty.usbmodem575E0032081` corresponding to your board.
+
+</hfoption>
+<hfoption id="Linux">
+
+On Linux, you might need to give access to the USB ports by running:
+
+```bash
+sudo chmod 666 /dev/ttyACM0
+sudo chmod 666 /dev/ttyACM1
+```
+
+Example output:
+
+```
+Finding all available ports for the MotorBus.
+['/dev/ttyACM0']
+Remove the usb cable from your MotorsBus and press Enter when done.
+
+[...Disconnect corresponding leader or follower arm and press Enter...]
+
+The port of this MotorsBus is /dev/ttyACM0
+Reconnect the USB cable.
+```
+
+Where the found port is: `/dev/ttyACM0` corresponding to your board.
+
+</hfoption>
+</hfoptions>
+
+### Configure motors
+
+The instructions for configuring the motors can be found in the SO101 [docs](./so101#configure-the-motors). Besides the ids for the arm motors, we also need to set the motor ids for the mobile base. These need to be in a specific order to work. Below an image of the motor ids and motor mounting positions for the mobile base. Note that we only use one Motor Control board on LeKiwi. This means the motor ids for the wheels are 7, 8 and 9.
+
+You can run this command to setup motors for LeKiwi. It will first setup the motors for arm (id 6..1) and then setup motors for wheels (9,8,7)
+
+```bash
+lerobot-setup-motors \
+    --robot.type=lekiwi \
+    --robot.port=/dev/tty.usbmodem58760431551 # <- paste here the port found at previous step
+```
+
+<img src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/motor_ids.webp" alt="Motor ID's for mobile robot" title="Motor ID's for mobile robot" width="60%">
+
+### Troubleshoot communication
+
+If you are having trouble connecting to the Mobile SO100, follow these steps to diagnose and resolve the issue.
+
+#### 1. Verify IP Address Configuration
+
+Make sure that the correct IP for the Pi is used in the commands or in your code. To check the Raspberry Pi's IP address, run (on the Pi command line):
+
+```bash
+hostname -I
+```
+
+#### 2. Check if Pi is reachable from laptop/pc
+
+Try pinging the Raspberry Pi from your laptop:
+
+```bach
+ping <your_pi_ip_address>
+```
+
+If the ping fails:
+
+- Ensure the Pi is powered on and connected to the same network.
+- Check if SSH is enabled on the Pi.
+
+#### 3. Try SSH connection
+
+If you can't SSH into the Pi, it might not be properly connected. Use:
+
+```bash
+ssh <your_pi_user_name>@<your_pi_ip_address>
+```
+
+If you get a connection error:
+
+- Ensure SSH is enabled on the Pi by running:
+  ```bash
+  sudo raspi-config
+  ```
+  Then navigate to: **Interfacing Options -> SSH** and enable it.
+
+### Calibration
+
+Now we have to calibrate the leader arm and the follower arm. The wheel motors don't have to be calibrated.
+The calibration process is very important because it allows a neural network trained on one robot to work on another.
+
+### Calibrate follower arm (on mobile base)
+
+Make sure the arm is connected to the Raspberry Pi and run this script or API example (on the Raspberry Pi via SSH) to launch calibration of the follower arm:
+
+```bash
+lerobot-calibrate \
+    --robot.type=lekiwi \
+    --robot.id=my_awesome_kiwi # <- Give the robot a unique name
+```
+
+We unified the calibration method for most robots, thus, the calibration steps for this SO100 arm are the same as the steps for the Koch and SO101. First, we have to move the robot to the position where each joint is in the middle of its range, then we press `Enter`. Secondly, we move all joints through their full range of motion. A video of this same process for the SO101 as reference can be found [here](https://huggingface.co/docs/lerobot/en/so101#calibration-video).
+
+### Wired version
+
+If you have the **wired** LeKiwi version, please run all commands on your laptop.
+
+### Calibrate leader arm
+
+Then, to calibrate the leader arm (which is attached to the laptop/pc). Run the following command of API example on your laptop:
+
+<hfoptions id="calibrate_leader">
+<hfoption id="Command">
+
+```bash
+lerobot-calibrate \
+    --teleop.type=so100_leader \
+    --teleop.port=/dev/tty.usbmodem58760431551 \ # <- The port of your robot
+    --teleop.id=my_awesome_leader_arm # <- Give the robot a unique name
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.teleoperators.so_leader import SO100LeaderConfig, SO100Leader
+
+config = SO100LeaderConfig(
+    port="/dev/tty.usbmodem58760431551",
+    id="my_awesome_leader_arm",
+)
+
+leader = SO100Leader(config)
+leader.connect(calibrate=False)
+leader.calibrate()
+leader.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+## Teleoperate LeKiwi
+
+> [!TIP]
+> If you're using a Mac, you might need to give Terminal permission to access your keyboard for teleoperation. Go to System Preferences > Security & Privacy > Input Monitoring and check the box for Terminal.
+
+To teleoperate, SSH into your Raspberry Pi, and run `conda activate lerobot` and this command:
+
+```bash
+python -m lerobot.robots.lekiwi.lekiwi_host --robot.id=my_awesome_kiwi
+```
+
+Then on your laptop, also run `conda activate lerobot` and run the API example, make sure you set the correct `remote_ip` and `port` in `examples/lekiwi/teleoperate.py`.
+
+```bash
+python examples/lekiwi/teleoperate.py
+```
+
+You should see on your laptop something like this: `[INFO] Connected to remote robot at tcp://172.17.133.91:5555 and video stream at tcp://172.17.133.91:5556.` Now you can move the leader arm and use the keyboard (w,a,s,d) to drive forward, left, backwards, right. And use (z,x) to turn left or turn right. You can use (r,f) to increase and decrease the speed of the mobile robot. There are three speed modes, see the table below:
+
+| Speed Mode | Linear Speed (m/s) | Rotation Speed (deg/s) |
+| ---------- | ------------------ | ---------------------- |
+| Fast       | 0.4                | 90                     |
+| Medium     | 0.25               | 60                     |
+| Slow       | 0.1                | 30                     |
+
+| Key | Action         |
+| --- | -------------- |
+| W   | Move forward   |
+| A   | Move left      |
+| S   | Move backward  |
+| D   | Move right     |
+| Z   | Turn left      |
+| X   | Turn right     |
+| R   | Increase speed |
+| F   | Decrease speed |
+
+> [!TIP]
+> If you use a different keyboard, you can change the keys for each command in the [`LeKiwiClientConfig`](https://github.com/huggingface/lerobot/blob/main/src/lerobot/robots/lekiwi/config_lekiwi.py).
+
+### Wired version
+
+If you have the **wired** LeKiwi version, please run all commands on your laptop.
+
+## Record a dataset
+
+Once you're familiar with teleoperation, you can record your first dataset.
+
+We use the Hugging Face hub features for uploading your dataset. If you haven't previously used the Hub, make sure you can login via the cli using a write-access token, this token can be generated from the [Hugging Face settings](https://huggingface.co/settings/tokens).
+
+Add your token to the CLI by running this command:
+
+```bash
+hf auth login --token ${HUGGINGFACE_TOKEN} --add-to-git-credential
+```
+
+Then store your Hugging Face repository name in a variable:
+
+```bash
+HF_USER=$(hf auth whoami | awk -F': *' 'NR==1 {print $2}')
+echo $HF_USER
+```
+
+Now you can record a dataset. To record episodes and upload your dataset to the hub, execute this API example tailored for LeKiwi. Make sure to first adapt the `remote_ip`, `repo_id`, `port` and `task` in the script. If you would like to run the script for longer you can increase `NB_CYCLES_CLIENT_CONNECTION`.
+
+```bash
+python examples/lekiwi/record.py
+```
+
+#### Dataset upload
+
+Locally, your dataset is stored in this folder: `~/.cache/huggingface/lerobot/{repo-id}`. At the end of data recording, your dataset will be uploaded on your Hugging Face page (e.g. https://huggingface.co/datasets/cadene/so101_test) that you can obtain by running:
+
+```bash
+echo https://huggingface.co/datasets/${HF_USER}/so101_test
+```
+
+Your dataset will be automatically tagged with `LeRobot` for the community to find it easily, and you can also add custom tags (in this case `tutorial` for example).
+
+You can look for other LeRobot datasets on the hub by searching for `LeRobot` [tags](https://huggingface.co/datasets?other=LeRobot).
+
+#### Tips for gathering data
+
+Once you're comfortable with data recording, you can create a larger dataset for training. A good starting task is grasping an object at different locations and placing it in a bin. We suggest recording at least 50 episodes, with 10 episodes per location. Keep the cameras fixed and maintain consistent grasping behavior throughout the recordings. Also make sure the object you are manipulating is visible on the camera's. A good rule of thumb is you should be able to do the task yourself by only looking at the camera images.
+
+In the following sections, you’ll train your neural network. After achieving reliable grasping performance, you can start introducing more variations during data collection, such as additional grasp locations, different grasping techniques, and altering camera positions.
+
+Avoid adding too much variation too quickly, as it may hinder your results.
+
+If you want to dive deeper into this important topic, you can check out the [blog post](https://huggingface.co/blog/lerobot-datasets#what-makes-a-good-dataset) we wrote on what makes a good dataset.
+
+#### Troubleshooting:
+
+- On Linux, if the left and right arrow keys and escape key don't have any effect during data recording, make sure you've set the `$DISPLAY` environment variable. See [pynput limitations](https://pynput.readthedocs.io/en/latest/limitations.html#linux).
+
+## Replay an episode
+
+To replay an episode run the API example below, make sure to change `remote_ip`, `port`, LeRobotDatasetId and episode index.
+
+```bash
+python examples/lekiwi/replay.py
+```
+
+Congrats 🎉, your robot is all set to learn a task on its own. Start training it by the training part of this tutorial: [Getting started with real-world robots](./il_robots)
+
+## Evaluate your policy
+
+To evaluate your policy run the `evaluate.py` API example, make sure to change `remote_ip`, `port`, model..
+
+```bash
+python examples/lekiwi/evaluate.py
+```
+
+> [!TIP]
+> If you have any questions or need help, please reach out on [Discord](https://discord.com/invite/s3KuuzsPFb).
diff --git a/lerobot/docs/source/lerobot-dataset-v3.mdx b/lerobot/docs/source/lerobot-dataset-v3.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..235a355bd44e6798a0a65f18f4bd42e198af61ca
--- /dev/null
+++ b/lerobot/docs/source/lerobot-dataset-v3.mdx
@@ -0,0 +1,317 @@
+# LeRobotDataset v3.0
+
+`LeRobotDataset v3.0` is a standardized format for robot learning data. It provides unified access to multi-modal time-series data, sensorimotor signals and multi‑camera video, as well as rich metadata for indexing, search, and visualization on the Hugging Face Hub.
+
+This docs will guide you to:
+
+- Understand the v3.0 design and directory layout
+- Record a dataset and push it to the Hub
+- Load datasets for training with `LeRobotDataset`
+- Stream datasets without downloading using `StreamingLeRobotDataset`
+- Apply image transforms for data augmentation during training
+- Migrate existing `v2.1` datasets to `v3.0`
+
+## What’s new in `v3`
+
+- **File-based storage**: Many episodes per Parquet/MP4 file (v2 used one file per episode).
+- **Relational metadata**: Episode boundaries and lookups are resolved through metadata, not filenames.
+- **Hub-native streaming**: Consume datasets directly from the Hub with `StreamingLeRobotDataset`.
+- **Lower file-system pressure**: Fewer, larger files ⇒ faster initialization and fewer issues at scale.
+- **Unified organization**: Clean directory layout with consistent path templates across data and videos.
+
+## Installation
+
+`LeRobotDataset v3.0` will be included in `lerobot >= 0.4.0`.
+
+Until that stable release, you can use the main branch by following the [build from source instructions](./installation#from-source).
+
+## Record a dataset
+
+Run the command below to record a dataset with the SO-101 and push to the Hub:
+
+```bash
+lerobot-record \
+  --robot.type=so101_follower \
+  --robot.port=/dev/tty.usbmodem585A0076841 \
+  --robot.id=my_awesome_follower_arm \
+  --robot.cameras="{ front: {type: opencv, index_or_path: 0, width: 1920, height: 1080, fps: 30}}" \
+  --teleop.type=so101_leader \
+  --teleop.port=/dev/tty.usbmodem58760431551 \
+  --teleop.id=my_awesome_leader_arm \
+  --display_data=true \
+  --dataset.repo_id=${HF_USER}/record-test \
+  --dataset.num_episodes=5 \
+  --dataset.single_task="Grab the black cube" \
+  --dataset.streaming_encoding=true \
+  # --dataset.vcodec=auto \
+  --dataset.encoder_threads=2
+```
+
+See the [recording guide](./il_robots#record-a-dataset) for more details.
+
+## Format design
+
+A core v3 principle is **decoupling storage from the user API**: data is stored efficiently (few large files), while the public API exposes intuitive episode-level access.
+
+`v3` has three pillars:
+
+1. **Tabular data**: Low‑dimensional, high‑frequency signals (states, actions, timestamps) stored in **Apache Parquet**. Access is memory‑mapped or streamed via the `datasets` stack.
+2. **Visual data**: Camera frames concatenated and encoded into **MP4**. Frames from the same episode are grouped; videos are sharded per camera for practical sizes.
+3. **Metadata**: JSON/Parquet records describing schema (feature names, dtypes, shapes), frame rates, normalization stats, and **episode segmentation** (start/end offsets into shared Parquet/MP4 files).
+
+> To scale to millions of episodes, tabular rows and video frames from multiple episodes are **concatenated** into larger files. Episode‑specific views are reconstructed **via metadata**, not file boundaries.
+
+<div style="display:flex; justify-content:center; gap:12px; flex-wrap:wrap;">
+  <figure style="margin:0; text-align:center;">
+    <img
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobotdataset-v3/asset1datasetv3.png"
+      alt="LeRobotDataset v3 diagram"
+      width="220"
+    />
+    <figcaption style="font-size:0.9em; color:#666;">
+      From episode‑based to file‑based datasets
+    </figcaption>
+  </figure>
+</div>
+
+### Directory layout (simplified)
+
+- **`meta/info.json`**: canonical schema (features, shapes/dtypes), FPS, codebase version, and **path templates** to locate data/video shards.
+- **`meta/stats.json`**: global feature statistics (mean/std/min/max) used for normalization; exposed as `dataset.meta.stats`.
+- **`meta/tasks.jsonl`**: natural‑language task descriptions mapped to integer IDs for task‑conditioned policies.
+- **`meta/episodes/`**: per‑episode records (lengths, tasks, offsets) stored as **chunked Parquet** for scalability.
+- **`data/`**: frame‑by‑frame **Parquet** shards; each file typically contains **many episodes**.
+- **`videos/`**: **MP4** shards per camera; each file typically contains **many episodes**.
+
+## Load a dataset for training
+
+`LeRobotDataset` returns Python dictionaries of PyTorch tensors and integrates with `torch.utils.data.DataLoader`. Here is a code example showing its use:
+
+```python
+import torch
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+repo_id = "yaak-ai/L2D-v3"
+
+# 1) Load from the Hub (cached locally)
+dataset = LeRobotDataset(repo_id)
+
+# 2) Random access by index
+sample = dataset[100]
+print(sample)
+# {
+#   'observation.state': tensor([...]),
+#   'action': tensor([...]),
+#   'observation.images.front_left': tensor([C, H, W]),
+#   'timestamp': tensor(1.234),
+#   ...
+# }
+
+# 3) Temporal windows via delta_timestamps (seconds relative to t)
+delta_timestamps = {
+    "observation.images.front_left": [-0.2, -0.1, 0.0]  # 0.2s and 0.1s before current frame
+}
+
+dataset = LeRobotDataset(repo_id, delta_timestamps=delta_timestamps)
+
+# Accessing an index now returns a stack for the specified key(s)
+sample = dataset[100]
+print(sample["observation.images.front_left"].shape)  # [T, C, H, W], where T=3
+
+# 4) Wrap with a DataLoader for training
+batch_size = 16
+data_loader = torch.utils.data.DataLoader(dataset, batch_size=batch_size)
+
+device = "cuda" if torch.cuda.is_available() else "cpu"
+for batch in data_loader:
+    observations = batch["observation.state"].to(device)
+    actions = batch["action"].to(device)
+    images = batch["observation.images.front_left"].to(device)
+    # model.forward(batch)
+```
+
+## Stream a dataset (no downloads)
+
+Use `StreamingLeRobotDataset` to iterate directly from the Hub without local copies. This allows to stream large datasets without the need to downloading them onto disk or loading them onto memory, and is a key feature of the new dataset format.
+
+```python
+from lerobot.datasets.streaming_dataset import StreamingLeRobotDataset
+
+repo_id = "yaak-ai/L2D-v3"
+dataset = StreamingLeRobotDataset(repo_id)  # streams directly from the Hub
+```
+
+<div style="display:flex; justify-content:center; gap:12px; flex-wrap:wrap;">
+  <figure style="margin:0; text-align:center;">
+    <img
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobotdataset-v3/streaming-lerobot.png"
+      alt="StreamingLeRobotDataset"
+      width="520"
+    />
+    <figcaption style="font-size:0.9em; color:#666;">
+      Stream directly from the Hub for on‑the‑fly training.
+    </figcaption>
+  </figure>
+</div>
+
+## Image transforms
+
+Image transforms are data augmentations applied to camera frames during training to improve model robustness and generalization. LeRobot supports various transforms including brightness, contrast, saturation, hue, and sharpness adjustments.
+
+### Using transforms during dataset creation/recording
+
+Currently, transforms are applied during **training time only**, not during recording. When you create or record a dataset, the raw images are stored without transforms. This allows you to experiment with different augmentations later without re-recording data.
+
+### Adding transforms to existing datasets (API)
+
+Use the `image_transforms` parameter when loading a dataset for training:
+
+```python
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.transforms import ImageTransforms, ImageTransformsConfig, ImageTransformConfig
+
+# Option 1: Use default transform configuration (disabled by default)
+transforms_config = ImageTransformsConfig(
+    enable=True,  # Enable transforms
+    max_num_transforms=3,  # Apply up to 3 transforms per frame
+    random_order=False,  # Apply in standard order
+)
+transforms = ImageTransforms(transforms_config)
+
+dataset = LeRobotDataset(
+    repo_id="your-username/your-dataset",
+    image_transforms=transforms
+)
+
+# Option 2: Create custom transform configuration
+custom_transforms_config = ImageTransformsConfig(
+    enable=True,
+    max_num_transforms=2,
+    random_order=True,
+    tfs={
+        "brightness": ImageTransformConfig(
+            weight=1.0,
+            type="ColorJitter",
+            kwargs={"brightness": (0.7, 1.3)}  # Adjust brightness range
+        ),
+        "contrast": ImageTransformConfig(
+            weight=2.0,  # Higher weight = more likely to be selected
+            type="ColorJitter",
+            kwargs={"contrast": (0.8, 1.2)}
+        ),
+        "sharpness": ImageTransformConfig(
+            weight=0.5,  # Lower weight = less likely to be selected
+            type="SharpnessJitter",
+            kwargs={"sharpness": (0.3, 2.0)}
+        ),
+    }
+)
+
+dataset = LeRobotDataset(
+    repo_id="your-username/your-dataset",
+    image_transforms=ImageTransforms(custom_transforms_config)
+)
+
+# Option 3: Use pure torchvision transforms
+from torchvision.transforms import v2
+
+torchvision_transforms = v2.Compose([
+    v2.ColorJitter(brightness=0.2, contrast=0.2, saturation=0.2, hue=0.1),
+    v2.GaussianBlur(kernel_size=3, sigma=(0.1, 2.0)),
+])
+
+dataset = LeRobotDataset(
+    repo_id="your-username/your-dataset",
+    image_transforms=torchvision_transforms
+)
+```
+
+### Available transform types
+
+LeRobot provides several transform types:
+
+- **`ColorJitter`**: Adjusts brightness, contrast, saturation, and hue
+- **`SharpnessJitter`**: Randomly adjusts image sharpness
+- **`Identity`**: No transformation (useful for testing)
+
+You can also use any `torchvision.transforms.v2` transform by passing it directly to the `image_transforms` parameter.
+
+### Configuration options
+
+- **`enable`**: Enable/disable transforms (default: `False`)
+- **`max_num_transforms`**: Maximum number of transforms applied per frame (default: `3`)
+- **`random_order`**: Apply transforms in random order vs. standard order (default: `False`)
+- **`weight`**: Sampling probability for each transform (higher = more likely, if sum of weights is not 1, they will be normalized)
+- **`kwargs`**: Transform-specific parameters (e.g., brightness range)
+
+### Visualizing transforms
+
+Use the visualization script to preview how transforms affect your data:
+
+```bash
+lerobot-imgtransform-viz \
+  --repo-id=your-username/your-dataset \
+  --output-dir=./transform_examples \
+  --n-examples=5
+```
+
+This saves example images showing the effect of each transform, helping you tune parameters.
+
+### Best practices
+
+- **Start conservative**: Begin with small ranges (e.g., brightness 0.9-1.1) and increase gradually
+- **Test first**: Use the visualization script to ensure transforms look reasonable
+- **Monitor training**: Strong augmentations can hurt performance if too aggressive
+- **Match your domain**: If your robot operates in varying lighting, use brightness/contrast transforms
+- **Combine wisely**: Using too many transforms simultaneously can make training unstable
+
+## Migrate `v2.1` → `v3.0`
+
+A converter aggregates per‑episode files into larger shards and writes episode offsets/metadata. Convert your dataset using the instructions below.
+
+```bash
+# Pre-release build with v3 support:
+pip install "https://github.com/huggingface/lerobot/archive/33cad37054c2b594ceba57463e8f11ee374fa93c.zip"
+
+# Convert an existing v2.1 dataset hosted on the Hub:
+python -m lerobot.datasets.v30.convert_dataset_v21_to_v30 --repo-id=<HF_USER/DATASET_ID>
+```
+
+**What it does**
+
+- Aggregates parquet files: `episode-0000.parquet`, `episode-0001.parquet`, … → **`file-0000.parquet`**, …
+- Aggregates mp4 files: `episode-0000.mp4`, `episode-0001.mp4`, … → **`file-0000.mp4`**, …
+- Updates `meta/episodes/*` (chunked Parquet) with per‑episode lengths, tasks, and byte/frame offsets.
+
+## Common Issues
+
+### Always call `finalize()` before pushing
+
+When creating or recording datasets, you **must** call `dataset.finalize()` to properly close parquet writers. See the [PR #1903](https://github.com/huggingface/lerobot/pull/1903) for more details.
+
+```python
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+# Create dataset and record episodes
+dataset = LeRobotDataset.create(...)
+
+for episode in range(num_episodes):
+    # Record frames
+    for frame in episode_data:
+        dataset.add_frame(frame)
+    dataset.save_episode()
+
+# Call finalize() when done recording and before push_to_hub()
+dataset.finalize()  # Closes parquet writers, writes metadata footers
+dataset.push_to_hub()
+```
+
+**Why is this necessary?**
+
+Dataset v3.0 uses incremental parquet writing with buffered metadata for efficiency. The `finalize()` method:
+
+- Flushes any buffered episode metadata to disk
+- Closes parquet writers to write footer metadata, otherwise the parquet files will be corrupt
+- Ensures the dataset is valid for loading
+
+Without calling `finalize()`, your parquet files will be incomplete and the dataset won't load properly.
diff --git a/lerobot/docs/source/libero.mdx b/lerobot/docs/source/libero.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..def974531c4415f44b76e742618fcbe6cce3cdda
--- /dev/null
+++ b/lerobot/docs/source/libero.mdx
@@ -0,0 +1,172 @@
+# LIBERO
+
+**LIBERO** is a benchmark designed to study **lifelong robot learning**. The idea is that robots won’t just be pretrained once in a factory, they’ll need to keep learning and adapting with their human users over time. This ongoing adaptation is called **lifelong learning in decision making (LLDM)**, and it’s a key step toward building robots that become truly personalized helpers.
+
+- 📄 [LIBERO paper](https://arxiv.org/abs/2306.03310)
+- 💻 [Original LIBERO repo](https://github.com/Lifelong-Robot-Learning/LIBERO)
+
+To make progress on this challenge, LIBERO provides a set of standardized tasks that focus on **knowledge transfer**: how well a robot can apply what it has already learned to new situations. By evaluating on LIBERO, different algorithms can be compared fairly and researchers can build on each other’s work.
+
+LIBERO includes **five task suites**:
+
+- **LIBERO-Spatial (`libero_spatial`)** – tasks that require reasoning about spatial relations.
+- **LIBERO-Object (`libero_object`)** – tasks centered on manipulating different objects.
+- **LIBERO-Goal (`libero_goal`)** – goal-conditioned tasks where the robot must adapt to changing targets.
+- **LIBERO-90 (`libero_90`)** – 90 short-horizon tasks from the LIBERO-100 collection.
+- **LIBERO-Long (`libero_10`)** – 10 long-horizon tasks from the LIBERO-100 collection.
+
+Together, these suites cover **130 tasks**, ranging from simple object manipulations to complex multi-step scenarios. LIBERO is meant to grow over time, and to serve as a shared benchmark where the community can test and improve lifelong learning algorithms.
+
+![An overview of the LIBERO benchmark](https://libero-project.github.io/assets/img/libero/fig1.png)
+
+## Evaluating with LIBERO
+
+At **LeRobot**, we ported [LIBERO](https://github.com/Lifelong-Robot-Learning/LIBERO) into our framework and used it mainly to **evaluate [SmolVLA](https://huggingface.co/docs/lerobot/en/smolvla)**, our lightweight Vision-Language-Action model.
+
+LIBERO is now part of our **multi-eval supported simulation**, meaning you can benchmark your policies either on a **single suite of tasks** or across **multiple suites at once** with just a flag.
+
+To Install LIBERO, after following LeRobot official instructions, just do:
+`pip install -e ".[libero]"`
+
+### Single-suite evaluation
+
+Evaluate a policy on one LIBERO suite:
+
+```bash
+lerobot-eval \
+  --policy.path="your-policy-id" \
+  --env.type=libero \
+  --env.task=libero_object \
+  --eval.batch_size=2 \
+  --eval.n_episodes=3
+```
+
+- `--env.task` picks the suite (`libero_object`, `libero_spatial`, etc.).
+- `--env.task_ids` picks task ids to run (`[0]`, `[1,2,3]`, etc.). Omit this flag (or set it to `null`) to run all tasks in the suite.
+- `--eval.batch_size` controls how many environments run in parallel.
+- `--eval.n_episodes` sets how many episodes to run in total.
+
+---
+
+### Multi-suite evaluation
+
+Benchmark a policy across multiple suites at once:
+
+```bash
+lerobot-eval \
+  --policy.path="your-policy-id" \
+  --env.type=libero \
+  --env.task=libero_object,libero_spatial \
+  --eval.batch_size=1 \
+  --eval.n_episodes=2
+```
+
+- Pass a comma-separated list to `--env.task` for multi-suite evaluation.
+
+### Control Mode
+
+LIBERO now supports two control modes: relative and absolute. This matters because different VLA checkpoints are trained with different mode of action to output hence control parameterizations.
+You can switch them with: `env.control_mode = "relative"` and `env.control_mode = "absolute"`
+
+### Policy inputs and outputs
+
+When using LIBERO through LeRobot, policies interact with the environment via **observations** and **actions**:
+
+- **Observations**
+  - `observation.state` – proprioceptive features (agent state).
+  - `observation.images.image` – main camera view (`agentview_image`).
+  - `observation.images.image2` – wrist camera view (`robot0_eye_in_hand_image`).
+
+  ⚠️ **Note:** LeRobot enforces the `.images.*` prefix for any multi-modal visual features. Always ensure that your policy config `input_features` use the same naming keys, and that your dataset metadata keys follow this convention during evaluation.
+  If your data contains different keys, you must rename the observations to match what the policy expects, since naming keys are encoded inside the normalization statistics layer.
+  This will be fixed with the upcoming Pipeline PR.
+
+- **Actions**
+  - Continuous control values in a `Box(-1, 1, shape=(7,))` space.
+
+We also provide a notebook for quick testing:
+Training with LIBERO
+
+## Training with LIBERO
+
+When training on LIBERO tasks, make sure your dataset parquet and metadata keys follow the LeRobot convention.
+
+The environment expects:
+
+- `observation.state` → 8-dim agent state
+- `observation.images.image` → main camera (`agentview_image`)
+- `observation.images.image2` → wrist camera (`robot0_eye_in_hand_image`)
+
+⚠️ Cleaning the dataset upfront is **cleaner and more efficient** than remapping keys inside the code.
+To avoid potential mismatches and key errors, we provide a **preprocessed LIBERO dataset** that is fully compatible with the current LeRobot codebase and requires no additional manipulation:
+👉 [HuggingFaceVLA/libero](https://huggingface.co/datasets/HuggingFaceVLA/libero)
+
+For reference, here is the **original dataset** published by Physical Intelligence:
+👉 [physical-intelligence/libero](https://huggingface.co/datasets/physical-intelligence/libero)
+
+---
+
+### Example training command
+
+```bash
+lerobot-train \
+  --policy.type=smolvla \
+  --policy.repo_id=${HF_USER}/libero-test \
+  --policy.load_vlm_weights=true \
+  --dataset.repo_id=HuggingFaceVLA/libero \
+  --env.type=libero \
+  --env.task=libero_10 \
+  --output_dir=./outputs/ \
+  --steps=100000 \
+  --batch_size=4 \
+  --eval.batch_size=1 \
+  --eval.n_episodes=1 \
+  --eval_freq=1000 \
+```
+
+---
+
+### Note on rendering
+
+LeRobot uses MuJoCo for simulation. You need to set the rendering backend before training or evaluation:
+
+- `export MUJOCO_GL=egl` → for headless servers (e.g. HPC, cloud)
+
+## Reproducing π₀.₅ results
+
+We reproduce the results of π₀.₅ on the LIBERO benchmark using the LeRobot implementation. We take the Physical Intelligence LIBERO base model (`pi05_libero`) and finetune for an additional 6k steps in bfloat16, with batch size of 256 on 8 H100 GPUs using the [HuggingFace LIBERO dataset](https://huggingface.co/datasets/HuggingFaceVLA/libero).
+
+The finetuned model can be found here:
+
+- **π₀.₅ LIBERO**: [lerobot/pi05_libero_finetuned](https://huggingface.co/lerobot/pi05_libero_finetuned)
+
+We then evaluate the finetuned model using the LeRobot LIBERO implementation, by running the following command:
+
+```bash
+lerobot-eval \
+  --output_dir=/logs/ \
+  --env.type=libero \
+  --env.task=libero_spatial,libero_object,libero_goal,libero_10 \
+  --eval.batch_size=1 \
+  --eval.n_episodes=10 \
+  --policy.path=pi05_libero_finetuned \
+  --policy.n_action_steps=10 \
+  --output_dir=./eval_logs/ \
+  --env.max_parallel_tasks=1
+```
+
+**Note:** We set `n_action_steps=10`, similar to the original OpenPI implementation.
+
+### Results
+
+We obtain the following results on the LIBERO benchmark:
+
+| Model    | LIBERO Spatial | LIBERO Object | LIBERO Goal | LIBERO 10 | Average  |
+| -------- | -------------- | ------------- | ----------- | --------- | -------- |
+| **π₀.₅** | 97.0           | 99.0          | 98.0        | 96.0      | **97.5** |
+
+These results are consistent with the original [results](https://github.com/Physical-Intelligence/openpi/tree/main/examples/libero#results) reported by Physical Intelligence:
+
+| Model    | LIBERO Spatial | LIBERO Object | LIBERO Goal | LIBERO 10 | Average   |
+| -------- | -------------- | ------------- | ----------- | --------- | --------- |
+| **π₀.₅** | 98.8           | 98.2          | 98.0        | 92.4      | **96.85** |
diff --git a/lerobot/docs/source/metaworld.mdx b/lerobot/docs/source/metaworld.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..da90bd51d84f31ae6a192d37319c1b611d6a0636
--- /dev/null
+++ b/lerobot/docs/source/metaworld.mdx
@@ -0,0 +1,80 @@
+# Meta-World
+
+Meta-World is a well-designed, open-source simulation benchmark for multi-task and meta reinforcement learning in continuous-control robotic manipulation. It gives researchers a shared, realistic playground to test whether algorithms can _learn many different tasks_ and _generalize quickly to new ones_ — two central challenges for real-world robotics.
+
+- 📄 [MetaWorld paper](https://arxiv.org/pdf/1910.10897)
+- 💻 [Original MetaWorld repo](https://github.com/Farama-Foundation/Metaworld)
+
+![MetaWorld MT10 demo](https://meta-world.github.io/figures/ml45.gif)
+
+## Why Meta-World matters
+
+- **Diverse, realistic tasks.** Meta-World bundles a large suite of simulated manipulation tasks (50 in the MT50 suite) using everyday objects and a common tabletop Sawyer arm. This diversity exposes algorithms to a wide variety of dynamics, contacts and goal specifications while keeping a consistent control and observation structure.
+- **Focus on generalization and multi-task learning.** By evaluating across task distributions that share structure but differ in goals and objects, Meta-World reveals whether an agent truly learns transferable skills rather than overfitting to a narrow task.
+- **Standardized evaluation protocol.** It provides clear evaluation modes and difficulty splits, so different methods can be compared fairly across easy, medium, hard and very-hard regimes.
+- **Empirical insight.** Past evaluations on Meta-World show impressive progress on some fronts, but also highlight that current multi-task and meta-RL methods still struggle with large, diverse task sets. That gap points to important research directions.
+
+## What it enables in LeRobot
+
+In LeRobot, you can evaluate any policy or vision-language-action (VLA) model on Meta-World tasks and get a clear success-rate measure. The integration is designed to be straightforward:
+
+- We provide a LeRobot-ready dataset for Meta-World (MT50) on the HF Hub: `https://huggingface.co/datasets/lerobot/metaworld_mt50`.
+  - This dataset is formatted for the MT50 evaluation that uses all 50 tasks (the most challenging multi-task setting).
+  - MT50 gives the policy a one-hot task vector and uses fixed object/goal positions for consistency.
+
+- Task descriptions and the exact keys required for evaluation are available in the repo/dataset — use these to ensure your policy outputs the right success signals.
+
+## Quick start, train a SmolVLA policy on Meta-World
+
+Example command to train a SmolVLA policy on a subset of tasks:
+
+```bash
+lerobot-train \
+  --policy.type=smolvla \
+  --policy.repo_id=${HF_USER}/metaworld-test \
+  --policy.load_vlm_weights=true \
+  --dataset.repo_id=lerobot/metaworld_mt50 \
+  --env.type=metaworld \
+  --env.task=assembly-v3,dial-turn-v3,handle-press-side-v3 \
+  --output_dir=./outputs/ \
+  --steps=100000 \
+  --batch_size=4 \
+  --eval.batch_size=1 \
+  --eval.n_episodes=1 \
+  --eval_freq=1000
+```
+
+Notes:
+
+- `--env.task` accepts explicit task lists (comma separated) or difficulty groups (e.g., `env.task="hard"`).
+- Adjust `batch_size`, `steps`, and `eval_freq` to match your compute budget.
+- **Gymnasium Assertion Error**: if you encounter an error like
+  `AssertionError: ['human', 'rgb_array', 'depth_array']` when running MetaWorld environments, this comes from a mismatch between MetaWorld and your Gymnasium version.
+  We recommend using:
+
+```bash
+  pip install "gymnasium==1.1.0"
+```
+
+to ensure proper compatibility.
+
+## Quick start — evaluate a trained policy
+
+To evaluate a trained policy on the Meta-World medium difficulty split:
+
+```bash
+lerobot-eval \
+  --policy.path="your-policy-id" \
+  --env.type=metaworld \
+  --env.task=medium \
+  --eval.batch_size=1 \
+  --eval.n_episodes=2
+```
+
+This will run episodes and return per-task success rates using the standard Meta-World evaluation keys.
+
+## Practical tips
+
+- If you care about generalization, run on the full MT50 suite — it’s intentionally challenging and reveals strengths/weaknesses better than a few narrow tasks.
+- Use the one-hot task conditioning for multi-task training (MT10 / MT50 conventions) so policies have explicit task context.
+- Inspect the dataset task descriptions and the `info["is_success"]` keys when writing post-processing or logging so your success metrics line up with the benchmark.
diff --git a/lerobot/docs/source/multi_gpu_training.mdx b/lerobot/docs/source/multi_gpu_training.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..122670f697f123f5ce51cd22a3d235070ef071ee
--- /dev/null
+++ b/lerobot/docs/source/multi_gpu_training.mdx
@@ -0,0 +1,125 @@
+# Multi-GPU Training
+
+This guide shows you how to train policies on multiple GPUs using [Hugging Face Accelerate](https://huggingface.co/docs/accelerate).
+
+## Installation
+
+First, ensure you have accelerate installed:
+
+```bash
+pip install accelerate
+```
+
+## Training with Multiple GPUs
+
+You can launch training in two ways:
+
+### Option 1: Without config (specify parameters directly)
+
+You can specify all parameters directly in the command without running `accelerate config`:
+
+```bash
+accelerate launch \
+  --multi_gpu \
+  --num_processes=2 \
+  $(which lerobot-train) \
+  --dataset.repo_id=${HF_USER}/my_dataset \
+  --policy.type=act \
+  --policy.repo_id=${HF_USER}/my_trained_policy \
+  --output_dir=outputs/train/act_multi_gpu \
+  --job_name=act_multi_gpu \
+  --wandb.enable=true
+```
+
+**Key accelerate parameters:**
+
+- `--multi_gpu`: Enable multi-GPU training
+- `--num_processes=2`: Number of GPUs to use
+- `--mixed_precision=fp16`: Use fp16 mixed precision (or `bf16` if supported)
+
+### Option 2: Using accelerate config
+
+If you prefer to save your configuration, you can optionally configure accelerate for your hardware setup by running:
+
+```bash
+accelerate config
+```
+
+This interactive setup will ask you questions about your training environment (number of GPUs, mixed precision settings, etc.) and saves the configuration for future use. For a simple multi-GPU setup on a single machine, you can use these recommended settings:
+
+- Compute environment: This machine
+- Number of machines: 1
+- Number of processes: (number of GPUs you want to use)
+- GPU ids to use: (leave empty to use all)
+- Mixed precision: fp16 or bf16 (recommended for faster training)
+
+Then launch training with:
+
+```bash
+accelerate launch $(which lerobot-train) \
+  --dataset.repo_id=${HF_USER}/my_dataset \
+  --policy.type=act \
+  --policy.repo_id=${HF_USER}/my_trained_policy \
+  --output_dir=outputs/train/act_multi_gpu \
+  --job_name=act_multi_gpu \
+  --wandb.enable=true
+```
+
+## How It Works
+
+When you launch training with accelerate:
+
+1. **Automatic detection**: LeRobot automatically detects if it's running under accelerate
+2. **Data distribution**: Your batch is automatically split across GPUs
+3. **Gradient synchronization**: Gradients are synchronized across GPUs during backpropagation
+4. **Single process logging**: Only the main process logs to wandb and saves checkpoints
+
+## Learning Rate and Training Steps Scaling
+
+**Important:** LeRobot does **NOT** automatically scale learning rates or training steps based on the number of GPUs. This gives you full control over your training hyperparameters.
+
+### Why No Automatic Scaling?
+
+Many distributed training frameworks automatically scale the learning rate by the number of GPUs (e.g., `lr = base_lr × num_gpus`).
+However, LeRobot keeps the learning rate exactly as you specify it.
+
+### When and How to Scale
+
+If you want to scale your hyperparameters when using multiple GPUs, you should do it manually:
+
+**Learning Rate Scaling:**
+
+```bash
+# Example: 2 GPUs with linear LR scaling
+# Base LR: 1e-4, with 2 GPUs -> 2e-4
+accelerate launch --num_processes=2 $(which lerobot-train) \
+  --optimizer.lr=2e-4 \
+  --dataset.repo_id=lerobot/pusht \
+  --policy=act
+```
+
+**Training Steps Scaling:**
+
+Since the effective batch size `bs` increases with multiple GPUs (batch_size × num_gpus), you may want to reduce the number of training steps proportionally:
+
+```bash
+# Example: 2 GPUs with effective batch size 2x larger
+# Original: batch_size=8, steps=100000
+# With 2 GPUs: batch_size=8 (16 in total), steps=50000
+accelerate launch --num_processes=2 $(which lerobot-train) \
+  --batch_size=8 \
+  --steps=50000 \
+  --dataset.repo_id=lerobot/pusht \
+  --policy=act
+```
+
+## Notes
+
+- The `--policy.use_amp` flag in `lerobot-train` is only used when **not** running with accelerate. When using accelerate, mixed precision is controlled by accelerate's configuration.
+- Training logs, checkpoints, and hub uploads are only done by the main process to avoid conflicts. Non-main processes have console logging disabled to prevent duplicate output.
+- The effective batch size is `batch_size × num_gpus`. If you use 4 GPUs with `--batch_size=8`, your effective batch size is 32.
+- Learning rate scheduling is handled correctly across multiple processes—LeRobot sets `step_scheduler_with_optimizer=False` to prevent accelerate from adjusting scheduler steps based on the number of processes.
+- When saving or pushing models, LeRobot automatically unwraps the model from accelerate's distributed wrapper to ensure compatibility.
+- WandB integration automatically initializes only on the main process, preventing multiple runs from being created.
+
+For more advanced configurations and troubleshooting, see the [Accelerate documentation](https://huggingface.co/docs/accelerate). If you want to learn more about how to train on a large number of GPUs, checkout this awesome guide: [Ultrascale Playbook](https://huggingface.co/spaces/nanotron/ultrascale-playbook).
diff --git a/lerobot/docs/source/notebooks.mdx b/lerobot/docs/source/notebooks.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..6a9c3b103ce264011ceee732f0530219c4ba2cd6
--- /dev/null
+++ b/lerobot/docs/source/notebooks.mdx
@@ -0,0 +1,29 @@
+# 🤗 LeRobot Notebooks
+
+This repository contains example notebooks for using LeRobot. These notebooks demonstrate how to train policies on real or simulation datasets using standardized policies.
+
+---
+
+### Training ACT
+
+[ACT](https://huggingface.co/papers/2304.13705) (Action Chunking Transformer) is a transformer-based policy architecture for imitation learning that processes robot states and camera inputs to generate smooth, chunked action sequences.
+
+We provide a ready-to-run Google Colab notebook to help you train ACT policies using datasets from the Hugging Face Hub, with optional logging to Weights & Biases.
+
+| Notebook                                                                                                | Colab                                                                                                                                                                             |
+| :------------------------------------------------------------------------------------------------------ | :-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| [Train ACT with LeRobot](https://github.com/huggingface/notebooks/blob/main/lerobot/training-act.ipynb) | [![Open in Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/huggingface/notebooks/blob/main/lerobot/training-act.ipynb) |
+
+Expected training time for 100k steps: ~1.5 hours on an NVIDIA A100 GPU with batch size of `64`.
+
+### Training SmolVLA
+
+[SmolVLA](https://huggingface.co/papers/2506.01844) is a small but efficient Vision-Language-Action model. It is compact in size with 450 M-parameter and is developed by Hugging Face.
+
+We provide a ready-to-run Google Colab notebook to help you train SmolVLA policies using datasets from the Hugging Face Hub, with optional logging to Weights & Biases.
+
+| Notebook                                                                                                        | Colab                                                                                                                                                                                 |
+| :-------------------------------------------------------------------------------------------------------------- | :------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
+| [Train SmolVLA with LeRobot](https://github.com/huggingface/notebooks/blob/main/lerobot/training-smolvla.ipynb) | [![Open in Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/huggingface/notebooks/blob/main/lerobot/training-smolvla.ipynb) |
+
+Expected training time for 20k steps: ~5 hours on an NVIDIA A100 GPU with batch size of `64`.
diff --git a/lerobot/docs/source/omx.mdx b/lerobot/docs/source/omx.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..4617ac7bd5bff28379f19cb7838acbfe6c5a131b
--- /dev/null
+++ b/lerobot/docs/source/omx.mdx
@@ -0,0 +1,197 @@
+## Order and Assemble the parts
+
+First, assemble the OMX hardware following the official assembly guide.
+
+OMX Assembly Guide: https://ai.robotis.com/omx/assembly_guide_omx.html
+
+OMX robots are shipped preconfigured from the factory. Motor IDs, communication parameters, and joint offsets are already set, so no additional motor setup or calibration is required before using LeRobot.
+
+## Install LeRobot 🤗
+
+To install LeRobot, follow our [Installation Guide](./installation)
+
+In addition to these instructions, you need to install the Dynamixel SDK:
+
+```bash
+pip install -e ".[dynamixel]"
+```
+
+## Connect the robot
+
+To find the port for each bus servo adapter, run this script:
+
+```bash
+lerobot-find-port
+```
+
+This command runs and when prompted, disconnect the USB cable from either the leader or follower arm and press Enter. The output will show 'The port of this MotorsBus is [port]'. This identifies the port for the disconnected arm. Repeat for the other arm to identify both ports.
+
+<hfoptions id="find_port">
+<hfoption id="Mac">
+
+Example output on macOS:
+
+```
+Finding all available ports for the MotorBus.
+['/dev/tty.usbmodem575E0032081', '/dev/tty.usbmodem575E0031751']
+Remove the USB cable from your MotorsBus and press Enter when done.
+
+[...Disconnect corresponding leader or follower arm and press Enter...]
+
+The port of this MotorsBus is /dev/tty.usbmodem575E0032081
+Reconnect the USB cable.
+```
+
+Where the found port is: `/dev/tty.usbmodem575E0032081` corresponding to your leader or follower arm.
+
+</hfoption>
+<hfoption id="Linux">
+
+On Linux, we strongly recommend using udev rules to assign persistent and human-readable device names to the OMX leader and follower arms. This avoids issues where device names such as ttyACM0 and ttyACM1 change when the robot is unplugged, replugged, or when the system is rebooted.
+
+#### 1. Find your device serial numbers
+
+You should have obtained the port numbers like ../../ttyACM? for the leader and follower using `lerobot-find-port`. You can match those results with the serial numbers using the `ls -l /dev/serial/by-id/` command.
+To create udev rules, you need the unique serial number for each OMX device. The easiest way is to list devices under:
+
+```bash
+ls -l /dev/serial/by-id/
+```
+
+You will see output similar to:
+
+```bash
+usb-ROBOTIS_OpenRB-150_228BDD7B503059384C2E3120FF0A2B19-if00 -> ../../ttyACM0
+usb-ROBOTIS_OpenRB-150_67E1ED68503059384C2E3120FF092234-if00 -> ../../ttyACM1
+```
+
+In each line, the serial number is the long string after `usb-ROBOTIS_OpenRB-150_` and before `-if00`.
+
+Follower serial: `228BDD7B503059384C2E3120FF0A2B19`
+
+Leader serial: `67E1ED68503059384C2E3120FF092234`
+
+#### 2. Create the udev rule
+
+Create a new udev rule file:
+
+```bash
+sudo nano /etc/udev/rules.d/99-omx.rules
+```
+
+Paste the following lines, replacing the serial numbers with the values you found above:
+
+```bash
+SUBSYSTEM=="tty", ATTRS{idVendor}=="0403", ATTRS{serial}=="228BDD7B503059384C2E3120FF0A2B19", SYMLINK+="omx_follower"
+SUBSYSTEM=="tty", ATTRS{idVendor}=="0403", ATTRS{serial}=="67E1ED68503059384C2E3120FF092234", SYMLINK+="omx_leader"
+```
+
+Save the file and reload udev rules:
+
+```bash
+sudo udevadm control --reload-rules
+sudo udevadm trigger
+```
+
+Now unplug and replug both devices once.
+
+#### 3. Verify the symlinks
+
+Check that the persistent device names exist:
+
+```bash
+ls -l /dev/omx_follower /dev/omx_leader
+```
+
+You should see them pointing to ttyACM\* devices:
+
+```bash
+/dev/omx_follower -> ttyACM*
+/dev/omx_leader   -> ttyACM*
+```
+
+These names remain stable across reboots and reconnections.
+
+</hfoption>
+</hfoptions>
+
+## Teleoperate
+
+After identifying the correct ports, you can directly teleoperate the follower arm using the leader arm.
+
+<hfoptions id="teleoperate">
+<hfoption id="Mac">
+
+### Teleoperate without camera
+
+```bash
+lerobot-teleoperate \
+  --robot.type=omx_follower \
+  --robot.port=<your_follower_port> \
+  --robot.id=omx_follower_arm \
+  --teleop.type=omx_leader \
+  --teleop.port=<your_leader_port> \
+  --teleop.id=omx_leader_arm
+```
+
+During teleoperation, motions of the leader arm are mirrored in real time by the follower arm. OMX is already preconfigured, teleoperation can begin immediately without any calibration steps.
+
+### Teleoperate with camera
+
+You can also enable camera input during teleoperation by providing a camera configuration for the follower arm.
+
+```bash
+lerobot-teleoperate \
+  --robot.type=omx_follower \
+  --robot.port=<your_follower_port> \
+  --robot.id=omx_follower_arm \
+  --robot.cameras="{front: {type: opencv, index_or_path: '/dev/video0', width: 640, height: 480, fps: 30}}" \
+  --teleop.type=omx_leader \
+  --teleop.port=<your_leader_port> \
+  --teleop.id=omx_leader_arm \
+  --display_data=true
+```
+
+When the camera is enabled, the camera stream is displayed in real time and synchronized with the robot state. This setup is useful for visual monitoring and can be reused later for demonstration recording and imitation learning.
+
+</hfoption>
+<hfoption id="Linux">
+
+### Teleoperate without camera
+
+```bash
+lerobot-teleoperate \
+  --robot.type=omx_follower \
+  --robot.port=/dev/omx_follower \
+  --robot.id=omx_follower_arm \
+  --teleop.type=omx_leader \
+  --teleop.port=/dev/omx_leader \
+  --teleop.id=omx_leader_arm
+```
+
+During teleoperation, motions of the leader arm are mirrored in real time by the follower arm. OMX is already preconfigured, teleoperation can begin immediately without any calibration steps.
+
+### Teleoperate with camera
+
+You can also enable camera input during teleoperation by providing a camera configuration for the follower arm.
+
+```bash
+lerobot-teleoperate \
+  --robot.type=omx_follower \
+  --robot.port=/dev/omx_follower \
+  --robot.id=omx_follower_arm \
+  --robot.cameras="{front: {type: opencv, index_or_path: '/dev/video0', width: 640, height: 480, fps: 30}}" \
+  --teleop.type=omx_leader \
+  --teleop.port=/dev/omx_leader \
+  --teleop.id=omx_leader_arm \
+  --display_data=true
+```
+
+When the camera is enabled, the camera stream is displayed in real time and synchronized with the robot state. This setup is useful for visual monitoring and can be reused later for demonstration recording and imitation learning.
+
+</hfoption>
+</hfoptions>
+
+Congrats 🎉, your robot is all set to learn a task on its own.
+
+> If you have any questions or need help, please reach out on [Discord](https://discord.com/invite/robotis).
diff --git a/lerobot/docs/source/openarm.mdx b/lerobot/docs/source/openarm.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..cd4ace9121da2ba8fc5d1384a1407f6a04da6455
--- /dev/null
+++ b/lerobot/docs/source/openarm.mdx
@@ -0,0 +1,276 @@
+# OpenArm
+
+[OpenArm](https://openarm.dev) is an open-source 7DOF humanoid arm designed for physical AI research and deployment.
+
+To get your OpenArm, assembled or DIY, and join the global community, browse verified and certified manufacturers worldwide at [openarm.dev](https://openarm.dev).
+
+## What's Unique?
+
+- **Human-Scale Design**: OpenArm is designed with human-like proportions, scaled for a person around 160-165cm tall. This provides an optimal balance between practical reach and manageable inertia for safe, responsive operation.
+
+- **Safety-First Architecture**: Built with QDD backdrivable motors and high compliance, OpenArm prioritizes safe human-robot interaction while maintaining practical payload capabilities (6.0kg peak / 4.1kg nominal) for real-world tasks.
+
+- **Built for Durability**: Critical structural components use aluminum and stainless steel construction, ensuring robust performance for repetitive data collection and continuous research use.
+
+- **Fully Accessible & Buildable**: Every component, from CNC parts and 3D-printed casings to electrical wiring is designed to be purchasable and buildable by individual researchers and labs, with complete fabrication data provided.
+
+- **Practical & Affordable**: At $6,500 USD for a complete bimanual system, OpenArm delivers research-grade capabilities at a fraction of traditional humanoid robot costs.
+
+## Platform Requirements
+
+<Tip warning={true}>
+  **Linux Only**: OpenArm currently only works on Linux. The CAN bus USB adapter
+  does not have macOS drivers and has not been tested on Windows.
+</Tip>
+
+## Safety Guide
+
+Before operating OpenArm, please read the [official safety guide](https://docs.openarm.dev/getting-started/safety-guide). Key points:
+
+- **Secure installation**: Fasten the arm to a flat, stable surface with screws or clamps
+- **Safe distance**: Keep body parts and objects outside the range of motion during operation
+- **Protective equipment**: Always wear safety goggles; use additional PPE as needed
+- **Payload limits**: Do not exceed specified payload limits (6.0kg peak / 4.1kg nominal per arm)
+- **Emergency stop**: Know the location and operation of the emergency stop device
+- **Regular inspection**: Check for loose screws, damaged mechanical limits, unusual noises, and wiring damage
+
+## Hardware Setup
+
+Follow the official [OpenArm hardware documentation](https://docs.openarm.dev) for:
+
+- Bill of materials and sourcing
+- 3D printing instructions
+- Mechanical assembly
+- Electrical wiring
+
+The hardware repositories are available at [github.com/enactic/openarm](https://github.com/enactic/openarm).
+
+## CAN Bus Setup
+
+OpenArm uses CAN bus communication with Damiao motors. Once you have the CAN bus USB adapter plugged into your Linux PC, follow the [Damiao Motors and CAN Bus guide](./damiao) to configure the interface.
+
+Quick setup:
+
+```bash
+# Setup CAN interfaces
+lerobot-setup-can --mode=setup --interfaces=can0,can1
+
+# Test motor communication
+lerobot-setup-can --mode=test --interfaces=can0,can1
+```
+
+## Install LeRobot 🤗
+
+Follow our [Installation Guide](./installation), then install the Damiao motor support:
+
+```bash
+pip install -e ".[damiao]"
+```
+
+## Usage
+
+### Follower Arm (Robot)
+
+<hfoptions id="follower">
+<hfoption id="Command">
+
+```bash
+lerobot-calibrate \
+    --robot.type=openarm_follower \
+    --robot.port=can0 \
+    --robot.side=right \
+    --robot.id=my_openarm_follower
+```
+
+</hfoption>
+<hfoption id="API example">
+
+```python
+from lerobot.robots.openarm_follower import OpenArmFollower, OpenArmFollowerConfig
+
+config = OpenArmFollowerConfig(
+    port="can0",
+    side="right",  # or "left" for left arm
+    id="my_openarm_follower",
+)
+
+follower = OpenArmFollower(config)
+follower.connect()
+
+# Read current state
+obs = follower.get_observation()
+print(obs)
+
+# Send action (position in degrees)
+action = {
+    "joint_1.pos": 0.0,
+    "joint_2.pos": 0.0,
+    "joint_3.pos": 0.0,
+    "joint_4.pos": 45.0,
+    "joint_5.pos": 0.0,
+    "joint_6.pos": 0.0,
+    "joint_7.pos": 0.0,
+    "gripper.pos": 0.0,
+}
+follower.send_action(action)
+
+follower.disconnect()
+```
+
+</hfoption>
+</hfoptions>
+
+### Leader Arm (Teleoperator)
+
+The leader arm is used for teleoperation - manually moving it to control the follower arm.
+
+<hfoptions id="leader">
+<hfoption id="Command">
+
+```bash
+lerobot-calibrate \
+    --teleop.type=openarm_leader \
+    --teleop.port=can1 \
+    --teleop.id=my_openarm_leader
+```
+
+</hfoption>
+<hfoption id="API example">
+
+```python
+from lerobot.teleoperators.openarm_leader import OpenArmLeader, OpenArmLeaderConfig
+
+config = OpenArmLeaderConfig(
+    port="can1",
+    id="my_openarm_leader",
+    manual_control=True,  # Disable torque for manual movement
+)
+
+leader = OpenArmLeader(config)
+leader.connect()
+
+# Read current position (as action to send to follower)
+action = leader.get_action()
+print(action)
+
+leader.disconnect()
+```
+
+</hfoption>
+</hfoptions>
+
+### Teleoperation
+
+To teleoperate OpenArm with leader-follower control:
+
+```bash
+lerobot-teleoperate \
+    --robot.type=openarm_follower \
+    --robot.port=can0 \
+    --robot.side=right \
+    --robot.id=my_follower \
+    --teleop.type=openarm_leader \
+    --teleop.port=can1 \
+    --teleop.id=my_leader
+```
+
+### Bimanual Teleoperation
+
+To teleoperate a bimanual OpenArm setup with two leader and two follower arms:
+
+```bash
+lerobot-teleoperate \
+    --robot.type=bi_openarm_follower \
+    --robot.left_arm_config.port=can0 \
+    --robot.left_arm_config.side=left \
+    --robot.right_arm_config.port=can1 \
+    --robot.right_arm_config.side=right \
+    --robot.id=my_bimanual_follower \
+    --teleop.type=bi_openarm_leader \
+    --teleop.left_arm_config.port=can2 \
+    --teleop.right_arm_config.port=can3 \
+    --teleop.id=my_bimanual_leader
+```
+
+### Recording Data
+
+To record a dataset during teleoperation:
+
+```bash
+lerobot-record \
+    --robot.type=openarm_follower \
+    --robot.port=can0 \
+    --robot.side=right \
+    --robot.id=my_follower \
+    --teleop.type=openarm_leader \
+    --teleop.port=can1 \
+    --teleop.id=my_leader \
+    --repo-id=my_hf_username/my_openarm_dataset \
+    --fps=30 \
+    --num-episodes=10
+```
+
+## Configuration Options
+
+### Follower Configuration
+
+| Parameter             | Default   | Description                                                |
+| --------------------- | --------- | ---------------------------------------------------------- |
+| `port`                | -         | CAN interface (e.g., `can0`)                               |
+| `side`                | `None`    | Arm side: `"left"`, `"right"`, or `None` for custom limits |
+| `use_can_fd`          | `True`    | Enable CAN FD for higher data rates                        |
+| `can_bitrate`         | `1000000` | Nominal bitrate (1 Mbps)                                   |
+| `can_data_bitrate`    | `5000000` | CAN FD data bitrate (5 Mbps)                               |
+| `max_relative_target` | `None`    | Safety limit for relative target positions                 |
+| `position_kp`         | Per-joint | Position control proportional gains                        |
+| `position_kd`         | Per-joint | Position control derivative gains                          |
+
+### Leader Configuration
+
+| Parameter          | Default   | Description                         |
+| ------------------ | --------- | ----------------------------------- |
+| `port`             | -         | CAN interface (e.g., `can1`)        |
+| `manual_control`   | `True`    | Disable torque for manual movement  |
+| `use_can_fd`       | `True`    | Enable CAN FD for higher data rates |
+| `can_bitrate`      | `1000000` | Nominal bitrate (1 Mbps)            |
+| `can_data_bitrate` | `5000000` | CAN FD data bitrate (5 Mbps)        |
+
+## Motor Configuration
+
+OpenArm uses Damiao motors with the following default configuration:
+
+| Joint                       | Motor Type | Send ID | Recv ID |
+| --------------------------- | ---------- | ------- | ------- |
+| joint_1 (Shoulder pan)      | DM8009     | 0x01    | 0x11    |
+| joint_2 (Shoulder lift)     | DM8009     | 0x02    | 0x12    |
+| joint_3 (Shoulder rotation) | DM4340     | 0x03    | 0x13    |
+| joint_4 (Elbow flex)        | DM4340     | 0x04    | 0x14    |
+| joint_5 (Wrist roll)        | DM4310     | 0x05    | 0x15    |
+| joint_6 (Wrist pitch)       | DM4310     | 0x06    | 0x16    |
+| joint_7 (Wrist rotation)    | DM4310     | 0x07    | 0x17    |
+| gripper                     | DM4310     | 0x08    | 0x18    |
+
+## Troubleshooting
+
+### No Response from Motors
+
+1. Check power supply connections
+2. Verify CAN wiring (CAN-H, CAN-L, GND)
+3. Run diagnostics: `lerobot-setup-can --mode=test --interfaces=can0`
+4. See the [Damiao troubleshooting guide](./damiao#troubleshooting) for more details
+
+### CAN Interface Not Found
+
+Ensure the CAN interface is configured:
+
+```bash
+ip link show can0
+```
+
+## Resources
+
+- [OpenArm Website](https://openarm.dev)
+- [OpenArm Documentation](https://docs.openarm.dev)
+- [OpenArm GitHub](https://github.com/enactic/openarm)
+- [Safety Guide](https://docs.openarm.dev/getting-started/safety-guide)
+- [Damiao Motors and CAN Bus](./damiao)
diff --git a/lerobot/docs/source/peft_training.mdx b/lerobot/docs/source/peft_training.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..dd0b1007560ee7278dd9cd272c5d234bb3d51c3a
--- /dev/null
+++ b/lerobot/docs/source/peft_training.mdx
@@ -0,0 +1,62 @@
+# Parameter efficient fine-tuning with 🤗 PEFT
+
+[🤗 PEFT](https://github.com/huggingface/peft) (Parameter-Efficient Fine-Tuning) is a library for efficiently adapting
+large pretrained models such as pre-trained policies (e.g., SmolVLA, π₀, ...) to new tasks without training all
+of the model's parameters while yielding comparable performance.
+
+Install the `lerobot[peft]` optional package to enable PEFT support.
+
+To read about all the possible methods of adaption, please refer to the [🤗 PEFT docs](https://huggingface.co/docs/peft/index).
+
+## Training SmolVLA
+
+In this section we'll show you how to train a pre-trained SmolVLA policy with PEFT on the libero dataset.
+For brevity we're only training on the `libero_spatial` subset. We will use `lerobot/smolvla_base` as the model
+to parameter efficiently fine-tune:
+
+```
+lerobot-train \
+ --policy.path=lerobot/smolvla_base \
+ --policy.repo_id=your_hub_name/my_libero_smolvla \
+ --dataset.repo_id=HuggingFaceVLA/libero \
+ --policy.output_features=null \
+ --policy.input_features=null \
+ --policy.optimizer_lr=1e-3 \
+ --policy.scheduler_decay_lr=1e-4 \
+ --env.type=libero \
+ --env.task=libero_spatial \
+ --steps=100000 \
+ --batch_size=32 \
+ --peft.method_type=LORA \
+ --peft.r=64
+```
+
+Note the `--peft.method_type` parameter that let's you select which PEFT method to use. Here we use
+[LoRA](https://huggingface.co/docs/peft/main/en/package_reference/lora) (Low-Rank Adapter) which is probably the most
+popular fine-tuning method to date. Low-rank adaption means that we only fine-tune a matrix with comparably low rank
+instead of the full weight matrix. This rank can be specified using the `--peft.r` parameter. The higher the rank
+the closer you get to full fine-tuning
+
+There are more complex methods that have more parameters. These are not yet supported, feel free to raise an issue
+if you want to see a specific PEFT method supported.
+
+By default, PEFT will target the `q_proj` and `v_proj` layers of the LM expert in SmolVLA. It will also target the
+state and action projection matrices as they are most likely task-dependent. If you need to target different layers
+you can use `--peft.target_modules` to specify which layers to target. You can refer to the respective PEFT method's
+documentation to see what inputs are supported, (e.g., [LoRA's target_modules documentation](https://huggingface.co/docs/peft/main/en/package_reference/lora#peft.LoraConfig.target_modules)).
+Usually a list of suffixes or a regex are supported. For example, to target the MLPs of the `lm_expert` instead of
+the `q` and `v` projections, use:
+
+```
+--peft.target_modules='(model\.vlm_with_expert\.lm_expert\..*\.(down|gate|up)_proj|.*\.(state_proj|action_in_proj|action_out_proj|action_time_mlp_in|action_time_mlp_out))'
+```
+
+In case you need to fully fine-tune a layer instead of just adapting it, you can supply a list of layer suffixes
+to the `--peft.full_training_modules` parameter:
+
+```
+--peft.full_training_modules=["state_proj"]
+```
+
+The learning rate and the scheduled target learning rate can usually be scaled by a factor of 10 compared to the
+learning rate used for full fine-tuning (e.g., 1e-4 normal, so 1e-3 using LoRA).
diff --git a/lerobot/docs/source/phone_teleop.mdx b/lerobot/docs/source/phone_teleop.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..678783e7b22b572a518fd9b07616d75557642797
--- /dev/null
+++ b/lerobot/docs/source/phone_teleop.mdx
@@ -0,0 +1,195 @@
+# Phone
+
+Use your phone (iOS or Android) to control your robot.
+
+**In this guide you'll learn:**
+
+- How to connect an iOS/Android phone
+- How phone pose is mapped to robot end‑effector (EE) targets
+- How to tweak safety limits, gripper control, and IK settings
+
+To use phone to control your robot, install the relevant dependencies with:
+
+```bash
+pip install lerobot[phone]
+```
+
+## Get started
+
+### Supported platforms
+
+- iOS: Uses the HEBI Mobile I/O app (ARKit pose + buttons). Download the app first, open it and the examples will discover it on your network and stream the phone pose and inputs.
+- Android: Uses the `teleop` package (WebXR). When you start the Python process, it prints a local URL. Open the link on your phone, tap Start, then use Move to stream pose.
+
+Links:
+
+- Android WebXR library: [`teleop` on PyPI](https://pypi.org/project/teleop/)
+- iOS app: [HEBI Mobile I/O](https://docs.hebi.us/tools.html#mobile-io)
+
+### Phone orientation and controls
+
+- Orientation: hold the phone with the screen facing up and the top edge pointing in the same direction as the robot gripper. This ensures calibration aligns the phone’s frame with the robot frame so motion feels natural, see the image below for reference.
+- Enable/disable:
+  - iOS: Hold `B1` to enable teleoperation, release to stop. The first press captures a reference pose.
+  - Android: Press and hold the `Move` button, release to stop. The first press captures a reference pose.
+- Gripper control:
+  - iOS: Analog input `A3` controls the gripper as velocity input.
+  - Android: Buttons `A` and `B` act like increment/decrement (A opens, B closes). You can tune velocity in the `GripperVelocityToJoint` step.
+
+<img src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/phone_teleop.webp" alt="Phone teleop orientation" title="Phone teleop orientation" width="40%">
+
+### Step 1: Choose the platform
+
+Modify the examples to use `PhoneOS.IOS` or `PhoneOS.ANDROID` in `PhoneConfig`. The API is identical across platforms, only the input source differs. All examples are under `examples/` and have `phone_so100_*.py` variants.
+
+Teleoperation example:
+
+```python
+from lerobot.teleoperators.phone.config_phone import PhoneConfig, PhoneOS
+
+teleop_config = PhoneConfig(phone_os=PhoneOS.IOS)  # or PhoneOS.ANDROID
+teleop_device = Phone(teleop_config)
+```
+
+### Step 2: Connect and calibrate
+
+When `Phone(teleop_config)` is created and `connect()` is called, calibration is prompted automatically. Hold the phone in the orientation described above, then:
+
+- iOS: press and hold `B1` to capture the reference pose.
+- Android: press `Move` button on the WebXR page to capture the reference pose.
+
+Why calibrate? We capture the current pose so subsequent poses are expressed in a robot aligned frame. When you again press the button to enable control, the position is recaptured to avoid drift when your phone is repositioned while it was disabled.
+
+### Step 3: Run an example
+
+Run on of the examples scripts to teleoperate, record a dataset, replay a dataset or evaluate a policy.
+
+All scripts assume you configured your robot (e.g., SO-100 follower) and set the correct serial port.
+
+Additionally you need to **copy the URDF of the robot into the examples folder**. For the examples in this tutorial (using SO100/SO101), copy the `SO101` folder from the [SO-ARM100 repo](https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101) into the `examples/phone_to_so100/` directory, so that the URDF file path becomes `examples/phone_to_so100/SO101/so101_new_calib.urdf`.
+
+- Run this example to teleoperate:
+
+  ```bash
+  cd examples/phone_to_so100
+  python teleoperate.py
+  ```
+
+After running the example:
+
+- Android: after starting the script, open the printed local URL on your phone, tap Start, then press and hold Move.
+- iOS: open HEBI Mobile I/O first; B1 enables motion. A3 controls the gripper.
+
+Additionally you can customize mapping or safety limits by editing the processor steps shown in the examples. You can also remap inputs (e.g., use a different analog input) or adapt the pipeline to other robots (e.g., LeKiwi) by modifying the input and kinematics steps. More about this in the [Processors for Robots and Teleoperators](./processors_robots_teleop) guide.
+
+- Run this example to record a dataset, which saves absolute end effector observations and actions:
+
+  ```bash
+  cd examples/phone_to_so100
+  python record.py
+  ```
+
+- Run this example to replay recorded episodes:
+
+  ```bash
+  cd examples/phone_to_so100
+  python replay.py
+  ```
+
+- Run this example to evaluate a pretrained policy:
+
+  ```bash
+  cd examples/phone_to_so100
+  python evaluate.py
+  ```
+
+### Important pipeline steps and options
+
+- Kinematics are used in multiple steps. We use [Placo](https://github.com/Rhoban/placo) which is a wrapper around Pinocchio for handling our kinematics. We construct the kinematics object by passing the robot's URDF and target frame. We set `target_frame_name` to the gripper frame.
+
+  ```python
+  kinematics_solver = RobotKinematics(
+    urdf_path="./SO101/so101_new_calib.urdf",
+    target_frame_name="gripper_frame_link",
+    joint_names=list(robot.bus.motors.keys()),
+  )
+
+  ```
+
+- The `MapPhoneActionToRobotAction` step converts the calibrated phone pose and inputs into target deltas and gripper commands, below is shown what the step outputs.
+
+  ```python
+  action["enabled"] = enabled
+        action["target_x"] = -pos[1] if enabled else 0.0
+        action["target_y"] = pos[0] if enabled else 0.0
+        action["target_z"] = pos[2] if enabled else 0.0
+        action["target_wx"] = rotvec[1] if enabled else 0.0
+        action["target_wy"] = rotvec[0] if enabled else 0.0
+        action["target_wz"] = -rotvec[2] if enabled else 0.0
+        action["gripper_vel"] = gripper_vel  # Still send gripper action when disabled
+  ```
+
+- The `EEReferenceAndDelta` step converts target deltas to an absolute desired EE pose, storing a reference on enable, the `end_effector_step_sizes` are the step sizes for the EE pose and can be modified to change the motion speed.
+
+  ```python
+  EEReferenceAndDelta(
+      kinematics=kinematics_solver,
+      end_effector_step_sizes={"x": 0.5, "y": 0.5, "z": 0.5},
+      motor_names=list(robot.bus.motors.keys()),
+      use_latched_reference=True,
+  ),
+  ```
+
+- The `EEBoundsAndSafety` step clamps EE motion to a workspace and checks for large ee step jumps to ensure safety. The `end_effector_bounds` are the bounds for the EE pose and can be modified to change the workspace. The `max_ee_step_m` are the step limits for the EE pose and can be modified to change the safety limits.
+
+  ```python
+  EEBoundsAndSafety(
+      end_effector_bounds={"min": [-1.0, -1.0, -1.0], "max": [1.0, 1.0, 1.0]},
+      max_ee_step_m=0.10,
+  )
+  ```
+
+- The `GripperVelocityToJoint` step turns a velocity‑like gripper input into absolute gripper position using the current measured state. The `speed_factor` is the factor by which the velocity is multiplied.
+
+  ```python
+  GripperVelocityToJoint(speed_factor=20.0)
+  ```
+
+#### Different IK initial guesses
+
+We use different IK initial guesses in the kinematic steps. As initial guess either the current measured joints or the previous IK solution is used.
+
+- Closed loop (used in record/eval): sets `initial_guess_current_joints=True` so IK starts from the measured joints each frame.
+
+  ```python
+  InverseKinematicsEEToJoints(
+      kinematics=kinematics_solver,
+      motor_names=list(robot.bus.motors.keys()),
+      initial_guess_current_joints=True,  # closed loop
+  )
+  ```
+
+- Open loop (used in replay): sets `initial_guess_current_joints=False` so IK continues from the previous IK solution rather than the measured state. This preserves action stability when we replay without feedback.
+
+  ```python
+  InverseKinematicsEEToJoints(
+      kinematics=kinematics_solver,
+      motor_names=list(robot.bus.motors.keys()),
+      initial_guess_current_joints=False,  # open loop
+  )
+  ```
+
+### Pipeline steps explained
+
+- MapPhoneActionToRobotAction: converts calibrated phone pose and inputs into target deltas and a gripper command. Motion is gated by an enable signal (B1 on iOS, Move on Android).
+- EEReferenceAndDelta: latches a reference EE pose on enable and combines it with target deltas to produce an absolute desired EE pose each frame. When disabled, it keeps sending the last commanded pose.
+- EEBoundsAndSafety: clamps the EE pose to a workspace and rate‑limits jumps for safety. Also declares `action.ee.*` features.
+- InverseKinematicsEEToJoints: turns an EE pose into joint positions with IK. `initial_guess_current_joints=True` is recommended for closed‑loop control; set `False` for open‑loop replay for stability.
+- GripperVelocityToJoint: integrates a velocity‑like gripper input into an absolute gripper position using the current measured state.
+- ForwardKinematicsJointsToEE: computes `observation.state.ee.*` from observed joints for logging and training on EE state.
+
+### Troubleshooting
+
+- iOS not discovered: ensure HEBI Mobile I/O is open and your laptop/phone are on the same network.
+- Android URL not reachable: check local you used `https` instead of `http`, use the exact IP printed by the script and allow your browser to enter and ignore the certificate issue.
+- Motion feels inverted: adjust the sign flips in `MapPhoneActionToRobotAction` or swap axes to match your setup.
diff --git a/lerobot/docs/source/pi0.mdx b/lerobot/docs/source/pi0.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..be7792b2819639167a968940b69ada47b653fd04
--- /dev/null
+++ b/lerobot/docs/source/pi0.mdx
@@ -0,0 +1,96 @@
+# π₀ (Pi0)
+
+π₀ is a **Vision-Language-Action model for general robot control**, from Physical Intelligence. The LeRobot implementation is adapted from their open source [OpenPI](https://github.com/Physical-Intelligence/openpi) repository.
+
+## Model Overview
+
+π₀ represents a breakthrough in robotics as the first general-purpose robot foundation model developed by [Physical Intelligence](https://www.physicalintelligence.company/blog/pi0). Unlike traditional robot programs that are narrow specialists programmed for repetitive motions, π₀ is designed to be a generalist policy that can understand visual inputs, interpret natural language instructions, and control a variety of different robots across diverse tasks.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/lerobot-pi0%20(1).png"
+  alt="An overview of Pi0"
+  width="85%"
+/>
+
+### The Vision for Physical Intelligence
+
+As described by Physical Intelligence, while AI has achieved remarkable success in digital domains, from chess-playing to drug discovery, human intelligence still dramatically outpaces AI in the physical world. To paraphrase Moravec's paradox, winning a game of chess represents an "easy" problem for AI, but folding a shirt or cleaning up a table requires solving some of the most difficult engineering problems ever conceived. π₀ represents a first step toward developing artificial physical intelligence that enables users to simply ask robots to perform any task they want, just like they can with large language models.
+
+### Architecture and Approach
+
+π₀ combines several key innovations:
+
+- **Flow Matching**: Uses a novel method to augment pre-trained VLMs with continuous action outputs via flow matching (a variant of diffusion models)
+- **Cross-Embodiment Training**: Trained on data from 8 distinct robot platforms including UR5e, Bimanual UR5e, Franka, Bimanual Trossen, Bimanual ARX, Mobile Trossen, and Mobile Fibocom
+- **Internet-Scale Pre-training**: Inherits semantic knowledge from a pre-trained 3B parameter Vision-Language Model
+- **High-Frequency Control**: Outputs motor commands at up to 50 Hz for real-time dexterous manipulation
+
+## Installation Requirements
+
+1. Install LeRobot by following our [Installation Guide](./installation).
+2. Install Pi0 dependencies by running:
+
+   ```bash
+   pip install -e ".[pi]"
+   ```
+
+## Training Data and Capabilities
+
+π₀ is trained on the largest robot interaction dataset to date, combining three key data sources:
+
+1. **Internet-Scale Pre-training**: Vision-language data from the web for semantic understanding
+2. **Open X-Embodiment Dataset**: Open-source robot manipulation datasets
+3. **Physical Intelligence Dataset**: Large and diverse dataset of dexterous tasks across 8 distinct robots
+
+## Usage
+
+To use π₀ in LeRobot, specify the policy type as:
+
+```python
+policy.type=pi0
+```
+
+## Training
+
+For training π₀, you can use the standard LeRobot training script with the appropriate configuration:
+
+```bash
+lerobot-train \
+    --dataset.repo_id=your_dataset \
+    --policy.type=pi0 \
+    --output_dir=./outputs/pi0_training \
+    --job_name=pi0_training \
+    --policy.pretrained_path=lerobot/pi0_base \
+    --policy.repo_id=your_repo_id \
+    --policy.compile_model=true \
+    --policy.gradient_checkpointing=true \
+    --policy.dtype=bfloat16 \
+    --policy.freeze_vision_encoder=false \
+    --policy.train_expert_only=false \
+    --steps=3000 \
+    --policy.device=cuda \
+    --batch_size=32
+```
+
+### Key Training Parameters
+
+- **`--policy.compile_model=true`**: Enables model compilation for faster training
+- **`--policy.gradient_checkpointing=true`**: Reduces memory usage significantly during training
+- **`--policy.dtype=bfloat16`**: Use mixed precision training for efficiency
+- **`--batch_size=32`**: Batch size for training, adapt this based on your GPU memory
+- **`--policy.pretrained_path=lerobot/pi0_base`**: The base π₀ model you want to finetune, options are:
+  - [lerobot/pi0_base](https://huggingface.co/lerobot/pi0_base)
+  - [lerobot/pi0_libero](https://huggingface.co/lerobot/pi0_libero) (specifically trained on the Libero dataset)
+
+### Training Parameters Explained
+
+| Parameter               | Default | Description                                 |
+| ----------------------- | ------- | ------------------------------------------- |
+| `freeze_vision_encoder` | `false` | Do not freeze the vision encoder            |
+| `train_expert_only`     | `false` | Do not freeze the VLM, train all parameters |
+
+**💡 Tip**: Setting `train_expert_only=true` freezes the VLM and trains only the action expert and projections, allowing finetuning with reduced memory usage.
+
+## License
+
+This model follows the **Apache 2.0 License**, consistent with the original [OpenPI repository](https://github.com/Physical-Intelligence/openpi).
diff --git a/lerobot/docs/source/pi05.mdx b/lerobot/docs/source/pi05.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..f586f0dc15b8fd49283640660cf1b7852346724a
--- /dev/null
+++ b/lerobot/docs/source/pi05.mdx
@@ -0,0 +1,118 @@
+# π₀.₅ (Pi05) Policy
+
+π₀.₅ is a **Vision-Language-Action model with open-world generalization**, from Physical Intelligence. The LeRobot implementation is adapted from their open source [OpenPI](https://github.com/Physical-Intelligence/openpi) repository.
+
+## Model Overview
+
+π₀.₅ represents a significant evolution from π₀, developed by [Physical Intelligence](https://www.physicalintelligence.company/blog/pi05) to address a big challenge in robotics: **open-world generalization**. While robots can perform impressive tasks in controlled environments, π₀.₅ is designed to generalize to entirely new environments and situations that were never seen during training.
+
+### The Generalization Challenge
+
+As Physical Intelligence explains, the fundamental challenge isn't performing tasks of agility or dexterity, but generalization, the ability to correctly perform tasks in new settings with new objects. Consider a robot cleaning different homes: each home has different objects in different places. Generalization must occur at multiple levels:
+
+- **Physical Level**: Understanding how to pick up a spoon (by the handle) or plate (by the edge), even with unseen objects in cluttered environments
+- **Semantic Level**: Understanding task semantics, where to put clothes and shoes (laundry hamper, not on the bed), and what tools are appropriate for cleaning spills
+- **Environmental Level**: Adapting to "messy" real-world environments like homes, grocery stores, offices, and hospitals
+
+### Co-Training on Heterogeneous Data
+
+The breakthrough innovation in π₀.₅ is **co-training on heterogeneous data sources**. The model learns from:
+
+1. **Multimodal Web Data**: Image captioning, visual question answering, object detection
+2. **Verbal Instructions**: Humans coaching robots through complex tasks step-by-step
+3. **Subtask Commands**: High-level semantic behavior labels (e.g., "pick up the pillow" for an unmade bed)
+4. **Cross-Embodiment Robot Data**: Data from various robot platforms with different capabilities
+5. **Multi-Environment Data**: Static robots deployed across many different homes
+6. **Mobile Manipulation Data**: ~400 hours of mobile robot demonstrations
+
+This diverse training mixture creates a "curriculum" that enables generalization across physical, visual, and semantic levels simultaneously.
+
+## Installation Requirements
+
+1. Install LeRobot by following our [Installation Guide](./installation).
+2. Install Pi0.5 dependencies by running:
+
+   ```bash
+   pip install -e ".[pi]"
+   ```
+
+## Usage
+
+To use π₀.₅ in your LeRobot configuration, specify the policy type as:
+
+```python
+policy.type=pi05
+```
+
+## Training
+
+### Training Command Example
+
+Here's a complete training command for finetuning the base π₀.₅ model on your own dataset:
+
+```bash
+lerobot-train \
+    --dataset.repo_id=your_dataset \
+    --policy.type=pi05 \
+    --output_dir=./outputs/pi05_training \
+    --job_name=pi05_training \
+    --policy.repo_id=your_repo_id \
+    --policy.pretrained_path=lerobot/pi05_base \
+    --policy.compile_model=true \
+    --policy.gradient_checkpointing=true \
+    --wandb.enable=true \
+    --policy.dtype=bfloat16 \
+    --policy.freeze_vision_encoder=false \
+    --policy.train_expert_only=false \
+    --steps=3000 \
+    --policy.device=cuda \
+    --batch_size=32
+```
+
+### Key Training Parameters
+
+- **`--policy.compile_model=true`**: Enables model compilation for faster training
+- **`--policy.gradient_checkpointing=true`**: Reduces memory usage significantly during training
+- **`--policy.dtype=bfloat16`**: Use mixed precision training for efficiency
+- **`--batch_size=32`**: Batch size for training, adapt this based on your GPU memory
+- **`--policy.pretrained_path=lerobot/pi05_base`**: The base π₀.₅ model you want to finetune, options are:
+  - [lerobot/pi05_base](https://huggingface.co/lerobot/pi05_base)
+  - [lerobot/pi05_libero](https://huggingface.co/lerobot/pi05_libero) (specifically trained on the Libero dataset)
+
+### Training Parameters Explained
+
+| Parameter               | Default | Description                                 |
+| ----------------------- | ------- | ------------------------------------------- |
+| `freeze_vision_encoder` | `false` | Do not freeze the vision encoder            |
+| `train_expert_only`     | `false` | Do not freeze the VLM, train all parameters |
+
+**💡 Tip**: Setting `train_expert_only=true` freezes the VLM and trains only the action expert and projections, allowing finetuning with reduced memory usage.
+
+If your dataset is not converted with `quantiles`, you can convert it with the following command:
+
+```bash
+python src/lerobot/datasets/v30/augment_dataset_quantile_stats.py \
+    --repo-id=your_dataset \
+```
+
+Or train pi05 with this normalization mapping: `--policy.normalization_mapping='{"ACTION": "MEAN_STD", "STATE": "MEAN_STD", "VISUAL": "IDENTITY"}'`
+
+## Performance Results
+
+### Libero Benchmark Results
+
+π₀.₅ has demonstrated strong performance on the Libero benchmark suite. To compare and test its LeRobot implementation, we finetuned the libero base model for an additional 6k steps on the Libero dataset and compared the results to the OpenPI reference results.
+
+| Benchmark          | LeRobot Implementation | OpenPI Reference |
+| ------------------ | ---------------------- | ---------------- |
+| **Libero Spatial** | 97.0%                  | 98.8%            |
+| **Libero Object**  | 99.0%                  | 98.2%            |
+| **Libero Goal**    | 98.0%                  | 98.0%            |
+| **Libero 10**      | 96.0%                  | 92.4%            |
+| **Average**        | 97.5%                  | 96.85%           |
+
+These results demonstrate π₀.₅'s strong generalization capabilities across diverse robotic manipulation tasks. To reproduce these results, you can follow the instructions in the [Libero](https://huggingface.co/docs/lerobot/libero) section.
+
+## License
+
+This model follows the **Apache 2.0 License**, consistent with the original [OpenPI repository](https://github.com/Physical-Intelligence/openpi).
diff --git a/lerobot/docs/source/pi0fast.mdx b/lerobot/docs/source/pi0fast.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..f7272acc5477da9a583be6a935eba027f9fcb3d9
--- /dev/null
+++ b/lerobot/docs/source/pi0fast.mdx
@@ -0,0 +1,241 @@
+# π₀-FAST (Pi0-FAST)
+
+π₀-FAST is a **Vision-Language-Action model for general robot control** that uses autoregressive next-token prediction to model continuous robot actions.
+
+## Model Overview
+
+π₀-FAST combines the power of Vision-Language Models with a novel action tokenization approach called **FAST (Frequency-space Action Sequence Tokenization)**. This enables training autoregressive VLAs on highly dexterous tasks that are impossible with standard binning-based discretization, while training **up to 5x faster** than diffusion-based approaches like π₀.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/lerobot-pifast.png"
+  alt="An overview of Pi0-FAST"
+  width="85%"
+/>
+
+### Why FAST?
+
+Standard approaches for robot action tokenization use simple per-dimension, per-timestep binning schemes. While passable for simple behaviors, this rapidly breaks down for complex and dexterous skills that require precision and high-frequency control.
+
+FAST solves this by compressing action sequences using signal processing techniques, resulting in a dense sequence of action tokens that can be predicted autoregressively—just like language tokens.
+
+### How FAST Tokenization Works
+
+The FAST tokenizer compresses action sequences through the following steps:
+
+1. **Normalize**: Take a continuous action chunk of shape `(H, D)` where `H` is the horizon and `D` is the action dimension. Normalize using one of the supported normalization methods (Quantiles recommended to handle outliers).
+
+2. **Discrete Cosine Transform (DCT)**: Apply DCT (via scipy) to each action dimension separately. DCT is a compression algorithm commonly used in image and audio codecs (JPEG, MP3).
+
+3. **Quantization**: Round and remove insignificant coefficients for each action dimension, producing a sparse frequency matrix.
+
+4. **Flatten**: Flatten the matrix into a 1D vector, with low-frequency components first.
+
+5. **Byte Pair Encoding (BPE)**: Train a BPE tokenizer to compress the DCT coefficients into dense action tokens, typically achieving **10x compression** over prior tokenization approaches.
+
+This approach can transform **any existing VLM** into a VLA by training it to predict these FAST tokens.
+
+## Installation Requirements
+
+1. Install LeRobot by following our [Installation Guide](./installation).
+2. Install π₀-FAST dependencies by running:
+
+   ```bash
+   pip install -e ".[pi]"
+   ```
+
+## Training a Custom FAST Tokenizer
+
+You have two options for the FAST tokenizer:
+
+1. **Use the pre-trained tokenizer**: The `lerobot/fast-action-tokenizer` tokenizer was trained on 1M+ real robot action sequences and works as a general-purpose tokenizer.
+
+2. **Train your own tokenizer**: For maximum performance on your specific dataset, you can finetune the tokenizer on your own data.
+
+### Training Your Own Tokenizer
+
+```bash
+lerobot-train-tokenizer \
+    --repo_id "user/my-lerobot-dataset" \
+    --action_horizon 10 \
+    --encoded_dims "0:6" \
+    --vocab_size 1024 \
+    --scale 10.0 \
+    --normalization_mode QUANTILES \
+    --output_dir "./my_fast_tokenizer" \
+    --push_to_hub \
+    --hub_repo_id "username/my-action-tokenizer"
+```
+
+### Key Tokenizer Parameters
+
+| Parameter              | Description                                                                       | Default      |
+| ---------------------- | --------------------------------------------------------------------------------- | ------------ |
+| `--repo_id`            | LeRobot dataset repository ID                                                     | Required     |
+| `--action_horizon`     | Number of future actions in each chunk                                            | `10`         |
+| `--encoded_dims`       | Comma-separated dimension ranges to encode (e.g., `"0:6,7:23"`)                   | `"0:6,7:23"` |
+| `--vocab_size`         | BPE vocabulary size                                                               | `1024`       |
+| `--scale`              | DCT scaling factor for quantization                                               | `10.0`       |
+| `--normalization_mode` | Normalization mode (`MEAN_STD`, `MIN_MAX`, `QUANTILES`, `QUANTILE10`, `IDENTITY`) | `QUANTILES`  |
+| `--sample_fraction`    | Fraction of chunks to sample per episode                                          | `0.1`        |
+
+## Usage
+
+To use π₀-FAST in LeRobot, specify the policy type as:
+
+```python
+policy.type=pi0_fast
+```
+
+## Training
+
+For training π₀-FAST, you can use the LeRobot training script:
+
+```bash
+lerobot-train \
+    --dataset.repo_id=your_dataset \
+    --policy.type=pi0_fast \
+    --output_dir=./outputs/pi0fast_training \
+    --job_name=pi0fast_training \
+    --policy.pretrained_path=lerobot/pi0_fast_base \
+    --policy.dtype=bfloat16 \
+    --policy.gradient_checkpointing=true \
+    --policy.chunk_size=10 \
+    --policy.n_action_steps=10 \
+    --policy.max_action_tokens=256 \
+    --steps=100000 \
+    --batch_size=4 \
+    --policy.device=cuda
+```
+
+### Key Training Parameters
+
+| Parameter                              | Description                                        | Default                         |
+| -------------------------------------- | -------------------------------------------------- | ------------------------------- |
+| `--policy.gradient_checkpointing=true` | Reduces memory usage significantly during training | `false`                         |
+| `--policy.dtype=bfloat16`              | Use mixed precision training for efficiency        | `float32`                       |
+| `--policy.chunk_size`                  | Number of action steps to predict (action horizon) | `50`                            |
+| `--policy.n_action_steps`              | Number of action steps to execute                  | `50`                            |
+| `--policy.max_action_tokens`           | Maximum number of FAST tokens per action chunk     | `256`                           |
+| `--policy.action_tokenizer_name`       | FAST tokenizer to use                              | `lerobot/fast-action-tokenizer` |
+| `--policy.compile_model=true`          | Enable torch.compile for faster training           | `false`                         |
+
+## Inference
+
+### KV-Caching for Fast Inference
+
+π₀-FAST supports **KV-caching**, a widely used optimization in LLM inference. This caches the key-value pairs from the attention mechanism, avoiding redundant computation during autoregressive decoding.
+
+```python
+# KV-caching is enabled by default
+policy.use_kv_cache=true
+```
+
+### Inference Example
+
+```python
+from lerobot.policies.pi0_fast import PI0FastPolicy, PI0FastConfig
+
+# Load the policy
+policy = PI0FastPolicy.from_pretrained("your-model-path")
+
+# During inference
+actions = policy.predict_action_chunk(batch)
+```
+
+## Model Architecture
+
+π₀-FAST uses a PaliGemma-based architecture:
+
+- **Vision Encoder**: SigLIP vision tower for image understanding
+- **Language Model**: Gemma 2B for processing language instructions and predicting action tokens
+
+The model takes images, text instructions, and robot state as input, and outputs discrete FAST tokens that are decoded back to continuous actions.
+
+## Configuration Options
+
+| Parameter            | Description                                     | Default    |
+| -------------------- | ----------------------------------------------- | ---------- |
+| `paligemma_variant`  | VLM backbone variant (`gemma_300m`, `gemma_2b`) | `gemma_2b` |
+| `max_state_dim`      | Maximum state vector dimension (padded)         | `32`       |
+| `max_action_dim`     | Maximum action vector dimension (padded)        | `32`       |
+| `temperature`        | Sampling temperature (0.0 for greedy)           | `0.0`      |
+| `max_decoding_steps` | Maximum decoding steps                          | `256`      |
+| `use_kv_cache`       | Enable KV caching for faster inference          | `true`     |
+
+## Comparison with π₀
+
+| Feature               | π₀                        | π₀-FAST                      |
+| --------------------- | ------------------------- | ---------------------------- |
+| Action Representation | Flow Matching (Diffusion) | Autoregressive Tokens (FAST) |
+| Training Speed        | 1x                        | **5x faster**                |
+| Dexterity             | High                      | High                         |
+| Inference Method      | Iterative Denoising       | Autoregressive Decoding      |
+| KV-Caching            | N/A                       | Supported                    |
+
+## Reproducing π₀Fast results
+
+We reproduce the results of π₀Fast on the LIBERO benchmark using the LeRobot implementation. We take the LeRobot PiFast base model [lerobot/pi0fast-base](https://huggingface.co/lerobot/pi0fast-base) and finetune for an additional 40kk steps in bfloat16, with batch size of 256 on 8 H100 GPUs using the [HuggingFace LIBERO dataset](https://huggingface.co/datasets/HuggingFaceVLA/libero).
+
+The finetuned model can be found here:
+
+- **π₀Fast LIBERO**: [lerobot/pi0fast-libero](https://huggingface.co/lerobot/pi0fast-libero)
+
+With the following training command:
+
+```bash
+lerobot-train \
+  --dataset.repo_id=lerobot/libero \
+  --output_dir=outputs/libero_pi0fast \
+  --job_name=libero_pi0fast \
+  --policy.path=lerobot/pi0fast_base \
+  --policy.dtype=bfloat16 \
+  --steps=100000 \
+  --save_freq=20000 \
+  --batch_size=4 \
+  --policy.device=cuda \
+  --policy.scheduler_warmup_steps=4000 \
+  --policy.scheduler_decay_steps=100000 \
+  --policy.scheduler_decay_lr=1e-5 \
+  --policy.gradient_checkpointing=true \
+  --policy.chunk_size=10 \
+  --policy.n_action_steps=10 \
+  --policy.max_action_tokens=256 \
+  --policy.empty_cameras=1 \
+```
+
+We then evaluate the finetuned model using the LeRobot LIBERO implementation, by running the following command:
+
+```bash
+tasks="libero_object,libero_spatial,libero_goal,libero_10"
+lerobot-eval \
+  --policy.path=lerobot/pi0fast-libero \
+  --policy.max_action_tokens=256 \
+  --env.type=libero \
+  --policy.gradient_checkpointing=false \
+  --env.task=${tasks} \
+  --eval.batch_size=1 \
+  --eval.n_episodes=1 \
+  --rename_map='{"observation.images.image":"observation.images.base_0_rgb","observation.images.image2":"observation.images.left_wrist_0_rgb"}'
+```
+
+**Note:** We set `n_action_steps=10`, similar to the original OpenPI implementation.
+
+### Results
+
+We obtain the following results on the LIBERO benchmark:
+
+| Model       | LIBERO Spatial | LIBERO Object | LIBERO Goal | LIBERO 10 | Average  |
+| ----------- | -------------- | ------------- | ----------- | --------- | -------- |
+| **π₀-fast** | 70.0           | 100.0         | 100.0       | 60.0      | **82.5** |
+
+The full evaluation output folder, including videos, is available [here](https://drive.google.com/drive/folders/1HXpwPTRm4hx6g1sF2P7OOqGG0TwPU7LQ?usp=sharing)
+
+## License
+
+This model follows the **Apache 2.0 License**, consistent with the original [OpenPI repository](https://github.com/Physical-Intelligence/openpi).
+
+## References
+
+- [FAST: Efficient Robot Action Tokenization](https://www.physicalintelligence.company/research/fast) - Physical Intelligence Blog
+- [OpenPI Repository](https://github.com/Physical-Intelligence/openpi) - Original implementation
+- [FAST Tokenizer on Hugging Face](https://huggingface.co/physical-intelligence/fast) - Pre-trained tokenizer
diff --git a/lerobot/docs/source/policy_act_README.md b/lerobot/docs/source/policy_act_README.md
new file mode 100644
index 0000000000000000000000000000000000000000..371a9136fe7aa3a69b496739a46d9946240eaed2
--- /dev/null
+++ b/lerobot/docs/source/policy_act_README.md
@@ -0,0 +1,14 @@
+## Paper
+
+https://tonyzhaozh.github.io/aloha
+
+## Citation
+
+```bibtex
+@article{zhao2023learning,
+  title={Learning fine-grained bimanual manipulation with low-cost hardware},
+  author={Zhao, Tony Z and Kumar, Vikash and Levine, Sergey and Finn, Chelsea},
+  journal={arXiv preprint arXiv:2304.13705},
+  year={2023}
+}
+```
diff --git a/lerobot/docs/source/policy_diffusion_README.md b/lerobot/docs/source/policy_diffusion_README.md
new file mode 100644
index 0000000000000000000000000000000000000000..9ec934add67dff09d5f650dfac8401d665e9b73a
--- /dev/null
+++ b/lerobot/docs/source/policy_diffusion_README.md
@@ -0,0 +1,14 @@
+## Paper
+
+https://diffusion-policy.cs.columbia.edu
+
+## Citation
+
+```bibtex
+@article{chi2024diffusionpolicy,
+	author = {Cheng Chi and Zhenjia Xu and Siyuan Feng and Eric Cousineau and Yilun Du and Benjamin Burchfiel and Russ Tedrake and Shuran Song},
+	title ={Diffusion Policy: Visuomotor Policy Learning via Action Diffusion},
+	journal = {The International Journal of Robotics Research},
+	year = {2024},
+}
+```
diff --git a/lerobot/docs/source/policy_groot_README.md b/lerobot/docs/source/policy_groot_README.md
new file mode 100644
index 0000000000000000000000000000000000000000..efcd76ebeb617cf158b12a97c400fa0e22f0baf1
--- /dev/null
+++ b/lerobot/docs/source/policy_groot_README.md
@@ -0,0 +1,27 @@
+## Research Paper
+
+Paper: https://research.nvidia.com/labs/gear/gr00t-n1_5/
+
+## Repository
+
+Code: https://github.com/NVIDIA/Isaac-GR00T
+
+## Citation
+
+```bibtex
+@inproceedings{gr00tn1_2025,
+  archivePrefix = {arxiv},
+  eprint     = {2503.14734},
+  title      = {{GR00T} {N1}: An Open Foundation Model for Generalist Humanoid Robots},
+  author     = {NVIDIA and Johan Bjorck andFernando Castañeda, Nikita Cherniadev and Xingye Da and Runyu Ding and Linxi "Jim" Fan and Yu Fang and Dieter Fox and Fengyuan Hu and Spencer Huang and Joel Jang and Zhenyu Jiang and Jan Kautz and Kaushil Kundalia and Lawrence Lao and Zhiqi Li and Zongyu Lin and Kevin Lin and Guilin Liu and Edith Llontop and Loic Magne and Ajay Mandlekar and Avnish Narayan and Soroush Nasiriany and Scott Reed and You Liang Tan and Guanzhi Wang and Zu Wang and Jing Wang and Qi Wang and Jiannan Xiang and Yuqi Xie and Yinzhen Xu and Zhenjia Xu and Seonghyeon Ye and Zhiding Yu and Ao Zhang and Hao Zhang and Yizhou Zhao and Ruijie Zheng and Yuke Zhu},
+  month      = {March},
+  year       = {2025},
+  booktitle  = {ArXiv Preprint},
+}
+```
+
+## Additional Resources
+
+Blog: https://developer.nvidia.com/isaac/gr00t
+
+Hugging Face Model: https://huggingface.co/nvidia/GR00T-N1.5-3B
diff --git a/lerobot/docs/source/policy_smolvla_README.md b/lerobot/docs/source/policy_smolvla_README.md
new file mode 100644
index 0000000000000000000000000000000000000000..ee567ee83613785c4b3f1a1d07eeb3dde520fbe4
--- /dev/null
+++ b/lerobot/docs/source/policy_smolvla_README.md
@@ -0,0 +1,14 @@
+## Paper
+
+https://arxiv.org/abs/2506.01844
+
+## Citation
+
+```bibtex
+@article{shukor2025smolvla,
+  title={SmolVLA: A Vision-Language-Action Model for Affordable and Efficient Robotics},
+  author={Shukor, Mustafa and Aubakirova, Dana and Capuano, Francesco and Kooijmans, Pepijn and Palma, Steven and Zouitine, Adil and Aractingi, Michel and Pascal, Caroline and Russi, Martino and Marafioti, Andres and Alibert, Simon and Cord, Matthieu and Wolf, Thomas and Cadene, Remi},
+  journal={arXiv preprint arXiv:2506.01844},
+  year={2025}
+}
+```
diff --git a/lerobot/docs/source/policy_tdmpc_README.md b/lerobot/docs/source/policy_tdmpc_README.md
new file mode 100644
index 0000000000000000000000000000000000000000..804f166c8f9b232d47a1346fa1d869c6248d9775
--- /dev/null
+++ b/lerobot/docs/source/policy_tdmpc_README.md
@@ -0,0 +1,14 @@
+## Paper
+
+https://www.nicklashansen.com/td-mpc/
+
+## Citation
+
+```bibtex
+@inproceedings{Hansen2022tdmpc,
+	title={Temporal Difference Learning for Model Predictive Control},
+	author={Nicklas Hansen and Xiaolong Wang and Hao Su},
+	booktitle={ICML},
+	year={2022}
+}
+```
diff --git a/lerobot/docs/source/policy_vqbet_README.md b/lerobot/docs/source/policy_vqbet_README.md
new file mode 100644
index 0000000000000000000000000000000000000000..02f95b7c2f5e5d9ba0c0559e21c66a44dad04a2b
--- /dev/null
+++ b/lerobot/docs/source/policy_vqbet_README.md
@@ -0,0 +1,14 @@
+## Paper
+
+https://sjlee.cc/vq-bet/
+
+## Citation
+
+```bibtex
+@article{lee2024behavior,
+  title={Behavior generation with latent actions},
+  author={Lee, Seungjae and Wang, Yibin and Etukuru, Haritheja and Kim, H Jin and Shafiullah, Nur Muhammad Mahi and Pinto, Lerrel},
+  journal={arXiv preprint arXiv:2403.03181},
+  year={2024}
+}
+```
diff --git a/lerobot/docs/source/policy_walloss_README.md b/lerobot/docs/source/policy_walloss_README.md
new file mode 100644
index 0000000000000000000000000000000000000000..93c0ad3927776875df77d496bd171a0963146b9a
--- /dev/null
+++ b/lerobot/docs/source/policy_walloss_README.md
@@ -0,0 +1,45 @@
+# WALL-OSS
+
+This repository contains the Hugging Face port of [**WALL-OSS**](https://x2robot.com/en/research/68bc2cde8497d7f238dde690), a Vision-Language-Action model for cross-embodiment robotic control based on Qwen2.5-VL with flow matching/FAST action prediction.
+
+---
+
+## Model Overview
+
+| Feature            | Description                                           |
+| ------------------ | ----------------------------------------------------- |
+| Base Model         | Qwen2.5-VL (Vision-Language Model)                    |
+| Action Prediction  | Flow Matching (diffusion) or FAST (discrete tokens)   |
+| Architecture       | Mixture of Experts (MoE) with action-specific routing |
+| Multi-Modal Inputs | Vision (images/videos), Language, Proprioception      |
+
+---
+
+## Additional Resources
+
+Paper: https://arxiv.org/pdf/2509.11766
+
+Official Repository: https://github.com/X-Square-Robot/wall-x
+
+Hugging Face: https://huggingface.co/x-square-robot
+
+---
+
+## Citation
+
+If you use this work, please cite:
+
+```bibtex
+@article{zhai2025igniting,
+    title   = {Igniting VLMs Toward the Embodied Space},
+    author  = {Zhai, Andy and Liu, Brae and Fang, Bruno and Cai, Chalse and Ma, Ellie and Yin, Ethan and Wang, Hao and Zhou, Hugo and Wang, James and Shi, Lights and Liang, Lucy and Wang, Make and Wang, Qian and Gan, Roy and Yu, Ryan and Li, Shalfun and Liu, Starrick and Chen, Sylas and Chen, Vincent and Xu, Zach},
+    journal = {arXiv preprint arXiv:2509.11766},
+    year    = {2025}
+}
+```
+
+---
+
+## License
+
+This model follows the **Apache 2.0 License**, consistent with the original [WallX repository](https://github.com/X-Square-Robot/wall-x).
diff --git a/lerobot/docs/source/porting_datasets_v3.mdx b/lerobot/docs/source/porting_datasets_v3.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..46793265eb154419544460373a21f9098bdded41
--- /dev/null
+++ b/lerobot/docs/source/porting_datasets_v3.mdx
@@ -0,0 +1,321 @@
+# Porting Large Datasets to LeRobot Dataset v3.0
+
+This tutorial explains how to port large-scale robotic datasets to the LeRobot Dataset v3.0 format. We'll use the **DROID 1.0.1** dataset as our primary example, which demonstrates handling multi-terabyte datasets with thousands of shards across SLURM clusters.
+
+## File Organization: v2.1 vs v3.0
+
+Dataset v3.0 fundamentally changes how data is organized and stored:
+
+**v2.1 Structure (Episode-based)**:
+
+```
+dataset/
+├── data/chunk-000/episode_000000.parquet
+├── data/chunk-000/episode_000001.parquet
+├── videos/chunk-000/camera/episode_000000.mp4
+└── meta/episodes.jsonl
+```
+
+**v3.0 Structure (File-based)**:
+
+```
+dataset/
+├── data/chunk-000/file-000.parquet        # Multiple episodes per file
+├── videos/camera/chunk-000/file-000.mp4   # Consolidated video chunks
+└── meta/episodes/chunk-000/file-000.parquet  # Structured metadata
+```
+
+This transition from individual episode files to file-based chunks dramatically improves performance and reduces storage overhead.
+
+## What's New in Dataset v3.0
+
+Dataset v3.0 introduces significant improvements for handling large datasets:
+
+### 🏗️ **Enhanced File Organization**
+
+- **File-based structure**: Episodes are now grouped into chunked files rather than individual episode files
+- **Configurable file sizes**: for data and video files
+- **Improved storage efficiency**: Better compression and reduced overhead
+
+### 📊 **Modern Metadata Management**
+
+- **Parquet-based metadata**: Replaced JSON Lines with efficient parquet format
+- **Structured episode access**: Direct pandas DataFrame access via `dataset.meta.episodes`
+- **Per-episode statistics**: Enhanced statistics tracking at episode level
+
+### 🚀 **Performance Enhancements**
+
+- **Memory-mapped access**: Improved RAM usage through PyArrow memory mapping
+- **Faster loading**: Significantly reduced dataset initialization time
+- **Better scalability**: Designed for datasets with millions of episodes
+
+## Prerequisites
+
+Before porting large datasets, ensure you have:
+
+- **LeRobot installed** with v3.0 support. Follow our [Installation Guide](./installation).
+- **Sufficient storage**: Raw datasets can be very large (e.g., DROID requires 2TB)
+- **Cluster access** (recommended for large datasets): SLURM or similar job scheduler
+- **Dataset-specific dependencies**: For DROID, you'll need TensorFlow Dataset utilities
+
+## Understanding the DROID Dataset
+
+[DROID 1.0.1](https://droid-dataset.github.io/droid/the-droid-dataset) is an excellent example of a large-scale robotic dataset:
+
+- **Size**: 1.7TB (RLDS format), 8.7TB (raw data)
+- **Structure**: 2048 pre-defined TensorFlow dataset shards
+- **Content**: 76,000+ robot manipulation trajectories from Franka Emika Panda robots
+- **Scope**: Real-world manipulation tasks across multiple environments and objects
+- **Format**: Originally in TensorFlow Records/RLDS format, requiring conversion to LeRobot format
+- **Hosting**: Google Cloud Storage with public access via `gsutil`
+
+The dataset contains diverse manipulation demonstrations with:
+
+- Multiple camera views (wrist camera, exterior cameras)
+- Natural language task descriptions
+- Robot proprioceptive state and actions
+- Success/failure annotations
+
+### DROID Features Schema
+
+```python
+DROID_FEATURES = {
+    # Episode markers
+    "is_first": {"dtype": "bool", "shape": (1,)},
+    "is_last": {"dtype": "bool", "shape": (1,)},
+    "is_terminal": {"dtype": "bool", "shape": (1,)},
+
+    # Language instructions
+    "language_instruction": {"dtype": "string", "shape": (1,)},
+    "language_instruction_2": {"dtype": "string", "shape": (1,)},
+    "language_instruction_3": {"dtype": "string", "shape": (1,)},
+
+    # Robot state
+    "observation.state.gripper_position": {"dtype": "float32", "shape": (1,)},
+    "observation.state.cartesian_position": {"dtype": "float32", "shape": (6,)},
+    "observation.state.joint_position": {"dtype": "float32", "shape": (7,)},
+
+    # Camera observations
+    "observation.images.wrist_left": {"dtype": "image"},
+    "observation.images.exterior_1_left": {"dtype": "image"},
+    "observation.images.exterior_2_left": {"dtype": "image"},
+
+    # Actions
+    "action.gripper_position": {"dtype": "float32", "shape": (1,)},
+    "action.cartesian_position": {"dtype": "float32", "shape": (6,)},
+    "action.joint_position": {"dtype": "float32", "shape": (7,)},
+
+    # Standard LeRobot format
+    "observation.state": {"dtype": "float32", "shape": (8,)},  # joints + gripper
+    "action": {"dtype": "float32", "shape": (8,)},  # joints + gripper
+}
+```
+
+## Approach 1: Single Computer Porting
+
+### Step 1: Install Dependencies
+
+For DROID specifically:
+
+```bash
+pip install tensorflow
+pip install tensorflow_datasets
+```
+
+For other datasets, install the appropriate readers for your source format.
+
+### Step 2: Download Raw Data
+
+Download DROID from Google Cloud Storage using `gsutil`:
+
+```bash
+# Install Google Cloud SDK if not already installed
+# https://cloud.google.com/sdk/docs/install
+
+# Download the full RLDS dataset (1.7TB)
+gsutil -m cp -r gs://gresearch/robotics/droid/1.0.1 /your/data/
+
+# Or download just the 100-episode sample (2GB) for testing
+gsutil -m cp -r gs://gresearch/robotics/droid_100 /your/data/
+```
+
+> [!WARNING]
+> Large datasets require substantial time and storage:
+>
+> - **Full DROID (1.7TB)**: Several days to download depending on bandwidth
+> - **Processing time**: 7+ days for local porting of full dataset
+> - **Upload time**: 3+ days to push to Hugging Face Hub
+> - **Local storage**: ~400GB for processed LeRobot format
+
+### Step 3: Port the Dataset
+
+```bash
+python examples/port_datasets/port_droid.py \
+    --raw-dir /your/data/droid/1.0.1 \
+    --repo-id your_id/droid_1.0.1 \
+    --push-to-hub
+```
+
+### Development and Testing
+
+For development, you can port a single shard:
+
+```bash
+python examples/port_datasets/port_droid.py \
+    --raw-dir /your/data/droid/1.0.1 \
+    --repo-id your_id/droid_1.0.1_test \
+    --num-shards 2048 \
+    --shard-index 0
+```
+
+This approach works for smaller datasets or testing, but large datasets require cluster computing.
+
+## Approach 2: SLURM Cluster Porting (Recommended)
+
+For large datasets like DROID, parallel processing across multiple nodes dramatically reduces processing time.
+
+### Step 1: Install Cluster Dependencies
+
+```bash
+pip install datatrove  # Hugging Face's distributed processing library
+```
+
+### Step 2: Configure Your SLURM Environment
+
+Find your partition information:
+
+```bash
+sinfo --format="%R"  # List available partitions
+sinfo -N -p your_partition -h -o "%N cpus=%c mem=%m"  # Check resources
+```
+
+Choose a **CPU partition** - no GPU needed for dataset porting.
+
+### Step 3: Launch Parallel Porting Jobs
+
+```bash
+python examples/port_datasets/slurm_port_shards.py \
+    --raw-dir /your/data/droid/1.0.1 \
+    --repo-id your_id/droid_1.0.1 \
+    --logs-dir /your/logs \
+    --job-name port_droid \
+    --partition your_partition \
+    --workers 2048 \
+    --cpus-per-task 8 \
+    --mem-per-cpu 1950M
+```
+
+#### Parameter Guidelines
+
+- **`--workers`**: Number of parallel jobs (max 2048 for DROID's shard count)
+- **`--cpus-per-task`**: 8 CPUs recommended for frame encoding parallelization
+- **`--mem-per-cpu`**: ~16GB total RAM (8×1950M) for loading raw frames
+
+> [!TIP]
+> Start with fewer workers (e.g., 100) to test your cluster configuration before launching thousands of jobs.
+
+### Step 4: Monitor Progress
+
+Check running jobs:
+
+```bash
+squeue -u $USER
+```
+
+Monitor overall progress:
+
+```bash
+jobs_status /your/logs
+```
+
+Inspect individual job logs:
+
+```bash
+less /your/logs/port_droid/slurm_jobs/JOB_ID_WORKER_ID.out
+```
+
+Debug failed jobs:
+
+```bash
+failed_logs /your/logs/port_droid
+```
+
+### Step 5: Aggregate Shards
+
+Once all porting jobs complete:
+
+```bash
+python examples/port_datasets/slurm_aggregate_shards.py \
+    --repo-id your_id/droid_1.0.1 \
+    --logs-dir /your/logs \
+    --job-name aggr_droid \
+    --partition your_partition \
+    --workers 2048 \
+    --cpus-per-task 8 \
+    --mem-per-cpu 1950M
+```
+
+### Step 6: Upload to Hub
+
+```bash
+python examples/port_datasets/slurm_upload.py \
+    --repo-id your_id/droid_1.0.1 \
+    --logs-dir /your/logs \
+    --job-name upload_droid \
+    --partition your_partition \
+    --workers 50 \
+    --cpus-per-task 4 \
+    --mem-per-cpu 1950M
+```
+
+> [!NOTE]
+> Upload uses fewer workers (50) since it's network-bound rather than compute-bound.
+
+## Dataset v3.0 File Structure
+
+Your completed dataset will have this modern structure:
+
+```
+dataset/
+├── meta/
+│   ├── episodes/
+│   │   └── chunk-000/
+│   │       └── file-000.parquet    # Episode metadata
+│   ├── tasks.parquet               # Task definitions
+│   ├── stats.json                  # Aggregated statistics
+│   └── info.json                   # Dataset information
+├── data/
+│   └── chunk-000/
+│       └── file-000.parquet        # Consolidated episode data
+└── videos/
+    └── camera_key/
+        └── chunk-000/
+            └── file-000.mp4        # Consolidated video files
+```
+
+This replaces the old episode-per-file structure with efficient, optimally-sized chunks.
+
+## Migrating from Dataset v2.1
+
+If you have existing datasets in v2.1 format, use the migration tool:
+
+```bash
+python src/lerobot/datasets/v30/convert_dataset_v21_to_v30.py \
+    --repo-id your_id/existing_dataset
+```
+
+This automatically:
+
+- Converts file structure to v3.0 format
+- Migrates metadata from JSON Lines to parquet
+- Aggregates statistics and creates per-episode stats
+- Updates version information
+
+## Performance Benefits
+
+Dataset v3.0 provides significant improvements for large datasets:
+
+- **Faster loading**: 3-5x reduction in initialization time
+- **Memory efficiency**: Better RAM usage through memory mapping
+- **Scalable processing**: Handles millions of episodes efficiently
+- **Storage optimization**: Reduced file count and improved compression
diff --git a/lerobot/docs/source/processors_robots_teleop.mdx b/lerobot/docs/source/processors_robots_teleop.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..093a8e0e3084a55c4dbbb14d7f36e8de8fb49c71
--- /dev/null
+++ b/lerobot/docs/source/processors_robots_teleop.mdx
@@ -0,0 +1,151 @@
+# Processors for Robots and Teleoperators
+
+This guide shows how to build and modify processing pipelines that connect teleoperators (e.g., phone) to robots and datasets. Pipelines standardize conversions between different action/observation spaces so you can swap teleops and robots without rewriting glue code.
+
+We use the Phone to SO‑100 follower examples for concreteness, but the same patterns apply to other robots.
+
+**What you'll learn**
+
+- Absolute vs. relative EE control: What each means, trade‑offs, and how to choose for your task.
+- Three-pipeline pattern: How to map teleop actions → dataset actions → robot commands, and robot observations → dataset observations.
+- Adapters (`to_transition` / `to_output`): How these convert raw dicts to `EnvTransition` and back to reduce boilerplate.
+- Dataset feature contracts: How steps declare features via `transform_features(...)`, and how to aggregate/merge them for recording.
+- Choosing a representation: When to store joints, absolute EE poses, or relative EE deltas—and how that affects training.
+- Pipeline customization guidance: How to swap robots/URDFs safely and tune bounds, step sizes, and options like IK initialization.
+
+### Absolute vs relative EE control
+
+The examples in this guide use absolute end effector (EE) poses because they are easy to reason about. In practice, relative EE deltas or joint position are often preferred as learning features.
+
+With processors, you choose the learning features you want to use for your policy. This could be joints positions/velocities, absolute EE, or relative EE positions. You can also choose to store other features, such as joint torques, motor currents, etc.
+
+## Three pipelines
+
+We often compose three pipelines. Depending on your setup, some can be empty if action and observation spaces already match.
+Each of these pipelines handle different conversions between different action and observation spaces. Below is a quick explanation of each pipeline.
+
+1. Pipeline 1: Teleop action space → dataset action space (phone pose → EE targets)
+2. Pipeline 2: Dataset action space → robot command space (EE targets → joints)
+3. Pipeline 3: Robot observation space → dataset observation space (joints → EE pose)
+
+Below is an example of the three pipelines that we use in the phone to SO-100 follower examples:
+
+```python
+phone_to_robot_ee_pose_processor = RobotProcessorPipeline[RobotAction, RobotAction]( # teleop -> dataset action
+    steps=[
+        MapPhoneActionToRobotAction(platform=teleop_config.phone_os),
+        EEReferenceAndDelta(
+            kinematics=kinematics_solver, end_effector_step_sizes={"x": 0.5, "y": 0.5, "z": 0.5}, motor_names=list(robot.bus.motors.keys()),
+        ),
+        EEBoundsAndSafety(
+            end_effector_bounds={"min": [-1.0, -1.0, -1.0], "max": [1.0, 1.0, 1.0]}, max_ee_step_m=0.20,
+        ),
+        GripperVelocityToJoint(),
+    ],
+    to_transition=robot_action_to_transition,
+    to_output=transition_to_robot_action,
+)
+
+robot_ee_to_joints_processor = RobotProcessorPipeline[RobotAction, RobotAction]( # dataset action -> robot
+    steps=[
+        InverseKinematicsEEToJoints(
+            kinematics=kinematics_solver, motor_names=list(robot.bus.motors.keys()), initial_guess_current_joints=True,
+        ),
+    ],
+    to_transition=robot_action_to_transition,
+    to_output=transition_to_robot_action,
+)
+
+robot_joints_to_ee_pose = RobotProcessorPipeline[RobotObservation, RobotObservation]( # robot obs -> dataset obs
+    steps=[
+        ForwardKinematicsJointsToEE(kinematics=kinematics_solver, motor_names=list(robot.bus.motors.keys()))
+    ],
+    to_transition=observation_to_transition,
+    to_output=transition_to_observation,
+)
+```
+
+## Why to_transition / to_output
+
+To convert from robot/teleoperator to pipeline and back, we use the `to_transition` and `to_output` pipeline adapters.
+They standardize conversions to reduce boilerplate code, and form the bridge between the robot and teleoperators raw dictionaries and the pipeline’s `EnvTransition` format.
+In the phone to SO-100 follower examples we use the following adapters:
+
+- `robot_action_to_transition`: transforms the teleop action dict to a pipeline transition.
+- `transition_to_robot_action`: transforms the pipeline transition to a robot action dict.
+- `observation_to_transition`: transforms the robot observation dict to a pipeline transition.
+- `transition_to_observation`: transforms the pipeline transition to a observation dict.
+
+Checkout [src/lerobot/processor/converters.py](https://github.com/huggingface/lerobot/blob/main/src/lerobot/processor/converters.py) for more details.
+
+## Dataset feature contracts
+
+Dataset features are determined by the keys saved in the dataset. Each step can declare what features it modifies in a contract called `transform_features(...)`. Once you build a processor, the processor can then aggregate all of these features with `aggregate_pipeline_dataset_features()` and merge multiple feature dicts with `combine_feature_dicts(...)`.
+
+Below is and example of how we declare features with the `transform_features` method in the phone to SO-100 follower examples:
+
+```python
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We only use the ee pose in the dataset, so we don't need the joint positions
+        for n in self.motor_names:
+            features[PipelineFeatureType.ACTION].pop(f"{n}.pos", None)
+        # We specify the dataset features of this step that we want to be stored in the dataset
+        for k in ["x", "y", "z", "wx", "wy", "wz", "gripper_pos"]:
+            features[PipelineFeatureType.ACTION][f"ee.{k}"] = PolicyFeature(
+                type=FeatureType.STATE, shape=(1,)
+            )
+        return features
+```
+
+Here we declare what PolicyFeatures we modify in this step, so we know what features we can expect when we run the processor. These features can then be aggregated and used to create the dataset features.
+
+Below is an example of how we aggregate and merge features in the phone to SO-100 record example:
+
+```python
+features=combine_feature_dicts(
+        # Run the feature contract of the pipelines
+        # This tells you how the features would look like after the pipeline steps
+        aggregate_pipeline_dataset_features(
+            pipeline=phone_to_robot_ee_pose_processor,
+            initial_features=create_initial_features(action=phone.action_features), # <- Action features we can expect, these come from our teleop device (phone) and action processor
+            use_videos=True,
+        ),
+        aggregate_pipeline_dataset_features(
+            pipeline=robot_joints_to_ee_pose,
+            initial_features=create_initial_features(observation=robot.observation_features), # <- Observation features we can expect, these come from our robot and observation processor
+            use_videos=True,
+            patterns=["observation.state.ee"], # <- Here you could optionally filter the features we want to store in the dataset, with a specific pattern
+
+        ),
+    ),
+```
+
+How it works:
+
+- `aggregate_pipeline_dataset_features(...)`: applies `transform_features` across the pipeline and filters by patterns (images included when `use_videos=True`, and state features included when `patterns` is specified).
+- `combine_feature_dicts(...)`: combine multiple feature dicts.
+- Recording with `record_loop(...)` uses `build_dataset_frame(...)` to build frames consistent with `dataset.features` before we call `add_frame(...)` to add the frame to the dataset.
+
+## Guidance when customizing robot pipelines
+
+You can store any of the following features as your action/observation space:
+
+- Joint positions
+- Absolute EE poses
+- Relative EE deltas
+- Other features: joint velocity, torques, etc.
+
+Pick what you want to use for your policy action and observation space and configure/modify the pipelines and steps accordingly.
+
+### Different robots
+
+- You can easily reuse pipelines, for example to use another robot with phone teleop, modify the examples and swap the robot `RobotKinematics` (URDF) and `motor_names` to use your own robot with Phone teleop. Additionally you should ensure `target_frame_name` points to your gripper/wrist.
+
+### Safety first
+
+- When changing pipelines, start with tight bounds, implement safety steps when working with real robots.
+- Its advised to start with simulation first and then move to real robots.
+
+Thats it! We hope this guide helps you get started with customizing your robot pipelines, If you run into any issues at any point, jump into our [Discord community](https://discord.com/invite/s3KuuzsPFb) for support.
diff --git a/lerobot/docs/source/reachy2.mdx b/lerobot/docs/source/reachy2.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..1b868711a1c4b9f23aaeb682908fbbe8a9f868b0
--- /dev/null
+++ b/lerobot/docs/source/reachy2.mdx
@@ -0,0 +1,309 @@
+# Reachy 2
+
+Reachy 2 is an open-source humanoid robot made by Pollen Robotics, specifically designed for the development of embodied AI and real-world applications.
+Check out [Pollen Robotics website](https://www.pollen-robotics.com/reachy/), or access [Reachy 2 documentation](https://docs.pollen-robotics.com/) for more information on the platform!
+
+## Teleoperate Reachy 2
+
+Currently, there are two ways to teleoperate Reachy 2:
+
+- Pollen Robotics’ VR teleoperation (not included in LeRobot).
+- Robot-to-robot teleoperation (use one Reachy 2 to control another).
+
+## Reachy 2 Simulation
+
+**(Linux only)** You can run Reachy 2 in simulation (Gazebo or MuJoCo) using the provided [Docker image](https://hub.docker.com/r/pollenrobotics/reachy2_core).
+
+1. Install [Docker Engine](https://docs.docker.com/engine/).
+2. Run (for MuJoCo):
+
+```
+docker run --rm -it \
+  --name reachy \
+  --privileged \
+  --network host \
+  --ipc host \
+  --device-cgroup-rule='c 189:* rwm' \
+  --group-add audio \
+  -e ROS_DOMAIN_ID="$ROS_DOMAIN_ID" \
+  -e DISPLAY="$DISPLAY" \
+  -e RCUTILS_CONSOLE_OUTPUT_FORMAT="[{severity}]: {message}" \
+  -e REACHY2_CORE_SERVICE_FAKE="${REACHY2_CORE_SERVICE_FAKE:-true}" \
+  -v /dev:/dev \
+  -v "$HOME/.reachy_config":/home/reachy/.reachy_config_override \
+  -v "$HOME/.reachy.log":/home/reachy/.ros/log \
+  -v /usr/lib/x86_64-linux-gnu:/opt/host-libs \
+  --entrypoint /package/launch.sh \
+  pollenrobotics/reachy2_core:1.7.5.9_deploy \
+  start_rviz:=true start_sdk_server:=true mujoco:=true
+```
+
+> [!NOTE]
+> If MuJoCo runs slowly (low simulation frequency), append `-e LD_LIBRARY_PATH="/opt/host-libs:$LD_LIBRARY_PATH" \` to the previous command to improve performance:
+>
+> ```
+> docker run --rm -it \
+>   --name reachy \
+>   --privileged \
+>   --network host \
+>   --ipc host \
+>   --device-cgroup-rule='c 189:* rwm' \
+>   --group-add audio \
+>   -e ROS_DOMAIN_ID="$ROS_DOMAIN_ID" \
+>   -e DISPLAY="$DISPLAY" \
+>   -e RCUTILS_CONSOLE_OUTPUT_FORMAT="[{severity}]: {message}" \
+>   -e REACHY2_CORE_SERVICE_FAKE="${REACHY2_CORE_SERVICE_FAKE:-true}" \
+>   -e LD_LIBRARY_PATH="/opt/host-libs:$LD_LIBRARY_PATH" \
+>   -v /dev:/dev \
+>   -v "$HOME/.reachy_config":/home/reachy/.reachy_config_override \
+>   -v "$HOME/.reachy.log":/home/reachy/.ros/log \
+>   -v /usr/lib/x86_64-linux-gnu:/opt/host-libs \
+>   --entrypoint /package/launch.sh \
+>   pollenrobotics/reachy2_core:1.7.5.9_deploy \
+>   start_rviz:=true start_sdk_server:=true mujoco:=true
+> ```
+
+## Setup
+
+### Prerequisites
+
+- On your robot, check the **service images** meet the minimum versions:
+  - **reachy2-core >= 1.7.5.2**
+  - **webrtc >= 2.0.1.1**
+
+Then, if you want to use VR teleoperation:
+
+- Install the [Reachy 2 teleoperation application](https://docs.pollen-robotics.com/teleoperation/teleoperation-introduction/discover-teleoperation/).
+  Use version **>=v1.2.0**
+
+We recommend using two computers: one for teleoperation (Windows required) and another for recording with LeRobot.
+
+### Install LeRobot
+
+Follow the [installation instructions](https://github.com/huggingface/lerobot#installation) to install LeRobot.
+
+Install LeRobot with Reachy 2 dependencies:
+
+```bash
+pip install -e ".[reachy2]"
+```
+
+### (Optional but recommended) Install pollen_data_acquisition_server
+
+How you manage Reachy 2 recording sessions is up to you, but the **easiest** way is to use this server so you can control sessions directly from the VR teleoperation app.
+
+> **Note:** Currently, only the VR teleoperation application works as a client for this server, so this step primarily targets teleoperation. You’re free to develop custom clients to manage sessions to your needs.
+
+In your LeRobot environment, install the server from source:
+
+```bash
+git clone https://github.com/pollen-robotics/pollen_data_acquisition_server.git
+cd pollen_data_acquisition_server
+pip install -e .
+```
+
+Find the [pollen_data_acquisition_server documentation here](https://github.com/pollen-robotics/pollen_data_acquisition_server).
+
+## Step 1: Recording
+
+### Get Reachy 2 IP address
+
+Before starting teleoperation and data recording, find the [robot's IP address](https://docs.pollen-robotics.com/getting-started/setup-reachy2/connect-reachy2/).
+We strongly recommend connecting all devices (PC and robot) via **Ethernet**.
+
+### Launch recording
+
+There are two ways to manage recording sessions when using the Reachy 2 VR teleoperation application:
+
+- **Using the data acquisition server (recommended for VR teleop)**: The VR app orchestrates sessions (via the server it tells LeRobot when to create datasets, start/stop episodes) while also controlling the robot’s motions.
+- **Using LeRobot’s record script**: LeRobot owns session control and decides when to start/stop episodes. If you also use the VR teleop app, it’s only for motion control.
+
+### Option 1: Using Pollen data acquisition server (recommended for VR teleop)
+
+Make sure you have installed pollen_data_acquisition_server, as explained in the Setup section.
+
+Launch the data acquisition server to be able to manage your session directly from the teleoperation application:
+
+```bash
+python -m pollen_data_acquisition_server.server
+```
+
+Then get into the teleoperation application and choose "Data acquisition session".
+You can finally setup your session by following the screens displayed.
+
+> Even without the VR app, you can use the `pollen_data_acquisition_server` with your own client implementation.
+
+### Option 2: Using lerobot.record
+
+Reachy 2 is fully supported by LeRobot’s recording features.
+If you choose this option but still want to use the VR teleoperation application, select "Standard session" in the app.
+
+**Example: start a recording without the mobile base:**
+First add reachy2 and reachy2_teleoperator to the imports of the record script. Then you can use the following command:
+
+```bash
+lerobot-record \
+    --robot.type=reachy2 \
+    --robot.ip_address=192.168.0.200 \
+    --robot.id=r2-0000 \
+    --robot.use_external_commands=true \
+    --robot.with_mobile_base=false \
+    --teleop.type=reachy2_teleoperator \
+    --teleop.ip_address=192.168.0.200 \
+    --teleop.with_mobile_base=false \
+    --robot.with_torso_camera=true \
+    --dataset.repo_id=pollen_robotics/record_test \
+    --dataset.single_task="Reachy 2 recording test" \
+    --dataset.num_episodes=1 \
+    --dataset.episode_time_s=5 \
+    --dataset.fps=15 \
+    --dataset.push_to_hub=true \
+    --dataset.private=true \
+    --dataset.streaming_encoding=true \
+    --dataset.encoder_threads=2 \
+    # --dataset.vcodec=auto \
+    --display_data=true
+```
+
+#### Specific Options
+
+**Extended setup overview (all options included):**
+
+```bash
+lerobot-record \
+    --robot.type=reachy2 \
+    --robot.ip_address=192.168.0.200 \
+    --robot.use_external_commands=true \
+    --robot.with_mobile_base=true \
+    --robot.with_l_arm=true \
+    --robot.with_r_arm=true \
+    --robot.with_neck=true \
+    --robot.with_antennas=true \
+    --robot.with_left_teleop_camera=true \
+    --robot.with_right_teleop_camera=true \
+    --robot.with_torso_camera=false \
+    --robot.camera_width=640 \
+    --robot.camera_height=480 \
+    --robot.disable_torque_on_disconnect=false \
+    --robot.max_relative_target=5.0 \
+    --teleop.type=reachy2_teleoperator \
+    --teleop.ip_address=192.168.0.200 \
+    --teleop.use_present_position=false \
+    --teleop.with_mobile_base=false \
+    --teleop.with_l_arm=true \
+    --teleop.with_r_arm=true \
+    --teleop.with_neck=true \
+    --teleop.with_antennas=true \
+    --dataset.repo_id=pollen_robotics/record_test \
+    --dataset.single_task="Reachy 2 recording test" \
+    --dataset.num_episodes=1 \
+    --dataset.episode_time_s=5 \
+    --dataset.fps=15 \
+    --dataset.push_to_hub=true \
+    --dataset.private=true \
+    --dataset.streaming_encoding=true \
+    --dataset.encoder_threads=2 \
+    # --dataset.vcodec=auto \
+    --display_data=true
+```
+
+##### `--robot.use_external_commands`
+
+Determine whether LeRobot robot.send_action() sends commands to the robot.
+**Must** be set to false while using the VR teleoperation application, as the app already sends commands.
+
+##### `--teleop.use_present_position`
+
+Determine whether the teleoperator reads the goal or present position of the robot.
+Must be set to true if a compliant Reachy 2 is used to control another one.
+
+##### Use the relevant parts
+
+From our initial tests, recording **all** joints when only some are moving can reduce model quality with certain policies.
+To avoid this, you can exclude specific parts from recording and replay using:
+
+```bash
+--robot.with_<part>=false
+```
+
+with `<part>` being one of : `mobile_base`, `l_arm`, `r_arm", `neck`, `antennas`.
+It determine whether the corresponding part is recorded in the observations. True if not set.
+
+By default, **all parts are recorded**.
+
+The same per-part mechanism is available in `reachy2_teleoperator` as well.
+
+```bash
+--teleop.with\_<part>
+```
+
+with `<part>` being one of : `mobile_base`, `l_arm`, `r_arm", `neck`, `antennas`.
+Determine whether the corresponding part is recorded in the actions. True if not set.
+
+> **Important:** In a given session, the **enabled parts must match** on both the robot and the teleoperator.
+> For example, if the robot runs with `--robot.with_mobile_base=false`, the teleoperator must disable the same part `--teleoperator.with_mobile_base=false`.
+
+##### Use the relevant cameras
+
+You can do the same for **cameras**. Enable or disable each camera with default parameters using:
+
+```bash
+--robot.with_left_teleop_camera=<true|false> \
+--robot.with_right_teleop_camera=<true|false> \
+--robot.with_torso_camera=<true|false>
+```
+
+By default, no camera is recorded, all camera arguments are set to `false`.
+If you want to, you can use custom `width` and `height` parameters for Reachy 2's cameras using the `--robot.camera_width` & `--robot.camera_height` argument:
+
+```bash
+--robot.camera_width=1920 \
+--robot.camera_height=1080
+```
+
+This will change the resolution of all 3 default robot cameras (enabled by the above bool arguments).
+
+If you want, you can add additional cameras other than the ones in the robot as usual with:
+
+```bash
+--robot.cameras="{ extra: {type: opencv, index_or_path: 42, width: 640, height: 480, fps: 30}}" \
+```
+
+## Step 2: Replay
+
+Make sure the robot is configured with the same parts as the dataset:
+
+```bash
+lerobot-replay \
+    --robot.type=reachy2 \
+    --robot.ip_address=192.168.0.200 \
+    --robot.use_external_commands=false \
+    --robot.with_mobile_base=false \
+    --dataset.repo_id=pollen_robotics/record_test \
+    --dataset.episode=0
+```
+
+## Step 3: Train
+
+```bash
+lerobot-train \
+  --dataset.repo_id=pollen_robotics/record_test \
+  --policy.type=act \
+  --output_dir=outputs/train/reachy2_test \
+  --job_name=reachy2 \
+  --policy.device=mps \
+  --wandb.enable=true \
+  --policy.repo_id=pollen_robotics/record_test_policy
+```
+
+## Step 4: Evaluate
+
+```bash
+lerobot-eval \
+  --robot.type=reachy2 \
+  --robot.ip_address=192.168.0.200 \
+  --dataset.repo_id=pollen_robotics/eval_record_test \
+  --dataset.single_task="Evaluate reachy2 policy" \
+  --dataset.num_episodes=10 \
+  --policy.path=outputs/train/reachy2_test/checkpoints/last/pretrained_model
+```
diff --git a/lerobot/docs/source/rename_map.mdx b/lerobot/docs/source/rename_map.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..6249faaca94e2f0bb74d06b64d1e72d24f462f64
--- /dev/null
+++ b/lerobot/docs/source/rename_map.mdx
@@ -0,0 +1,114 @@
+# Rename Map and Empty Cameras
+
+When you train, evaluate, or record with a robot policy, your **dataset** or **environment** provides observations under one set of keys (e.g. `observation.images.front`, `observation.images.eagle`), while your **policy** expects another (e.g. `observation.images.image`, `observation.images.image2`). The **rename map** bridges that gap without changing the policy or data source.
+
+> **Scope:** The rename map only renames **observation** keys (images and state). Action keys are not affected.
+
+## Why observation keys don't always match
+
+Policies have a fixed set of **input feature names** baked into their pretrained config. For example:
+
+- [pi0fast-libero](https://huggingface.co/lerobot/pi0fast-libero) expects `observation.images.base_0_rgb` and `observation.images.left_wrist_0_rgb`.
+- [xvla-base](https://huggingface.co/lerobot/xvla-base) expects `observation.images.image`, `observation.images.image2`, and `observation.images.image3`.
+
+Your dataset might use different names entirely (e.g. `observation.images.front`, `observation.images.eagle`, `observation.images.glove`), and your eval environment might use yet another set. Rather than editing the policy config or renaming columns in the dataset, you pass a **rename map**: a JSON dictionary that maps source keys to the keys the policy expects. Renaming happens inside the preprocessor pipeline, so the policy always sees its expected keys.
+
+## Using the rename map
+
+Pass the mapping as a JSON string on the command line. The convention is always:
+
+```
+--rename_map='{"source_key": "policy_key", ...}'
+```
+
+where **source_key** is what the dataset or environment provides, and **policy_key** is what the policy expects.
+
+Only listed keys are renamed; everything else passes through unchanged. Order of entries doesn't matter.
+
+Supported policies: **PI0**, **PI05**, **PI0Fast**, **SmolVLA**, and **XVLA**.
+
+### Training
+
+Suppose you fine-tune [lerobot/xvla-base](https://huggingface.co/lerobot/xvla-base) on a dataset with images under `observation.images.front`, `observation.images.eagle`, and `observation.images.glove`. XVLA expects `observation.images.image`, `observation.images.image2`, and `observation.images.image3`:
+
+```bash
+lerobot-train \
+  --dataset.repo_id=YOUR_DATASET \
+  --output_dir=./outputs/xvla_training \
+  --job_name=xvla_training \
+  --policy.path="lerobot/xvla-base" \
+  --policy.repo_id="HF_USER/xvla-your-robot" \
+  --policy.dtype=bfloat16 \
+  --policy.action_mode=auto \
+  --steps=20000 \
+  --policy.device=cuda \
+  --policy.freeze_vision_encoder=false \
+  --policy.freeze_language_encoder=false \
+  --policy.train_policy_transformer=true \
+  --policy.train_soft_prompts=true \
+  --rename_map='{"observation.images.front": "observation.images.image", "observation.images.eagle": "observation.images.image2", "observation.images.glove": "observation.images.image3"}'
+```
+
+### Evaluation
+
+A policy that expects `observation.images.base_0_rgb` and `observation.images.left_wrist_0_rgb` (e.g. [pi0fast-libero](https://huggingface.co/lerobot/pi0fast-libero)), but the LIBERO environment returns `observation.images.image` and `observation.images.image2`:
+
+```bash
+lerobot-eval \
+  --policy.path=lerobot/pi0fast-libero \
+  --env.type=libero \
+  ... \
+  --rename_map='{"observation.images.image": "observation.images.base_0_rgb", "observation.images.image2": "observation.images.left_wrist_0_rgb"}'
+```
+
+### Recording
+
+`lerobot-record` also supports rename maps, nested under the dataset config:
+
+```bash
+lerobot-record \ # When running inference
+  --policy.path="<user>/smolVLA_finetuned" \
+  ... \
+  --dataset.rename_map='{"observation.images.glove2": "observation.images.image"}'
+```
+
+## Alternative: edit the policy config directly
+
+If you always use the same dataset or environment, you can **edit the policy's `config.json`** so its observation keys match your data source. Then no rename map is needed.
+
+The tradeoff: modifying the policy config ties it to one data source. A rename map keeps one policy usable across many datasets and environments.
+
+## Empty cameras: fewer views than the policy expects
+
+Some policies are built for a fixed number of image inputs. If your dataset has fewer cameras, you can set **`empty_cameras`** in the policy config instead of modifying the model architecture.
+
+### How it works
+
+Setting `empty_cameras=N` adds N placeholder image features to the policy config, named:
+
+```
+observation.images.empty_camera_0
+observation.images.empty_camera_1
+...
+```
+
+At runtime, these keys have no corresponding data in the batch. The policy fills them with masked dummy tensors (padded with `-1` for SigLIP-based vision encoders, with a zero attention mask), so the extra image slots are effectively ignored during training and inference.
+
+### Example
+
+XVLA-base has three visual inputs and `empty_cameras=0` by default. Your dataset only has two cameras:
+
+1. Set `--policy.empty_cameras=1`.
+2. The config adds a third key: `observation.images.empty_camera_0`.
+3. Use the rename map for your two real cameras as usual.
+4. The third slot is masked out — no fake images needed in your dataset.
+
+## Quick reference
+
+| Goal                                      | What to do                                                                  |
+| ----------------------------------------- | --------------------------------------------------------------------------- |
+| Dataset keys ≠ policy keys                | `--rename_map='{"dataset_key": "policy_key", ...}'`                         |
+| Env keys ≠ policy keys (eval)             | `--rename_map='{"env_key": "policy_key", ...}'`                             |
+| Recording with different keys (inference) | `--dataset.rename_map='{"source_key": "policy_key", ...}'`.                 |
+| Fewer cameras than policy expects         | `--policy.empty_cameras=N` (supported by PI0, PI05, PI0Fast, SmolVLA, XVLA) |
+| Avoid passing a rename map                | Edit the policy's `config.json` so its keys match your data source          |
diff --git a/lerobot/docs/source/rtc.mdx b/lerobot/docs/source/rtc.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..f63c00fcaf6d0fd3963cca6d5207ce17caeea261
--- /dev/null
+++ b/lerobot/docs/source/rtc.mdx
@@ -0,0 +1,188 @@
+# Real-Time Chunking (RTC)
+
+Real-Time Chunking (RTC) is an inference-time method that allows large, flow-matching based robotic policies, such as [Pi0](./pi0), [Pi0.5](./pi05), and [SmolVLA](./smolvla), to produce smooth, continuous, and reactive motion despite having high inference latency.
+
+These policies generate chunks of future actions (e.g., 50 steps at a time) instead of single actions.
+Because the models are large, producing each chunk takes longer than the time it takes the robot to execute it.
+Naively executing chunks leads to problems such as pauses, jerky transitions, or sudden changes in strategy whenever the next chunk arrives late or disagrees with the previously executed actions.
+
+RTC solves this by asynchronously generating the next chunk while the robot continues executing the current one, and by guiding the new chunk so it aligns smoothly with the portion of the previous chunk that has already been executed.
+
+## How RTC Works (simplified)
+
+RTC lets the robot think ahead while it’s still moving. When the robot is carrying out one chunk of actions, RTC starts creating the next chunk early.
+But since the robot has already moved a bit by the time the new chunk is ready, RTC has to make sure the new chunk still lines up smoothly with what the robot is currently doing.
+
+To do this, RTC treats the beginning of the new chunk like an inpainting or “fill-in-the-gaps” problem:
+it gently adjusts the first part of the new chunk so it blends naturally with the robot’s ongoing motion. The result is no pauses, no sudden jumps.
+
+In technical terms, RTC adds a guidance term to the flow-matching denoising process that forces the overlapping timesteps of the new chunk to stay close to the executed portion of the previous chunk, typically using a soft transition mask.
+
+## Quick Start
+
+### Installation
+
+RTC is built into LeRobot. Just install the policy dependencies you need:
+
+```bash
+# For Pi0 or Pi0.5
+pip install -e ".[pi]"
+
+# For SmolVLA
+pip install -e ".[smolvla]"
+```
+
+### Using RTC with Pi0
+
+You can find a complete reference implementation in [eval_with_real_robot.py](examples/rtc/eval_with_real_robot.py).
+The snippet below provides a simplified pseudo-example of how RTC operates with Pi0 in your pipeline:
+
+```python
+from lerobot.policies.pi0 import PI0Policy, PI0Config
+from lerobot.configs.types import RTCAttentionSchedule
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+from lerobot.policies.rtc.action_queue import ActionQueue
+
+# Load Pi0 with RTC enabled
+policy_cfg = PI0Config()
+
+# Enable RTC
+policy_cfg.rtc_config = RTCConfig(
+    enabled=True,
+    execution_horizon=10,  # How many steps to blend with previous chunk
+    max_guidance_weight=10.0,  # How strongly to enforce consistency
+    prefix_attention_schedule=RTCAttentionSchedule.EXP,  # Exponential blend
+)
+
+# Load the policy
+policy = PI0Policy.from_pretrained("lerobot/pi0_base", policy_cfg=policy_cfg, device="cuda")
+
+# Now use predict_action_chunk with RTC parameters
+inference_delay = 4  # How many steps of inference latency, this values should be calculated based on the inference latency of the policy
+
+# Initialize the action queue
+action_queue = ActionQueue(policy_cfg.rtc_config)
+
+# Start in a separate thread with the following function
+def get_actions():
+  while True:
+    if should_get_actions:
+
+      prev_actions = action_queue.get_left_over()
+      obs = get_robot_observations(robot)
+
+      # Generate actions WITH RTC
+      actions = policy.predict_action_chunk(
+          obs,
+          inference_delay=inference_delay,
+          prev_chunk_left_over=prev_actions,
+      )
+
+      action_queue.merge(
+          actions, actions, inference_delay
+      )
+
+for step in range(num_steps):
+    action = action_queue.get()
+
+    # Execute the first N actions
+    execute_actions(action)
+```
+
+## Key Parameters
+
+`RTCConfig` has the following parameters to tune:
+
+**`execution_horizon`**: How many timesteps from the previous chunk to maintain consistency with. Higher values mean smoother transitions but potentially less reactivity.
+
+Typical values: 8-12 steps
+
+```python
+RTCConfig(execution_horizon=10)
+```
+
+**`max_guidance_weight`**: How strongly to enforce consistency with the previous chunk. This is a hyperparameter that can be tuned to balance the smoothness of the transitions and the reactivity of the policy. For 10 steps flow matching (SmolVLA, Pi0, Pi0.5), a value of 10.0 is a optimal value.
+
+**`prefix_attention_schedule`**: How to weight consistency across the overlap region.
+
+- `LINEAR`: Linear decay from inference_delay to execution_horizon
+- `EXP`: Exponential decay (recommended for getting started)
+- `ONES`: Full weight across entire execution_horizon
+- `ZEROS`: Binary (full weight up to inference_delay, then zero)
+
+**`inference_delay`**: How many timesteps of inference latency your system has. This is passed to `predict_action_chunk()` rather than the config, since it may vary at runtime.
+
+## Testing RTC Offline
+
+Before running on a real robot, test RTC with dataset samples to visualize how it works:
+
+```bash
+python examples/rtc/eval_dataset.py \
+    --policy.path=lerobot/pi0_libero_finetuned \
+    --dataset.repo_id=HuggingFaceVLA/libero \
+    --rtc.execution_horizon=10 \
+    --rtc.max_guidance_weight=10.0 \
+    --device=cuda
+```
+
+The script generates a visualization of the denoising process, comparing standard generation (left) with RTC (right). In the RTC plots, you can see how the first few steps (blue/purple lines) are guided to match the red ground truth trajectory (previous chunk's tail), ensuring a smooth transition between chunks.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/flow_matching.png"
+    alt="Denoising steps with and without RTC"
+    width="100%"
+  />
+</p>
+
+## Testing RTC with a Real Robot
+
+```bash
+python examples/rtc/eval_with_real_robot.py \
+    --policy.path=${HF_USERNAME}/policy_repo_id \
+    --robot.type=so100_follower \
+    --robot.port=/dev/tty.usbmodem58FA0834591 \
+    --robot.cameras="{ gripper: {type: opencv, index_or_path: 1, width: 640, height: 480, fps: 30}, front: {type: opencv, index_or_path: 0, width: 640, height: 480, fps: 30}}" \
+    --task="Move green small object into the purple platform" \
+    --duration=120 \
+    --device=cuda
+```
+
+## How It Differs from the Async Inference in LeRobot
+
+Both RTC and [async inference](./async) improve real-time robot control, but they solve different problems.
+
+| Aspect        | Async Inference                                                            | RTC                                                 |
+| ------------- | -------------------------------------------------------------------------- | --------------------------------------------------- |
+| **Problem**   | Idle frames while waiting for inference                                    | Discontinuities between action chunks               |
+| **Solution**  | Decouple prediction from execution                                         | Guide new chunks to continue smoothly from previous |
+| **Benefit**   | No waiting, continuous action                                              | Smooth transitions, natural motion                  |
+| **Best Used** | Async inference is best used with large models with high inference latency | Flow-matching based policies                        |
+
+**Use both together** for maximum smoothness and reactivity!
+
+## Advanced: Debug Tracking
+
+RTC includes built-in debug tracking to help you understand what's happening during inference:
+
+```python
+# Enable debug tracking
+policy_cfg.rtc_config.debug = True
+policy_cfg.rtc_config.debug_maxlen = 100
+
+# After inference, access debug data
+debug_data = policy.rtc_processor.get_debug_data()
+
+# Visualize denoising steps, corrections, etc.
+from lerobot.policies.rtc.debug_visualizer import RTCDebugVisualizer
+visualizer = RTCDebugVisualizer()
+# ... create plots
+```
+
+See `examples/rtc/eval_dataset.py` for a complete example of visualization.
+
+## References
+
+- [Smooth-As-Butter Robot Policies](https://alexander-soare.github.io/robotics/2025/08/05/smooth-as-butter-robot-policies.html) - Excellent technical explanation with real robot results
+- [Physical Intelligence - Real-Time Chunking](https://www.physicalintelligence.company/research/real_time_chunking) - Original paper and research
+- [Kinetix RTC Implementation](https://github.com/Physical-Intelligence/real-time-chunking-kinetix) - Reference implementation from Physical Intelligence
diff --git a/lerobot/docs/source/sarm.mdx b/lerobot/docs/source/sarm.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..cd488fe1f965b331ca85859c2a1c51b4d147a463
--- /dev/null
+++ b/lerobot/docs/source/sarm.mdx
@@ -0,0 +1,592 @@
+# SARM: Stage-Aware Reward Modeling
+
+SARM (Stage-Aware Reward Modeling) is a video-based reward modeling framework for long-horizon robot manipulation tasks. This guide covers how to train SARM reward models and optionally use them with Reward-Aligned Behavior Cloning (RA-BC).
+
+**Paper**: [SARM: Stage-Aware Reward Modeling for Long Horizon Robot Manipulation](https://arxiv.org/abs/2509.25358)
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/lerobot-sarm.png"
+  alt="An overview of SARM"
+  width="80%"
+/>
+
+## Why Reward Models?
+
+Standard behavior cloning treats all demonstration frames equally, but real-world robot datasets are messy. They contain hesitations, corrections, and variable-quality trajectories. Reward models solve this by learning a generalizable notion of **task progress** from demonstrations: given video frames and a task description, they predict how close the robot is to completing the task (0→1). This learned "progress signal" can be used in multiple ways, two promising applications are: (1) **weighted imitation learning** (RA-BC), where high-progress frames receive more weight during policy training, and (2) **reinforcement learning**, where the reward model provides dense rewards for online or offline policy improvement.
+
+## Overview
+
+SARM has following features:
+
+1. **Stage-aware architecture**: Jointly predicts the high-level task stage and fine-grained progress within each stage
+2. **Subtask annotations**: Uses natural language subtask annotations to derive consistent progress labels
+3. **Temporal proportions**: Computes dataset-level priors (α̅\_k) for each subtask to normalize progress across variable-length demonstrations
+
+SARM trains on a compact **stage+tau** target for each frame:
+
+- **stage**: integer stage index `k ∈ {0, ..., K-1}`
+- **τ (tau)**: within-stage progress `τ ∈ [0, 1]`
+- **target encoding**: `y = k + τ` (this is what the dataset processor produces)
+
+At inference time (and in downstream RA-BC), SARM converts the raw `k + τ` value into a **normalized progress** in `[0, 1]` using dataset-level **temporal proportions** `α̅_k` (stored in `meta/temporal_proportions_*.json`).
+
+This matches **Formula (2)** from the paper:
+
+```
+progress_t = P_{k-1} + α̅_k × τ_t
+```
+
+Where:
+
+- `τ_t = (t - s_k) / (e_k - s_k)` is within-subtask normalized time
+- `P_{k-1}` is cumulative prior (sum of previous subtask proportions)
+- `α̅_k` is the temporal proportion for subtask k
+
+This ensures identical task states map to consistent progress values, even across demonstrations of different lengths.
+
+## Inputs and Targets (What the new code expects)
+
+SARM is trained through its processor (`src/lerobot/policies/sarm/processor_sarm.py`), which:
+
+- **Encodes** images and task text with CLIP (ViT-B/32) into `video_features` and `text_features`
+- **Pads/truncates** robot state into `state_features` (up to `max_state_dim`)
+- **Builds targets** as `sparse_targets` (and `dense_targets` in `dense_only`/`dual`) using the stage+tau encoding `y = k + τ`
+- **Masks rewind frames** using a per-sample `lengths` tensor (rewind is a training-time augmentation)
+
+At minimum, each training sample needs:
+
+- `task` (string): task description
+- `policy.image_key` images and `policy.state_key` states from the dataset
+
+---
+
+## Annotation Modes
+
+You can choose from **3 annotation modes** that determine how progress labels are computed:
+
+| Mode           | Annotations Required | Heads                        | Use Case                                                     |
+| -------------- | -------------------- | ---------------------------- | ------------------------------------------------------------ |
+| `single_stage` | None                 | Sparse only                  | Simple tasks, quick experiments, no VLM needed               |
+| `dense_only`   | Dense (VLM)          | Dual (sparse auto-generated) | Detailed subtask tracking without defining high-level stages |
+| `dual`         | Sparse + Dense (VLM) | Dual                         | Full SARM paper setup with both granularities                |
+
+### Mode Details
+
+<hfoptions id="mode_explanation">
+<hfoption id="single_stage">
+
+**No annotations required.** The entire episode is treated as a single stage called `"task"`, and progress is linear from 0 to 1 over the episode duration.
+
+- **Sparse head**: 1 stage ("task"), linear progress
+- **Dense head**: Not used
+- **Best for**: Simple tasks, quick experiments, or when VLM annotation is not available
+
+## Set Up Your Environment
+
+1. Install LeRobot by following our [Installation Guide](./installation).
+2. Install SARM dependencies by running:
+
+```bash
+pip install -e ".[sarm]"
+```
+
+Workflow:
+
+```
+1. Train SARM → 2. Visualize predictions → 3. (Optional) Train policy with RA-BC
+```
+
+</hfoption>
+<hfoption id="dense_only">
+
+**Only dense (fine-grained) annotations from a VLM.** The sparse head automatically uses a single `"task"` stage covering the full episode, while the dense head learns detailed subtask progression.
+
+- **Sparse head**: 1 stage ("task"), linear progress (auto-generated)
+- **Dense head**: Multiple fine-grained stages from VLM annotations
+- **Best for**: When you want detailed subtask tracking but don't need to define high-level stages
+
+Workflow:
+
+```
+1. Annotate (dense) → 2. Verify → 3. Train SARM → 4. Visualize → 5. (Optional) Train policy with RA-BC
+```
+
+</hfoption>
+<hfoption id="dual">
+
+**Both sparse and dense annotations from VLM.** Full dual-head mode as described in the SARM paper, with both high-level (sparse) and fine-grained (dense) stage predictions.
+
+- **Sparse head**: High-level stages from VLM annotations
+- **Dense head**: Fine-grained stages from VLM annotations
+- **Best for**: Complex multi-stage tasks where both granularities are useful
+
+Workflow:
+
+```
+1. Annotate (sparse+dense) → 2. Verify → 3. Train SARM → 4. Visualize → 5. (Optional) Train policy with RA-BC
+```
+
+</hfoption>
+</hfoptions>
+
+---
+
+## Step 1: Subtask Annotation
+
+<hfoptions id="annotation_mode">
+<hfoption id="single_stage">
+
+**No annotation required!** Skip this step entirely. The model will use the episode's task description and compute linear progress automatically.
+
+</hfoption>
+<hfoption id="dense_only">
+
+Generate **dense (fine-grained) annotations only** using a VLM. The sparse stage will be auto-generated.
+
+```bash
+python src/lerobot/data_processing/sarm_annotations/subtask_annotation.py \
+  --repo-id your-username/your-dataset \
+  --dense-only \
+  --dense-subtasks "Bring robot arms up from starting position,Grab near side and do 1st fold,Grab side and do 2nd fold,Grab side and do 3rd fold to finish folding" \
+  --video-key observation.images.base \
+  --num-workers 4 \
+  --push-to-hub
+```
+
+**What gets saved:**
+
+- `meta/temporal_proportions_sparse.json` - Auto-generated sparse proportions (`{"task": 1.0}`)
+- `meta/temporal_proportions_dense.json` - Dense temporal proportions
+- Per-episode columns in `episodes/*.parquet`:
+  - `dense_subtask_names`, `dense_subtask_start_frames`, `dense_subtask_end_frames`
+  - (also time-based columns: `dense_subtask_start_times`, `dense_subtask_end_times`)
+
+</hfoption>
+<hfoption id="dual">
+
+Generate **both sparse (high-level) and dense (fine-grained) annotations** using a VLM.
+
+```bash
+python src/lerobot/data_processing/sarm_annotations/subtask_annotation.py \
+  --repo-id your-username/your-dataset \
+  --sparse-subtasks "Bring arms up from starting position,Fold the towel (3 folds in total)" \
+  --dense-subtasks "Bring robot arms up from starting position,Grab near side and do 1st fold,Grab side and do 2nd fold,Grab side and do 3rd fold to finish folding" \
+  --video-key observation.images.base \
+  --num-workers 4 \
+  --push-to-hub
+```
+
+**What gets saved:**
+
+- `meta/temporal_proportions_sparse.json` - Sparse temporal proportions
+- `meta/temporal_proportions_dense.json` - Dense temporal proportions
+- Per-episode columns in `episodes/*.parquet`:
+  - `sparse_subtask_names`, `sparse_subtask_start_frames`, `sparse_subtask_end_frames`
+  - `dense_subtask_names`, `dense_subtask_start_frames`, `dense_subtask_end_frames`
+  - (also time-based columns: `*_subtask_start_times`, `*_subtask_end_times`)
+
+</hfoption>
+</hfoptions>
+
+### Annotation Arguments
+
+| Argument               | Description                                                                     |
+| ---------------------- | ------------------------------------------------------------------------------- |
+| `--repo-id`            | HuggingFace dataset repository ID                                               |
+| `--sparse-subtasks`    | Comma-separated list of high-level subtask names                                |
+| `--dense-subtasks`     | Comma-separated list of fine-grained subtask names                              |
+| `--dense-only`         | Generate only dense annotations (auto-creates sparse "task" stage)              |
+| `--video-key`          | Camera/video key to use (e.g., `observation.images.top`)                        |
+| `--num-workers`        | Number of parallel GPU workers (default: 1)                                     |
+| `--episodes`           | Specific episode indices to annotate (default: all)                             |
+| `--skip-existing`      | Skip episodes that already have annotations                                     |
+| `--model`              | VLM model (default: `Qwen/Qwen3-VL-30B-A3B-Instruct`)                           |
+| `--num-visualizations` | Number of episodes to visualize after annotation (default: 5, set to 0 to skip) |
+
+> **Note**: After annotation completes, 5 episodes are automatically visualized by default. Use `--num-visualizations 0` to skip this step.
+
+---
+
+## Step 2: Verify Annotations
+
+<hfoptions id="verify_mode">
+<hfoption id="single_stage">
+
+**No verification needed!** Skip this step.
+
+</hfoption>
+<hfoption id="dense_only">
+
+Visualize annotations using the `--visualize-only` flag:
+
+```bash
+python src/lerobot/data_processing/sarm_annotations/subtask_annotation.py \
+  --repo-id your-username/your-dataset \
+  --visualize-only \
+  --visualize-type dense \
+  --num-visualizations 5 \
+  --video-key observation.images.base \
+  --output-dir ./subtask_viz
+```
+
+</hfoption>
+<hfoption id="dual">
+
+Visualize annotations using the `--visualize-only` flag:
+
+```bash
+python src/lerobot/data_processing/sarm_annotations/subtask_annotation.py \
+  --repo-id your-username/your-dataset \
+  --visualize-only \
+  --visualize-type both \
+  --num-visualizations 5 \
+  --video-key observation.images.base \
+  --output-dir ./subtask_viz
+```
+
+</hfoption>
+</hfoptions>
+
+This generates visualizations showing video frames with subtask boundaries overlaid and timeline of subtasks.
+
+### Visualization Arguments
+
+| Argument               | Description                                                    |
+| ---------------------- | -------------------------------------------------------------- |
+| `--visualize-only`     | Only visualize existing annotations (no generation)            |
+| `--num-visualizations` | Number of episodes to visualize (default: 5)                   |
+| `--visualize-type`     | Type of annotations to visualize: `sparse`, `dense`, or `both` |
+
+**Tip**: If annotations are inaccurate, adjust your subtask descriptions to be more specific and re-run.
+
+---
+
+## Step 3: Train SARM
+
+<hfoptions id="train_mode">
+<hfoption id="single_stage">
+
+Train with **no annotations** - uses linear progress from 0 to 1:
+
+```bash
+lerobot-train \
+  --dataset.repo_id=your-username/your-dataset \
+  --policy.type=sarm \
+  --policy.annotation_mode=single_stage \
+  --policy.image_key=observation.images.base \
+  --output_dir=outputs/train/sarm_single \
+  --batch_size=32 \
+  --steps=5000 \
+  --wandb.enable=true \
+  --wandb.project=sarm \
+  --policy.repo_id=your-username/your-model-name
+```
+
+</hfoption>
+<hfoption id="dense_only">
+
+Train with **dense annotations only** (sparse auto-generated):
+
+```bash
+lerobot-train \
+  --dataset.repo_id=your-username/your-dataset \
+  --policy.type=sarm \
+  --policy.annotation_mode=dense_only \
+  --policy.image_key=observation.images.base \
+  --output_dir=outputs/train/sarm_dense \
+  --batch_size=32 \
+  --steps=5000 \
+  --wandb.enable=true \
+  --wandb.project=sarm \
+  --policy.repo_id=your-username/your-model-name
+```
+
+</hfoption>
+<hfoption id="dual">
+
+Train with **both sparse and dense annotations**:
+
+```bash
+lerobot-train \
+  --dataset.repo_id=your-username/your-dataset \
+  --policy.type=sarm \
+  --policy.annotation_mode=dual \
+  --policy.image_key=observation.images.base \
+  --output_dir=outputs/train/sarm_dual \
+  --batch_size=32 \
+  --steps=5000 \
+  --wandb.enable=true \
+  --wandb.project=sarm \
+  --policy.repo_id=your-username/your-model-name
+```
+
+</hfoption>
+</hfoptions>
+
+### Multi-GPU Training
+
+Add `accelerate launch --multi_gpu --num_processes=4` to use multiple GPUs for training.
+
+### Training Arguments
+
+| Argument                   | Description                                                       | Default                  |
+| -------------------------- | ----------------------------------------------------------------- | ------------------------ |
+| `--policy.annotation_mode` | `single_stage`, `dense_only`, or `dual`                           | `single_stage`           |
+| `--policy.image_key`       | Camera key for images                                             | `observation.images.top` |
+| `--policy.state_key`       | Key for joint states                                              | `observation.state`      |
+| `--policy.n_obs_steps`     | Observation history steps (total obs frames = `n_obs_steps + 1`)  | `8`                      |
+| `--policy.frame_gap`       | Gap (in frames) between sampled observations (at 30 fps: 30 ≈ 1s) | `30`                     |
+
+---
+
+## Step 4: Visualize Predictions
+
+Use `compute_rabc_weights.py` with `--visualize-only` to visualize model predictions (and, if available, annotation-derived targets) without writing a parquet file.
+
+<hfoptions id="viz_mode">
+<hfoption id="single_stage">
+
+```bash
+python src/lerobot/policies/sarm/compute_rabc_weights.py \
+  --dataset-repo-id your-username/your-dataset \
+  --reward-model-path your-username/sarm-model \
+  --visualize-only \
+  --num-visualizations 5 \
+  --head-mode sparse \
+  --output-dir ./sarm_viz
+```
+
+</hfoption>
+<hfoption id="dense_only">
+
+```bash
+python src/lerobot/policies/sarm/compute_rabc_weights.py \
+  --dataset-repo-id your-username/your-dataset \
+  --reward-model-path your-username/sarm-model \
+  --visualize-only \
+  --num-visualizations 5 \
+  --head-mode dense \
+  --output-dir ./sarm_viz
+```
+
+</hfoption>
+<hfoption id="dual">
+
+```bash
+python src/lerobot/policies/sarm/compute_rabc_weights.py \
+  --dataset-repo-id your-username/your-dataset \
+  --reward-model-path your-username/sarm-model \
+  --visualize-only \
+  --num-visualizations 5 \
+  --head-mode both \
+  --output-dir ./sarm_viz
+```
+
+</hfoption>
+</hfoptions>
+
+The visualization shows:
+
+- **Progress plot**: Predicted progress (and optional annotation-derived “GT” when available and `--stride 1`)
+- **Stage probabilities**: Stacked area plot of predicted stage probabilities
+- **Sample frames**: Key frames from the episode with progress/stage labels
+
+### Visualization Arguments
+
+| Argument               | Description                                               |
+| ---------------------- | --------------------------------------------------------- |
+| `--visualize-only`     | Only visualize predictions (no RABC computation)          |
+| `--num-visualizations` | Number of episodes to visualize (default: 5)              |
+| `--head-mode`          | SARM head to use: `sparse`, `dense`, or `both`            |
+| `--stride`             | Compute every N frames, interpolate the rest (default: 1) |
+
+---
+
+## Step 5 (Optional): Train Policy with RA-BC
+
+Reward-Aligned Behavior Cloning (RA-BC) uses the trained SARM model to weight training samples based on predicted progress improvement. This requires two steps:
+
+1. **Precompute progress values** for all frames using the trained SARM model
+2. **Train policy** with RA-BC weighting using the precomputed values
+
+### How RA-BC Works
+
+For each training sample, RA-BC computes the progress delta:
+
+```
+r_i = φ(o_{t+Δ}) - φ(o_t)
+```
+
+Where `φ` is the SARM progress prediction and `Δ` is the policy's `chunk_size`. Samples with positive progress (good demonstrations) get higher weights, while samples with negative or zero progress get down-weighted.
+
+The weighting follows **Equations 8-9** from the paper:
+
+- **Soft weight**: `w̃_i = clip((r_i − (μ − 2σ)) / (4σ + ε), 0, 1)`
+- **Final weight**: `w_i = 𝟙{r_i > κ} + 𝟙{0 ≤ r_i ≤ κ} × w̃_i`
+
+### Step 5a: Compute SARM Progress Values
+
+First, run the SARM model on all frames in your dataset to compute progress values:
+
+```bash
+python src/lerobot/policies/sarm/compute_rabc_weights.py \
+  --dataset-repo-id your-username/your-dataset \
+  --reward-model-path your-username/sarm-model \
+  --head-mode sparse \
+  --num-visualizations 5 \
+  --push-to-hub
+```
+
+This script:
+
+- Processes all frames and computes progress values
+- Saves progress values to a parquet file next to the dataset on disk (defaults to `<dataset_root>/sarm_progress.parquet`)
+- Generates visualizations of the first N episodes (default: 5)
+
+**Arguments:**
+
+| Argument               | Description                                                    | Default    |
+| ---------------------- | -------------------------------------------------------------- | ---------- |
+| `--reward-model-path`  | Path to trained SARM model                                     | (required) |
+| `--head-mode`          | SARM head to use: `sparse`, `dense`, or `both`                 | `sparse`   |
+| `--device`             | Device for inference                                           | `cuda`     |
+| `--visualize-only`     | Only visualize predictions (no RA-BC computation)              | `false`    |
+| `--num-visualizations` | Number of episodes to visualize (default: 5, set to 0 to skip) | `5`        |
+
+**Output format** (`sarm_progress.parquet`):
+
+| Column            | Description                                    |
+| ----------------- | ---------------------------------------------- |
+| `index`           | Global frame index in dataset                  |
+| `episode_index`   | Episode number                                 |
+| `frame_index`     | Local frame index within episode               |
+| `progress_sparse` | Sparse head progress value [0, 1]              |
+| `progress_dense`  | Dense head progress value [0, 1] (if computed) |
+
+### Step 5b: Train Policy with RA-BC
+
+Once you have the progress file, train your policy with RA-BC weighting. The progress file is auto-detected from the dataset path (`sarm_progress.parquet`). Currently PI0, PI0.5 and SmolVLA are supported with RA-BC:
+
+```bash
+lerobot-train \
+  --dataset.repo_id=your-username/your-dataset \
+  --policy.type=pi0 \
+  --use_rabc=true \
+  --rabc_head_mode=sparse \
+  --rabc_kappa=0.01 \
+  --output_dir=outputs/train/policy_rabc \
+  --batch_size=32 \
+  --steps=40000
+```
+
+The training script automatically:
+
+- Loads the precomputed progress values from the parquet file
+- Uses the policy's `chunk_size` to compute progress deltas (Δ)
+- Computes sample weights based on progress improvement
+- Applies weighted loss during training
+
+**RA-BC Arguments:**
+
+| Argument               | Description                                                | Default                            |
+| ---------------------- | ---------------------------------------------------------- | ---------------------------------- |
+| `--use_rabc`           | Enable RA-BC sample weighting                              | `false`                            |
+| `--rabc_progress_path` | Path to progress parquet file (auto-detected from dataset) | `sarm_progress.parquet` in dataset |
+| `--rabc_head_mode`     | Which SARM head's progress to use: `sparse` or `dense`     | `sparse`                           |
+| `--rabc_kappa`         | Threshold κ for high-quality samples                       | `0.01`                             |
+
+### Tuning RA-BC Kappa
+
+The `kappa` parameter is the threshold that determines which samples get full weight (w=1). Understanding how to tune it is critical for RA-BC to work effectively.
+
+**How the weighting works:**
+
+| Condition           | Weight                  |
+| ------------------- | ----------------------- |
+| `delta > kappa`     | 1.0 (hard threshold)    |
+| `0 ≤ delta ≤ kappa` | Soft weight from Eq. 8  |
+| `delta < 0`         | 0.0 (negative progress) |
+
+**Diagnosing kappa issues:**
+
+Monitor these WandB metrics during training:
+
+| Metric             | Healthy Range | Problem Indicator         |
+| ------------------ | ------------- | ------------------------- |
+| `rabc_mean_weight` | 0.3 - 0.8     | ≈ 1.0 means kappa too low |
+| `rabc_delta_mean`  | > 0           | Should be positive        |
+| `rabc_delta_std`   | > 0           | Variance in data quality  |
+
+**If `rabc_mean_weight ≈ 1.0`:** Your kappa is too low. Most samples have `delta > kappa` and bypass the soft-weighting entirely. RA-BC becomes equivalent to vanilla BC.
+
+**Setting kappa based on your data:**
+
+The default `kappa=0.01` was tuned for the paper's T-shirt folding task (~90s episodes at 30fps). For your dataset, check the logged `rabc_delta_mean` and `rabc_delta_std`:
+
+```
+# If delta_mean ≈ 0.03 and delta_std ≈ 0.02:
+# Most deltas fall in range [0.01, 0.05]
+
+# Option 1: Set kappa = delta_mean (medium selectivity)
+--rabc_kappa=0.03
+
+# Option 2: Set kappa = delta_mean + delta_std (high selectivity)
+--rabc_kappa=0.05
+
+# Option 3: Set kappa = delta_mean + 2*delta_std (very selective)
+--rabc_kappa=0.07
+```
+
+**When RA-BC may not help:**
+
+If your dataset is already high quality (consistent progress across all demonstrations), RA-BC won't provide much benefit since there's nothing to filter.
+
+### Multi-GPU Training with RA-BC
+
+```bash
+accelerate launch \
+  --multi_gpu \
+  --num_processes=4 \
+  src/lerobot/scripts/lerobot_train.py \
+  --dataset.repo_id=your-username/your-dataset \
+  --policy.type=pi0 \
+  --use_rabc=true \
+  --rabc_kappa=0.01 \
+  --output_dir=outputs/train/policy_rabc \
+  --batch_size=32 \
+  --steps=40000
+```
+
+---
+
+## Tips & Best Practices
+
+### Choosing a Mode
+
+- **Start with `single_stage`** for quick experiments - no annotation overhead
+- Use **`dense_only`** when you want detailed progress tracking but tasks don't have clear high-level stages
+- Use **`dual`** for complex tasks where both coarse and fine-grained progress is meaningful
+
+### Annotation Quality
+
+1. **Be specific with subtask names**: Instead of "fold", use "grab near side and fold toward center"
+2. **Verify with visualization**: Always check a few episodes before training
+3. **Consistent naming**: Use the same subtask names across all episodes
+
+### RA-BC
+
+1. **Train SARM first**: RA-BC quality depends entirely on SARM quality
+2. **Monitor `rabc_mean_weight`**: If it's ≈ 1.0, increase kappa (see [Tuning RA-BC Kappa](#tuning-ra-bc-kappa))
+
+---
+
+## Citation
+
+```bibtex
+@article{chen2025sarm,
+  title={SARM: Stage-Aware Reward Modeling for Long Horizon Robot Manipulation},
+  author={Chen, Qianzhong and Yu, Justin and Schwager, Mac and Abbeel, Pieter and Shentu, Yide and Wu, Philipp},
+  journal={arXiv preprint arXiv:2509.25358},
+  year={2025}
+}
+```
diff --git a/lerobot/docs/source/smolvla.mdx b/lerobot/docs/source/smolvla.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..bf8a0d2f01b2888e58d65e7619131da9fcf66068
--- /dev/null
+++ b/lerobot/docs/source/smolvla.mdx
@@ -0,0 +1,119 @@
+# SmolVLA
+
+SmolVLA is Hugging Face’s lightweight foundation model for robotics. Designed for easy fine-tuning on LeRobot datasets, it helps accelerate your development!
+
+<p align="center">
+  <img
+    src="https://cdn-uploads.huggingface.co/production/uploads/640e21ef3c82bd463ee5a76d/aooU0a3DMtYmy_1IWMaIM.png"
+    alt="SmolVLA architecture."
+    width="500"
+  />
+  <br />
+  <em>
+    Figure 1. SmolVLA takes as input (i) multiple cameras views, (ii) the
+    robot’s current sensorimotor state, and (iii) a natural language
+    instruction, encoded into contextual features used to condition the action
+    expert when generating an action chunk.
+  </em>
+</p>
+
+## Set Up Your Environment
+
+1. Install LeRobot by following our [Installation Guide](./installation).
+2. Install SmolVLA dependencies by running:
+
+   ```bash
+   pip install -e ".[smolvla]"
+   ```
+
+## Collect a dataset
+
+SmolVLA is a base model, so fine-tuning on your own data is required for optimal performance in your setup.
+We recommend recording ~50 episodes of your task as a starting point. Follow our guide to get started: [Recording a Dataset](./il_robots)
+
+<Tip>
+
+In your dataset, make sure to have enough demonstrations per each variation (e.g. the cube position on the table if it is cube pick-place task) you are introducing.
+
+We recommend checking out the dataset linked below for reference that was used in the [SmolVLA paper](https://huggingface.co/papers/2506.01844):
+
+🔗 [SVLA SO100 PickPlace](https://huggingface.co/spaces/lerobot/visualize_dataset?path=%2Flerobot%2Fsvla_so100_pickplace%2Fepisode_0)
+
+In this dataset, we recorded 50 episodes across 5 distinct cube positions. For each position, we collected 10 episodes of pick-and-place interactions. This structure, repeating each variation several times, helped the model generalize better. We tried similar dataset with 25 episodes, and it was not enough leading to a bad performance. So, the data quality and quantity is definitely a key.
+After you have your dataset available on the Hub, you are good to go to use our finetuning script to adapt SmolVLA to your application.
+
+</Tip>
+
+## Finetune SmolVLA on your data
+
+Use [`smolvla_base`](https://hf.co/lerobot/smolvla_base), our pretrained 450M model, and fine-tune it on your data.
+Training the model for 20k steps will roughly take ~4 hrs on a single A100 GPU. You should tune the number of steps based on performance and your use-case.
+
+If you don't have a gpu device, you can train using our notebook on [![Google Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/huggingface/notebooks/blob/main/lerobot/training-smolvla.ipynb)
+
+Pass your dataset to the training script using `--dataset.repo_id`. If you want to test your installation, run the following command where we use one of the datasets we collected for the [SmolVLA Paper](https://huggingface.co/papers/2506.01844).
+
+```bash
+cd lerobot && lerobot-train \
+  --policy.path=lerobot/smolvla_base \
+  --dataset.repo_id=${HF_USER}/mydataset \
+  --batch_size=64 \
+  --steps=20000 \
+  --output_dir=outputs/train/my_smolvla \
+  --job_name=my_smolvla_training \
+  --policy.device=cuda \
+  --wandb.enable=true
+```
+
+<Tip>
+  You can start with a small batch size and increase it incrementally, if the
+  GPU allows it, as long as loading times remain short.
+</Tip>
+
+Fine-tuning is an art. For a complete overview of the options for finetuning, run
+
+```bash
+lerobot-train --help
+```
+
+<p align="center">
+  <img
+    src="https://cdn-uploads.huggingface.co/production/uploads/640e21ef3c82bd463ee5a76d/S-3vvVCulChREwHDkquoc.gif"
+    alt="Comparison of SmolVLA across task variations."
+    width="500"
+  />
+  <br />
+  <em>
+    Figure 2: Comparison of SmolVLA across task variations. From left to right:
+    (1) pick-place cube counting, (2) pick-place cube counting, (3) pick-place
+    cube counting under perturbations, and (4) generalization on pick-and-place
+    of the lego block with real-world SO101.
+  </em>
+</p>
+
+## Evaluate the finetuned model and run it in real-time
+
+Similarly for when recording an episode, it is recommended that you are logged in to the HuggingFace Hub. You can follow the corresponding steps: [Record a dataset](./il_robots).
+Once you are logged in, you can run inference in your setup by doing:
+
+```bash
+lerobot-record \
+  --robot.type=so101_follower \
+  --robot.port=/dev/ttyACM0 \ # <- Use your port
+  --robot.id=my_blue_follower_arm \ # <- Use your robot id
+  --robot.cameras="{ front: {type: opencv, index_or_path: 8, width: 640, height: 480, fps: 30}}" \ # <- Use your cameras
+  --dataset.single_task="Grasp a lego block and put it in the bin." \ # <- Use the same task description you used in your dataset recording
+  --dataset.repo_id=${HF_USER}/eval_DATASET_NAME_test \  # <- This will be the dataset name on HF Hub
+  --dataset.episode_time_s=50 \
+  --dataset.num_episodes=10 \
+  --dataset.streaming_encoding=true \
+  --dataset.encoder_threads=2 \
+  # --dataset.vcodec=auto \
+  # <- Teleop optional if you want to teleoperate in between episodes \
+  # --teleop.type=so100_leader \
+  # --teleop.port=/dev/ttyACM0 \
+  # --teleop.id=my_red_leader_arm \
+  --policy.path=HF_USER/FINETUNE_MODEL_NAME # <- Use your fine-tuned model
+```
+
+Depending on your evaluation setup, you can configure the duration and the number of episodes to record for your evaluation suite.
diff --git a/lerobot/docs/source/so100.mdx b/lerobot/docs/source/so100.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..399781ef4bcb07875f24628f593397b03bcd7b68
--- /dev/null
+++ b/lerobot/docs/source/so100.mdx
@@ -0,0 +1,640 @@
+# SO-100
+
+In the steps below, we explain how to assemble the SO-100 robot.
+
+## Source the parts
+
+Follow this [README](https://github.com/TheRobotStudio/SO-ARM100/blob/main/SO100.md). It contains the bill of materials, with a link to source the parts, as well as the instructions to 3D print the parts. And advise if it's your first time printing or if you don't own a 3D printer.
+
+## Install LeRobot 🤗
+
+To install LeRobot, follow our [Installation Guide](./installation)
+
+In addition to these instructions, you need to install the Feetech SDK:
+
+```bash
+pip install -e ".[feetech]"
+```
+
+## Configure the motors
+
+**Note:**
+Unlike the SO-101, the motor connectors are not easily accessible once the arm is assembled, so the configuration step must be done beforehand.
+
+### 1. Find the USB ports associated with each arm
+
+To find the port for each bus servo adapter, run this script:
+
+```bash
+lerobot-find-port
+```
+
+<hfoptions id="example">
+<hfoption id="Mac">
+
+Example output:
+
+```
+Finding all available ports for the MotorBus.
+['/dev/tty.usbmodem575E0032081', '/dev/tty.usbmodem575E0031751']
+Remove the USB cable from your MotorsBus and press Enter when done.
+
+[...Disconnect corresponding leader or follower arm and press Enter...]
+
+The port of this MotorsBus is /dev/tty.usbmodem575E0032081
+Reconnect the USB cable.
+```
+
+Where the found port is: `/dev/tty.usbmodem575E0032081` corresponding to your leader or follower arm.
+
+</hfoption>
+<hfoption id="Linux">
+
+On Linux, you might need to give access to the USB ports by running:
+
+```bash
+sudo chmod 666 /dev/ttyACM0
+sudo chmod 666 /dev/ttyACM1
+```
+
+Example output:
+
+```
+Finding all available ports for the MotorBus.
+['/dev/ttyACM0', '/dev/ttyACM1']
+Remove the usb cable from your MotorsBus and press Enter when done.
+
+[...Disconnect corresponding leader or follower arm and press Enter...]
+
+The port of this MotorsBus is /dev/ttyACM1
+Reconnect the USB cable.
+```
+
+Where the found port is: `/dev/ttyACM1` corresponding to your leader or follower arm.
+
+</hfoption>
+</hfoptions>
+
+### 2. Set the motors ids and baudrates
+
+Each motor is identified by a unique id on the bus. When brand new, motors usually come with a default id of `1`. For the communication to work properly between the motors and the controller, we first need to set a unique, different id to each motor. Additionally, the speed at which data is transmitted on the bus is determined by the baudrate. In order to talk to each other, the controller and all the motors need to be configured with the same baudrate.
+
+To that end, we first need to connect to each motor individually with the controller in order to set these. Since we will write these parameters in the non-volatile section of the motors' internal memory (EEPROM), we'll only need to do this once.
+
+If you are repurposing motors from another robot, you will probably also need to perform this step as the ids and baudrate likely won't match.
+
+#### Follower
+
+Connect the usb cable from your computer and the power supply to the follower arm's controller board. Then, run the following command or run the API example with the port you got from the previous step. You'll also need to give your leader arm a name with the `id` parameter.
+
+For a visual reference on how to set the motor ids please refer to [this video](https://huggingface.co/docs/lerobot/en/so101#setup-motors-video) where we follow the process for the SO101 arm.
+
+<hfoptions id="setup_motors">
+<hfoption id="Command">
+
+```bash
+lerobot-setup-motors \
+    --robot.type=so100_follower \
+    --robot.port=/dev/tty.usbmodem585A0076841  # <- paste here the port found at previous step
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+
+config = SO100FollowerConfig(
+    port="/dev/tty.usbmodem585A0076841",
+    id="my_awesome_follower_arm",
+)
+follower = SO100Follower(config)
+follower.setup_motors()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+You should see the following instruction
+
+```
+Connect the controller board to the 'gripper' motor only and press enter.
+```
+
+As instructed, plug the gripper's motor. Make sure it's the only motor connected to the board, and that the motor itself is not yet daisy-chained to any other motor. As you press `[Enter]`, the script will automatically set the id and baudrate for that motor.
+
+<details>
+<summary>Troubleshooting</summary>
+
+If you get an error at that point, check your cables and make sure they are plugged in properly:
+
+<ul>
+  <li>Power supply</li>
+  <li>USB cable between your computer and the controller board</li>
+  <li>The 3-pin cable from the controller board to the motor</li>
+</ul>
+
+If you are using a Waveshare controller board, make sure that the two jumpers are set on the `B` channel (USB).
+
+</details>
+
+You should then see the following message:
+
+```
+'gripper' motor id set to 6
+```
+
+Followed by the next instruction:
+
+```
+Connect the controller board to the 'wrist_roll' motor only and press enter.
+```
+
+You can disconnect the 3-pin cable from the controller board, but you can leave it connected to the gripper motor on the other end, as it will already be in the right place. Now, plug in another 3-pin cable to the wrist roll motor and connect it to the controller board. As with the previous motor, make sure it is the only motor connected to the board and that the motor itself isn't connected to any other one.
+
+Repeat the operation for each motor as instructed.
+
+> [!TIP]
+> Check your cabling at each step before pressing Enter. For instance, the power supply cable might disconnect as you manipulate the board.
+
+When you are done, the script will simply finish, at which point the motors are ready to be used. You can now plug the 3-pin cable from each motor to the next one, and the cable from the first motor (the 'shoulder pan' with id=1) to the controller board, which can now be attached to the base of the arm.
+
+#### Leader
+
+Do the same steps for the leader arm.
+
+<hfoptions id="setup_motors">
+<hfoption id="Command">
+```bash
+lerobot-setup-motors \
+    --teleop.type=so100_leader \
+    --teleop.port=/dev/tty.usbmodem575E0031751  # <- paste here the port found at previous step
+```
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.teleoperators.so_leader import SO100Leader, SO100LeaderConfig
+
+config = SO100LeaderConfig(
+    port="/dev/tty.usbmodem585A0076841",
+    id="my_awesome_leader_arm",
+)
+leader = SO100Leader(config)
+leader.setup_motors()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+## Step-by-Step Assembly Instructions
+
+## Remove the gears of the 6 leader motors
+
+<details>
+<summary><strong>Video removing gears</strong></summary>
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://github.com/user-attachments/assets/0c95b88c-5b85-413d-ba19-aee2f864f2a7"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+</details>
+
+Follow the video for removing gears. You need to remove the gear for the motors of the leader arm. As a result, you will only use the position encoding of the motor and reduce friction to more easily operate the leader arm.
+
+### Clean Parts
+
+Remove all support material from the 3D-printed parts. The easiest way to do this is using a small screwdriver to get underneath the support material.
+
+### Additional Guidance
+
+<details>
+<summary><strong>Video assembling arms</strong></summary>
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://github.com/user-attachments/assets/488a39de-0189-4461-9de3-05b015f90cca"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+</details>
+
+**Note:**
+This video provides visual guidance for assembling the arms, but it doesn't specify when or how to do the wiring. Inserting the cables beforehand is much easier than doing it afterward. The first arm may take a bit more than 1 hour to assemble, but once you get used to it, you can assemble the second arm in under 1 hour.
+
+---
+
+### First Motor
+
+**Step 2: Insert Wires**
+
+- Insert two wires into the first motor.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_1.webp"
+  style="height:300px;"
+/>
+
+**Step 3: Install in Base**
+
+- Place the first motor into the base.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_2.webp"
+  style="height:300px;"
+/>
+
+**Step 4: Secure Motor**
+
+- Fasten the motor with 4 screws. Two from the bottom and two from top.
+
+**Step 5: Attach Motor Holder**
+
+- Slide over the first motor holder and fasten it using two screws (one on each side).
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_4.webp"
+  style="height:300px;"
+/>
+
+**Step 6: Attach Motor Horns**
+
+- Install both motor horns, securing the top horn with a screw. Try not to move the motor position when attaching the motor horn, especially for the leader arms, where we removed the gears.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_5.webp"
+  style="height:300px;"
+/>
+
+<details>
+  <summary>
+    <strong>Video adding motor horn</strong>
+  </summary>
+  <video src="https://github.com/user-attachments/assets/ef3391a4-ad05-4100-b2bd-1699bf86c969"></video>
+</details>
+
+**Step 7: Attach Shoulder Part**
+
+- Route one wire to the back of the robot and the other to the left or towards you (see photo).
+- Attach the shoulder part.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_6.webp"
+  style="height:300px;"
+/>
+
+**Step 8: Secure Shoulder**
+
+- Tighten the shoulder part with 4 screws on top and 4 on the bottom
+  _(access bottom holes by turning the shoulder)._
+
+---
+
+### Second Motor Assembly
+
+**Step 9: Install Motor 2**
+
+- Slide the second motor in from the top and link the wire from motor 1 to motor 2.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_8.webp"
+  style="height:300px;"
+/>
+
+**Step 10: Attach Shoulder Holder**
+
+- Add the shoulder motor holder.
+- Ensure the wire from motor 1 to motor 2 goes behind the holder while the other wire is routed upward (see photo).
+- This part can be tight to assemble, you can use a workbench like the image or a similar setup to push the part around the motor.
+
+<div style="display: flex;">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_9.webp"
+    style="height:250px;"
+  />
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_10.webp"
+    style="height:250px;"
+  />
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_12.webp"
+    style="height:250px;"
+  />
+</div>
+
+**Step 11: Secure Motor 2**
+
+- Fasten the second motor with 4 screws.
+
+**Step 12: Attach Motor Horn**
+
+- Attach both motor horns to motor 2, again use the horn screw.
+
+**Step 13: Attach Base**
+
+- Install the base attachment using 2 screws.
+
+<img src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_11.webp" style="height:300px;">
+
+**Step 14: Attach Upper Arm**
+
+- Attach the upper arm with 4 screws on each side.
+
+<img src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_13.webp" style="height:300px;">
+
+---
+
+### Third Motor Assembly
+
+**Step 15: Install Motor 3**
+
+- Route the motor cable from motor 2 through the cable holder to motor 3, then secure motor 3 with 4 screws.
+
+**Step 16: Attach Motor Horn**
+
+- Attach both motor horns to motor 3 and secure one again with a horn screw.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_14.webp"
+  style="height:300px;"
+/>
+
+**Step 17: Attach Forearm**
+
+- Connect the forearm to motor 3 using 4 screws on each side.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_15.webp"
+  style="height:300px;"
+/>
+
+---
+
+### Fourth Motor Assembly
+
+**Step 18: Install Motor 4**
+
+- Slide in motor 4, attach the cable from motor 3, and secure the cable in its holder with a screw.
+
+<div style="display: flex;">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_16.webp"
+    style="height:300px;"
+  />
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_19.webp"
+    style="height:300px;"
+  />
+</div>
+
+**Step 19: Attach Motor Holder 4**
+
+- Install the fourth motor holder (a tight fit). Ensure one wire is routed upward and the wire from motor 3 is routed downward (see photo).
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_17.webp"
+  style="height:300px;"
+/>
+
+**Step 20: Secure Motor 4 & Attach Horn**
+
+- Fasten motor 4 with 4 screws and attach its motor horns, use for one a horn screw.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_18.webp"
+  style="height:300px;"
+/>
+
+---
+
+### Wrist Assembly
+
+**Step 21: Install Motor 5**
+
+- Insert motor 5 into the wrist holder and secure it with 2 front screws.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_20.webp"
+  style="height:300px;"
+/>
+
+**Step 22: Attach Wrist**
+
+- Connect the wire from motor 4 to motor 5. And already insert the other wire for the gripper.
+- Secure the wrist to motor 4 using 4 screws on both sides.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_22.webp"
+  style="height:300px;"
+/>
+
+**Step 23: Attach Wrist Horn**
+
+- Install only one motor horn on the wrist motor and secure it with a horn screw.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_23.webp"
+  style="height:300px;"
+/>
+
+---
+
+### Follower Configuration
+
+**Step 24: Attach Gripper**
+
+- Attach the gripper to motor 5.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_24.webp"
+  style="height:300px;"
+/>
+
+**Step 25: Install Gripper Motor**
+
+- Insert the gripper motor, connect the motor wire from motor 5 to motor 6, and secure it with 3 screws on each side.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_25.webp"
+  style="height:300px;"
+/>
+
+**Step 26: Attach Gripper Horn & Claw**
+
+- Attach the motor horns and again use a horn screw.
+- Install the gripper claw and secure it with 4 screws on both sides.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_26.webp"
+  style="height:300px;"
+/>
+
+**Step 27: Mount Controller**
+
+- Attach the motor controller to the back of the robot.
+
+<div style="display: flex;">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_27.webp"
+    style="height:300px;"
+  />
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_28.webp"
+    style="height:300px;"
+  />
+</div>
+
+_Assembly complete – proceed to Leader arm assembly._
+
+---
+
+### Leader Configuration
+
+For the leader configuration, perform **Steps 1–23**. Make sure that you removed the motor gears from the motors.
+
+**Step 24: Attach Leader Holder**
+
+- Mount the leader holder onto the wrist and secure it with a screw.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_29.webp"
+  style="height:300px;"
+/>
+
+**Step 25: Attach Handle**
+
+- Attach the handle to motor 5 using 4 screws.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_30.webp"
+  style="height:300px;"
+/>
+
+**Step 26: Install Gripper Motor**
+
+- Insert the gripper motor, secure it with 3 screws on each side, attach a motor horn using a horn screw, and connect the motor wire.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_31.webp"
+  style="height:300px;"
+/>
+
+**Step 27: Attach Trigger**
+
+- Attach the follower trigger with 4 screws.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_32.webp"
+  style="height:300px;"
+/>
+
+**Step 28: Mount Controller**
+
+- Attach the motor controller to the back of the robot.
+
+<div style="display: flex;">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_27.webp"
+    style="height:300px;"
+  />
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/so100_assembly_28.webp"
+    style="height:300px;"
+  />
+</div>
+
+## Calibrate
+
+Next, you'll need to calibrate your robot to ensure that the leader and follower arms have the same position values when they are in the same physical position.
+The calibration process is very important because it allows a neural network trained on one robot to work on another.
+
+#### Follower
+
+Run the following command or API example to calibrate the follower arm:
+
+<hfoptions id="calibrate_follower">
+<hfoption id="Command">
+
+```bash
+lerobot-calibrate \
+    --robot.type=so100_follower \
+    --robot.port=/dev/tty.usbmodem58760431551 \ # <- The port of your robot
+    --robot.id=my_awesome_follower_arm # <- Give the robot a unique name
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.robots.so_follower import SO100FollowerConfig, SO100Follower
+
+config = SO100FollowerConfig(
+    port="/dev/tty.usbmodem585A0076891",
+    id="my_awesome_follower_arm",
+)
+
+follower = SO100Follower(config)
+follower.connect(calibrate=False)
+follower.calibrate()
+follower.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+We unified the calibration method for most robots. Thus, the calibration steps for this SO100 arm are the same as the steps for the Koch and SO101. First, we have to move the robot to the position where each joint is in the middle of its range, then we press `Enter`. Secondly, we move all joints through their full range of motion. A video of this same process for the SO101 as reference can be found [here](https://huggingface.co/docs/lerobot/en/so101#calibration-video)
+
+#### Leader
+
+Do the same steps to calibrate the leader arm, run the following command or API example:
+
+<hfoptions id="calibrate_leader">
+<hfoption id="Command">
+
+```bash
+lerobot-calibrate \
+    --teleop.type=so100_leader \
+    --teleop.port=/dev/tty.usbmodem58760431551 \ # <- The port of your robot
+    --teleop.id=my_awesome_leader_arm # <- Give the robot a unique name
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.teleoperators.so_leader import SO100LeaderConfig, SO100Leader
+
+config = SO100LeaderConfig(
+    port="/dev/tty.usbmodem58760431551",
+    id="my_awesome_leader_arm",
+)
+
+leader = SO100Leader(config)
+leader.connect(calibrate=False)
+leader.calibrate()
+leader.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+Congrats 🎉, your robot is all set to learn a task on its own. Start training it by following this tutorial: [Getting started with real-world robots](./il_robots)
+
+> [!TIP]
+> If you have any questions or need help, please reach out on [Discord](https://discord.com/invite/s3KuuzsPFb).
diff --git a/lerobot/docs/source/so101.mdx b/lerobot/docs/source/so101.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..7c9df588a1129374213bf66072a8b8ef6c8901d3
--- /dev/null
+++ b/lerobot/docs/source/so101.mdx
@@ -0,0 +1,449 @@
+# SO-101
+
+<div style="display: flex; align-items: center; gap: 10px;">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/SO101_Follower.webp"
+    alt="SO-101"
+    width="60%"
+  />
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/SO101_Leader.webp"
+    alt="SO-101"
+    width="60%"
+  />
+</div>
+
+In the steps below, we explain how to assemble our flagship robot, the SO-101.
+
+## Source the parts
+
+Follow this [README](https://github.com/TheRobotStudio/SO-ARM100). It contains the bill of materials, with a link to source the parts, as well as the instructions to 3D print the parts.
+And advise if it's your first time printing or if you don't own a 3D printer.
+
+## Install LeRobot 🤗
+
+To install LeRobot, follow our [Installation Guide](./installation)
+
+In addition to these instructions, you need to install the Feetech SDK:
+
+```bash
+pip install -e ".[feetech]"
+```
+
+## Step-by-Step Assembly Instructions
+
+The follower arm uses 6x STS3215 motors with 1/345 gearing. The leader, however, uses three differently geared motors to make sure it can both sustain its own weight and it can be moved without requiring much force. Which motor is needed for which joint is shown in the table below.
+
+| Leader-Arm Axis     | Motor | Gear Ratio |
+| ------------------- | :---: | :--------: |
+| Base / Shoulder Pan |   1   |  1 / 191   |
+| Shoulder Lift       |   2   |  1 / 345   |
+| Elbow Flex          |   3   |  1 / 191   |
+| Wrist Flex          |   4   |  1 / 147   |
+| Wrist Roll          |   5   |  1 / 147   |
+| Gripper             |   6   |  1 / 147   |
+
+## Configure the motors
+
+### 1. Find the USB ports associated with each arm
+
+To find the port for each bus servo adapter, connect MotorBus to your computer via USB and power. Run the following script and disconnect the MotorBus when prompted:
+
+```bash
+lerobot-find-port
+```
+
+<hfoptions id="example">
+<hfoption id="Mac">
+
+Example output:
+
+```
+Finding all available ports for the MotorBus.
+['/dev/tty.usbmodem575E0032081', '/dev/tty.usbmodem575E0031751']
+Remove the USB cable from your MotorsBus and press Enter when done.
+
+[...Disconnect corresponding leader or follower arm and press Enter...]
+
+The port of this MotorsBus is /dev/tty.usbmodem575E0032081
+Reconnect the USB cable.
+```
+
+Where the found port is: `/dev/tty.usbmodem575E0032081` corresponding to your leader or follower arm.
+
+</hfoption>
+<hfoption id="Linux">
+
+On Linux, you might need to give access to the USB ports by running:
+
+```bash
+sudo chmod 666 /dev/ttyACM0
+sudo chmod 666 /dev/ttyACM1
+```
+
+Example output:
+
+```
+Finding all available ports for the MotorBus.
+['/dev/ttyACM0', '/dev/ttyACM1']
+Remove the usb cable from your MotorsBus and press Enter when done.
+
+[...Disconnect corresponding leader or follower arm and press Enter...]
+
+The port of this MotorsBus is /dev/ttyACM1
+Reconnect the USB cable.
+```
+
+Where the found port is: `/dev/ttyACM1` corresponding to your leader or follower arm.
+
+</hfoption>
+</hfoptions>
+
+### 2. Set the motors ids and baudrates
+
+Each motor is identified by a unique id on the bus. When brand new, motors usually come with a default id of `1`. For the communication to work properly between the motors and the controller, we first need to set a unique, different id to each motor. Additionally, the speed at which data is transmitted on the bus is determined by the baudrate. In order to talk to each other, the controller and all the motors need to be configured with the same baudrate.
+
+To that end, we first need to connect to each motor individually with the controller in order to set these. Since we will write these parameters in the non-volatile section of the motors' internal memory (EEPROM), we'll only need to do this once.
+
+If you are repurposing motors from another robot, you will probably also need to perform this step as the ids and baudrate likely won't match.
+
+The video below shows the sequence of steps for setting the motor ids.
+
+##### Setup motors video
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/setup_motors_so101_2.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+#### Follower
+
+Connect the usb cable from your computer and the power supply to the follower arm's controller board. Then, run the following command or run the API example with the port you got from the previous step. You'll also need to give your leader arm a name with the `id` parameter.
+
+<hfoptions id="setup_motors">
+<hfoption id="Command">
+
+```bash
+lerobot-setup-motors \
+    --robot.type=so101_follower \
+    --robot.port=/dev/tty.usbmodem585A0076841  # <- paste here the port found at previous step
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.robots.so_follower import SO101Follower, SO101FollowerConfig
+
+config = SO101FollowerConfig(
+    port="/dev/tty.usbmodem585A0076841",
+    id="my_awesome_follower_arm",
+)
+follower = SO101Follower(config)
+follower.setup_motors()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+You should see the following instruction
+
+```bash
+Connect the controller board to the 'gripper' motor only and press enter.
+```
+
+As instructed, plug the gripper's motor. Make sure it's the only motor connected to the board, and that the motor itself is not yet daisy-chained to any other motor. As you press `[Enter]`, the script will automatically set the id and baudrate for that motor.
+
+<details>
+<summary>Troubleshooting</summary>
+
+If you get an error at that point, check your cables and make sure they are plugged in properly:
+
+<ul>
+  <li>Power supply</li>
+  <li>USB cable between your computer and the controller board</li>
+  <li>The 3-pin cable from the controller board to the motor</li>
+</ul>
+
+If you are using a Waveshare controller board, make sure that the two jumpers are set on the `B` channel (USB).
+
+</details>
+
+You should then see the following message:
+
+```bash
+'gripper' motor id set to 6
+```
+
+Followed by the next instruction:
+
+```bash
+Connect the controller board to the 'wrist_roll' motor only and press enter.
+```
+
+You can disconnect the 3-pin cable from the controller board, but you can leave it connected to the gripper motor on the other end, as it will already be in the right place. Now, plug in another 3-pin cable to the wrist roll motor and connect it to the controller board. As with the previous motor, make sure it is the only motor connected to the board and that the motor itself isn't connected to any other one.
+
+Repeat the operation for each motor as instructed.
+
+> [!TIP]
+> Check your cabling at each step before pressing Enter. For instance, the power supply cable might disconnect as you manipulate the board.
+
+When you are done, the script will simply finish, at which point the motors are ready to be used. You can now plug the 3-pin cable from each motor to the next one, and the cable from the first motor (the 'shoulder pan' with id=1) to the controller board, which can now be attached to the base of the arm.
+
+#### Leader
+
+Do the same steps for the leader arm.
+
+<hfoptions id="setup_motors">
+<hfoption id="Command">
+
+```bash
+lerobot-setup-motors \
+    --teleop.type=so101_leader \
+    --teleop.port=/dev/tty.usbmodem575E0031751  # <- paste here the port found at previous step
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.teleoperators.so_leader import SO101Leader, SO101LeaderConfig
+
+config = SO101LeaderConfig(
+    port="/dev/tty.usbmodem585A0076841",
+    id="my_awesome_leader_arm",
+)
+leader = SO101Leader(config)
+leader.setup_motors()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+### Clean Parts
+
+Remove all support material from the 3D-printed parts. The easiest way to do this is using a small screwdriver to get underneath the support material.
+
+It is advisable to install one 3-pin cable in the motor after placing them before continuing assembly.
+
+### Joint 1
+
+- Place the first motor into the base.
+- Fasten the motor with 4 M2x6mm screws (smallest screws). Two from the top and two from the bottom.
+- Slide over the first motor holder and fasten it using two M2x6mm screws (one on each side).
+- Install both motor horns, securing the top horn with a M3x6mm screw.
+- Attach the shoulder part.
+- Tighten the shoulder part with 4 M3x6mm screws on top and 4 M3x6mm screws on the bottom
+- Add the shoulder motor holder.
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/Joint1_v2.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+### Joint 2
+
+- Slide the second motor in from the top.
+- Fasten the second motor with 4 M2x6mm screws.
+- Attach both motor horns to motor 2, again use the M3x6mm horn screw.
+- Attach the upper arm with 4 M3x6mm screws on each side.
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/Joint2_v2.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+### Joint 3
+
+- Insert motor 3 and fasten using 4 M2x6mm screws
+- Attach both motor horns to motor 3 and secure one again with a M3x6mm horn screw.
+- Connect the forearm to motor 3 using 4 M3x6mm screws on each side.
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/Joint3_v2.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+### Joint 4
+
+- Slide over motor holder 4.
+- Slide in motor 4.
+- Fasten motor 4 with 4 M2x6mm screws and attach its motor horns, use a M3x6mm horn screw.
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/Joint4_v2.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+### Joint 5
+
+- Insert motor 5 into the wrist holder and secure it with 2 M2x6mm front screws.
+- Install only one motor horn on the wrist motor and secure it with a M3x6mm horn screw.
+- Secure the wrist to motor 4 using 4 M3x6mm screws on both sides.
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/Joint5_v2.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+### Gripper / Handle
+
+<hfoptions id="assembly">
+<hfoption id="Follower">
+
+- Attach the gripper to motor 5, attach it to the motor horn on the wrist using 4 M3x6mm screws.
+- Insert the gripper motor and secure it with 2 M2x6mm screws on each side.
+- Attach the motor horns and again use a M3x6mm horn screw.
+- Install the gripper claw and secure it with 4 M3x6mm screws on both sides.
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/Gripper_v2.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+</hfoption>
+<hfoption id="Leader">
+
+- Mount the leader holder onto the wrist and secure it with 4 M3x6mm screws.
+- Attach the handle to motor 5 using 1 M2x6mm screw.
+- Insert the gripper motor, secure it with 2 M2x6mm screws on each side, attach a motor horn using a M3x6mm horn screw.
+- Attach the follower trigger with 4 M3x6mm screws.
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/Leader_v2.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+</hfoption>
+</hfoptions>
+
+## Calibrate
+
+Next, you'll need to calibrate your robot to ensure that the leader and follower arms have the same position values when they are in the same physical position.
+The calibration process is very important because it allows a neural network trained on one robot to work on another.
+
+#### Follower
+
+Run the following command or API example to calibrate the follower arm:
+
+<hfoptions id="calibrate_follower">
+<hfoption id="Command">
+
+```bash
+lerobot-calibrate \
+    --robot.type=so101_follower \
+    --robot.port=/dev/tty.usbmodem58760431551 \ # <- The port of your robot
+    --robot.id=my_awesome_follower_arm # <- Give the robot a unique name
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.robots.so_follower import SO101FollowerConfig, SO101Follower
+
+config = SO101FollowerConfig(
+    port="/dev/tty.usbmodem585A0076891",
+    id="my_awesome_follower_arm",
+)
+
+follower = SO101Follower(config)
+follower.connect(calibrate=False)
+follower.calibrate()
+follower.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+The video below shows how to perform the calibration. First you need to move the robot to the position where all joints are in the middle of their ranges. Then after pressing enter you have to move each joint through its full range of motion.
+
+##### Calibration video
+
+<div class="video-container">
+  <video controls width="600">
+    <source
+      src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/calibrate_so101_2.mp4"
+      type="video/mp4"
+    />
+  </video>
+</div>
+
+#### Leader
+
+Do the same steps to calibrate the leader arm, run the following command or API example:
+
+<hfoptions id="calibrate_leader">
+<hfoption id="Command">
+
+```bash
+lerobot-calibrate \
+    --teleop.type=so101_leader \
+    --teleop.port=/dev/tty.usbmodem58760431551 \ # <- The port of your robot
+    --teleop.id=my_awesome_leader_arm # <- Give the robot a unique name
+```
+
+</hfoption>
+<hfoption id="API example">
+
+<!-- prettier-ignore-start -->
+```python
+from lerobot.teleoperators.so_leader import SO101LeaderConfig, SO101Leader
+
+config = SO101LeaderConfig(
+    port="/dev/tty.usbmodem58760431551",
+    id="my_awesome_leader_arm",
+)
+
+leader = SO101Leader(config)
+leader.connect(calibrate=False)
+leader.calibrate()
+leader.disconnect()
+```
+<!-- prettier-ignore-end -->
+
+</hfoption>
+</hfoptions>
+
+Congrats 🎉, your robot is all set to learn a task on its own. Start training it by following this tutorial: [Getting started with real-world robots](./il_robots)
+
+> [!TIP]
+> If you have any questions or need help, please reach out on [Discord](https://discord.com/invite/s3KuuzsPFb).
diff --git a/lerobot/docs/source/streaming_video_encoding.mdx b/lerobot/docs/source/streaming_video_encoding.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..40004200e94f06bcb337eac6257d5593466a2603
--- /dev/null
+++ b/lerobot/docs/source/streaming_video_encoding.mdx
@@ -0,0 +1,155 @@
+# Streaming Video Encoding Guide
+
+## 1. Overview
+
+Streaming video encoding eliminates the traditional PNG round-trip during video dataset recording. Instead of:
+
+1. Capture frame -> write PNG to disk -> (at episode end) read PNG's -> encode to MP4 -> delete PNG's
+
+Frames can be encoded in real-time during capture:
+
+1. Capture frame -> queue to encoder thread -> encode to MP4 directly
+
+This makes `save_episode()` near-instant (the video is already encoded by the time the episode ends) and removes the blocking wait that previously occurred between episodes, especially with multiple cameras in long episodes.
+
+## 2. Tuning Parameters
+
+| Parameter               | CLI Flag                          | Type          | Default       | Description                                                       |
+| ----------------------- | --------------------------------- | ------------- | ------------- | ----------------------------------------------------------------- |
+| `streaming_encoding`    | `--dataset.streaming_encoding`    | `bool`        | `True`        | Enable real-time encoding during capture                          |
+| `vcodec`                | `--dataset.vcodec`                | `str`         | `"libsvtav1"` | Video codec. `"auto"` detects best HW encoder                     |
+| `encoder_threads`       | `--dataset.encoder_threads`       | `int \| None` | `None` (auto) | Threads per encoder instance. `None` will leave the vcoded decide |
+| `encoder_queue_maxsize` | `--dataset.encoder_queue_maxsize` | `int`         | `60`          | Max buffered frames per camera (~2s at 30fps). Consumes RAM       |
+
+## 3. Performance Considerations
+
+Streaming encoding means the CPU is encoding video **during** the capture loop, not after. This creates a CPU budget that must be shared between:
+
+- **Control loop** (reading cameras, control the robot, writing non-video data)
+- **Encoder threads** (one pool per camera)
+- **Rerun visualization** (if enabled)
+- **OS and other processes**
+
+### Resolution & Number of Cameras Impact
+
+| Setup                     | Throughput (px/sec) | CPU Encoding Load | Notes                          |
+| ------------------------- | ------------------- | ----------------- | ------------------------------ |
+| 2camsx 640x480x3 @30fps   | 55M                 | Low               | Works on most systems          |
+| 2camsx 1280x720x3 @30fps  | 165M                | Moderate          | Comfortable on modern systems  |
+| 2camsx 1920x1080x3 @30fps | 373M                | High              | Requires powerful high-end CPU |
+
+### `encoder_threads` Tuning
+
+This parameter controls how many threads each encoder instance uses internally:
+
+- **Higher values** (e.g., 4-5): Faster encoding, but uses more CPU cores per camera. Good for high-end systems with many cores.
+- **Lower values** (e.g., 1-2): Less CPU per camera, freeing cores for capture and visualization. Good for low-res images and capable CPUs.
+- **`None` (default)**: Lets the codec decide. Information available in the codec logs.
+
+### Backpressure and Frame Dropping
+
+Each camera has a bounded queue (`encoder_queue_maxsize`, default 60 frames). When the encoder can't keep up:
+
+1. The queue fills up (consuming RAM)
+2. New frames are **dropped** (not blocked) — the capture loop continues uninterrupted
+3. A warning is logged: `"Encoder queue full for {camera}, dropped N frame(s)"`
+4. At episode end, total dropped frames per camera are reported
+
+### Symptoms of Encoder Falling Behind
+
+- **System feels laggy and freezes**: all CPUs are at 100%
+- **Dropped frame warnings** in the log or lower frames/FPS than expected in the recorded dataset
+- **Choppy robot movement**: If CPU is severely overloaded, even the capture loop may be affected
+- **Accumulated rerun lag**: Visualization falls behind real-time
+
+## 4. Hardware-Accelerated Encoding
+
+### When to Use
+
+Use HW encoding when:
+
+- CPU is the bottleneck (dropped frames, choppy robot, rerun lag)
+- You have compatible hardware (GPU or dedicated encoder)
+- You're recording at high throughput (high resolution or with many cameras)
+
+### Choosing a Codec
+
+| Codec                 | CPU Usage | File Size      | Quality | Notes                                                            |
+| --------------------- | --------- | -------------- | ------- | ---------------------------------------------------------------- |
+| `libsvtav1` (default) | High      | Smallest       | Best    | Default. Best compression but most CPU-intensive                 |
+| `h264`                | Medium    | ~30-50% larger | Good    | Software H.264. Lower CPU                                        |
+| HW encoders           | Very Low  | Largest        | Good    | Offloads to dedicated hardware. Best for CPU-constrained systems |
+
+### Available HW Encoders
+
+| Encoder             | Platform      | Hardware                                                                                         | CLI Value                            |
+| ------------------- | ------------- | ------------------------------------------------------------------------------------------------ | ------------------------------------ |
+| `h264_videotoolbox` | macOS         | Apple Silicon / Intel                                                                            | `--dataset.vcodec=h264_videotoolbox` |
+| `hevc_videotoolbox` | macOS         | Apple Silicon / Intel                                                                            | `--dataset.vcodec=hevc_videotoolbox` |
+| `h264_nvenc`        | Linux/Windows | NVIDIA GPU                                                                                       | `--dataset.vcodec=h264_nvenc`        |
+| `hevc_nvenc`        | Linux/Windows | NVIDIA GPU                                                                                       | `--dataset.vcodec=hevc_nvenc`        |
+| `h264_vaapi`        | Linux         | Intel/AMD GPU                                                                                    | `--dataset.vcodec=h264_vaapi`        |
+| `h264_qsv`          | Linux/Windows | Intel Quick Sync                                                                                 | `--dataset.vcodec=h264_qsv`          |
+| `auto`              | Any           | Probes the system for available HW encoders. Falls back to `libsvtav1` if no HW encoder is found | `--dataset.vcodec=auto`              |
+
+> [!NOTE]
+> In order to use the HW accelerated encoders you might need to upgrade your GPU drivers.
+
+> [!NOTE]
+> `libsvtav1` is the default because it provides the best training performance; other vcodecs can reduce CPU usage and be faster, but they typically produce larger files and may affect training time.
+
+## 5. Troubleshooting
+
+| Symptom                                                            | Likely Cause                                 | Fix                                                                                                                                                                                                                                                                                  |
+| ------------------------------------------------------------------ | -------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
+| System freezes or choppy robot movement or Rerun visualization lag | CPU starved (100% load usage)                | Close other apps, reduce encoding throughput, lower `encoder_threads`, use `h264`, use `display_data=False`. If the CPU continues to be at 100% then it might be insufficient for your setup, consider `--dataset.streaming_encoding=false` or HW encoding (`--dataset.vcodec=auto`) |
+| "Encoder queue full" warnings or dropped frames in dataset         | Encoder can't keep up (Queue overflow)       | If CPU is not at 100%: Increase `encoder_threads`, increase `encoder_queue_maxsize` or use HW encoding (`--dataset.vcodec=auto`).                                                                                                                                                    |
+| High RAM usage                                                     | Queue filling faster than encoding           | `encoder_threads` too low or CPU insufficient. Reduce `encoder_queue_maxsize` or use HW encoding                                                                                                                                                                                     |
+| Large video files                                                  | Using HW encoder or H.264                    | Expected trade-off. Switch to `libsvtav1` if CPU allows                                                                                                                                                                                                                              |
+| `save_episode()` still slow                                        | `streaming_encoding` is `False`              | Set `--dataset.streaming_encoding=true`                                                                                                                                                                                                                                              |
+| Encoder thread crash                                               | Codec not available or invalid settings      | Check `vcodec` is installed, try `--dataset.vcodec=auto`                                                                                                                                                                                                                             |
+| Recorded dataset is missing frames                                 | CPU/GPU starvation or occasional load spikes | If ~5% of frames are missing, your system is likely overloaded — follow the recommendations above. If fewer frames are missing (~2%), they are probably due to occasional transient load spikes (often at startup) and can be considered expected.                                   |
+
+## 6. Recommended Configurations
+
+These estimates are conservative; we recommend testing them on your setup—start with a low load and increase it gradually.
+
+### High-End Systems: modern 12+ cores (24+ threads)
+
+A throughput between ~250-500M px/sec should be comfortable in CPU. For even better results try HW encoding if available.
+
+```bash
+# 3camsx 1280x720x3 @30fps: Defaults work well. Optionally increase encoder parallelism.
+# 2camsx 1920x1080x3 @30fps: Defaults work well. Optionally increase encoder parallelism.
+lerobot-record --dataset.encoder_threads=5 ...
+
+# 3camsx 1920x1080x3 @30fps: Might require some tuning.
+```
+
+### Mid-Range Systems: modern 8+ cores (16+ threads) or Apple Silicon
+
+A throughput between ~80-300M px/sec should be possible in CPU.
+
+```bash
+# 3camsx 640x480x3 @30fps: Defaults work well. Optionally decrease encoder parallelism.
+# 2camsx 1280x720x3 @30fps: Defaults work well. Optionally decrease encoder parallelism.
+lerobot-record --dataset.encoder_threads=2 ...
+
+# 2camsx 1920x1080x3 @30fps: Might require some tuning.
+```
+
+### Low-Resource Systems: modern 4+ cores (8+ threads) or Raspberry Pi 5
+
+On very constrained systems, streaming encoding may compete too heavily with the capture loop. Disabling it falls back to the PNG-based approach where encoding happens between episodes (blocking, but doesn't interfere with capture). Alternatively, record at a lower throughput to reduce both capture and encoding load. Consider also changing codec to `h264` and using batch encoding.
+
+```bash
+# 2camsx 640x480x3 @30fps: Requires some tuning.
+
+# Use H.264, disable streaming, consider batching encoding
+lerobot-record --dataset.vcodec=h264 --dataset.streaming_encoding=false ...
+```
+
+## 7. Closing note
+
+Performance ultimately depends on your exact setup — frames-per-second, resolution, CPU cores and load, available memory, episode length, and the encoder you choose. Always test with your target workload, be mindful about your CPU & system capabilities and tune `encoder_threads`, `encoder_queue_maxsize`, and
+`vcodec` reasonably. That said, a common practical configuration (for many applications) is three cameras at 640×480x3 @30fps; this usually runs fine with the default streaming video encoding settings in modern systems. Always verify your recorded dataset is healthy by comparing the video duration to the CLI episode duration and confirming the row count equals FPS × CLI duration.
diff --git a/lerobot/docs/source/torch_accelerators.mdx b/lerobot/docs/source/torch_accelerators.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..6dfbfbad5320300146354e6902b21c0fe6e3a815
--- /dev/null
+++ b/lerobot/docs/source/torch_accelerators.mdx
@@ -0,0 +1,42 @@
+# PyTorch accelerators
+
+LeRobot supports multiple hardware acceleration options for both training and inference.
+
+These options include:
+
+- **CPU**: CPU executes all computations, no dedicated accelerator is used
+- **CUDA**: acceleration with NVIDIA & AMD GPUs
+- **MPS**: acceleration with Apple Silicon GPUs
+- **XPU**: acceleration with Intel integrated and discrete GPUs
+
+## Getting Started
+
+To use particular accelerator, a suitable version of PyTorch should be installed.
+
+For CPU, CUDA, and MPS backends follow instructions provided on [PyTorch installation page](https://pytorch.org/get-started/locally).
+For XPU backend, follow instructions from [PyTorch documentation](https://docs.pytorch.org/docs/stable/notes/get_start_xpu.html).
+
+### Verifying the installation
+
+After installation, accelerator availability can be verified by running
+
+```python
+import torch
+print(torch.<backend_name>.is_available())  # <backend_name> is cuda, mps, or xpu
+```
+
+## How to run training or evaluation
+
+To select the desired accelerator, use the `--policy.device` flag when running `lerobot-train` or `lerobot-eval`. For example, to use MPS on Apple Silicon, run:
+
+```bash
+lerobot-train
+    --policy.device=mps ...
+```
+
+```bash
+lerobot-eval \
+    --policy.device=mps ...
+```
+
+However, in most cases, presence of an accelerator is detected automatically and `policy.device` parameter can be omitted from CLI commands.
diff --git a/lerobot/docs/source/unitree_g1.mdx b/lerobot/docs/source/unitree_g1.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..2e615085e098ad9d19a0da2171f36320ee64d070
--- /dev/null
+++ b/lerobot/docs/source/unitree_g1.mdx
@@ -0,0 +1,302 @@
+# Unitree G1
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/unitree_thumbnail.jpg"
+  alt="Unitree G1 locomanipulation demo"
+  style={{ width: "100%" }}
+/>
+
+The Unitree G1 humanoid is now supported in LeRobot! You can teleoperate, train locomanipulation policies, test in sim, and more. Both 29 and 23 DoF variants are supported.
+
+---
+
+## Part 1: Getting Started
+
+### Install the Unitree SDK
+
+Follow the [unitree_sdk2_python installation guide](https://github.com/unitreerobotics/unitree_sdk2_python#installation). Tested with `unitree_sdk2py==1.0.1` and `cyclonedds==0.10.2`:
+
+```bash
+conda create -y -n lerobot python=3.12
+conda activate lerobot
+git clone https://github.com/unitreerobotics/unitree_sdk2_python.git
+cd unitree_sdk2_python
+pip install -e .
+cd ..
+```
+
+### Install LeRobot
+
+```bash
+conda install ffmpeg -c conda-forge
+conda install -c conda-forge "pinocchio>=3.0.0,<4.0.0"
+git clone https://github.com/huggingface/lerobot.git
+cd lerobot
+pip install -e '.[unitree_g1]'
+```
+
+<Tip>
+  For now, pinocchio must be installed from conda-forge (not pip) to include the
+  CasADi bindings needed for arm IK.
+</Tip>
+
+### Test the Installation (Simulation)
+
+The simulation environment has its own dependencies. Check the Simulation environment dependencies: [Unitree G1 Mujoco EnvHub](https://huggingface.co/lerobot/unitree-g1-mujoco/tree/main).
+
+```bash
+pip install mujoco loguru msgpack msgpack-numpy
+```
+
+```bash
+lerobot-teleoperate \
+  --robot.type=unitree_g1 \
+  --robot.is_simulation=true \
+  --teleop.type=unitree_g1 \
+  --teleop.id=wbc_unitree \
+  --robot.cameras='{"global_view": {"type": "zmq", "server_address": "localhost", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30, "warmup_s": 5}}' \
+  --display_data=true \
+  --robot.controller=GrootLocomotionController
+```
+
+This will launch a [MuJoCo sim instance](https://huggingface.co/lerobot/unitree-g1-mujoco/tree/main) for the G1. You can connect a gamepad to your machine before launching in order to control the robot's locomotion in sim. We support both [HolosomaLocomotionController](https://github.com/amazon-far/holosoma) and [GrootLocomotionController](https://github.com/NVlabs/GR00T-WholeBodyControl) via `--robot.controller`.
+
+- Press `9` to release the robot
+- Press `7` / `8` to increase / decrease waist height
+
+### Connect to the Physical Robot
+
+The G1's Ethernet IP is fixed at `192.168.123.164`. Your machine must have a static IP on the same subnet: `192.168.123.x` where `x ≠ 164`.
+
+```bash
+# Replace 'enp131s0' with your ethernet interface name (check with `ip a`)
+sudo ip addr flush dev enp131s0
+sudo ip addr add 192.168.123.200/24 dev enp131s0
+sudo ip link set enp131s0 up
+```
+
+### SSH into the Robot
+
+```bash
+ssh unitree@192.168.123.164
+# Password: 123
+```
+
+### Share Internet via Ethernet
+
+The G1 needs internet access to clone repos and install packages. Share your laptop's connection over Ethernet:
+
+**On your laptop:**
+
+```bash
+sudo sysctl -w net.ipv4.ip_forward=1
+
+# Replace wlp132s0f0 with your WiFi interface name
+sudo iptables -t nat -A POSTROUTING -o wlp132s0f0 -s 192.168.123.0/24 -j MASQUERADE
+sudo iptables -A FORWARD -i wlp132s0f0 -o enp131s0 -m state --state RELATED,ESTABLISHED -j ACCEPT
+sudo iptables -A FORWARD -i enp131s0 -o wlp132s0f0 -j ACCEPT
+```
+
+**On the G1:**
+
+```bash
+sudo ip route del default 2>/dev/null || true
+sudo ip route add default via 192.168.123.200 dev eth0
+echo "nameserver 8.8.8.8" | sudo tee /etc/resolv.conf
+
+# Verify
+ping -c 3 8.8.8.8
+```
+
+### Install the Unitree SDK on the G1
+
+Follow the [unitree_sdk2_python installation guide](https://github.com/unitreerobotics/unitree_sdk2_python#installation):
+
+```bash
+conda create -y -n lerobot python=3.12
+conda activate lerobot
+git clone https://github.com/unitreerobotics/unitree_sdk2_python.git
+cd unitree_sdk2_python
+python -m pip install -e .
+cd ..
+```
+
+### Install LeRobot on the G1
+
+```bash
+git clone https://github.com/huggingface/lerobot.git
+cd lerobot
+conda install -c conda-forge "pinocchio>=3.0.0,<4.0.0"
+python -m pip install -e '.[unitree_g1]'
+```
+
+<Tip>
+  For now, pinocchio must be installed from conda-forge (not pip) to include the
+  CasADi bindings needed for arm IK.
+</Tip>
+
+### (Optional) Enable WiFi on the Robot
+
+For wireless SSH access, you can enable WiFi on the G1 (it's blocked by default):
+
+```bash
+sudo rfkill unblock all
+sudo ip link set wlan0 up
+sudo nmcli radio wifi on
+sudo nmcli device set wlan0 managed yes
+sudo systemctl restart NetworkManager
+```
+
+**Connect to a WiFi network:**
+
+```bash
+nmcli device wifi list
+
+sudo nmcli connection add type wifi ifname wlan0 con-name "YourNetwork" ssid "YourNetwork"
+sudo nmcli connection modify "YourNetwork" wifi-sec.key-mgmt wpa-psk
+sudo nmcli connection modify "YourNetwork" wifi-sec.psk "YourPassword"
+sudo nmcli connection modify "YourNetwork" connection.autoconnect yes
+sudo nmcli connection up "YourNetwork"
+
+ip a show wlan0
+```
+
+You can then SSH over WiFi instead of Ethernet:
+
+```bash
+ssh unitree@<ROBOT_WIFI_IP>
+# Password: 123
+```
+
+---
+
+## Part 2: Teleoperation & Locomotion
+
+### Run the Robot Server
+
+On the robot (from `~/lerobot`):
+
+```bash
+cd ~/lerobot
+python src/lerobot/robots/unitree_g1/run_g1_server.py --camera
+```
+
+### Run the Locomotion Policy
+
+You can run the teleoperation client from your laptop over Ethernet, over WiFi (experimental), or directly on the robot itself. Mind potential latency introduced by your network.
+
+**From your laptop:**
+
+```bash
+lerobot-teleoperate \
+  --robot.type=unitree_g1 \
+  --robot.is_simulation=false \
+  --robot.robot_ip=<ROBOT_IP> \
+  --teleop.type=unitree_g1 \
+  --teleop.id=wbc_unitree \
+  --robot.cameras='{"global_view": {"type": "zmq", "server_address": "<ROBOT_IP>", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30}}' \
+  --display_data=true \
+  --robot.controller=HolosomaLocomotionController
+```
+
+We support both [GrootLocomotionController](https://github.com/NVlabs/GR00T-WholeBodyControl) and [HolosomaLocomotionController](https://github.com/amazon-far/holosoma) via `--robot.controller`.
+
+---
+
+## Part 3: Loco-Manipulation with the Homunculus Exoskeleton
+
+We provide a loco-manipulation solution via the Homunculus Exoskeleton — an open-source 7 DoF exoskeleton for whole-body control. Check it out [here](https://github.com/nepyope/hmc_exo).
+
+### Calibrate
+
+```bash
+lerobot-calibrate \
+  --teleop.type=unitree_g1 \
+  --teleop.left_arm_config.port=/dev/ttyACM1 \
+  --teleop.right_arm_config.port=/dev/ttyACM0 \
+  --teleop.id=exo
+```
+
+During calibration move each joint through its entire range. After fitting, move the joint in a neutral position and press `n` to advance.
+
+### Record a Dataset
+
+```bash
+lerobot-record \
+  --robot.type=unitree_g1 \
+  --robot.is_simulation=true \
+  --robot.cameras='{"global_view": {"type": "zmq", "server_address": "localhost", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30}}' \
+  --teleop.type=unitree_g1 \
+  --teleop.left_arm_config.port=/dev/ttyACM1 \
+  --teleop.right_arm_config.port=/dev/ttyACM0 \
+  --teleop.id=exo \
+  --dataset.repo_id=your-username/dataset-name \
+  --dataset.single_task="Test" \
+  --dataset.num_episodes=2 \
+  --dataset.episode_time_s=5 \
+  --dataset.reset_time_s=5 \
+  --dataset.push_to_hub=true \
+  --dataset.streaming_encoding=true \
+  --dataset.encoder_threads=2
+```
+
+> **Note:** Omit `--teleop.left_arm_config.port` and `--teleop.right_arm_config.port` if you're only using the joystick.
+
+Example dataset: [nepyope/unitree_box_move_blue_full](https://huggingface.co/datasets/nepyope/unitree_box_move_blue_full)
+
+---
+
+## Part 4: Training & Inference
+
+### Train
+
+```bash
+python src/lerobot/scripts/lerobot_train.py \
+  --dataset.repo_id=your-username/dataset-name  \
+  --policy.type=pi05 \
+  --output_dir=./outputs/pi05_training \
+  --job_name=pi05_training \
+  --policy.repo_id=your-username/your-repo-id \
+  --policy.pretrained_path=lerobot/pi05_base \
+  --policy.compile_model=true \
+  --policy.gradient_checkpointing=true \
+  --wandb.enable=true \
+  --policy.dtype=bfloat16 \
+  --policy.freeze_vision_encoder=false \
+  --policy.train_expert_only=false \
+  --steps=3000 \
+  --policy.device=cuda \
+  --batch_size=32
+```
+
+### Inference with RTC
+
+Once trained, we recommend deploying policies using inference-time RTC:
+
+```bash
+python examples/rtc/eval_with_real_robot.py \
+  --policy.path=your-username/your-repo-id \
+  --policy.device=cuda \
+  --robot.type=unitree_g1 \
+  --robot.is_simulation=false \
+  --robot.controller=HolosomaLocomotionController \
+  --robot.cameras='{"global_view": {"type": "zmq", "server_address": "<ROBOT_IP>", "port": 5555, "camera_name": "head_camera", "width": 640, "height": 480, "fps": 30}}' \
+  --task="task_description" \
+  --duration=1000 \
+  --fps=30 \
+  --rtc.enabled=true
+```
+
+---
+
+## Additional Resources
+
+- [Unitree SDK Documentation](https://github.com/unitreerobotics/unitree_sdk2_python)
+- [GR00T-WholeBodyControl](https://github.com/NVlabs/GR00T-WholeBodyControl)
+- [Holosoma](https://github.com/amazon-far/holosoma)
+- [LeRobot Documentation](https://github.com/huggingface/lerobot)
+- [Unitree IL LeRobot](https://github.com/unitreerobotics/unitree_IL_lerobot)
+
+---
+
+_Last updated: March 2026_
diff --git a/lerobot/docs/source/using_dataset_tools.mdx b/lerobot/docs/source/using_dataset_tools.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..f7fc9be2094503f06518ab5d323f994d737a2e11
--- /dev/null
+++ b/lerobot/docs/source/using_dataset_tools.mdx
@@ -0,0 +1,235 @@
+# Using Dataset Tools
+
+This guide covers the dataset tools utilities available in LeRobot for modifying and editing existing datasets.
+
+## Overview
+
+LeRobot provides several utilities for manipulating datasets:
+
+1. **Delete Episodes** - Remove specific episodes from a dataset
+2. **Split Dataset** - Divide a dataset into multiple smaller datasets
+3. **Merge Datasets** - Combine multiple datasets into one. The datasets must have identical features, and episodes are concatenated in the order specified in `repo_ids`
+4. **Add Features** - Add new features to a dataset
+5. **Remove Features** - Remove features from a dataset
+6. **Convert to Video** - Convert image-based datasets to video format for efficient storage
+7. **Show the Info of Datasets** - Show the summary of datasets information such as number of episode etc.
+
+The core implementation is in `lerobot.datasets.dataset_tools`.
+An example script detailing how to use the tools API is available in `examples/dataset/use_dataset_tools.py`.
+
+## Command-Line Tool: lerobot-edit-dataset
+
+`lerobot-edit-dataset` is a command-line script for editing datasets. It can be used to delete episodes, split datasets, merge datasets, add features, remove features, and convert image datasets to video format.
+
+Run `lerobot-edit-dataset --help` for more information on the configuration of each operation.
+
+### Usage Examples
+
+#### Delete Episodes
+
+Remove specific episodes from a dataset. This is useful for filtering out undesired data.
+
+```bash
+# Delete episodes 0, 2, and 5 (modifies original dataset)
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht \
+    --operation.type delete_episodes \
+    --operation.episode_indices "[0, 2, 5]"
+
+# Delete episodes and save to a new dataset (preserves original dataset)
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht \
+    --new_repo_id lerobot/pusht_after_deletion \
+    --operation.type delete_episodes \
+    --operation.episode_indices "[0, 2, 5]"
+```
+
+#### Split Dataset
+
+Divide a dataset into multiple subsets.
+
+```bash
+# Split by fractions (e.g. 80% train, 20% test, 20% val)
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht \
+    --operation.type split \
+    --operation.splits '{"train": 0.8, "test": 0.2, "val": 0.2}'
+
+# Split by specific episode indices
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht \
+    --operation.type split \
+    --operation.splits '{"task1": [0, 1, 2, 3], "task2": [4, 5]}'
+```
+
+There are no constraints on the split names, they can be determined by the user. Resulting datasets are saved under the repo id with the split name appended, e.g. `lerobot/pusht_train`, `lerobot/pusht_task1`, `lerobot/pusht_task2`.
+
+#### Merge Datasets
+
+Combine multiple datasets into a single dataset.
+
+```bash
+# Merge train and validation splits back into one dataset
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_merged \
+    --operation.type merge \
+    --operation.repo_ids "['lerobot/pusht_train', 'lerobot/pusht_val']"
+```
+
+#### Remove Features
+
+Remove features from a dataset.
+
+```bash
+# Remove a camera feature
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht \
+    --operation.type remove_feature \
+    --operation.feature_names "['observation.images.top']"
+```
+
+#### Convert to Video
+
+Convert an image-based dataset to video format, creating a new LeRobotDataset where images are stored as videos. This is useful for reducing storage requirements and improving data loading performance. The new dataset will have the exact same structure as the original, but with images encoded as MP4 videos in the proper LeRobot format.
+
+```bash
+# Local-only: Save to a custom output directory (no hub push)
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_image \
+    --operation.type convert_image_to_video \
+    --operation.output_dir /path/to/output/pusht_video
+
+# Save with new repo_id (local storage)
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_image \
+    --new_repo_id lerobot/pusht_video \
+    --operation.type convert_image_to_video
+
+# Convert and push to Hugging Face Hub
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_image \
+    --new_repo_id lerobot/pusht_video \
+    --operation.type convert_image_to_video \
+    --push_to_hub true
+
+# Convert with custom video codec and quality settings
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_image \
+    --operation.type convert_image_to_video \
+    --operation.output_dir outputs/pusht_video \
+    --operation.vcodec libsvtav1 \
+    --operation.pix_fmt yuv420p \
+    --operation.g 2 \
+    --operation.crf 30
+
+# Convert only specific episodes
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_image \
+    --operation.type convert_image_to_video \
+    --operation.output_dir outputs/pusht_video \
+    --operation.episode_indices "[0, 1, 2, 5, 10]"
+
+# Convert with multiple workers for parallel processing
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_image \
+    --operation.type convert_image_to_video \
+    --operation.output_dir outputs/pusht_video \
+    --operation.num_workers 8
+
+# For memory-constrained systems, users can now specify limits:
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_image \
+    --operation.type convert_to_video \
+    --operation.max_episodes_per_batch 50 \
+    --operation.max_frames_per_batch 10000
+```
+
+**Parameters:**
+
+- `output_dir`: Custom output directory (optional - by default uses `new_repo_id` or `{repo_id}_video`)
+- `vcodec`: Video codec to use - options: `h264`, `hevc`, `libsvtav1` (default: `libsvtav1`)
+- `pix_fmt`: Pixel format - options: `yuv420p`, `yuv444p` (default: `yuv420p`)
+- `g`: Group of pictures (GOP) size - lower values give better quality but larger files (default: 2)
+- `crf`: Constant rate factor - lower values give better quality but larger files, 0 is lossless (default: 30)
+- `fast_decode`: Fast decode tuning option (default: 0)
+- `episode_indices`: List of specific episodes to convert (default: all episodes)
+- `num_workers`: Number of parallel workers for processing (default: 4)
+
+**Note:** The resulting dataset will be a proper LeRobotDataset with all cameras encoded as videos in the `videos/` directory, with parquet files containing only metadata (no raw image data). All episodes, stats, and tasks are preserved.
+
+### Show the information of datasets
+
+Show the information of datasets such as number of episode, number of frame, File size and so on.
+No change will be made to the dataset
+
+```bash
+
+# Show dataset information without feature details
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_image \
+    --operation.type info \
+
+# Show dataset information with feature details
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht_image \
+    --operation.type info \
+    --operation.show_features true
+
+```
+
+**Parameters:**
+
+- `parameters`: The flag to control show or no show dataset information with feature details.(default=false)
+
+### Push to Hub
+
+Add the `--push_to_hub true` flag to any command to automatically upload the resulting dataset to the Hugging Face Hub:
+
+```bash
+lerobot-edit-dataset \
+    --repo_id lerobot/pusht \
+    --new_repo_id lerobot/pusht_after_deletion \
+    --operation.type delete_episodes \
+    --operation.episode_indices "[0, 2, 5]" \
+    --push_to_hub true
+```
+
+There is also a tool for adding features to a dataset that is not yet covered in `lerobot-edit-dataset`.
+
+# Dataset Visualization
+
+## Online Visualization
+
+When you record a dataset using `lerobot`, it automatically uploads to the Hugging Face Hub unless you specify otherwise. To view the dataset online, use our **LeRobot Dataset Visualizer**, available at:
+https://huggingface.co/spaces/lerobot/visualize_dataset
+
+## Local Visualization
+
+You can also visualize episodes from a dataset locally using our command-line tool.
+
+**From the Hugging Face Hub:**
+
+```bash
+lerobot-dataset-viz \
+    --repo-id lerobot/pusht \
+    --episode-index 0
+```
+
+**From a local folder:**
+Add the `--root` option and set `--mode local`. For example, to search in `./my_local_data_dir/lerobot/pusht`:
+
+```bash
+lerobot-dataset-viz \
+    --repo-id lerobot/pusht \
+    --root ./my_local_data_dir \
+    --mode local \
+    --episode-index 0
+```
+
+Once executed, the tool opens `rerun.io` and displays the camera streams, robot states, and actions for the selected episode.
+
+For advanced usage—including visualizing datasets stored on a remote server—run:
+
+```bash
+lerobot-dataset-viz --help
+```
diff --git a/lerobot/docs/source/walloss.mdx b/lerobot/docs/source/walloss.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..e9785cc93a48bfb6b86a4d0980cee2de6114961b
--- /dev/null
+++ b/lerobot/docs/source/walloss.mdx
@@ -0,0 +1,80 @@
+# WALL-OSS
+
+WALL-OSS is an open-source foundation model for embodied intelligence, proposed by the [XSquare Robot](https://x2robot.com/en/research/68bc2cde8497d7f238dde690) team in 2025. The LeRobot implementation is adapted from their open-source [WallX](https://github.com/X-Square-Robot/wall-x) repository.
+
+X Square Robot’s WALL-OSS is now integrated into Hugging Face’s LeRobot ecosystem. This is an exciting collaborative project between the LeRobot and X Square Robot teams. You can now post-train, evaluate, and deploy WALL-OSS directly through LeRobot. With this, we’re aiming to make it easier for the open-source robotics community to customize and deploy WALL-OSS foundation models. Read and explore WALL-OSS [paper](https://arxiv.org/pdf/2509.11766) and [code](https://github.com/X-Square-Robot/wall-x).
+
+## Model Overview
+
+The WALL-OSS team is building the embodied foundation model to capture and compress the world's most valuable data: the continuous, high-fidelity stream of physical interaction. By creating a direct feedback loop between the model's decisions and the body's lived experience, the emergence of a truly generalizable intelligence is enabled—one that understands not just how the world works, but how to act effectively within it.
+
+<img
+  src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/walloss-lerobot-paper.png"
+  alt="An overview of WALL-OSS"
+  width="85%"
+/>
+
+Technically, WALL-OSS introduces a tightly coupled multimodal architecture (tightly-coupled MoE structure) that integrates both discrete and continuous action modeling strategies. Through a two-stage training pipeline (Inspiration → Integration), the model gradually unifies semantic reasoning and high-frequency action generation. Its core innovations include:
+
+- **Embodied perception–enhanced multimodal pretraining**: Large-scale training on unified vision–language–action data to strengthen spatial, causal, and manipulation understanding.
+- **Unified Cross-Level Chain-of-Thought (Uni-CoT)**: A single differentiable framework that unifies high-level instruction reasoning, sub-task decomposition, and fine-grained action synthesis, forming a continuous chain from “understanding” to “execution.”
+- **Mixture-of-Experts (MoE) action heads**: Dynamically activating experts depending on the task phase and modeling actions in discrete or continuous space to maintain stable VLM priors.
+- **Two-stage training paradigm**:
+  - **Inspiration stage**: Injecting discrete action priors to strengthen spatial understanding and semantic-action alignment.
+  - **Integration stage**: Using flow matching to achieve high-frequency continuous control.
+
+## Installation Requirements
+
+1. Install LeRobot by following our [Installation Guide](./installation).
+2. Install WallX dependencies by running:
+
+   ```bash
+   pip install -e ".[wallx]"
+   ```
+
+## Usage
+
+To use WallX in LeRobot, specify the policy type as:
+
+```python
+policy.type=wall_x
+```
+
+## Training
+
+For training WallX, you can use the standard LeRobot training script with the appropriate configuration:
+
+```bash
+lerobot-train \
+    --dataset.repo_id=your_dataset \
+    --policy.type=wall_x \
+    --output_dir=./outputs/wallx_training \
+    --job_name=wallx_training \
+    --policy.repo_id=your_repo_id \
+    --policy.pretrained_name_or_path=x-square-robot/wall-oss-flow \
+    --policy.prediction_mode=diffusion \
+    --policy.attn_implementation=eager \
+    --steps=3000 \
+    --policy.device=cuda \
+    --batch_size=32
+```
+
+### Training Arguments
+
+| Argument                       | Description                                                                                                                                                   |
+| ------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `--dataset.repo_id`            | The Hugging Face Hub repository ID for your training dataset (e.g., `lerobot/aloha_sim_insertion_human`)                                                      |
+| `--policy.type`                | Specifies using the WallX policy architecture                                                                                                                 |
+| `--output_dir`                 | Local directory where training checkpoints and logs will be saved                                                                                             |
+| `--job_name`                   | A name identifier for this training run (used in logging/tracking)                                                                                            |
+| `--policy.repo_id`             | Your Hugging Face Hub repo ID where the trained model will be pushed                                                                                          |
+| `--policy.pretrained_path`     | Path to pretrained WallX weights to initialize from (the official WALL-OSS checkpoint)                                                                        |
+| `--policy.prediction_mode`     | The action prediction strategy: `diffusion` or `fast` - `diffusion` uses iterative denoising for action generation, `fast` uses next token prediction instead |
+| `--policy.attn_implementation` | Attention implementation backend - `eager` uses standard PyTorch attention (alternatives include `flash_attention_2` or `sdpa`)                               |
+| `--steps`                      | Total number of training steps to run                                                                                                                         |
+| `--policy.device`              | Device to train on (`cuda` for GPU, `cpu` for CPU)                                                                                                            |
+| `--batch_size`                 | Number of samples per training batch                                                                                                                          |
+
+## License
+
+This model follows the **Apache 2.0 License**, consistent with the original [WallX repository](https://github.com/X-Square-Robot/wall-x).
diff --git a/lerobot/docs/source/xvla.mdx b/lerobot/docs/source/xvla.mdx
new file mode 100644
index 0000000000000000000000000000000000000000..97e04d4ec326f4db526e1509704a5030c35f0aed
--- /dev/null
+++ b/lerobot/docs/source/xvla.mdx
@@ -0,0 +1,528 @@
+# X-VLA: The First Soft-Prompted Robot Foundation Model for Any Robot, Any Task
+
+## Overview
+
+For years, robotics has aspired to build agents that can follow natural human instructions and operate dexterously across many environments and robot bodies. Recent breakthroughs in LLMs and VLMs suggest a path forward: extend these foundation-model architectures to embodied control by grounding them in actions. This has led to the rise of Vision-Language-Action (VLA) models, with the hope that a single generalist model could combine broad semantic understanding with robust manipulation skills.
+
+But training such models is difficult. Robot data is fragmented across platforms, sensors, embodiments, and collection protocols. Heterogeneity appears everywhere: different arm configurations, different action spaces, different camera setups, different visual domains, and different task distributions. These inconsistencies create major distribution shifts that make pretraining unstable and adaptation unreliable.
+
+Inspired by meta-learning and prompt learning, we ask: **"What if a VLA model could learn the structure of each robot and dataset the same way LLMs learn tasks, through prompts?"**
+
+**X-VLA** is a soft-prompted, flow-matching VLA framework that treats each hardware setup as a "task" and encodes it using a small set of learnable embeddings. These **Soft Prompts** capture embodiment and domain-specific variations, guiding the Transformer from the earliest stages of multimodal fusion. With this mechanism, X-VLA can reconcile diverse robot morphologies, data types, and sensor setups within a single unified architecture.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/xvla-architecture.png"
+    alt="XVLA Architecture"
+    style="max-width: 100%; height: auto; width: 800px;"
+  />
+</p>
+
+Built from pure Transformer encoders, X-VLA scales naturally with model size and dataset diversity. Across 6 simulation benchmarks and 3 real robots, Soft Prompts consistently outperform existing methods in handling hardware and domain differences. X-VLA-0.9B, trained on 290K episodes spanning seven robotic platforms, learns an embodiment-agnostic generalist policy in Phase I, and adapts efficiently to new robots in Phase II simply by learning a new set of prompts, while keeping the backbone frozen.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/xvla-architecture2.png"
+    alt="XVLA Architecture 2"
+    style="width: 60%; height: auto;"
+  />
+</p>
+
+With only 1% of parameters tuned (9M), X-VLA-0.9B achieves near-π₀ performance on LIBERO and Simpler-WidowX, despite using **300× fewer trainable parameters**. It also demonstrates strong real-world dexterity with minimal demonstrations, including folding cloths in under two minutes.
+
+<p align="center">
+  <img
+    src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/lerobot/xvla-fold.png"
+    alt="XVLA fold visualization"
+    style="width: 95%; max-width: 1100px; height: auto;"
+  />
+</p>
+
+X-VLA shows that generalist robot intelligence does not require increasingly complex architectures, only the right way to absorb heterogeneity. Soft Prompts offer a simple, scalable mechanism for unifying diverse robotic data, paving the way toward adaptable, cross-embodiment robot foundation models.
+
+## Installation
+
+After installing LeRobot, install the X-VLA dependencies:
+
+```bash
+pip install -e .[xvla]
+```
+
+After the new release, you'll be able to do:
+
+```bash
+pip install lerobot[xvla]
+```
+
+## Quick Start
+
+### Basic Usage
+
+To use X-VLA in your LeRobot configuration, specify the policy type as:
+
+```bash
+policy.type=xvla
+```
+
+### Evaluating Pre-trained Checkpoints
+
+Example evaluation with LIBERO:
+
+```bash
+lerobot-eval \
+  --policy.path="lerobot/xvla-libero" \
+  --env.type=libero \
+  --env.task=libero_spatial,libero_goal,libero_10 \
+  --env.control_mode=absolute \
+  --eval.batch_size=1 \
+  --eval.n_episodes=1 \
+  --env.episode_length=800 \
+  --seed=142
+```
+
+## Available Checkpoints
+
+### 🎯 Base Model
+
+**[lerobot/xvla-base](https://huggingface.co/lerobot/xvla-base)**
+
+A 0.9B parameter instantiation of X-VLA, trained with a carefully designed data processing and learning recipe. The training pipeline consists of two phases:
+
+- **Phase I: Pretraining** - Pretrained on 290K episodes from Droid, Robomind, and Agibot, spanning seven platforms across five types of robotic arms (single-arm to bi-manual setups). By leveraging soft prompts to absorb embodiment-specific variations, the model learns an embodiment-agnostic generalist policy.
+
+- **Phase II: Domain Adaptation** - Adapted to deployable policies for target domains. A new set of soft prompts is introduced and optimized to encode the hardware configuration of the novel domain, while the pretrained backbone remains frozen.
+
+### Simulation Checkpoints
+
+**[lerobot/xvla-libero](https://huggingface.co/lerobot/xvla-libero)**
+
+Achieves 93% success rate on LIBERO benchmarks. Fine-tuned from the base model for simulation tasks.
+
+**[lerobot/xvla-widowx](https://huggingface.co/lerobot/xvla-widowx)**
+
+Fine-tuned on BridgeData for pick-and-place experiments on compact WidowX platforms. Demonstrates robust manipulation capabilities.
+
+### 🤖 Real-World Checkpoints
+
+**[lerobot/xvla-folding](https://huggingface.co/lerobot/xvla-folding)**
+
+A fine-tuned dexterous manipulation model trained on the high-quality Soft-FOLD cloth folding dataset. Achieves 100% success rate over 2 hours of continuous cloth folding.
+
+**[lerobot/xvla-agibot-world](https://huggingface.co/lerobot/xvla-agibot-world)**
+
+Optimized for AgileX robot dexterous manipulation tasks.
+
+**[lerobot/xvla-google-robot](https://huggingface.co/lerobot/xvla-google-robot)**
+
+Adapted for Google Robot platforms.
+
+## Training X-VLA
+
+### Recommended Training Configuration
+
+When fine-tuning X-VLA for a new embodiment or task, we recommend not freezing the VLM, and also setting the `policy.dtype=bfloat16` to not hit OOM errors.
+
+```bash
+lerobot-train \
+  --dataset.repo_id=YOUR_DATASET \
+  --output_dir=./outputs/xvla_training \
+  --job_name=xvla_training \
+  --policy.path="lerobot/xvla-base" \
+  --policy.repo_id="HF_USER/xvla-your-robot" \
+  --policy.dtype=bfloat16 \
+  --policy.action_mode=auto \
+  --steps=20000 \
+  --policy.device=cuda \
+  --policy.freeze_vision_encoder=false \
+  --policy.freeze_language_encoder=false \
+  --policy.train_policy_transformer=true \
+  --policy.train_soft_prompts=true \
+```
+
+### Training Parameters Explained
+
+| Parameter                  | Default | Description                                    |
+| -------------------------- | ------- | ---------------------------------------------- |
+| `freeze_vision_encoder`    | `false` | Do not freeze the VLM vision encoder weights   |
+| `freeze_language_encoder`  | `false` | Do not freeze the VLM language encoder weights |
+| `train_policy_transformer` | `true`  | Allow policy transformer layers to train       |
+| `train_soft_prompts`       | `true`  | Allow soft prompts to train                    |
+
+**💡 Best Practice**: For Phase II adaptation to new embodiments, do not freeze the VLM encoders and also train the policy transformer and soft prompts.
+
+### Example: Training on Bimanual Robot
+
+```bash
+lerobot-train \
+  --dataset.repo_id=<USER>/bimanual-so100-handover-cube \
+  --output_dir=./outputs/xvla_bimanual \
+  --job_name=xvla_so101_training \
+  --policy.path="lerobot/xvla-base" \
+  --policy.dtype=bfloat16 \
+  --policy.repo_id="YOUR_USERNAME/xvla-biso101" \
+  --steps=3000 \
+  --policy.device=cuda \
+  --policy.action_mode=so101_bimanual \
+  --policy.freeze_vision_encoder=false \
+  --policy.freeze_language_encoder=false \
+  --policy.train_policy_transformer=true \
+  --policy.train_soft_prompts=true
+```
+
+💡 **Best Performance:** If you have sufficient computational resources and want to achieve best X-VLA finetuning performance, you should follow the official finetuning strategy:
+
+**🔥 Full-finetune all components with a custom learning-rate scheme**
+
+To ensure stable optimization, the Vision-Language Model (VLM) must be trained with only 1/10 of the base learning rate, while all other components use the full LR.
+This LR ratio is crucial for achieving strong and stable finetuning performance. This is already done for you by default.
+❕Note
+
+Completely matching the official reported performance may require an additional warm-up LR schedule for soft-prompts, which can bring minor improvements.
+We encourage implementing this in your customized training pipeline for optimal results.
+
+## Core Concepts
+
+### 1. Action Modes
+
+X-VLA uses an **Action Registry** system to handle different action spaces and embodiments. The `action_mode` parameter defines how actions are processed, what loss functions are used, and how predictions are post-processed.
+
+#### Available Action Modes
+
+| Action Mode      | Action Dim              | Description                                 | Use Case                             |
+| ---------------- | ----------------------- | ------------------------------------------- | ------------------------------------ |
+| `ee6d`           | 20                      | End-effector with xyz, 6D rotation, gripper | Dual-arm setups with spatial control |
+| `joint`          | 14                      | Joint-space with gripper                    | Direct joint control robots          |
+| `agibot_ee6d`    | 20                      | AGI-bot variant with MSE loss               | AGI-bot platforms                    |
+| `so101_bimanual` | 20 (model), 12 (real)   | SO101 bimanual robot                        | Bimanual manipulation tasks          |
+| `auto`           | 20 (model), auto (real) | Auto-detects action dim from dataset        | **Recommended** for new robots       |
+
+#### Why Action Modes Matter
+
+When you have a pretrained checkpoint like `lerobot/xvla-base` trained with `action_dim=20`, and you want to train on a dataset with a different action dimension (e.g., 14 for bimanual arms), you can't simply trim the action dimension. The action mode orchestrates:
+
+1. **Loss Computation**: Different loss functions for different action components (MSE for joints, BCE for grippers, etc.)
+2. **Preprocessing**: Zeroing out gripper channels, padding dimensions
+3. **Postprocessing**: Applying sigmoid to gripper logits, trimming padding
+
+#### Example: BimanualSO101 Action Space
+
+The `so101_bimanual` action mode handles the mismatch between model output (20D) and real robot control (12D):
+
+```python
+# Model outputs 20 dimensions for compatibility
+dim_action = 20
+
+# Real robot only needs 12 dimensions
+# [left_arm (6), right_arm (6)] = [joints (5) + gripper (1)] × 2
+REAL_DIM = 12
+
+# Preprocessing: Pad 12D actions to 20D for training
+# Postprocessing: Trim 20D predictions to 12D for deployment
+```
+
+See the [action_hub.py](/home/jade_choghari/robot/lerobot/src/lerobot/policies/xvla/action_hub.py) implementation for details.
+
+#### Auto Action Mode (Recommended)
+
+The `auto` action mode is the easiest way to use X-VLA with any robot. It automatically detects your dataset's action dimension and handles padding/trimming:
+
+```bash
+lerobot-train \
+  --policy.path="lerobot/xvla-base" \
+  --policy.action_mode=auto \
+  --policy.max_action_dim=20 \
+  ...
+```
+
+**How it works:**
+
+- Reads `action_feature.shape[-1]` from your dataset (e.g., 7 for Franka)
+- Model outputs `max_action_dim` (default 20) for pretrained compatibility
+- Loss is computed **only on the real dimensions**: `MSE(pred[:,:,:real_dim], target[:,:,:real_dim])`
+- Postprocess trims output back to `real_dim` for robot control
+
+This eliminates the need to create custom action modes for most robots.
+
+### 2. Domain IDs
+
+Domain IDs are learnable identifiers for different robot configurations and camera setups. They allow X-VLA to distinguish between:
+
+- Different robots (Robot 1 vs Robot 2)
+- Different camera configurations (cam1 vs cam2)
+- Different combinations (Robot1-cam1-cam2 vs Robot1-cam1 vs Robot2-cam1)
+
+#### Setting Domain IDs
+
+**During Training**: By default, domain_id is set to 0 for general training.
+
+**During Evaluation**: Specify the domain_id that matches your checkpoint's training configuration.
+
+```python
+# Example: LIBERO checkpoint uses domain_id=3
+domain_id = 3
+```
+
+The domain_id is automatically added to observations by the `XVLAAddDomainIdProcessorStep` in the preprocessing pipeline.
+
+The `lerobot/xvla-base` model has been trained on the following domain IDs. It is recommended to choose one that most resembles your robot/configuration:
+
+#### Fine-tuning Datasets
+
+| Dataset Name     | Domain ID |
+| ---------------- | --------- |
+| Bridge           | 0         |
+| RT1              | 1         |
+| Calvin           | 2         |
+| libero           | 3         |
+| widowx-air       | 4         |
+| AIR-AGILEX-HQ    | 5         |
+| robotwin2_abs_ee | 6         |
+| robotwin2_clean  | 6         |
+| robocasa-human   | 7         |
+| VLABench         | 8         |
+| AGIBOT-challenge | 9         |
+| AIR-AGILEX       | 10        |
+| AIRBOT           | 18        |
+
+### 3. Processor Steps
+
+X-VLA requires specific preprocessing and postprocessing steps for proper operation.
+
+#### Required Preprocessing Steps
+
+1. **XVLAImageToFloatProcessorStep**: Converts images from [0, 255] to [0, 1] range
+2. **XVLAImageNetNormalizeProcessorStep**: Applies ImageNet normalization (required for VLM backbone)
+3. **XVLAAddDomainIdProcessorStep**: Adds domain_id to observations
+
+#### Example Custom Processor
+
+For LIBERO environments, a custom processor handles the specific observation format:
+
+```python
+from lerobot.policies.xvla.processor_xvla import LiberoProcessorStep
+
+processor = LiberoProcessorStep()
+# Handles robot_state dictionary, converts rotation matrices to 6D representation
+# Applies 180° image rotation for camera convention
+```
+
+### 4. Configuration Parameters
+
+Key configuration parameters for X-VLA:
+
+```python
+# Observation and action
+n_obs_steps: int = 1          # Number of observation timesteps
+chunk_size: int = 32           # Action sequence length
+n_action_steps: int = 32       # Number of action steps to execute
+
+# Model architecture
+hidden_size: int = 1024        # Transformer hidden dimension
+depth: int = 24                # Number of transformer layers
+num_heads: int = 16            # Number of attention heads
+num_domains: int = 30          # Maximum number of domain IDs
+len_soft_prompts: int = 32     # Length of soft prompt embeddings
+
+# Action space
+action_mode: str = "ee6d"      # Action space type (use "auto" for auto-detection)
+use_proprio: bool = True       # Use proprioceptive state
+max_state_dim: int = 32        # Maximum state dimension
+max_action_dim: int = 20       # Max action dim for padding (used by "auto" mode)
+
+# Vision
+num_image_views: int | None    # Number of camera views
+resize_imgs_with_padding: tuple[int, int] | None  # Target image size with padding
+
+# Training
+num_denoising_steps: int = 10  # Flow matching denoising steps
+```
+
+## Creating Custom Action Modes
+
+If your robot has a unique action space, you can create a custom action mode:
+
+### Step 1: Define Your Action Space
+
+```python
+from lerobot.policies.xvla.action_hub import BaseActionSpace, register_action
+import torch.nn as nn
+
+@register_action("my_custom_robot")
+class MyCustomActionSpace(BaseActionSpace):
+    """Custom action space for my robot."""
+
+    dim_action = 15  # Your robot's action dimension
+    gripper_idx = (7, 14)  # Gripper channel indices
+
+    def __init__(self):
+        super().__init__()
+        self.mse = nn.MSELoss()
+        self.bce = nn.BCEWithLogitsLoss()
+
+    def compute_loss(self, pred, target):
+        """Define your loss computation."""
+        # Example: MSE for joints, BCE for grippers
+        joints_loss = self.mse(pred[:, :, :7], target[:, :, :7])
+        gripper_loss = self.bce(pred[:, :, self.gripper_idx],
+                                target[:, :, self.gripper_idx])
+
+        return {
+            "joints_loss": joints_loss,
+            "gripper_loss": gripper_loss,
+        }
+
+    def preprocess(self, proprio, action, mode="train"):
+        """Preprocess actions before training."""
+        # Example: Zero out grippers in proprioception
+        proprio_m = proprio.clone()
+        action_m = action.clone() if action is not None else None
+        proprio_m[..., self.gripper_idx] = 0.0
+        if action_m is not None:
+            action_m[..., self.gripper_idx] = 0.0
+        return proprio_m, action_m
+
+    def postprocess(self, action):
+        """Post-process predictions for deployment."""
+        # Example: Apply sigmoid to gripper logits
+        action[..., self.gripper_idx] = torch.sigmoid(action[..., self.gripper_idx])
+        return action
+```
+
+### Step 2: Use Your Custom Action Mode
+
+```bash
+lerobot-train \
+  --policy.action_mode=my_custom_robot \
+  --dataset.repo_id=YOUR_DATASET \
+  --policy.path="lerobot/xvla-base" \
+  ...
+```
+
+## Advanced Topics
+
+### Multi-Camera Support
+
+X-VLA supports multiple camera views through the `num_image_views` parameter:
+
+```python
+# Configure for 3 camera views
+policy.num_image_views=3
+
+# Add empty cameras if you have fewer physical cameras
+policy.empty_cameras=1  # Adds 1 zero-padded camera view
+```
+
+### Custom Preprocessing Pipeline
+
+Create a custom preprocessing pipeline for your environment:
+
+```python
+from lerobot.processor import PolicyProcessorPipeline
+from lerobot.policies.xvla.processor_xvla import (
+    XVLAImageToFloatProcessorStep,
+    XVLAImageNetNormalizeProcessorStep,
+    XVLAAddDomainIdProcessorStep,
+)
+
+# Build custom pipeline
+preprocessor = PolicyProcessorPipeline(
+    steps=[
+        YourCustomProcessorStep(),  # Your custom processing
+        XVLAImageToFloatProcessorStep(),  # Required: convert to float
+        XVLAImageNetNormalizeProcessorStep(),  # Required: ImageNet norm
+        XVLAAddDomainIdProcessorStep(domain_id=5),  # Your domain ID
+    ]
+)
+```
+
+### Handling Different Action Dimensions
+
+When your dataset has fewer action dimensions than the pretrained model:
+
+**Option 1 (Recommended)**: Use `auto` action mode
+
+```bash
+# Automatically detects your dataset's action dimension
+# Works with any robot without custom code
+policy.action_mode=auto
+policy.max_action_dim=20  # Match pretrained model
+```
+
+**Option 2**: Use a predefined action mode with built-in padding
+
+```python
+# Model expects 20D, dataset has 12D
+# Action mode handles padding internally
+action_mode = "so101_bimanual"  # Pads 12 → 20
+```
+
+**Option 2**: Create a custom action mode that maps dimensions explicitly
+
+```python
+@register_action("my_mapped_action")
+class MappedActionSpace(BaseActionSpace):
+    dim_action = 20
+    REAL_DIM = 12
+
+    def _pad_to_model_dim(self, x):
+        # Custom padding logic
+        ...
+```
+
+## Troubleshooting
+
+### Common Issues
+
+**Issue**: "Action dimension mismatch"
+
+- **Solution**: Check that your `action_mode` matches your robot's action space. Create a custom action mode if needed.
+
+**Issue**: "Image values outside [0, 1] range"
+
+- **Solution**: Ensure images are preprocessed with `XVLAImageToFloatProcessorStep` before normalization.
+
+**Issue**: "Domain ID not found"
+
+- **Solution**: Make sure `XVLAAddDomainIdProcessorStep` is in your preprocessing pipeline with the correct domain_id.
+
+**Issue**: "Low success rate on new embodiment"
+
+- **Solution**:
+  1. Verify your action_mode is correct
+  2. Check that soft prompts are being trained (`train_soft_prompts=True`)
+  3. Ensure proper preprocessing (ImageNet normalization, domain_id)
+  4. Consider increasing training steps
+
+**Issue**: "Out of memory during training"
+
+- **Solution**:
+  1. Reduce `chunk_size` (e.g., from 32 to 16)
+  2. Enable gradient checkpointing
+  3. Reduce batch size
+  4. Freeze more components
+
+## Citation
+
+If you use X-VLA in your research, please cite:
+
+```bibtex
+@article{zheng2025x,
+  title   = {X-VLA: Soft-Prompted Transformer as Scalable Cross-Embodiment Vision-Language-Action Model},
+  author  = {Zheng, Jinliang and Li, Jianxiong and Wang, Zhihao and Liu, Dongxiu and Kang, Xirui
+             and Feng, Yuchun and Zheng, Yinan and Zou, Jiayin and Chen, Yilun and Zeng, Jia and others},
+  journal = {arXiv preprint arXiv:2510.10274},
+  year    = {2025}
+}
+```
+
+## Additional Resources
+
+- [X-VLA Paper](https://arxiv.org/pdf/2510.10274)
+- [LeRobot Documentation](https://github.com/huggingface/lerobot)
+- [Action Registry Implementation](https://github.com/huggingface/lerobot/src/lerobot/policies/xvla/action_hub.py)
+- [Processor Implementation](https://github.com/huggingface/lerobot/src/lerobot/policies/xvla/processor_xvla.py)
+- [Model Configuration](https://github.com/huggingface/lerobot/src/lerobot/policies/xvla/configuration_xvla.py)
+
+## Contributing
+
+We welcome contributions! If you've implemented a new action mode or processor for your robot, please consider submitting a PR to help the community.
diff --git a/lerobot/examples/backward_compatibility/replay.py b/lerobot/examples/backward_compatibility/replay.py
new file mode 100644
index 0000000000000000000000000000000000000000..13fdfd5f514668597f98a2a5200a0b7a7097418e
--- /dev/null
+++ b/lerobot/examples/backward_compatibility/replay.py
@@ -0,0 +1,106 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Replays the actions of an episode from a dataset on a robot.
+
+Example:
+
+```shell
+lerobot-replay \
+    --robot.type=so100_follower \
+    --robot.port=/dev/tty.usbmodem58760431541 \
+    --robot.id=black \
+    --dataset.repo_id=<USER>/record-test \
+    --dataset.episode=2
+```
+"""
+
+import logging
+import time
+from dataclasses import asdict, dataclass
+from pathlib import Path
+from pprint import pformat
+
+import draccus
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.robots import (  # noqa: F401
+    Robot,
+    RobotConfig,
+    koch_follower,
+    make_robot_from_config,
+    so_follower,
+)
+from lerobot.utils.constants import ACTION
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import (
+    init_logging,
+    log_say,
+)
+
+
+@dataclass
+class DatasetReplayConfig:
+    # Dataset identifier. By convention it should match '{hf_username}/{dataset_name}' (e.g. `lerobot/test`).
+    repo_id: str
+    # Episode to replay.
+    episode: int
+    # Root directory where the dataset will be stored (e.g. 'dataset/path'). If None, defaults to $HF_LEROBOT_HOME/repo_id.
+    root: str | Path | None = None
+    # Limit the frames per second. By default, uses the policy fps.
+    fps: int = 30
+
+
+@dataclass
+class ReplayConfig:
+    robot: RobotConfig
+    dataset: DatasetReplayConfig
+    # Use vocal synthesis to read events.
+    play_sounds: bool = True
+
+
+@draccus.wrap()
+def replay(cfg: ReplayConfig):
+    init_logging()
+    logging.info(pformat(asdict(cfg)))
+
+    robot = make_robot_from_config(cfg.robot)
+    dataset = LeRobotDataset(cfg.dataset.repo_id, root=cfg.dataset.root, episodes=[cfg.dataset.episode])
+    actions = dataset.hf_dataset.select_columns(ACTION)
+    robot.connect()
+
+    try:
+        log_say("Replaying episode", cfg.play_sounds, blocking=True)
+        for idx in range(dataset.num_frames):
+            start_episode_t = time.perf_counter()
+
+            action_array = actions[idx][ACTION]
+            action = {}
+            for i, name in enumerate(dataset.features[ACTION]["names"]):
+                key = f"{name.removeprefix('main_')}.pos"
+                action[key] = action_array[i].item()
+
+            action["shoulder_lift.pos"] = -(action["shoulder_lift.pos"] - 90)
+            action["elbow_flex.pos"] -= 90
+            robot.send_action(action)
+
+            dt_s = time.perf_counter() - start_episode_t
+            precise_sleep(max(1 / dataset.fps - dt_s, 0.0))
+    finally:
+        robot.disconnect()
+
+
+if __name__ == "__main__":
+    replay()
diff --git a/lerobot/examples/dataset/load_lerobot_dataset.py b/lerobot/examples/dataset/load_lerobot_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..ea3516710190a2159dfffc28c51b0a3be08a6228
--- /dev/null
+++ b/lerobot/examples/dataset/load_lerobot_dataset.py
@@ -0,0 +1,152 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+This script demonstrates the use of `LeRobotDataset` class for handling and processing robotic datasets from Hugging Face.
+It illustrates how to load datasets, manipulate them, and apply transformations suitable for machine learning tasks in PyTorch.
+
+Features included in this script:
+- Viewing a dataset's metadata and exploring its properties.
+- Loading an existing dataset from the hub or a subset of it.
+- Accessing frames by episode number.
+- Using advanced dataset features like timestamp-based frame selection.
+- Demonstrating compatibility with PyTorch DataLoader for batch processing.
+
+The script ends with examples of how to batch process data using PyTorch's DataLoader.
+"""
+
+from pprint import pprint
+
+import torch
+from huggingface_hub import HfApi
+
+import lerobot
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+
+def main():
+    # We ported a number of existing datasets ourselves, use this to see the list:
+    print("List of available datasets:")
+    pprint(lerobot.available_datasets)
+
+    # You can also browse through the datasets created/ported by the community on the hub using the hub api:
+    hub_api = HfApi()
+    repo_ids = [info.id for info in hub_api.list_datasets(task_categories="robotics", tags=["LeRobot"])]
+    pprint(repo_ids)
+
+    # Or simply explore them in your web browser directly at:
+    # https://huggingface.co/datasets?other=LeRobot
+
+    # Let's take this one for this example
+    repo_id = "lerobot/aloha_mobile_cabinet"
+    # We can have a look and fetch its metadata to know more about it:
+    ds_meta = LeRobotDatasetMetadata(repo_id)
+
+    # By instantiating just this class, you can quickly access useful information about the content and the
+    # structure of the dataset without downloading the actual data yet (only metadata files — which are
+    # lightweight).
+    print(f"Total number of episodes: {ds_meta.total_episodes}")
+    print(f"Average number of frames per episode: {ds_meta.total_frames / ds_meta.total_episodes:.3f}")
+    print(f"Frames per second used during data collection: {ds_meta.fps}")
+    print(f"Robot type: {ds_meta.robot_type}")
+    print(f"keys to access images from cameras: {ds_meta.camera_keys=}\n")
+
+    print("Tasks:")
+    print(ds_meta.tasks)
+    print("Features:")
+    pprint(ds_meta.features)
+
+    # You can also get a short summary by simply printing the object:
+    print(ds_meta)
+
+    # You can then load the actual dataset from the hub.
+    # Either load any subset of episodes:
+    dataset = LeRobotDataset(repo_id, episodes=[0, 10, 11, 23])
+
+    # And see how many frames you have:
+    print(f"Selected episodes: {dataset.episodes}")
+    print(f"Number of episodes selected: {dataset.num_episodes}")
+    print(f"Number of frames selected: {dataset.num_frames}")
+
+    # Or simply load the entire dataset:
+    dataset = LeRobotDataset(repo_id)
+    print(f"Number of episodes selected: {dataset.num_episodes}")
+    print(f"Number of frames selected: {dataset.num_frames}")
+
+    # The previous metadata class is contained in the 'meta' attribute of the dataset:
+    print(dataset.meta)
+
+    # LeRobotDataset actually wraps an underlying Hugging Face dataset
+    # (see https://huggingface.co/docs/datasets for more information).
+    print(dataset.hf_dataset)
+
+    # LeRobot datasets also subclasses PyTorch datasets so you can do everything you know and love from working
+    # with the latter, like iterating through the dataset.
+    # The __getitem__ iterates over the frames of the dataset. Since our datasets are also structured by
+    # episodes, you can access the frame indices of any episode using dataset.meta.episodes. Here, we access
+    # frame indices associated to the first episode:
+    episode_index = 0
+    from_idx = dataset.meta.episodes["dataset_from_index"][episode_index]
+    to_idx = dataset.meta.episodes["dataset_to_index"][episode_index]
+
+    # Then we grab all the image frames from the first camera:
+    camera_key = dataset.meta.camera_keys[0]
+    frames = [dataset[idx][camera_key] for idx in range(from_idx, to_idx)]
+
+    # The objects returned by the dataset are all torch.Tensors
+    print(type(frames[0]))
+    print(frames[0].shape)
+
+    # Since we're using pytorch, the shape is in pytorch, channel-first convention (c, h, w).
+    # We can compare this shape with the information available for that feature
+    pprint(dataset.features[camera_key])
+    # In particular:
+    print(dataset.features[camera_key]["shape"])
+    # The shape is in (h, w, c) which is a more universal format.
+
+    # For many machine learning applications we need to load the history of past observations or trajectories of
+    # future actions. Our datasets can load previous and future frames for each key/modality, using timestamps
+    # differences with the current loaded frame. For instance:
+    delta_timestamps = {
+        # loads 4 images: 1 second before current frame, 500 ms before, 200 ms before, and current frame
+        camera_key: [-1, -0.5, -0.20, 0],
+        # loads 6 state vectors: 1.5 seconds before, 1 second before, ... 200 ms, 100 ms, and current frame
+        "observation.state": [-1.5, -1, -0.5, -0.20, -0.10, 0],
+        # loads 64 action vectors: current frame, 1 frame in the future, 2 frames, ... 63 frames in the future
+        "action": [t / dataset.fps for t in range(64)],
+    }
+    # Note that in any case, these delta_timestamps values need to be multiples of (1/fps) so that added to any
+    # timestamp, you still get a valid timestamp.
+
+    dataset = LeRobotDataset(repo_id, delta_timestamps=delta_timestamps)
+    print(f"\n{dataset[0][camera_key].shape=}")  # (4, c, h, w)
+    print(f"{dataset[0]['observation.state'].shape=}")  # (6, c)
+    print(f"{dataset[0]['action'].shape=}\n")  # (64, c)
+
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        num_workers=4,
+        batch_size=32,
+        shuffle=True,
+    )
+    for batch in dataloader:
+        print(f"{batch[camera_key].shape=}")  # (32, 4, c, h, w)
+        print(f"{batch['observation.state'].shape=}")  # (32, 6, c)
+        print(f"{batch['action'].shape=}")  # (32, 64, c)
+        break
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/dataset/slurm_compute_rabc.py b/lerobot/examples/dataset/slurm_compute_rabc.py
new file mode 100644
index 0000000000000000000000000000000000000000..2ddf84d07480890f8dfde164082233c188e35502
--- /dev/null
+++ b/lerobot/examples/dataset/slurm_compute_rabc.py
@@ -0,0 +1,490 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+SLURM-distributed SARM RA-BC annotation pipeline.
+
+Computes SARM progress values for all frames in a dataset, distributed across
+SLURM workers, then merges the shards into a single sarm_progress.parquet.
+
+Two subcommands, each a separate SLURM submission:
+
+  compute    – N workers, each computes progress for a subset of episodes
+  aggregate  – 1 worker, merges N shards into sarm_progress.parquet, pushes to hub
+
+Usage:
+    python slurm_compute_rabc.py compute \\
+        --repo-id user/dataset --reward-model-path user/sarm_model \\
+        --stride 10 --device cpu --workers 50 --partition cpu
+
+    python slurm_compute_rabc.py aggregate \\
+        --repo-id user/dataset --reward-model-path user/sarm_model \\
+        --partition cpu --push-to-hub
+"""
+
+import argparse
+from pathlib import Path
+
+from datatrove.executor import LocalPipelineExecutor
+from datatrove.executor.slurm import SlurmPipelineExecutor
+from datatrove.pipeline.base import PipelineStep
+
+
+class ComputeProgressShards(PipelineStep):
+    """Each worker computes SARM progress for its assigned episodes."""
+
+    def __init__(
+        self, repo_id, reward_model_path, stride=1, head_mode="sparse", device="cpu", shard_dir="rabc_shards"
+    ):
+        super().__init__()
+        if stride < 1:
+            raise ValueError(f"stride must be >= 1, got {stride}")
+        self.repo_id = repo_id
+        self.reward_model_path = reward_model_path
+        self.stride = stride
+        self.head_mode = head_mode
+        self.device = device
+        self.shard_dir = shard_dir
+
+    def run(self, data=None, rank: int = 0, world_size: int = 1):
+        import logging
+        from pathlib import Path
+
+        import numpy as np
+        import pyarrow as pa
+        import pyarrow.parquet as pq
+        import torch
+        from tqdm import tqdm
+
+        from lerobot.policies.sarm.compute_rabc_weights import (
+            generate_all_frame_indices,
+            interpolate_progress,
+            load_sarm_resources,
+        )
+        from lerobot.utils.utils import init_logging
+
+        init_logging()
+
+        dataset, reward_model, preprocess = load_sarm_resources(
+            self.repo_id,
+            self.reward_model_path,
+            self.device,
+        )
+
+        if hasattr(preprocess, "eval"):
+            preprocess.eval()
+        for step in preprocess.steps:
+            if hasattr(step, "eval"):
+                step.eval()
+
+        image_key = reward_model.config.image_key
+        state_key = reward_model.config.state_key
+        frame_gap = reward_model.config.frame_gap
+        center_idx = reward_model.config.n_obs_steps // 2
+
+        dual_mode = reward_model.config.uses_dual_heads
+        compute_sparse = self.head_mode in ("sparse", "both") or not dual_mode
+        compute_dense = self.head_mode in ("dense", "both") and dual_mode
+
+        my_episodes = list(range(dataset.num_episodes))[rank::world_size]
+        if not my_episodes:
+            logging.info(f"Rank {rank}: no episodes assigned")
+            return
+        logging.info(f"Rank {rank}: {len(my_episodes)} / {dataset.num_episodes} episodes")
+
+        all_rows = []
+
+        for ep_idx in tqdm(my_episodes, desc=f"Rank {rank}"):
+            ep = dataset.meta.episodes[ep_idx]
+            ep_start, ep_end = ep["dataset_from_index"], ep["dataset_to_index"]
+            task = dataset[ep_start].get("task", "perform the task")
+
+            all_ep_indices = generate_all_frame_indices(ep_start, ep_end, frame_gap)
+            if self.stride > 1:
+                compute_indices = [i for i in all_ep_indices if (i - ep_start) % self.stride == 0]
+                if (ep_end - 1) not in compute_indices:
+                    compute_indices.append(ep_end - 1)
+                compute_indices = sorted(set(compute_indices))
+            else:
+                compute_indices = all_ep_indices
+
+            frame_results = {}
+            for qi in tqdm(compute_indices, desc=f"  Ep {ep_idx}", leave=False):
+                try:
+                    sample = dataset[qi]
+                    batch = {
+                        image_key: sample[image_key],
+                        "task": task,
+                        "index": qi,
+                        "episode_index": ep_idx,
+                    }
+                    if state_key in sample:
+                        batch[state_key] = sample[state_key]
+
+                    with torch.no_grad():
+                        processed = preprocess(batch)
+                        vf = processed["video_features"].to(self.device)
+                        tf = processed["text_features"].to(self.device)
+                        sf = processed.get("state_features")
+                        if sf is not None:
+                            sf = sf.to(self.device)
+                        lengths = processed.get("lengths")
+
+                        sparse_val = dense_val = np.nan
+                        if compute_sparse:
+                            r = reward_model.calculate_rewards(
+                                text_embeddings=tf,
+                                video_embeddings=vf,
+                                state_features=sf,
+                                lengths=lengths,
+                                return_all_frames=True,
+                                head_mode="sparse",
+                            )
+                            sparse_val = float(r[0, center_idx] if r.ndim == 2 else r[center_idx])
+                        if compute_dense:
+                            r = reward_model.calculate_rewards(
+                                text_embeddings=tf,
+                                video_embeddings=vf,
+                                state_features=sf,
+                                lengths=lengths,
+                                return_all_frames=True,
+                                head_mode="dense",
+                            )
+                            dense_val = float(r[0, center_idx] if r.ndim == 2 else r[center_idx])
+
+                        frame_results[qi] = (sparse_val, dense_val)
+                except Exception as e:
+                    logging.warning(f"Failed frame {qi}: {e}")
+
+            if not frame_results:
+                logging.warning(f"Episode {ep_idx}: all frames failed, skipping")
+                continue
+
+            # Interpolate to all frames in this episode
+            computed_idx = np.array(sorted(frame_results.keys()))
+            all_frame_arr = np.arange(ep_start, ep_end)
+
+            sparse_vals = np.array([frame_results[i][0] for i in computed_idx]) if compute_sparse else None
+            dense_vals = np.array([frame_results[i][1] for i in computed_idx]) if compute_dense else None
+
+            if self.stride > 1 and len(computed_idx) > 1:
+                if compute_sparse:
+                    sparse_vals = interpolate_progress(computed_idx, sparse_vals, all_frame_arr)
+                if compute_dense:
+                    dense_vals = interpolate_progress(computed_idx, dense_vals, all_frame_arr)
+                output_frames = all_frame_arr
+            else:
+                # Use only successfully computed frames to avoid indexing mismatch on failures
+                output_frames = computed_idx
+
+            for i, fi in enumerate(output_frames):
+                row = {"index": int(fi), "episode_index": ep_idx, "frame_index": int(fi - ep_start)}
+                if compute_sparse:
+                    row["progress_sparse"] = float(sparse_vals[i])
+                if compute_dense:
+                    row["progress_dense"] = float(dense_vals[i])
+                all_rows.append(row)
+
+        if all_rows:
+            import pandas as pd
+
+            df = pd.DataFrame(all_rows).sort_values("index").reset_index(drop=True)
+            table = pa.Table.from_pandas(df, preserve_index=False)
+            table = table.replace_schema_metadata({b"reward_model_path": self.reward_model_path.encode()})
+            shard_dir = Path(self.shard_dir)
+            shard_dir.mkdir(parents=True, exist_ok=True)
+            out = shard_dir / f"shard_{rank:05d}.parquet"
+            pq.write_table(table, out)
+            logging.info(f"Rank {rank}: saved {len(df)} rows to {out}")
+
+
+class AggregateProgress(PipelineStep):
+    """Merge all shard parquets into final sarm_progress.parquet."""
+
+    def __init__(self, repo_id, reward_model_path, shard_dir="rabc_shards", push_to_hub=False):
+        super().__init__()
+        self.repo_id = repo_id
+        self.reward_model_path = reward_model_path
+        self.shard_dir = shard_dir
+        self.push_to_hub = push_to_hub
+
+    def run(self, data=None, rank: int = 0, world_size: int = 1):
+        import datetime
+        import logging
+        import os
+        from pathlib import Path
+
+        import pandas as pd
+        import pyarrow as pa
+        import pyarrow.parquet as pq
+
+        from lerobot.datasets.lerobot_dataset import LeRobotDataset
+        from lerobot.utils.utils import init_logging
+
+        init_logging()
+        if rank != 0:
+            return
+
+        shard_dir = Path(self.shard_dir)
+        shards = sorted(shard_dir.glob("shard_*.parquet"))
+        if not shards:
+            raise FileNotFoundError(f"No shards found in {shard_dir}")
+
+        # Log shard modification time range to help detect stale files
+        mtimes = [os.path.getmtime(s) for s in shards]
+        oldest = datetime.datetime.fromtimestamp(min(mtimes)).isoformat(timespec="seconds")
+        newest = datetime.datetime.fromtimestamp(max(mtimes)).isoformat(timespec="seconds")
+        logging.info(f"Aggregating {len(shards)} shards (oldest: {oldest}, newest: {newest})")
+
+        df = pd.concat([pd.read_parquet(s) for s in shards], ignore_index=True)
+        df = df.sort_values("index").reset_index(drop=True)
+
+        table = pa.Table.from_pandas(df, preserve_index=False)
+        table = table.replace_schema_metadata({b"reward_model_path": self.reward_model_path.encode()})
+
+        temp_ds = LeRobotDataset(self.repo_id, download_videos=False)
+        out_path = Path(temp_ds.root) / "sarm_progress.parquet"
+        out_path.parent.mkdir(parents=True, exist_ok=True)
+        pq.write_table(table, out_path)
+        logging.info(f"Saved {len(df)} rows to {out_path}")
+
+        for col in ["progress_sparse", "progress_dense"]:
+            if col in df.columns:
+                v = df[col].dropna()
+                logging.info(
+                    f"{col}: mean={v.mean():.4f} std={v.std():.4f} min={v.min():.4f} max={v.max():.4f}"
+                )
+
+        if self.push_to_hub:
+            from huggingface_hub import HfApi
+
+            api = HfApi()
+            hub_path = "sarm_progress.parquet"
+            logging.info(f"Uploading to {self.repo_id}/{hub_path}")
+            api.upload_file(
+                path_or_fileobj=str(out_path),
+                path_in_repo=hub_path,
+                repo_id=self.repo_id,
+                repo_type="dataset",
+            )
+            logging.info(f"Uploaded: https://huggingface.co/datasets/{self.repo_id}/blob/main/{hub_path}")
+
+
+def make_compute_executor(
+    repo_id,
+    reward_model_path,
+    stride,
+    head_mode,
+    device,
+    shard_dir,
+    logs_dir,
+    job_name,
+    slurm,
+    workers,
+    partition,
+    cpus_per_task,
+    mem_per_cpu,
+):
+    kwargs = {
+        "pipeline": [
+            ComputeProgressShards(repo_id, reward_model_path, stride, head_mode, device, str(shard_dir)),
+        ],
+        "logging_dir": str(logs_dir / job_name),
+    }
+
+    if slurm:
+        kwargs.update(
+            {
+                "job_name": job_name,
+                "tasks": workers,
+                "workers": workers,
+                "time": "24:00:00",
+                "partition": partition,
+                "cpus_per_task": cpus_per_task,
+                "sbatch_args": {"mem-per-cpu": mem_per_cpu},
+            }
+        )
+        return SlurmPipelineExecutor(**kwargs)
+
+    kwargs.update({"tasks": workers, "workers": 1})
+    return LocalPipelineExecutor(**kwargs)
+
+
+def make_aggregate_executor(
+    repo_id,
+    reward_model_path,
+    shard_dir,
+    logs_dir,
+    job_name,
+    slurm,
+    partition,
+    cpus_per_task,
+    mem_per_cpu,
+    push_to_hub,
+):
+    kwargs = {
+        "pipeline": [
+            AggregateProgress(repo_id, reward_model_path, str(shard_dir), push_to_hub),
+        ],
+        "logging_dir": str(logs_dir / job_name),
+    }
+
+    if slurm:
+        kwargs.update(
+            {
+                "job_name": job_name,
+                "tasks": 1,
+                "workers": 1,
+                "time": "02:00:00",
+                "partition": partition,
+                "cpus_per_task": cpus_per_task,
+                "sbatch_args": {"mem-per-cpu": mem_per_cpu},
+            }
+        )
+        return SlurmPipelineExecutor(**kwargs)
+
+    kwargs.update({"tasks": 1, "workers": 1})
+    return LocalPipelineExecutor(**kwargs)
+
+
+def _add_shared_args(p):
+    p.add_argument(
+        "--repo-id",
+        type=str,
+        required=True,
+        help="Hugging Face repository identifier, e.g. 'user/dataset'.",
+    )
+    p.add_argument(
+        "--shard-dir",
+        type=Path,
+        default=Path("rabc_shards"),
+        help="Directory to read/write per-rank parquet shards.",
+    )
+    p.add_argument(
+        "--logs-dir",
+        type=Path,
+        default=Path("logs"),
+        help="Directory for datatrove logs.",
+    )
+    p.add_argument(
+        "--job-name",
+        type=str,
+        default=None,
+        help="SLURM job name (defaults to rabc_<subcommand>).",
+    )
+    p.add_argument(
+        "--slurm",
+        type=int,
+        default=1,
+        help="1 = submit via SLURM; 0 = run locally (useful for debugging).",
+    )
+    p.add_argument(
+        "--partition",
+        type=str,
+        default=None,
+        help="SLURM partition to submit to.",
+    )
+    p.add_argument(
+        "--cpus-per-task",
+        type=int,
+        default=4,
+        help="Number of CPUs per SLURM task.",
+    )
+    p.add_argument(
+        "--mem-per-cpu",
+        type=str,
+        default="4G",
+        help="Memory per CPU, e.g. '4G' or '1950M'.",
+    )
+
+
+def main():
+    parser = argparse.ArgumentParser(
+        description="SLURM-distributed SARM RA-BC annotation pipeline",
+        formatter_class=argparse.RawDescriptionHelpFormatter,
+    )
+    sub = parser.add_subparsers(dest="command", required=True)
+
+    # compute subcommand
+    cp = sub.add_parser(
+        "compute",
+        help="Distribute progress computation across SLURM workers.",
+    )
+    _add_shared_args(cp)
+    cp.add_argument(
+        "--reward-model-path",
+        type=str,
+        required=True,
+        help="Path or HF repo id of the SARM reward model.",
+    )
+    cp.add_argument(
+        "--stride",
+        type=int,
+        default=1,
+        help="Compute every Nth frame; intermediate frames are interpolated (must be >= 1).",
+    )
+    cp.add_argument(
+        "--head-mode",
+        type=str,
+        default="sparse",
+        choices=["sparse", "dense", "both"],
+        help="Which reward head(s) to compute.",
+    )
+    cp.add_argument(
+        "--device",
+        type=str,
+        default="cpu",
+        help="Device for reward model inference, e.g. 'cpu' or 'cuda'.",
+    )
+    cp.add_argument(
+        "--workers",
+        type=int,
+        default=50,
+        help="Number of parallel SLURM tasks (one shard per worker).",
+    )
+
+    # aggregate subcommand
+    ap = sub.add_parser(
+        "aggregate",
+        help="Merge per-rank shards into a single sarm_progress.parquet.",
+    )
+    _add_shared_args(ap)
+    ap.add_argument(
+        "--reward-model-path",
+        type=str,
+        required=True,
+        help="Path or HF repo id of the SARM reward model (stored in parquet metadata).",
+    )
+    ap.add_argument(
+        "--push-to-hub",
+        action="store_true",
+        help="Upload sarm_progress.parquet to the Hugging Face Hub after aggregation.",
+    )
+
+    args = parser.parse_args()
+    job_name = args.job_name or f"rabc_{args.command}"
+    kwargs = vars(args)
+    kwargs["slurm"] = kwargs.pop("slurm") == 1
+    kwargs["job_name"] = job_name
+    command = kwargs.pop("command")
+
+    executor = make_compute_executor(**kwargs) if command == "compute" else make_aggregate_executor(**kwargs)
+
+    executor.run()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/dataset/use_dataset_image_transforms.py b/lerobot/examples/dataset/use_dataset_image_transforms.py
new file mode 100644
index 0000000000000000000000000000000000000000..c28f2ef0c6422b0d0449d95b0cdc8ca2ddf87f0e
--- /dev/null
+++ b/lerobot/examples/dataset/use_dataset_image_transforms.py
@@ -0,0 +1,177 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+This example demonstrates how to use image transforms with LeRobot datasets for data augmentation during training.
+
+Image transforms are applied to camera frames to improve model robustness and generalization. They are applied
+at training time only, not during dataset recording, allowing you to experiment with different augmentations
+without re-recording data.
+"""
+
+import torch
+from torchvision.transforms import v2
+from torchvision.transforms.functional import to_pil_image
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.transforms import ImageTransformConfig, ImageTransforms, ImageTransformsConfig
+
+
+def save_image(tensor, filename):
+    """Helper function to save a tensor as an image file."""
+    if tensor.dim() == 3:  # [C, H, W]
+        if tensor.max() > 1.0:
+            tensor = tensor / 255.0
+        tensor = torch.clamp(tensor, 0.0, 1.0)
+        pil_image = to_pil_image(tensor)
+        pil_image.save(filename)
+        print(f"Saved: {filename}")
+    else:
+        print(f"Skipped {filename}: unexpected tensor shape {tensor.shape}")
+
+
+def example_1_default_transforms():
+    """Example 1: Use default transform configuration and save original vs transformed images"""
+    print("\n Example 1: Default Transform Configuration with Image Saving")
+
+    repo_id = "pepijn223/record_main_0"  # Example dataset
+
+    try:
+        # Load dataset without transforms (original)
+        dataset_original = LeRobotDataset(repo_id=repo_id)
+
+        # Load dataset with transforms enabled
+        transforms_config = ImageTransformsConfig(
+            enable=True,  # Enable transforms (disabled by default)
+            max_num_transforms=2,  # Apply up to 2 transforms per frame
+            random_order=False,  # Apply in standard order
+        )
+        dataset_with_transforms = LeRobotDataset(
+            repo_id=repo_id, image_transforms=ImageTransforms(transforms_config)
+        )
+
+        # Save original and transformed images for comparison
+        if len(dataset_original) > 0:
+            frame_idx = 0  # Use first frame
+            original_sample = dataset_original[frame_idx]
+            transformed_sample = dataset_with_transforms[frame_idx]
+
+            print(f"Saving comparison images (frame {frame_idx}):")
+
+            for cam_key in dataset_original.meta.camera_keys:
+                if cam_key in original_sample and cam_key in transformed_sample:
+                    cam_name = cam_key.replace(".", "_").replace("/", "_")
+
+                    # Save original and transformed images
+                    save_image(original_sample[cam_key], f"{cam_name}_original.png")
+                    save_image(transformed_sample[cam_key], f"{cam_name}_transformed.png")
+
+    except Exception as e:
+        print(f"Could not load dataset '{repo_id}': {e}")
+
+
+def example_2_custom_transforms():
+    """Example 2: Create custom transform configuration and save examples"""
+    print("\n Example 2: Custom Transform Configuration")
+
+    repo_id = "pepijn223/record_main_0"  # Example dataset
+
+    try:
+        # Create custom transform configuration with strong effects
+        custom_transforms_config = ImageTransformsConfig(
+            enable=True,
+            max_num_transforms=2,  # Apply up to 2 transforms per frame
+            random_order=True,  # Apply transforms in random order
+            tfs={
+                "brightness": ImageTransformConfig(
+                    weight=1.0,
+                    type="ColorJitter",
+                    kwargs={"brightness": (0.5, 1.5)},  # Strong brightness range
+                ),
+                "contrast": ImageTransformConfig(
+                    weight=1.0,  # Higher weight = more likely to be selected
+                    type="ColorJitter",
+                    kwargs={"contrast": (0.6, 1.4)},  # Strong contrast
+                ),
+                "sharpness": ImageTransformConfig(
+                    weight=0.5,  # Lower weight = less likely to be selected
+                    type="SharpnessJitter",
+                    kwargs={"sharpness": (0.2, 2.0)},  # Strong sharpness variation
+                ),
+            },
+        )
+
+        dataset_with_custom_transforms = LeRobotDataset(
+            repo_id=repo_id, image_transforms=ImageTransforms(custom_transforms_config)
+        )
+
+        # Save examples with strong transforms
+        if len(dataset_with_custom_transforms) > 0:
+            sample = dataset_with_custom_transforms[0]
+            print("Saving custom transform examples:")
+
+            for cam_key in dataset_with_custom_transforms.meta.camera_keys:
+                if cam_key in sample:
+                    cam_name = cam_key.replace(".", "_").replace("/", "_")
+                    save_image(sample[cam_key], f"{cam_name}_custom_transforms.png")
+
+    except Exception as e:
+        print(f"Could not load dataset '{repo_id}': {e}")
+
+
+def example_3_torchvision_transforms():
+    """Example 3: Use pure torchvision transforms and save examples"""
+    print("\n Example 3: Pure Torchvision Transforms")
+
+    repo_id = "pepijn223/record_main_0"  # Example dataset
+
+    try:
+        # Create torchvision transform pipeline
+        torchvision_transforms = v2.Compose(
+            [
+                v2.ColorJitter(brightness=0.3, contrast=0.3, saturation=0.3, hue=0.1),
+                v2.GaussianBlur(kernel_size=3, sigma=(0.1, 2.0)),
+                v2.RandomRotation(degrees=10),  # Small rotation
+            ]
+        )
+
+        dataset_with_torchvision = LeRobotDataset(repo_id=repo_id, image_transforms=torchvision_transforms)
+
+        # Save examples with torchvision transforms
+        if len(dataset_with_torchvision) > 0:
+            sample = dataset_with_torchvision[0]
+            print("Saving torchvision transform examples:")
+
+            for cam_key in dataset_with_torchvision.meta.camera_keys:
+                if cam_key in sample:
+                    cam_name = cam_key.replace(".", "_").replace("/", "_")
+                    save_image(sample[cam_key], f"{cam_name}_torchvision.png")
+
+    except Exception as e:
+        print(f"Could not load dataset '{repo_id}': {e}")
+
+
+def main():
+    """Run all examples"""
+    print("LeRobot Dataset Image Transforms Examples")
+
+    example_1_default_transforms()
+    example_2_custom_transforms()
+    example_3_torchvision_transforms()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/dataset/use_dataset_tools.py b/lerobot/examples/dataset/use_dataset_tools.py
new file mode 100644
index 0000000000000000000000000000000000000000..bd7c389bc42d5217e8c9338e1a74e00e1d15fc0e
--- /dev/null
+++ b/lerobot/examples/dataset/use_dataset_tools.py
@@ -0,0 +1,124 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Example script demonstrating dataset tools utilities.
+
+This script shows how to:
+1. Delete episodes from a dataset
+2. Split a dataset into train/val sets
+3. Add/remove features
+4. Merge datasets
+
+Usage:
+    python examples/dataset/use_dataset_tools.py
+"""
+
+import numpy as np
+
+from lerobot.datasets.dataset_tools import (
+    add_features,
+    delete_episodes,
+    merge_datasets,
+    modify_features,
+    remove_feature,
+    split_dataset,
+)
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+
+def main():
+    dataset = LeRobotDataset("lerobot/pusht")
+
+    print(f"Original dataset: {dataset.meta.total_episodes} episodes, {dataset.meta.total_frames} frames")
+    print(f"Features: {list(dataset.meta.features.keys())}")
+
+    print("\n1. Deleting episodes 0 and 2...")
+    filtered_dataset = delete_episodes(dataset, episode_indices=[0, 2], repo_id="lerobot/pusht_filtered")
+    print(f"Filtered dataset: {filtered_dataset.meta.total_episodes} episodes")
+
+    print("\n2. Splitting dataset into train/val...")
+    splits = split_dataset(
+        dataset,
+        splits={"train": 0.8, "val": 0.2},
+    )
+    print(f"Train split: {splits['train'].meta.total_episodes} episodes")
+    print(f"Val split: {splits['val'].meta.total_episodes} episodes")
+
+    print("\n3. Adding features...")
+
+    reward_values = np.random.randn(dataset.meta.total_frames).astype(np.float32)
+
+    def compute_success(row_dict, episode_index, frame_index):
+        episode_length = 10
+        return float(frame_index >= episode_length - 10)
+
+    dataset_with_features = add_features(
+        dataset,
+        features={
+            "reward": (
+                reward_values,
+                {"dtype": "float32", "shape": (1,), "names": None},
+            ),
+            "success": (
+                compute_success,
+                {"dtype": "float32", "shape": (1,), "names": None},
+            ),
+        },
+        repo_id="lerobot/pusht_with_features",
+    )
+
+    print(f"New features: {list(dataset_with_features.meta.features.keys())}")
+
+    print("\n4. Removing the success feature...")
+    dataset_cleaned = remove_feature(
+        dataset_with_features, feature_names="success", repo_id="lerobot/pusht_cleaned"
+    )
+    print(f"Features after removal: {list(dataset_cleaned.meta.features.keys())}")
+
+    print("\n5. Using modify_features to add and remove features simultaneously...")
+    dataset_modified = modify_features(
+        dataset_with_features,
+        add_features={
+            "discount": (
+                np.ones(dataset.meta.total_frames, dtype=np.float32) * 0.99,
+                {"dtype": "float32", "shape": (1,), "names": None},
+            ),
+        },
+        remove_features="reward",
+        repo_id="lerobot/pusht_modified",
+    )
+    print(f"Modified features: {list(dataset_modified.meta.features.keys())}")
+
+    print("\n6. Merging train and val splits back together...")
+    merged = merge_datasets([splits["train"], splits["val"]], output_repo_id="lerobot/pusht_merged")
+    print(f"Merged dataset: {merged.meta.total_episodes} episodes")
+
+    print("\n7. Complex workflow example...")
+
+    if len(dataset.meta.camera_keys) > 1:
+        camera_to_remove = dataset.meta.camera_keys[0]
+        print(f"Removing camera: {camera_to_remove}")
+        dataset_no_cam = remove_feature(
+            dataset, feature_names=camera_to_remove, repo_id="pusht_no_first_camera"
+        )
+        print(f"Remaining cameras: {dataset_no_cam.meta.camera_keys}")
+
+    print("\nDone! Check ~/.cache/huggingface/lerobot/ for the created datasets.")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/lekiwi/evaluate.py b/lerobot/examples/lekiwi/evaluate.py
new file mode 100644
index 0000000000000000000000000000000000000000..ef98640aac65e2871055034f7c7b18c701edcf3a
--- /dev/null
+++ b/lerobot/examples/lekiwi/evaluate.py
@@ -0,0 +1,146 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from lerobot.datasets.feature_utils import hw_to_dataset_features
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.act.modeling_act import ACTPolicy
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.processor import make_default_processors
+from lerobot.robots.lekiwi import LeKiwiClient, LeKiwiClientConfig
+from lerobot.scripts.lerobot_record import record_loop
+from lerobot.utils.constants import ACTION, OBS_STR
+from lerobot.utils.control_utils import init_keyboard_listener
+from lerobot.utils.utils import log_say
+from lerobot.utils.visualization_utils import init_rerun
+
+NUM_EPISODES = 2
+FPS = 30
+EPISODE_TIME_SEC = 60
+TASK_DESCRIPTION = "My task description"
+HF_MODEL_ID = "<hf_username>/<model_repo_id>"
+HF_DATASET_ID = "<hf_username>/<eval_dataset_repo_id>"
+
+
+def main():
+    # Create the robot configuration & robot
+    robot_config = LeKiwiClientConfig(remote_ip="172.18.134.136", id="lekiwi")
+
+    robot = LeKiwiClient(robot_config)
+
+    # Create policy
+    policy = ACTPolicy.from_pretrained(HF_MODEL_ID)
+
+    # Configure the dataset features
+    action_features = hw_to_dataset_features(robot.action_features, ACTION)
+    obs_features = hw_to_dataset_features(robot.observation_features, OBS_STR)
+    dataset_features = {**action_features, **obs_features}
+
+    # Create the dataset
+    dataset = LeRobotDataset.create(
+        repo_id=HF_DATASET_ID,
+        fps=FPS,
+        features=dataset_features,
+        robot_type=robot.name,
+        use_videos=True,
+        image_writer_threads=4,
+    )
+
+    # Build Policy Processors
+    preprocessor, postprocessor = make_pre_post_processors(
+        policy_cfg=policy,
+        pretrained_path=HF_MODEL_ID,
+        dataset_stats=dataset.meta.stats,
+        # The inference device is automatically set to match the detected hardware, overriding any previous device settings from training to ensure compatibility.
+        preprocessor_overrides={"device_processor": {"device": str(policy.config.device)}},
+    )
+
+    # Connect the robot
+    # To connect you already should have this script running on LeKiwi: `python -m lerobot.robots.lekiwi.lekiwi_host --robot.id=my_awesome_kiwi`
+    robot.connect()
+
+    # TODO(Steven): Update this example to use pipelines
+    teleop_action_processor, robot_action_processor, robot_observation_processor = make_default_processors()
+
+    # Initialize the keyboard listener and rerun visualization
+    listener, events = init_keyboard_listener()
+    init_rerun(session_name="lekiwi_evaluate")
+
+    try:
+        if not robot.is_connected:
+            raise ValueError("Robot is not connected!")
+
+        print("Starting evaluate loop...")
+        recorded_episodes = 0
+        while recorded_episodes < NUM_EPISODES and not events["stop_recording"]:
+            log_say(f"Running inference, recording eval episode {recorded_episodes} of {NUM_EPISODES}")
+
+            # Main record loop
+            record_loop(
+                robot=robot,
+                events=events,
+                fps=FPS,
+                policy=policy,
+                preprocessor=preprocessor,  # Pass the pre and post policy processors
+                postprocessor=postprocessor,
+                dataset=dataset,
+                control_time_s=EPISODE_TIME_SEC,
+                single_task=TASK_DESCRIPTION,
+                display_data=True,
+                teleop_action_processor=teleop_action_processor,
+                robot_action_processor=robot_action_processor,
+                robot_observation_processor=robot_observation_processor,
+            )
+
+            # Reset the environment if not stopping or re-recording
+            if not events["stop_recording"] and (
+                (recorded_episodes < NUM_EPISODES - 1) or events["rerecord_episode"]
+            ):
+                log_say("Reset the environment")
+                record_loop(
+                    robot=robot,
+                    events=events,
+                    fps=FPS,
+                    control_time_s=EPISODE_TIME_SEC,
+                    single_task=TASK_DESCRIPTION,
+                    display_data=True,
+                    teleop_action_processor=teleop_action_processor,
+                    robot_action_processor=robot_action_processor,
+                    robot_observation_processor=robot_observation_processor,
+                )
+
+            if events["rerecord_episode"]:
+                log_say("Re-record episode")
+                events["rerecord_episode"] = False
+                events["exit_early"] = False
+                dataset.clear_episode_buffer()
+                continue
+
+            # Save episode
+            dataset.save_episode()
+            recorded_episodes += 1
+
+    finally:
+        # Clean up
+        log_say("Stop recording")
+        robot.disconnect()
+        listener.stop()
+
+        dataset.finalize()
+        dataset.push_to_hub()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/lekiwi/record.py b/lerobot/examples/lekiwi/record.py
new file mode 100644
index 0000000000000000000000000000000000000000..ace2e35b85de0520cf8312650fbe2a4057fd6292
--- /dev/null
+++ b/lerobot/examples/lekiwi/record.py
@@ -0,0 +1,142 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from lerobot.datasets.feature_utils import hw_to_dataset_features
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.processor import make_default_processors
+from lerobot.robots.lekiwi.config_lekiwi import LeKiwiClientConfig
+from lerobot.robots.lekiwi.lekiwi_client import LeKiwiClient
+from lerobot.scripts.lerobot_record import record_loop
+from lerobot.teleoperators.keyboard import KeyboardTeleop, KeyboardTeleopConfig
+from lerobot.teleoperators.so_leader import SO100Leader, SO100LeaderConfig
+from lerobot.utils.constants import ACTION, OBS_STR
+from lerobot.utils.control_utils import init_keyboard_listener
+from lerobot.utils.utils import log_say
+from lerobot.utils.visualization_utils import init_rerun
+
+NUM_EPISODES = 2
+FPS = 30
+EPISODE_TIME_SEC = 30
+RESET_TIME_SEC = 10
+TASK_DESCRIPTION = "My task description"
+HF_REPO_ID = "<hf_username>/<dataset_repo_id>"
+
+
+def main():
+    # Create the robot and teleoperator configurations
+    robot_config = LeKiwiClientConfig(remote_ip="172.18.134.136", id="lekiwi")
+    leader_arm_config = SO100LeaderConfig(port="/dev/tty.usbmodem585A0077581", id="my_awesome_leader_arm")
+    keyboard_config = KeyboardTeleopConfig()
+
+    # Initialize the robot and teleoperator
+    robot = LeKiwiClient(robot_config)
+    leader_arm = SO100Leader(leader_arm_config)
+    keyboard = KeyboardTeleop(keyboard_config)
+
+    # TODO(Steven): Update this example to use pipelines
+    teleop_action_processor, robot_action_processor, robot_observation_processor = make_default_processors()
+
+    # Configure the dataset features
+    action_features = hw_to_dataset_features(robot.action_features, ACTION)
+    obs_features = hw_to_dataset_features(robot.observation_features, OBS_STR)
+    dataset_features = {**action_features, **obs_features}
+
+    # Create the dataset
+    dataset = LeRobotDataset.create(
+        repo_id=HF_REPO_ID,
+        fps=FPS,
+        features=dataset_features,
+        robot_type=robot.name,
+        use_videos=True,
+        image_writer_threads=4,
+    )
+
+    # Connect the robot and teleoperator
+    # To connect you already should have this script running on LeKiwi: `python -m lerobot.robots.lekiwi.lekiwi_host --robot.id=my_awesome_kiwi`
+    robot.connect()
+    leader_arm.connect()
+    keyboard.connect()
+
+    # Initialize the keyboard listener and rerun visualization
+    listener, events = init_keyboard_listener()
+    init_rerun(session_name="lekiwi_record")
+
+    try:
+        if not robot.is_connected or not leader_arm.is_connected or not keyboard.is_connected:
+            raise ValueError("Robot or teleop is not connected!")
+
+        print("Starting record loop...")
+        recorded_episodes = 0
+        while recorded_episodes < NUM_EPISODES and not events["stop_recording"]:
+            log_say(f"Recording episode {recorded_episodes}")
+
+            # Main record loop
+            record_loop(
+                robot=robot,
+                events=events,
+                fps=FPS,
+                dataset=dataset,
+                teleop=[leader_arm, keyboard],
+                control_time_s=EPISODE_TIME_SEC,
+                single_task=TASK_DESCRIPTION,
+                display_data=True,
+                teleop_action_processor=teleop_action_processor,
+                robot_action_processor=robot_action_processor,
+                robot_observation_processor=robot_observation_processor,
+            )
+
+            # Reset the environment if not stopping or re-recording
+            if not events["stop_recording"] and (
+                (recorded_episodes < NUM_EPISODES - 1) or events["rerecord_episode"]
+            ):
+                log_say("Reset the environment")
+                record_loop(
+                    robot=robot,
+                    events=events,
+                    fps=FPS,
+                    teleop=[leader_arm, keyboard],
+                    control_time_s=RESET_TIME_SEC,
+                    single_task=TASK_DESCRIPTION,
+                    display_data=True,
+                    teleop_action_processor=teleop_action_processor,
+                    robot_action_processor=robot_action_processor,
+                    robot_observation_processor=robot_observation_processor,
+                )
+
+            if events["rerecord_episode"]:
+                log_say("Re-record episode")
+                events["rerecord_episode"] = False
+                events["exit_early"] = False
+                dataset.clear_episode_buffer()
+                continue
+
+            # Save episode
+            dataset.save_episode()
+            recorded_episodes += 1
+    finally:
+        # Clean up
+        log_say("Stop recording")
+        robot.disconnect()
+        leader_arm.disconnect()
+        keyboard.disconnect()
+        listener.stop()
+
+        dataset.finalize()
+        dataset.push_to_hub()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/lekiwi/replay.py b/lerobot/examples/lekiwi/replay.py
new file mode 100644
index 0000000000000000000000000000000000000000..cf89aea1642248555c4d60f8a04c67942f906719
--- /dev/null
+++ b/lerobot/examples/lekiwi/replay.py
@@ -0,0 +1,69 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import time
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.robots.lekiwi.config_lekiwi import LeKiwiClientConfig
+from lerobot.robots.lekiwi.lekiwi_client import LeKiwiClient
+from lerobot.utils.constants import ACTION
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import log_say
+
+EPISODE_IDX = 0
+
+
+def main():
+    # Initialize the robot config
+    robot_config = LeKiwiClientConfig(remote_ip="172.18.134.136", id="lekiwi")
+
+    # Initialize the robot
+    robot = LeKiwiClient(robot_config)
+
+    # Fetch the dataset to replay
+    dataset = LeRobotDataset("<hf_username>/<dataset_repo_id>", episodes=[EPISODE_IDX])
+    # Filter dataset to only include frames from the specified episode since episodes are chunked in dataset V3.0
+    episode_frames = dataset.hf_dataset.filter(lambda x: x["episode_index"] == EPISODE_IDX)
+    actions = episode_frames.select_columns(ACTION)
+
+    # Connect to the robot
+    robot.connect()
+
+    try:
+        if not robot.is_connected:
+            raise ValueError("Robot is not connected!")
+
+        print("Starting replay loop...")
+        log_say(f"Replaying episode {EPISODE_IDX}")
+        for idx in range(len(episode_frames)):
+            t0 = time.perf_counter()
+
+            # Get recorded action from dataset
+            action = {
+                name: float(actions[idx][ACTION][i])
+                for i, name in enumerate(dataset.features[ACTION]["names"])
+            }
+
+            # Send action to robot
+            _ = robot.send_action(action)
+
+            precise_sleep(max(1.0 / dataset.fps - (time.perf_counter() - t0), 0.0))
+    finally:
+        robot.disconnect()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/lekiwi/teleoperate.py b/lerobot/examples/lekiwi/teleoperate.py
new file mode 100644
index 0000000000000000000000000000000000000000..feb3cbb0138ad6e2d2c897653838c2e1ae761edf
--- /dev/null
+++ b/lerobot/examples/lekiwi/teleoperate.py
@@ -0,0 +1,78 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import time
+
+from lerobot.robots.lekiwi import LeKiwiClient, LeKiwiClientConfig
+from lerobot.teleoperators.keyboard.teleop_keyboard import KeyboardTeleop, KeyboardTeleopConfig
+from lerobot.teleoperators.so_leader import SO100Leader, SO100LeaderConfig
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.visualization_utils import init_rerun, log_rerun_data
+
+FPS = 30
+
+
+def main():
+    # Create the robot and teleoperator configurations
+    robot_config = LeKiwiClientConfig(remote_ip="172.18.134.136", id="my_lekiwi")
+    teleop_arm_config = SO100LeaderConfig(port="/dev/tty.usbmodem585A0077581", id="my_awesome_leader_arm")
+    keyboard_config = KeyboardTeleopConfig(id="my_laptop_keyboard")
+
+    # Initialize the robot and teleoperator
+    robot = LeKiwiClient(robot_config)
+    leader_arm = SO100Leader(teleop_arm_config)
+    keyboard = KeyboardTeleop(keyboard_config)
+
+    # Connect to the robot and teleoperator
+    # To connect you already should have this script running on LeKiwi: `python -m lerobot.robots.lekiwi.lekiwi_host --robot.id=my_awesome_kiwi`
+    robot.connect()
+    leader_arm.connect()
+    keyboard.connect()
+
+    # Init rerun viewer
+    init_rerun(session_name="lekiwi_teleop")
+
+    if not robot.is_connected or not leader_arm.is_connected or not keyboard.is_connected:
+        raise ValueError("Robot or teleop is not connected!")
+
+    print("Starting teleop loop...")
+    while True:
+        t0 = time.perf_counter()
+
+        # Get robot observation
+        observation = robot.get_observation()
+
+        # Get teleop action
+        # Arm
+        arm_action = leader_arm.get_action()
+        arm_action = {f"arm_{k}": v for k, v in arm_action.items()}
+        # Keyboard
+        keyboard_keys = keyboard.get_action()
+        base_action = robot._from_keyboard_to_base_action(keyboard_keys)
+
+        action = {**arm_action, **base_action} if len(base_action) > 0 else arm_action
+
+        # Send action to robot
+        _ = robot.send_action(action)
+
+        # Visualize
+        log_rerun_data(observation=observation, action=action)
+
+        precise_sleep(max(1.0 / FPS - (time.perf_counter() - t0), 0.0))
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/phone_to_so100/evaluate.py b/lerobot/examples/phone_to_so100/evaluate.py
new file mode 100644
index 0000000000000000000000000000000000000000..9cd7a98c2bb3e22fd46b934b627764ae550c2c06
--- /dev/null
+++ b/lerobot/examples/phone_to_so100/evaluate.py
@@ -0,0 +1,208 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.datasets.feature_utils import combine_feature_dicts
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.policies.act.modeling_act import ACTPolicy
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.processor import (
+    RobotProcessorPipeline,
+    make_default_teleop_action_processor,
+)
+from lerobot.processor.converters import (
+    observation_to_transition,
+    robot_action_observation_to_transition,
+    transition_to_observation,
+    transition_to_robot_action,
+)
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+from lerobot.robots.so_follower.robot_kinematic_processor import (
+    ForwardKinematicsJointsToEE,
+    InverseKinematicsEEToJoints,
+)
+from lerobot.scripts.lerobot_record import record_loop
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.control_utils import init_keyboard_listener
+from lerobot.utils.utils import log_say
+from lerobot.utils.visualization_utils import init_rerun
+
+NUM_EPISODES = 5
+FPS = 30
+EPISODE_TIME_SEC = 60
+TASK_DESCRIPTION = "My task description"
+HF_MODEL_ID = "<hf_username>/<model_repo_id>"
+HF_DATASET_ID = "<hf_username>/<dataset_repo_id>"
+
+
+def main():
+    # Create the robot configuration & robot
+    camera_config = {"front": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=FPS)}
+    robot_config = SO100FollowerConfig(
+        port="/dev/tty.usbmodem58760434471",
+        id="my_awesome_follower_arm",
+        cameras=camera_config,
+        use_degrees=True,
+    )
+
+    robot = SO100Follower(robot_config)
+
+    # Create policy
+    policy = ACTPolicy.from_pretrained(HF_MODEL_ID)
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(robot.bus.motors.keys()),
+    )
+
+    # Build pipeline to convert EE action to joints action
+    robot_ee_to_joints_processor = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        steps=[
+            InverseKinematicsEEToJoints(
+                kinematics=kinematics_solver,
+                motor_names=list(robot.bus.motors.keys()),
+                initial_guess_current_joints=True,
+            ),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Build pipeline to convert joints observation to EE observation
+    robot_joints_to_ee_pose_processor = RobotProcessorPipeline[RobotObservation, RobotObservation](
+        steps=[
+            ForwardKinematicsJointsToEE(
+                kinematics=kinematics_solver, motor_names=list(robot.bus.motors.keys())
+            )
+        ],
+        to_transition=observation_to_transition,
+        to_output=transition_to_observation,
+    )
+
+    # Create the dataset
+    dataset = LeRobotDataset.create(
+        repo_id=HF_DATASET_ID,
+        fps=FPS,
+        features=combine_feature_dicts(
+            aggregate_pipeline_dataset_features(
+                pipeline=robot_joints_to_ee_pose_processor,
+                initial_features=create_initial_features(observation=robot.observation_features),
+                use_videos=True,
+            ),
+            # User for now should be explicit on the feature keys that were used for record
+            # Alternatively, the user can pass the processor step that has the right features
+            aggregate_pipeline_dataset_features(
+                pipeline=make_default_teleop_action_processor(),
+                initial_features=create_initial_features(
+                    action={
+                        f"ee.{k}": PolicyFeature(type=FeatureType.ACTION, shape=(1,))
+                        for k in ["x", "y", "z", "wx", "wy", "wz", "gripper_pos"]
+                    }
+                ),
+                use_videos=True,
+            ),
+        ),
+        robot_type=robot.name,
+        use_videos=True,
+        image_writer_threads=4,
+    )
+
+    # Build Policy Processors
+    preprocessor, postprocessor = make_pre_post_processors(
+        policy_cfg=policy,
+        pretrained_path=HF_MODEL_ID,
+        dataset_stats=dataset.meta.stats,
+        # The inference device is automatically set to match the detected hardware, overriding any previous device settings from training to ensure compatibility.
+        preprocessor_overrides={"device_processor": {"device": str(policy.config.device)}},
+    )
+
+    # Connect the robot
+    robot.connect()
+
+    # Initialize the keyboard listener and rerun visualization
+    listener, events = init_keyboard_listener()
+    init_rerun(session_name="phone_so100_evaluate")
+
+    try:
+        if not robot.is_connected:
+            raise ValueError("Robot is not connected!")
+
+        print("Starting evaluate loop...")
+        episode_idx = 0
+        for episode_idx in range(NUM_EPISODES):
+            log_say(f"Running inference, recording eval episode {episode_idx + 1} of {NUM_EPISODES}")
+
+            # Main record loop
+            record_loop(
+                robot=robot,
+                events=events,
+                fps=FPS,
+                policy=policy,
+                preprocessor=preprocessor,  # Pass the pre and post policy processors
+                postprocessor=postprocessor,
+                dataset=dataset,
+                control_time_s=EPISODE_TIME_SEC,
+                single_task=TASK_DESCRIPTION,
+                display_data=True,
+                teleop_action_processor=make_default_teleop_action_processor(),
+                robot_action_processor=robot_ee_to_joints_processor,
+                robot_observation_processor=robot_joints_to_ee_pose_processor,
+            )
+
+            # Reset the environment if not stopping or re-recording
+            if not events["stop_recording"] and (
+                (episode_idx < NUM_EPISODES - 1) or events["rerecord_episode"]
+            ):
+                log_say("Reset the environment")
+                record_loop(
+                    robot=robot,
+                    events=events,
+                    fps=FPS,
+                    control_time_s=EPISODE_TIME_SEC,
+                    single_task=TASK_DESCRIPTION,
+                    display_data=True,
+                    teleop_action_processor=make_default_teleop_action_processor(),
+                    robot_action_processor=robot_ee_to_joints_processor,
+                    robot_observation_processor=robot_joints_to_ee_pose_processor,
+                )
+
+            if events["rerecord_episode"]:
+                log_say("Re-record episode")
+                events["rerecord_episode"] = False
+                events["exit_early"] = False
+                dataset.clear_episode_buffer()
+                continue
+
+            # Save episode
+            dataset.save_episode()
+            episode_idx += 1
+    finally:
+        # Clean up
+        log_say("Stop recording")
+        robot.disconnect()
+        listener.stop()
+
+        dataset.finalize()
+        dataset.push_to_hub()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/phone_to_so100/record.py b/lerobot/examples/phone_to_so100/record.py
new file mode 100644
index 0000000000000000000000000000000000000000..f2a17cd337a7e2a3d2d32b92b6da0c2778100b76
--- /dev/null
+++ b/lerobot/examples/phone_to_so100/record.py
@@ -0,0 +1,217 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.datasets.feature_utils import combine_feature_dicts
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor.converters import (
+    observation_to_transition,
+    robot_action_observation_to_transition,
+    transition_to_observation,
+    transition_to_robot_action,
+)
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+from lerobot.robots.so_follower.robot_kinematic_processor import (
+    EEBoundsAndSafety,
+    EEReferenceAndDelta,
+    ForwardKinematicsJointsToEE,
+    GripperVelocityToJoint,
+    InverseKinematicsEEToJoints,
+)
+from lerobot.scripts.lerobot_record import record_loop
+from lerobot.teleoperators.phone.config_phone import PhoneConfig, PhoneOS
+from lerobot.teleoperators.phone.phone_processor import MapPhoneActionToRobotAction
+from lerobot.teleoperators.phone.teleop_phone import Phone
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.control_utils import init_keyboard_listener
+from lerobot.utils.utils import log_say
+from lerobot.utils.visualization_utils import init_rerun
+
+NUM_EPISODES = 2
+FPS = 30
+EPISODE_TIME_SEC = 60
+RESET_TIME_SEC = 30
+TASK_DESCRIPTION = "My task description"
+HF_REPO_ID = "<hf_username>/<dataset_repo_id>"
+
+
+def main():
+    # Create the robot and teleoperator configurations
+    camera_config = {"front": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=FPS)}
+    robot_config = SO100FollowerConfig(
+        port="/dev/tty.usbmodem5A460814411",
+        id="my_awesome_follower_arm",
+        cameras=camera_config,
+        use_degrees=True,
+    )
+    teleop_config = PhoneConfig(phone_os=PhoneOS.IOS)  # or PhoneOS.ANDROID
+
+    # Initialize the robot and teleoperator
+    robot = SO100Follower(robot_config)
+    phone = Phone(teleop_config)
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(robot.bus.motors.keys()),
+    )
+
+    # Build pipeline to convert phone action to EE action
+    phone_to_robot_ee_pose_processor = RobotProcessorPipeline[
+        tuple[RobotAction, RobotObservation], RobotAction
+    ](
+        steps=[
+            MapPhoneActionToRobotAction(platform=teleop_config.phone_os),
+            EEReferenceAndDelta(
+                kinematics=kinematics_solver,
+                end_effector_step_sizes={"x": 0.5, "y": 0.5, "z": 0.5},
+                motor_names=list(robot.bus.motors.keys()),
+                use_latched_reference=True,
+            ),
+            EEBoundsAndSafety(
+                end_effector_bounds={"min": [-1.0, -1.0, -1.0], "max": [1.0, 1.0, 1.0]},
+                max_ee_step_m=0.20,
+            ),
+            GripperVelocityToJoint(speed_factor=20.0),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Build pipeline to convert EE action to joints action
+    robot_ee_to_joints_processor = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        steps=[
+            InverseKinematicsEEToJoints(
+                kinematics=kinematics_solver,
+                motor_names=list(robot.bus.motors.keys()),
+                initial_guess_current_joints=True,
+            ),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Build pipeline to convert joint observation to EE observation
+    robot_joints_to_ee_pose = RobotProcessorPipeline[RobotObservation, RobotObservation](
+        steps=[
+            ForwardKinematicsJointsToEE(
+                kinematics=kinematics_solver, motor_names=list(robot.bus.motors.keys())
+            )
+        ],
+        to_transition=observation_to_transition,
+        to_output=transition_to_observation,
+    )
+
+    # Create the dataset
+    dataset = LeRobotDataset.create(
+        repo_id=HF_REPO_ID,
+        fps=FPS,
+        features=combine_feature_dicts(
+            # Run the feature contract of the pipelines
+            # This tells you how the features would look like after the pipeline steps
+            aggregate_pipeline_dataset_features(
+                pipeline=phone_to_robot_ee_pose_processor,
+                initial_features=create_initial_features(action=phone.action_features),
+                use_videos=True,
+            ),
+            aggregate_pipeline_dataset_features(
+                pipeline=robot_joints_to_ee_pose,
+                initial_features=create_initial_features(observation=robot.observation_features),
+                use_videos=True,
+            ),
+        ),
+        robot_type=robot.name,
+        use_videos=True,
+        image_writer_threads=4,
+    )
+
+    # Connect the robot and teleoperator
+    robot.connect()
+    phone.connect()
+
+    # Initialize the keyboard listener and rerun visualization
+    listener, events = init_keyboard_listener()
+    init_rerun(session_name="phone_so100_record")
+
+    try:
+        if not robot.is_connected or not phone.is_connected:
+            raise ValueError("Robot or teleop is not connected!")
+
+        print("Starting record loop. Move your phone to teleoperate the robot...")
+        episode_idx = 0
+        while episode_idx < NUM_EPISODES and not events["stop_recording"]:
+            log_say(f"Recording episode {episode_idx + 1} of {NUM_EPISODES}")
+
+            # Main record loop
+            record_loop(
+                robot=robot,
+                events=events,
+                fps=FPS,
+                teleop=phone,
+                dataset=dataset,
+                control_time_s=EPISODE_TIME_SEC,
+                single_task=TASK_DESCRIPTION,
+                display_data=True,
+                teleop_action_processor=phone_to_robot_ee_pose_processor,
+                robot_action_processor=robot_ee_to_joints_processor,
+                robot_observation_processor=robot_joints_to_ee_pose,
+            )
+
+            # Reset the environment if not stopping or re-recording
+            if not events["stop_recording"] and (
+                episode_idx < NUM_EPISODES - 1 or events["rerecord_episode"]
+            ):
+                log_say("Reset the environment")
+                record_loop(
+                    robot=robot,
+                    events=events,
+                    fps=FPS,
+                    teleop=phone,
+                    control_time_s=RESET_TIME_SEC,
+                    single_task=TASK_DESCRIPTION,
+                    display_data=True,
+                    teleop_action_processor=phone_to_robot_ee_pose_processor,
+                    robot_action_processor=robot_ee_to_joints_processor,
+                    robot_observation_processor=robot_joints_to_ee_pose,
+                )
+
+            if events["rerecord_episode"]:
+                log_say("Re-recording episode")
+                events["rerecord_episode"] = False
+                events["exit_early"] = False
+                dataset.clear_episode_buffer()
+                continue
+
+            # Save episode
+            dataset.save_episode()
+            episode_idx += 1
+    finally:
+        # Clean up
+        log_say("Stop recording")
+        robot.disconnect()
+        phone.disconnect()
+        listener.stop()
+
+        dataset.finalize()
+        dataset.push_to_hub()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/phone_to_so100/replay.py b/lerobot/examples/phone_to_so100/replay.py
new file mode 100644
index 0000000000000000000000000000000000000000..7b955cdb747efb6ec4097634684efd91d46559cd
--- /dev/null
+++ b/lerobot/examples/phone_to_so100/replay.py
@@ -0,0 +1,108 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import time
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor.converters import (
+    robot_action_observation_to_transition,
+    transition_to_robot_action,
+)
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+from lerobot.robots.so_follower.robot_kinematic_processor import (
+    InverseKinematicsEEToJoints,
+)
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.constants import ACTION
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import log_say
+
+EPISODE_IDX = 0
+HF_REPO_ID = "<hf_username>/<dataset_repo_id>"
+
+
+def main():
+    # Initialize the robot config
+    robot_config = SO100FollowerConfig(
+        port="/dev/tty.usbmodem5A460814411", id="my_awesome_follower_arm", use_degrees=True
+    )
+
+    # Initialize the robot
+    robot = SO100Follower(robot_config)
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(robot.bus.motors.keys()),
+    )
+
+    # Build pipeline to convert EE action to joints action
+    robot_ee_to_joints_processor = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        steps=[
+            InverseKinematicsEEToJoints(
+                kinematics=kinematics_solver,
+                motor_names=list(robot.bus.motors.keys()),
+                initial_guess_current_joints=False,  # Because replay is open loop
+            ),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Fetch the dataset to replay
+    dataset = LeRobotDataset(HF_REPO_ID, episodes=[EPISODE_IDX])
+    # Filter dataset to only include frames from the specified episode since episodes are chunked in dataset V3.0
+    episode_frames = dataset.hf_dataset.filter(lambda x: x["episode_index"] == EPISODE_IDX)
+    actions = episode_frames.select_columns(ACTION)
+
+    # Connect to the robot
+    robot.connect()
+
+    try:
+        if not robot.is_connected:
+            raise ValueError("Robot is not connected!")
+
+        print("Starting replay loop...")
+        log_say(f"Replaying episode {EPISODE_IDX}")
+        for idx in range(len(episode_frames)):
+            t0 = time.perf_counter()
+
+            # Get recorded action from dataset
+            ee_action = {
+                name: float(actions[idx][ACTION][i])
+                for i, name in enumerate(dataset.features[ACTION]["names"])
+            }
+
+            # Get robot observation
+            robot_obs = robot.get_observation()
+
+            # Dataset EE -> robot joints
+            joint_action = robot_ee_to_joints_processor((ee_action, robot_obs))
+
+            # Send action to robot
+            _ = robot.send_action(joint_action)
+
+            precise_sleep(max(1.0 / dataset.fps - (time.perf_counter() - t0), 0.0))
+    finally:
+        # Clean up
+        robot.disconnect()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/phone_to_so100/teleoperate.py b/lerobot/examples/phone_to_so100/teleoperate.py
new file mode 100644
index 0000000000000000000000000000000000000000..7242c39ce46eabdb49376ccf0bd447c145d4a375
--- /dev/null
+++ b/lerobot/examples/phone_to_so100/teleoperate.py
@@ -0,0 +1,121 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specif
+
+import time
+
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor.converters import (
+    robot_action_observation_to_transition,
+    transition_to_robot_action,
+)
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+from lerobot.robots.so_follower.robot_kinematic_processor import (
+    EEBoundsAndSafety,
+    EEReferenceAndDelta,
+    GripperVelocityToJoint,
+    InverseKinematicsEEToJoints,
+)
+from lerobot.teleoperators.phone.config_phone import PhoneConfig, PhoneOS
+from lerobot.teleoperators.phone.phone_processor import MapPhoneActionToRobotAction
+from lerobot.teleoperators.phone.teleop_phone import Phone
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.visualization_utils import init_rerun, log_rerun_data
+
+FPS = 30
+
+
+def main():
+    # Initialize the robot and teleoperator
+    robot_config = SO100FollowerConfig(
+        port="/dev/tty.usbmodem5A460814411", id="my_awesome_follower_arm", use_degrees=True
+    )
+    teleop_config = PhoneConfig(phone_os=PhoneOS.IOS)  # or PhoneOS.ANDROID
+
+    # Initialize the robot and teleoperator
+    robot = SO100Follower(robot_config)
+    teleop_device = Phone(teleop_config)
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(robot.bus.motors.keys()),
+    )
+
+    # Build pipeline to convert phone action to ee pose action to joint action
+    phone_to_robot_joints_processor = RobotProcessorPipeline[
+        tuple[RobotAction, RobotObservation], RobotAction
+    ](
+        steps=[
+            MapPhoneActionToRobotAction(platform=teleop_config.phone_os),
+            EEReferenceAndDelta(
+                kinematics=kinematics_solver,
+                end_effector_step_sizes={"x": 0.5, "y": 0.5, "z": 0.5},
+                motor_names=list(robot.bus.motors.keys()),
+                use_latched_reference=True,
+            ),
+            EEBoundsAndSafety(
+                end_effector_bounds={"min": [-1.0, -1.0, -1.0], "max": [1.0, 1.0, 1.0]},
+                max_ee_step_m=0.10,
+            ),
+            GripperVelocityToJoint(
+                speed_factor=20.0,
+            ),
+            InverseKinematicsEEToJoints(
+                kinematics=kinematics_solver,
+                motor_names=list(robot.bus.motors.keys()),
+                initial_guess_current_joints=True,
+            ),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Connect to the robot and teleoperator
+    robot.connect()
+    teleop_device.connect()
+
+    # Init rerun viewer
+    init_rerun(session_name="phone_so100_teleop")
+
+    if not robot.is_connected or not teleop_device.is_connected:
+        raise ValueError("Robot or teleop is not connected!")
+
+    print("Starting teleop loop. Move your phone to teleoperate the robot...")
+    while True:
+        t0 = time.perf_counter()
+
+        # Get robot observation
+        robot_obs = robot.get_observation()
+
+        # Get teleop action
+        phone_obs = teleop_device.get_action()
+
+        # Phone -> EE pose -> Joints transition
+        joint_action = phone_to_robot_joints_processor((phone_obs, robot_obs))
+
+        # Send action to robot
+        _ = robot.send_action(joint_action)
+
+        # Visualize
+        log_rerun_data(observation=phone_obs, action=joint_action)
+
+        precise_sleep(max(1.0 / FPS - (time.perf_counter() - t0), 0.0))
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/port_datasets/display_error_files.py b/lerobot/examples/port_datasets/display_error_files.py
new file mode 100644
index 0000000000000000000000000000000000000000..fffab5ff38cb8d57b3a0c131ac6d47d393759455
--- /dev/null
+++ b/lerobot/examples/port_datasets/display_error_files.py
@@ -0,0 +1,85 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import argparse
+import json
+from pathlib import Path
+
+
+def find_missing_workers(completions_dir, world_size):
+    """Find workers that are not completed and returns their indices."""
+    full = list(range(world_size))
+
+    completed = []
+    for path in completions_dir.glob("*"):
+        if path.name in [".", ".."]:
+            continue
+        index = path.name.lstrip("0")
+        index = 0 if index == "" else int(index)
+        completed.append(index)
+
+    missing_workers = set(full) - set(completed)
+    return missing_workers
+
+
+def find_output_files(slurm_dir, worker_indices):
+    """Find output files associated to worker indices, and return tuples
+    of (worker index, output file path)
+    """
+    out_files = []
+    for path in slurm_dir.glob("*.out"):
+        _, worker_id = path.name.replace(".out", "").split("_")
+        worker_id = int(worker_id)
+        if worker_id in worker_indices:
+            out_files.append((worker_id, path))
+    return out_files
+
+
+def display_error_files(logs_dir, job_name):
+    executor_path = Path(logs_dir) / job_name / "executor.json"
+    completions_dir = Path(logs_dir) / job_name / "completions"
+
+    with open(executor_path) as f:
+        executor = json.load(f)
+
+    missing_workers = find_missing_workers(completions_dir, executor["world_size"])
+
+    for missing in sorted(missing_workers)[::-1]:
+        print(missing)
+
+
+def main():
+    parser = argparse.ArgumentParser()
+
+    parser.add_argument(
+        "--logs-dir",
+        type=str,
+        help="Path to logs directory for `datatrove`.",
+    )
+    parser.add_argument(
+        "--job-name",
+        type=str,
+        default="port_droid",
+        help="Job name used in slurm, and name of the directory created inside the provided logs directory.",
+    )
+
+    args = parser.parse_args()
+
+    display_error_files(**vars(args))
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/port_datasets/port_droid.py b/lerobot/examples/port_datasets/port_droid.py
new file mode 100644
index 0000000000000000000000000000000000000000..f58bacbe095b6a9f7d06f648d267ace8953bf889
--- /dev/null
+++ b/lerobot/examples/port_datasets/port_droid.py
@@ -0,0 +1,433 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import argparse
+import logging
+import time
+from pathlib import Path
+
+import numpy as np
+import tensorflow_datasets as tfds
+
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.utils.utils import get_elapsed_time_in_days_hours_minutes_seconds
+
+DROID_SHARDS = 2048
+DROID_FPS = 15
+DROID_ROBOT_TYPE = "Franka"
+
+# Dataset schema slightly adapted from: https://droid-dataset.github.io/droid/the-droid-dataset.html#-dataset-schema
+DROID_FEATURES = {
+    # true on first step of the episode
+    "is_first": {
+        "dtype": "bool",
+        "shape": (1,),
+        "names": None,
+    },
+    # true on last step of the episode
+    "is_last": {
+        "dtype": "bool",
+        "shape": (1,),
+        "names": None,
+    },
+    # true on last step of the episode if it is a terminal step, True for demos
+    "is_terminal": {
+        "dtype": "bool",
+        "shape": (1,),
+        "names": None,
+    },
+    # language_instruction is also stored as "task" to follow LeRobot standard
+    "language_instruction": {
+        "dtype": "string",
+        "shape": (1,),
+        "names": None,
+    },
+    "language_instruction_2": {
+        "dtype": "string",
+        "shape": (1,),
+        "names": None,
+    },
+    "language_instruction_3": {
+        "dtype": "string",
+        "shape": (1,),
+        "names": None,
+    },
+    "observation.state.gripper_position": {
+        "dtype": "float32",
+        "shape": (1,),
+        "names": {
+            "axes": ["gripper"],
+        },
+    },
+    "observation.state.cartesian_position": {
+        "dtype": "float32",
+        "shape": (6,),
+        "names": {
+            "axes": ["x", "y", "z", "roll", "pitch", "yaw"],
+        },
+    },
+    "observation.state.joint_position": {
+        "dtype": "float32",
+        "shape": (7,),
+        "names": {
+            "axes": ["joint_0", "joint_1", "joint_2", "joint_3", "joint_4", "joint_5", "joint_6"],
+        },
+    },
+    # Add this new feature to follow LeRobot standard of using joint position + gripper
+    "observation.state": {
+        "dtype": "float32",
+        "shape": (8,),
+        "names": {
+            "axes": ["joint_0", "joint_1", "joint_2", "joint_3", "joint_4", "joint_5", "joint_6", "gripper"],
+        },
+    },
+    # Initially called wrist_image_left
+    "observation.images.wrist_left": {
+        "dtype": "video",
+        "shape": (180, 320, 3),
+        "names": [
+            "height",
+            "width",
+            "channels",
+        ],
+    },
+    # Initially called exterior_image_1_left
+    "observation.images.exterior_1_left": {
+        "dtype": "video",
+        "shape": (180, 320, 3),
+        "names": [
+            "height",
+            "width",
+            "channels",
+        ],
+    },
+    # Initially called exterior_image_2_left
+    "observation.images.exterior_2_left": {
+        "dtype": "video",
+        "shape": (180, 320, 3),
+        "names": [
+            "height",
+            "width",
+            "channels",
+        ],
+    },
+    "action.gripper_position": {
+        "dtype": "float32",
+        "shape": (1,),
+        "names": {
+            "axes": ["gripper"],
+        },
+    },
+    "action.gripper_velocity": {
+        "dtype": "float32",
+        "shape": (1,),
+        "names": {
+            "axes": ["gripper"],
+        },
+    },
+    "action.cartesian_position": {
+        "dtype": "float32",
+        "shape": (6,),
+        "names": {
+            "axes": ["x", "y", "z", "roll", "pitch", "yaw"],
+        },
+    },
+    "action.cartesian_velocity": {
+        "dtype": "float32",
+        "shape": (6,),
+        "names": {
+            "axes": ["x", "y", "z", "roll", "pitch", "yaw"],
+        },
+    },
+    "action.joint_position": {
+        "dtype": "float32",
+        "shape": (7,),
+        "names": {
+            "axes": ["joint_0", "joint_1", "joint_2", "joint_3", "joint_4", "joint_5", "joint_6"],
+        },
+    },
+    "action.joint_velocity": {
+        "dtype": "float32",
+        "shape": (7,),
+        "names": {
+            "axes": ["joint_0", "joint_1", "joint_2", "joint_3", "joint_4", "joint_5", "joint_6"],
+        },
+    },
+    # This feature was called "action" in RLDS dataset and consists of [6x joint velocities, 1x gripper position]
+    "action.original": {
+        "dtype": "float32",
+        "shape": (7,),
+        "names": {
+            "axes": ["x", "y", "z", "roll", "pitch", "yaw", "gripper"],
+        },
+    },
+    # Add this new feature to follow LeRobot standard of using joint position + gripper
+    "action": {
+        "dtype": "float32",
+        "shape": (8,),
+        "names": {
+            "axes": ["joint_0", "joint_1", "joint_2", "joint_3", "joint_4", "joint_5", "joint_6", "gripper"],
+        },
+    },
+    "discount": {
+        "dtype": "float32",
+        "shape": (1,),
+        "names": None,
+    },
+    "reward": {
+        "dtype": "float32",
+        "shape": (1,),
+        "names": None,
+    },
+    # Meta data that are the same for all frames in the episode
+    "task_category": {
+        "dtype": "string",
+        "shape": (1,),
+        "names": None,
+    },
+    "building": {
+        "dtype": "string",
+        "shape": (1,),
+        "names": None,
+    },
+    "collector_id": {
+        "dtype": "string",
+        "shape": (1,),
+        "names": None,
+    },
+    "date": {
+        "dtype": "string",
+        "shape": (1,),
+        "names": None,
+    },
+    "camera_extrinsics.wrist_left": {
+        "dtype": "float32",
+        "shape": (6,),
+        "names": {
+            "axes": ["x", "y", "z", "roll", "pitch", "yaw"],
+        },
+    },
+    "camera_extrinsics.exterior_1_left": {
+        "dtype": "float32",
+        "shape": (6,),
+        "names": {
+            "axes": ["x", "y", "z", "roll", "pitch", "yaw"],
+        },
+    },
+    "camera_extrinsics.exterior_2_left": {
+        "dtype": "float32",
+        "shape": (6,),
+        "names": {
+            "axes": ["x", "y", "z", "roll", "pitch", "yaw"],
+        },
+    },
+    "is_episode_successful": {
+        "dtype": "bool",
+        "shape": (1,),
+        "names": None,
+    },
+}
+
+
+def is_episode_successful(tf_episode_metadata):
+    # Adapted from: https://github.com/droid-dataset/droid_policy_learning/blob/dd1020eb20d981f90b5ff07dc80d80d5c0cb108b/robomimic/utils/rlds_utils.py#L8
+    return "/success/" in tf_episode_metadata["file_path"].numpy().decode()
+
+
+def generate_lerobot_frames(tf_episode):
+    m = tf_episode["episode_metadata"]
+    frame_meta = {
+        "task_category": m["building"].numpy().decode(),
+        "building": m["building"].numpy().decode(),
+        "collector_id": m["collector_id"].numpy().decode(),
+        "date": m["date"].numpy().decode(),
+        "camera_extrinsics.wrist_left": m["extrinsics_wrist_cam"].numpy(),
+        "camera_extrinsics.exterior_1_left": m["extrinsics_exterior_cam_1"].numpy(),
+        "camera_extrinsics.exterior_2_left": m["extrinsics_exterior_cam_2"].numpy(),
+        "is_episode_successful": np.array([is_episode_successful(m)]),
+    }
+    for f in tf_episode["steps"]:
+        # Dataset schema slightly adapted from: https://droid-dataset.github.io/droid/the-droid-dataset.html#-dataset-schema
+        frame = {
+            "is_first": np.array([f["is_first"].numpy()]),
+            "is_last": np.array([f["is_last"].numpy()]),
+            "is_terminal": np.array([f["is_terminal"].numpy()]),
+            "language_instruction": f["language_instruction"].numpy().decode(),
+            "language_instruction_2": f["language_instruction_2"].numpy().decode(),
+            "language_instruction_3": f["language_instruction_3"].numpy().decode(),
+            "observation.state.gripper_position": f["observation"]["gripper_position"].numpy(),
+            "observation.state.cartesian_position": f["observation"]["cartesian_position"].numpy(),
+            "observation.state.joint_position": f["observation"]["joint_position"].numpy(),
+            "observation.images.wrist_left": f["observation"]["wrist_image_left"].numpy(),
+            "observation.images.exterior_1_left": f["observation"]["exterior_image_1_left"].numpy(),
+            "observation.images.exterior_2_left": f["observation"]["exterior_image_2_left"].numpy(),
+            "action.gripper_position": f["action_dict"]["gripper_position"].numpy(),
+            "action.gripper_velocity": f["action_dict"]["gripper_velocity"].numpy(),
+            "action.cartesian_position": f["action_dict"]["cartesian_position"].numpy(),
+            "action.cartesian_velocity": f["action_dict"]["cartesian_velocity"].numpy(),
+            "action.joint_position": f["action_dict"]["joint_position"].numpy(),
+            "action.joint_velocity": f["action_dict"]["joint_velocity"].numpy(),
+            "discount": np.array([f["discount"].numpy()]),
+            "reward": np.array([f["reward"].numpy()]),
+            "action.original": f["action"].numpy(),
+        }
+
+        # language_instruction is also stored as "task" to follow LeRobot standard
+        frame["task"] = frame["language_instruction"]
+
+        # Add this new feature to follow LeRobot standard of using joint position + gripper
+        frame["observation.state"] = np.concatenate(
+            [frame["observation.state.joint_position"], frame["observation.state.gripper_position"]]
+        )
+        frame["action"] = np.concatenate([frame["action.joint_position"], frame["action.gripper_position"]])
+
+        # Meta data that are the same for all frames in the episode
+        frame.update(frame_meta)
+
+        # Cast fp64 to fp32
+        for key in frame:
+            if isinstance(frame[key], np.ndarray) and frame[key].dtype == np.float64:
+                frame[key] = frame[key].astype(np.float32)
+
+        yield frame
+
+
+def port_droid(
+    raw_dir: Path,
+    repo_id: str,
+    push_to_hub: bool = False,
+    num_shards: int | None = None,
+    shard_index: int | None = None,
+):
+    dataset_name = raw_dir.parent.name
+    version = raw_dir.name
+    data_dir = raw_dir.parent.parent
+
+    builder = tfds.builder(f"{dataset_name}/{version}", data_dir=data_dir, version="")
+
+    if num_shards is not None:
+        tfds_num_shards = builder.info.splits["train"].num_shards
+        if tfds_num_shards != DROID_SHARDS:
+            raise ValueError(
+                f"Number of shards of Droid dataset is expected to be {DROID_SHARDS} but is {tfds_num_shards}."
+            )
+        if num_shards != tfds_num_shards:
+            raise ValueError(
+                f"We only shard over the fixed number of shards provided by tensorflow dataset ({tfds_num_shards}), but {num_shards} shards provided instead."
+            )
+        if shard_index >= tfds_num_shards:
+            raise ValueError(
+                f"Shard index is greater than the num of shards ({shard_index} >= {num_shards})."
+            )
+
+        raw_dataset = builder.as_dataset(split=f"train[{shard_index}shard]")
+    else:
+        raw_dataset = builder.as_dataset(split="train")
+
+    lerobot_dataset = LeRobotDataset.create(
+        repo_id=repo_id,
+        robot_type=DROID_ROBOT_TYPE,
+        fps=DROID_FPS,
+        features=DROID_FEATURES,
+    )
+
+    start_time = time.time()
+    num_episodes = raw_dataset.cardinality().numpy().item()
+    logging.info(f"Number of episodes {num_episodes}")
+
+    for episode_index, episode in enumerate(raw_dataset):
+        elapsed_time = time.time() - start_time
+        d, h, m, s = get_elapsed_time_in_days_hours_minutes_seconds(elapsed_time)
+
+        logging.info(
+            f"{episode_index} / {num_episodes} episodes processed (after {d} days, {h} hours, {m} minutes, {s:.3f} seconds)"
+        )
+
+        for frame in generate_lerobot_frames(episode):
+            lerobot_dataset.add_frame(frame)
+
+        lerobot_dataset.save_episode()
+        logging.info("Save_episode")
+
+    lerobot_dataset.finalize()
+
+    if push_to_hub:
+        lerobot_dataset.push_to_hub(
+            # Add openx tag, since it belongs to the openx collection of datasets
+            tags=["openx"],
+            private=False,
+        )
+
+
+def validate_dataset(repo_id):
+    """Sanity check that ensure meta data can be loaded and all files are present."""
+    meta = LeRobotDatasetMetadata(repo_id)
+
+    if meta.total_episodes == 0:
+        raise ValueError("Number of episodes is 0.")
+
+    for ep_idx in range(meta.total_episodes):
+        data_path = meta.root / meta.get_data_file_path(ep_idx)
+
+        if not data_path.exists():
+            raise ValueError(f"Parquet file is missing in: {data_path}")
+
+        for vid_key in meta.video_keys:
+            vid_path = meta.root / meta.get_video_file_path(ep_idx, vid_key)
+            if not vid_path.exists():
+                raise ValueError(f"Video file is missing in: {vid_path}")
+
+
+def main():
+    parser = argparse.ArgumentParser()
+
+    parser.add_argument(
+        "--raw-dir",
+        type=Path,
+        required=True,
+        help="Directory containing input raw datasets (e.g. `path/to/dataset` or `path/to/dataset/version).",
+    )
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        help="Repositery identifier on Hugging Face: a community or a user name `/` the name of the dataset, required when push-to-hub is True",
+    )
+    parser.add_argument(
+        "--push-to-hub",
+        action="store_true",
+        help="Upload to hub.",
+    )
+    parser.add_argument(
+        "--num-shards",
+        type=int,
+        default=None,
+        help="Number of shards. Can be either None to load the full dataset, or 2048 to load one of the 2048 tensorflow dataset files.",
+    )
+    parser.add_argument(
+        "--shard-index",
+        type=int,
+        default=None,
+        help="Index of the shard. Can be either None to load the full dataset, or in [0,2047] to load one of the 2048 tensorflow dataset files.",
+    )
+
+    args = parser.parse_args()
+
+    port_droid(**vars(args))
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/port_datasets/slurm_aggregate_shards.py b/lerobot/examples/port_datasets/slurm_aggregate_shards.py
new file mode 100644
index 0000000000000000000000000000000000000000..af5473c79728af3edda2c7624172637f8f059f70
--- /dev/null
+++ b/lerobot/examples/port_datasets/slurm_aggregate_shards.py
@@ -0,0 +1,149 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import argparse
+from pathlib import Path
+
+from datatrove.executor import LocalPipelineExecutor
+from datatrove.executor.slurm import SlurmPipelineExecutor
+from datatrove.pipeline.base import PipelineStep
+from port_droid import DROID_SHARDS
+
+
+class AggregateDatasets(PipelineStep):
+    def __init__(
+        self,
+        repo_ids: list[str],
+        aggregated_repo_id: str,
+    ):
+        super().__init__()
+        self.repo_ids = repo_ids
+        self.aggr_repo_id = aggregated_repo_id
+
+    def run(self, data=None, rank: int = 0, world_size: int = 1):
+        import logging
+
+        from lerobot.datasets.aggregate import aggregate_datasets
+        from lerobot.utils.utils import init_logging
+
+        init_logging()
+
+        # Since aggregate_datasets already handles parallel processing internally,
+        # we only need one worker to run the entire aggregation
+        if rank == 0:
+            logging.info(f"Starting aggregation of {len(self.repo_ids)} datasets into {self.aggr_repo_id}")
+            aggregate_datasets(self.repo_ids, self.aggr_repo_id)
+            logging.info("Aggregation complete!")
+        else:
+            logging.info(f"Worker {rank} skipping - only worker 0 performs aggregation")
+
+
+def make_aggregate_executor(
+    repo_ids, repo_id, job_name, logs_dir, workers, partition, cpus_per_task, mem_per_cpu, slurm=True
+):
+    kwargs = {
+        "pipeline": [
+            AggregateDatasets(repo_ids, repo_id),
+        ],
+        "logging_dir": str(logs_dir / job_name),
+    }
+
+    if slurm:
+        # For aggregation, we only need 1 task since aggregate_datasets handles everything
+        kwargs.update(
+            {
+                "job_name": job_name,
+                "tasks": 1,  # Only need 1 task for aggregation
+                "workers": 1,  # Only need 1 worker
+                "time": "08:00:00",
+                "partition": partition,
+                "cpus_per_task": cpus_per_task,
+                "sbatch_args": {"mem-per-cpu": mem_per_cpu},
+            }
+        )
+        executor = SlurmPipelineExecutor(**kwargs)
+    else:
+        kwargs.update(
+            {
+                "tasks": 1,
+                "workers": 1,
+            }
+        )
+        executor = LocalPipelineExecutor(**kwargs)
+
+    return executor
+
+
+def main():
+    parser = argparse.ArgumentParser()
+
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        help="Repository identifier on Hugging Face: a community or a user name `/` the name of the dataset, required when push-to-hub is True.",
+    )
+    parser.add_argument(
+        "--logs-dir",
+        type=Path,
+        help="Path to logs directory for `datatrove`.",
+    )
+    parser.add_argument(
+        "--job-name",
+        type=str,
+        default="aggr_droid",
+        help="Job name used in slurm, and name of the directory created inside the provided logs directory.",
+    )
+    parser.add_argument(
+        "--slurm",
+        type=int,
+        default=1,
+        help="Launch over slurm. Use `--slurm 0` to launch sequentially (useful to debug).",
+    )
+    parser.add_argument(
+        "--workers",
+        type=int,
+        default=1,  # Changed default to 1 since aggregation doesn't need multiple workers
+        help="Number of slurm workers. For aggregation, this should be 1.",
+    )
+    parser.add_argument(
+        "--partition",
+        type=str,
+        help="Slurm partition. Ideally a CPU partition. No need for GPU partition.",
+    )
+    parser.add_argument(
+        "--cpus-per-task",
+        type=int,
+        default=8,
+        help="Number of cpus that each slurm worker will use.",
+    )
+    parser.add_argument(
+        "--mem-per-cpu",
+        type=str,
+        default="1950M",
+        help="Memory per cpu that each worker will use.",
+    )
+
+    args = parser.parse_args()
+    kwargs = vars(args)
+    kwargs["slurm"] = kwargs.pop("slurm") == 1
+
+    repo_ids = [f"{args.repo_id}_world_{DROID_SHARDS}_rank_{rank}" for rank in range(DROID_SHARDS)]
+    aggregate_executor = make_aggregate_executor(repo_ids, **kwargs)
+    aggregate_executor.run()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/port_datasets/slurm_port_shards.py b/lerobot/examples/port_datasets/slurm_port_shards.py
new file mode 100644
index 0000000000000000000000000000000000000000..657ea870c8de103eb43df17b667ca3fb9433b25a
--- /dev/null
+++ b/lerobot/examples/port_datasets/slurm_port_shards.py
@@ -0,0 +1,162 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import argparse
+from pathlib import Path
+
+from datatrove.executor import LocalPipelineExecutor
+from datatrove.executor.slurm import SlurmPipelineExecutor
+from datatrove.pipeline.base import PipelineStep
+from port_droid import DROID_SHARDS
+
+
+class PortDroidShards(PipelineStep):
+    def __init__(
+        self,
+        raw_dir: Path | str,
+        repo_id: str = None,
+    ):
+        super().__init__()
+        self.raw_dir = Path(raw_dir)
+        self.repo_id = repo_id
+
+    def run(self, data=None, rank: int = 0, world_size: int = 1):
+        from datasets.utils.tqdm import disable_progress_bars
+        from port_droid import port_droid, validate_dataset
+
+        from lerobot.utils.utils import init_logging
+
+        init_logging()
+        disable_progress_bars()
+
+        shard_repo_id = f"{self.repo_id}_world_{world_size}_rank_{rank}"
+
+        try:
+            validate_dataset(shard_repo_id)
+            return
+        except Exception:
+            pass  # nosec B110 - Dataset doesn't exist yet, continue with porting
+
+        port_droid(
+            self.raw_dir,
+            shard_repo_id,
+            push_to_hub=False,
+            num_shards=world_size,
+            shard_index=rank,
+        )
+
+        validate_dataset(shard_repo_id)
+
+
+def make_port_executor(
+    raw_dir, repo_id, job_name, logs_dir, workers, partition, cpus_per_task, mem_per_cpu, slurm=True
+):
+    kwargs = {
+        "pipeline": [
+            PortDroidShards(raw_dir, repo_id),
+        ],
+        "logging_dir": str(logs_dir / job_name),
+    }
+
+    if slurm:
+        kwargs.update(
+            {
+                "job_name": job_name,
+                "tasks": DROID_SHARDS,
+                "workers": workers,
+                "time": "08:00:00",
+                "partition": partition,
+                "cpus_per_task": cpus_per_task,
+                "sbatch_args": {"mem-per-cpu": mem_per_cpu},
+            }
+        )
+        executor = SlurmPipelineExecutor(**kwargs)
+    else:
+        kwargs.update(
+            {
+                "tasks": 1,
+                "workers": 1,
+            }
+        )
+        executor = LocalPipelineExecutor(**kwargs)
+
+    return executor
+
+
+def main():
+    parser = argparse.ArgumentParser()
+
+    parser.add_argument(
+        "--raw-dir",
+        type=Path,
+        required=True,
+        help="Directory containing input raw datasets (e.g. `path/to/dataset` or `path/to/dataset/version).",
+    )
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        help="Repositery identifier on Hugging Face: a community or a user name `/` the name of the dataset, required when push-to-hub is True.",
+    )
+    parser.add_argument(
+        "--logs-dir",
+        type=Path,
+        help="Path to logs directory for `datatrove`.",
+    )
+    parser.add_argument(
+        "--job-name",
+        type=str,
+        default="port_droid",
+        help="Job name used in slurm, and name of the directory created inside the provided logs directory.",
+    )
+    parser.add_argument(
+        "--slurm",
+        type=int,
+        default=1,
+        help="Launch over slurm. Use `--slurm 0` to launch sequentially (useful to debug).",
+    )
+    parser.add_argument(
+        "--workers",
+        type=int,
+        default=2048,
+        help="Number of slurm workers. It should be less than the maximum number of shards.",
+    )
+    parser.add_argument(
+        "--partition",
+        type=str,
+        help="Slurm partition. Ideally a CPU partition. No need for GPU partition.",
+    )
+    parser.add_argument(
+        "--cpus-per-task",
+        type=int,
+        default=8,
+        help="Number of cpus that each slurm worker will use.",
+    )
+    parser.add_argument(
+        "--mem-per-cpu",
+        type=str,
+        default="1950M",
+        help="Memory per cpu that each worker will use.",
+    )
+
+    args = parser.parse_args()
+    kwargs = vars(args)
+    kwargs["slurm"] = kwargs.pop("slurm") == 1
+    port_executor = make_port_executor(**kwargs)
+    port_executor.run()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/port_datasets/slurm_upload.py b/lerobot/examples/port_datasets/slurm_upload.py
new file mode 100644
index 0000000000000000000000000000000000000000..7fb01c11bc29e0491c524caab8429d2452dea2ed
--- /dev/null
+++ b/lerobot/examples/port_datasets/slurm_upload.py
@@ -0,0 +1,287 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import argparse
+import logging
+import os
+from pathlib import Path
+
+from datatrove.executor import LocalPipelineExecutor
+from datatrove.executor.slurm import SlurmPipelineExecutor
+from datatrove.pipeline.base import PipelineStep
+from huggingface_hub import HfApi
+from huggingface_hub.constants import REPOCARD_NAME
+from port_droid import DROID_SHARDS
+
+from lerobot.datasets.dataset_metadata import CODEBASE_VERSION, LeRobotDatasetMetadata
+from lerobot.datasets.utils import create_lerobot_dataset_card
+from lerobot.utils.utils import init_logging
+
+
+class UploadDataset(PipelineStep):
+    def __init__(
+        self,
+        repo_id: str,
+        branch: str | None = None,
+        revision: str | None = None,
+        tags: list | None = None,
+        license: str | None = "apache-2.0",
+        private: bool = False,
+        distant_repo_id: str | None = None,
+        **card_kwargs,
+    ):
+        super().__init__()
+        self.repo_id = repo_id
+        self.distant_repo_id = self.repo_id if distant_repo_id is None else distant_repo_id
+        self.branch = branch
+        self.tags = tags
+        self.license = license
+        self.private = private
+        self.card_kwargs = card_kwargs
+        self.revision = revision if revision else CODEBASE_VERSION
+
+        if os.environ.get("HF_HUB_ENABLE_HF_TRANSFER", "0") != "1":
+            logging.warning(
+                'HF_HUB_ENABLE_HF_TRANSFER is not set to "1". Install hf_transfer and set the env '
+                "variable for faster uploads:\npip install hf-transfer\nexport HF_HUB_ENABLE_HF_TRANSFER=1"
+            )
+
+        self.create_repo()
+
+    def create_repo(self):
+        logging.info(f"Loading meta data from {self.repo_id}...")
+        meta = LeRobotDatasetMetadata(self.repo_id)
+
+        logging.info(f"Creating repo {self.distant_repo_id}...")
+        hub_api = HfApi()
+        hub_api.create_repo(
+            repo_id=self.distant_repo_id,
+            private=self.private,
+            repo_type="dataset",
+            exist_ok=True,
+        )
+        if self.branch:
+            hub_api.create_branch(
+                repo_id=self.distant_repo_id,
+                branch=self.branch,
+                revision=self.revision,
+                repo_type="dataset",
+                exist_ok=True,
+            )
+
+        if not hub_api.file_exists(
+            self.distant_repo_id, REPOCARD_NAME, repo_type="dataset", revision=self.branch
+        ):
+            card = create_lerobot_dataset_card(
+                tags=self.tags, dataset_info=meta.info, license=self.license, **self.card_kwargs
+            )
+            card.push_to_hub(repo_id=self.distant_repo_id, repo_type="dataset", revision=self.branch)
+
+            hub_api.create_tag(self.distant_repo_id, tag=CODEBASE_VERSION, repo_type="dataset")
+
+        def list_files_recursively(directory):
+            base_path = Path(directory)
+            return [str(file.relative_to(base_path)) for file in base_path.rglob("*") if file.is_file()]
+
+        logging.info(f"Listing all local files from {self.repo_id}...")
+        self.file_paths = list_files_recursively(meta.root)
+        self.file_paths = sorted(self.file_paths)
+
+    def create_chunks(self, lst, n):
+        from itertools import islice
+
+        it = iter(lst)
+        return [list(islice(it, size)) for size in [len(lst) // n + (i < len(lst) % n) for i in range(n)]]
+
+    def create_commits(self, additions):
+        import logging
+        import math
+        import random
+        import time
+
+        from huggingface_hub import create_commit
+        from huggingface_hub.utils import HfHubHTTPError
+
+        FILES_BETWEEN_COMMITS = 10  # noqa: N806
+        BASE_DELAY = 0.1  # noqa: N806
+        MAX_RETRIES = 12  # noqa: N806
+
+        # Split the files into smaller chunks for faster commit
+        # and avoiding "A commit has happened since" error
+        num_chunks = math.ceil(len(additions) / FILES_BETWEEN_COMMITS)
+        chunks = self.create_chunks(additions, num_chunks)
+
+        for chunk in chunks:
+            retries = 0
+            while True:
+                try:
+                    create_commit(
+                        self.distant_repo_id,
+                        repo_type="dataset",
+                        operations=chunk,
+                        commit_message=f"DataTrove upload ({len(chunk)} files)",
+                        revision=self.branch,
+                    )
+                    # TODO: every 100 chunks super_squach_commits()
+                    logging.info("create_commit completed!")
+                    break
+                except HfHubHTTPError as e:
+                    if "A commit has happened since" in e.server_message:
+                        if retries >= MAX_RETRIES:
+                            logging.error(f"Failed to create commit after {MAX_RETRIES=}. Giving up.")
+                            raise e
+                        logging.info("Commit creation race condition issue. Waiting...")
+                        time.sleep(BASE_DELAY * 2**retries + random.uniform(0, 2))
+                        retries += 1
+                    else:
+                        raise e
+
+    def run(self, data=None, rank: int = 0, world_size: int = 1):
+        import logging
+
+        from datasets.utils.tqdm import disable_progress_bars
+        from huggingface_hub import CommitOperationAdd, preupload_lfs_files
+
+        from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+        from lerobot.utils.utils import init_logging
+
+        init_logging()
+        disable_progress_bars()
+
+        chunks = self.create_chunks(self.file_paths, world_size)
+        file_paths = chunks[rank]
+
+        if len(file_paths) == 0:
+            raise ValueError(file_paths)
+
+        logging.info("Pre-uploading LFS files...")
+        for i, path in enumerate(file_paths):
+            logging.info(f"{i}: {path}")
+
+        meta = LeRobotDatasetMetadata(self.repo_id)
+        additions = [
+            CommitOperationAdd(path_in_repo=path, path_or_fileobj=meta.root / path) for path in file_paths
+        ]
+        preupload_lfs_files(
+            repo_id=self.distant_repo_id, repo_type="dataset", additions=additions, revision=self.branch
+        )
+
+        logging.info("Creating commits...")
+        self.create_commits(additions)
+        logging.info("Done!")
+
+
+def make_upload_executor(
+    repo_id, job_name, logs_dir, workers, partition, cpus_per_task, mem_per_cpu, private=False, slurm=True
+):
+    kwargs = {
+        "pipeline": [
+            UploadDataset(repo_id, private=private),
+        ],
+        "logging_dir": str(logs_dir / job_name),
+    }
+
+    if slurm:
+        kwargs.update(
+            {
+                "job_name": job_name,
+                "tasks": DROID_SHARDS,
+                "workers": workers,
+                "time": "08:00:00",
+                "partition": partition,
+                "cpus_per_task": cpus_per_task,
+                "sbatch_args": {"mem-per-cpu": mem_per_cpu},
+            }
+        )
+        executor = SlurmPipelineExecutor(**kwargs)
+    else:
+        kwargs.update(
+            {
+                "tasks": DROID_SHARDS,
+                "workers": 1,
+            }
+        )
+        executor = LocalPipelineExecutor(**kwargs)
+
+    return executor
+
+
+def main():
+    parser = argparse.ArgumentParser()
+
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        help="Repositery identifier on Hugging Face: a community or a user name `/` the name of the dataset, required when push-to-hub is True.",
+    )
+    parser.add_argument(
+        "--logs-dir",
+        type=Path,
+        help="Path to logs directory for `datatrove`.",
+    )
+    parser.add_argument(
+        "--job-name",
+        type=str,
+        default="upload_droid",
+        help="Job name used in slurm, and name of the directory created inside the provided logs directory.",
+    )
+    parser.add_argument(
+        "--slurm",
+        type=int,
+        default=1,
+        help="Launch over slurm. Use `--slurm 0` to launch sequentially (useful to debug).",
+    )
+    parser.add_argument(
+        "--workers",
+        type=int,
+        default=50,
+        help="Number of slurm workers. It should be less than the maximum number of shards.",
+    )
+    parser.add_argument(
+        "--partition",
+        type=str,
+        help="Slurm partition. Ideally a CPU partition. No need for GPU partition.",
+    )
+    parser.add_argument(
+        "--cpus-per-task",
+        type=int,
+        default=8,
+        help="Number of cpus that each slurm worker will use.",
+    )
+    parser.add_argument(
+        "--mem-per-cpu",
+        type=str,
+        default="1950M",
+        help="Memory per cpu that each worker will use.",
+    )
+    parser.add_argument(
+        "--private",
+        action="store_true",
+        default=False,
+        help="Whether to create a private repository.",
+    )
+
+    init_logging()
+
+    args = parser.parse_args()
+    kwargs = vars(args)
+    kwargs["slurm"] = kwargs.pop("slurm") == 1
+    upload_executor = make_upload_executor(**kwargs)
+    upload_executor.run()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/rtc/eval_dataset.py b/lerobot/examples/rtc/eval_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..a94d4da48dd1a71c0eafa542221260547c6ce90e
--- /dev/null
+++ b/lerobot/examples/rtc/eval_dataset.py
@@ -0,0 +1,952 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Evaluate Real-Time Chunking (RTC) performance on dataset samples.
+
+This script takes two random samples from a dataset:
+- Uses actions from the first sample as previous chunk
+- Generates new actions for the second sample with and without RTC
+
+It compares action predictions with and without RTC on dataset samples,
+measuring consistency and ground truth alignment.
+
+Usage:
+    # Basic usage with smolvla policy
+    uv run python examples/rtc/eval_dataset.py \
+        --policy.path=<USER>/smolvla_check_rtc_last3 \
+        --dataset.repo_id=<USER>/check_rtc \
+        --rtc.execution_horizon=8 \
+        --device=mps \
+        --rtc.max_guidance_weight=10.0 \
+        --rtc.prefix_attention_schedule=EXP \
+        --seed=10
+
+    # Basic usage with pi0.5 policy
+    uv run python examples/rtc/eval_dataset.py \
+        --policy.path=lerobot/pi05_libero_finetuned \
+        --dataset.repo_id=HuggingFaceVLA/libero \
+        --rtc.execution_horizon=10 \
+        --device=mps
+        --seed=10
+
+    # Basic usage with pi0.5 policy with cuda device
+    uv run python examples/rtc/eval_dataset.py \
+        --policy.path=lerobot/pi05_libero_finetuned \
+        --dataset.repo_id=HuggingFaceVLA/libero \
+        --rtc.execution_horizon=8 \
+        --device=cuda
+
+    # Basic usage with pi0 policy with cuda device
+    uv run python examples/rtc/eval_dataset.py \
+        --policy.path=lerobot/pi0_libero_finetuned \
+        --dataset.repo_id=HuggingFaceVLA/libero \
+        --rtc.execution_horizon=8 \
+        --device=cuda
+
+    uv run python examples/rtc/eval_dataset.py \
+        --policy.path=<USER>/reuben_pi0 \
+        --dataset.repo_id=<USER>/so101_cube_in_cup \
+        --rtc.execution_horizon=8 \
+        --device=cuda
+
+    # With torch.compile for faster inference (PyTorch 2.0+)
+    # Note: CUDA graphs disabled by default due to in-place ops in denoising loop
+    uv run python examples/rtc/eval_dataset.py \
+        --policy.path=<USER>/smolvla_check_rtc_last3 \
+        --dataset.repo_id=<USER>/check_rtc \
+        --rtc.execution_horizon=8 \
+        --device=mps \
+        --use_torch_compile=true \
+        --torch_compile_mode=max-autotune
+
+    # With torch.compile on CUDA (CUDA graphs disabled by default)
+    uv run python examples/rtc/eval_dataset.py \
+        --policy.path=<USER>/smolvla_check_rtc_last3 \
+        --dataset.repo_id=<USER>/check_rtc \
+        --rtc.execution_horizon=8 \
+        --device=cuda \
+        --use_torch_compile=true \
+        --torch_compile_mode=reduce-overhead
+
+    # Enable CUDA graphs (advanced - may cause tensor aliasing errors)
+    uv run python examples/rtc/eval_dataset.py \
+        --policy.path=<USER>/smolvla_check_rtc_last3 \
+        --dataset.repo_id=<USER>/check_rtc \
+        --use_torch_compile=true \
+        --torch_compile_backend=inductor \
+        --torch_compile_mode=max-autotune \
+        --torch_compile_disable_cudagraphs=false
+"""
+
+import gc
+import logging
+import os
+import random
+from dataclasses import dataclass, field
+
+import numpy as np
+import torch
+
+try:
+    import matplotlib.pyplot as plt
+
+    MATPLOTLIB_AVAILABLE = True
+except ImportError:
+    MATPLOTLIB_AVAILABLE = False
+    plt = None
+
+from lerobot.configs import parser
+from lerobot.configs.default import DatasetConfig
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import RTCAttentionSchedule
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.factory import resolve_delta_timestamps
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.factory import get_policy_class, make_pre_post_processors
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+from lerobot.policies.rtc.debug_visualizer import RTCDebugVisualizer
+from lerobot.utils.hub import HubMixin
+from lerobot.utils.utils import init_logging
+
+
+def set_seed(seed: int):
+    """Set random seed for reproducibility."""
+    random.seed(seed)
+    np.random.seed(seed)
+    torch.manual_seed(seed)
+    if torch.cuda.is_available():
+        torch.cuda.manual_seed(seed)
+        torch.cuda.manual_seed_all(seed)
+    if torch.backends.mps.is_available():
+        torch.mps.manual_seed(seed)
+    torch.backends.cudnn.deterministic = True
+    torch.backends.cudnn.benchmark = False
+
+
+def _check_matplotlib_available():
+    """Check if matplotlib is available, raise helpful error if not."""
+    if not MATPLOTLIB_AVAILABLE:
+        raise ImportError(
+            "matplotlib is required for RTC debug visualizations. "
+            "Please install it by running:\n"
+            "  uv pip install matplotlib"
+        )
+
+
+@dataclass
+class RTCEvalConfig(HubMixin):
+    """Configuration for RTC evaluation."""
+
+    # Policy configuration
+    policy: PreTrainedConfig | None = None
+
+    # Dataset configuration
+    dataset: DatasetConfig = field(default_factory=DatasetConfig)
+
+    # RTC configuration
+    rtc: RTCConfig = field(
+        default_factory=lambda: RTCConfig(
+            enabled=True,
+            execution_horizon=20,
+            max_guidance_weight=10.0,
+            prefix_attention_schedule=RTCAttentionSchedule.EXP,
+            debug=True,
+            debug_maxlen=1000,
+        )
+    )
+
+    # Device configuration
+    device: str | None = field(
+        default=None,
+        metadata={"help": "Device to run on (cuda, cpu, mps, auto)"},
+    )
+
+    # Output configuration
+    output_dir: str = field(
+        default="rtc_debug_output",
+        metadata={"help": "Directory to save debug visualizations"},
+    )
+
+    # Seed configuration
+    seed: int = field(
+        default=42,
+        metadata={"help": "Random seed for reproducibility"},
+    )
+
+    inference_delay: int = field(
+        default=4,
+        metadata={"help": "Inference delay for RTC"},
+    )
+
+    # Torch compile configuration
+    use_torch_compile: bool = field(
+        default=False,
+        metadata={"help": "Use torch.compile for faster inference (PyTorch 2.0+)"},
+    )
+
+    torch_compile_backend: str = field(
+        default="inductor",
+        metadata={"help": "Backend for torch.compile (inductor, aot_eager, cudagraphs)"},
+    )
+
+    torch_compile_mode: str = field(
+        default="default",
+        metadata={"help": "Compilation mode (default, reduce-overhead, max-autotune)"},
+    )
+
+    torch_compile_disable_cudagraphs: bool = field(
+        default=True,
+        metadata={
+            "help": "Disable CUDA graphs in torch.compile. Required due to in-place tensor "
+            "operations in denoising loop (x_t += dt * v_t) which cause tensor aliasing issues."
+        },
+    )
+
+    def __post_init__(self):
+        # Parse policy path
+        policy_path = parser.get_path_arg("policy")
+        if policy_path:
+            cli_overrides = parser.get_cli_overrides("policy")
+            self.policy = PreTrainedConfig.from_pretrained(policy_path, cli_overrides=cli_overrides)
+            self.policy.pretrained_path = policy_path
+        else:
+            raise ValueError("Policy path is required (--policy.path)")
+
+        # Auto-detect device if not specified
+        if self.device is None or self.device == "auto":
+            if torch.cuda.is_available():
+                self.device = "cuda"
+            elif torch.backends.mps.is_available():
+                self.device = "mps"
+            else:
+                self.device = "cpu"
+            logging.info(f"Auto-detected device: {self.device}")
+
+    @classmethod
+    def __get_path_fields__(cls) -> list[str]:
+        """This enables the parser to load config from the policy using `--policy.path=local/dir`"""
+        return ["policy"]
+
+
+class RTCEvaluator:
+    """Evaluator for RTC on dataset samples."""
+
+    def __init__(self, cfg: RTCEvalConfig):
+        self.cfg = cfg
+        self.device = cfg.device
+
+        # Load dataset with proper delta_timestamps based on policy configuration
+        # Calculate delta_timestamps using the same logic as make_dataset factory
+        logging.info(f"Loading dataset: {cfg.dataset.repo_id}")
+
+        # Get dataset metadata to extract FPS
+        ds_meta = LeRobotDatasetMetadata(cfg.dataset.repo_id)
+
+        # Calculate delta_timestamps from policy's delta_indices
+        delta_timestamps = resolve_delta_timestamps(cfg.policy, ds_meta)
+
+        # Create dataset with calculated delta_timestamps
+        self.dataset = LeRobotDataset(
+            cfg.dataset.repo_id,
+            delta_timestamps=delta_timestamps,
+        )
+        logging.info(f"Dataset loaded: {len(self.dataset)} samples, {self.dataset.num_episodes} episodes")
+
+        # Create preprocessor/postprocessor
+        self.preprocessor, self.postprocessor = make_pre_post_processors(
+            policy_cfg=cfg.policy,
+            pretrained_path=cfg.policy.pretrained_path,
+            preprocessor_overrides={
+                "device_processor": {"device": self.device},
+            },
+        )
+
+        logging.info("=" * 80)
+        logging.info("Ready to run evaluation with sequential policy loading:")
+        logging.info("  1. policy_prev_chunk - Generate reference chunk, then destroy")
+        logging.info("  2. policy_no_rtc - Generate without RTC, then destroy")
+        logging.info("  3. policy_rtc - Generate with RTC, then destroy")
+        logging.info("  Note: Only one policy in memory at a time for efficient memory usage")
+        logging.info("=" * 80)
+
+    def _init_policy(self, name: str, rtc_enabled: bool, rtc_debug: bool):
+        """Initialize a single policy instance with specified RTC configuration.
+
+        Args:
+            name: Name identifier for logging purposes
+            rtc_enabled: Whether to enable RTC for this policy
+            rtc_debug: Whether to enable debug tracking for this policy
+
+        Returns:
+            Configured policy instance with optional torch.compile applied
+        """
+        logging.info(f"Initializing {name}...")
+
+        # Load policy from pretrained
+        policy_class = get_policy_class(self.cfg.policy.type)
+
+        config = PreTrainedConfig.from_pretrained(self.cfg.policy.pretrained_path)
+
+        if self.cfg.policy.type == "pi05" or self.cfg.policy.type == "pi0":
+            config.compile_model = self.cfg.use_torch_compile
+
+        policy = policy_class.from_pretrained(self.cfg.policy.pretrained_path, config=config)
+        policy = policy.to(self.device)
+        policy.eval()
+
+        # Configure RTC
+        rtc_config = RTCConfig(
+            enabled=rtc_enabled,
+            execution_horizon=self.cfg.rtc.execution_horizon,
+            max_guidance_weight=self.cfg.rtc.max_guidance_weight,
+            prefix_attention_schedule=self.cfg.rtc.prefix_attention_schedule,
+            debug=rtc_debug,
+            debug_maxlen=self.cfg.rtc.debug_maxlen,
+        )
+        policy.config.rtc_config = rtc_config
+        policy.init_rtc_processor()
+
+        logging.info(f"  RTC enabled: {rtc_enabled}")
+        logging.info(f"  RTC debug: {rtc_debug}")
+        logging.info(f"  Policy config: {config}")
+
+        # Apply torch.compile to predict_action_chunk method if enabled
+        if self.cfg.use_torch_compile:
+            policy = self._apply_torch_compile(policy, name)
+
+        logging.info(f"✓ {name} initialized successfully")
+        return policy
+
+    def _apply_torch_compile(self, policy, policy_name: str):
+        """Apply torch.compile to the policy's predict_action_chunk method.
+
+        Args:
+            policy: Policy instance to compile
+            policy_name: Name for logging purposes
+
+        Returns:
+            Policy with compiled predict_action_chunk method
+        """
+
+        # PI models handle their own compilation
+        if policy.type == "pi05" or policy.type == "pi0":
+            return policy
+
+        try:
+            # Check if torch.compile is available (PyTorch 2.0+)
+            if not hasattr(torch, "compile"):
+                logging.warning(
+                    f"  [{policy_name}] torch.compile is not available. Requires PyTorch 2.0+. "
+                    f"Current version: {torch.__version__}. Skipping compilation."
+                )
+                return policy
+
+            logging.info(f"  [{policy_name}] Applying torch.compile to predict_action_chunk...")
+            logging.info(f"    Backend: {self.cfg.torch_compile_backend}")
+            logging.info(f"    Mode: {self.cfg.torch_compile_mode}")
+            logging.info(f"    Disable CUDA graphs: {self.cfg.torch_compile_disable_cudagraphs}")
+            logging.info("    Note: Debug tracker excluded from compilation via @torch._dynamo.disable")
+
+            # Compile the predict_action_chunk method
+            # - Debug tracker is excluded from compilation via @torch._dynamo.disable
+            # - CUDA graphs disabled to prevent tensor aliasing from in-place ops (x_t += dt * v_t)
+            compile_kwargs = {
+                "backend": self.cfg.torch_compile_backend,
+                "mode": self.cfg.torch_compile_mode,
+            }
+
+            # Disable CUDA graphs if requested (prevents tensor aliasing issues)
+            if self.cfg.torch_compile_disable_cudagraphs:
+                compile_kwargs["options"] = {"triton.cudagraphs": False}
+
+            original_method = policy.predict_action_chunk
+            compiled_method = torch.compile(original_method, **compile_kwargs)
+            policy.predict_action_chunk = compiled_method
+            logging.info(f"  ✓ [{policy_name}] Successfully compiled predict_action_chunk")
+
+        except Exception as e:
+            logging.error(f"  [{policy_name}] Failed to apply torch.compile: {e}")
+            logging.warning(f"  [{policy_name}] Continuing without torch.compile")
+
+        return policy
+
+    def _destroy_policy(self, policy, policy_name: str):
+        """Explicitly destroy a policy and free all associated memory.
+
+        This method performs aggressive cleanup to ensure maximum memory is freed,
+        which is critical for large models (e.g., VLAs with billions of parameters).
+
+        Args:
+            policy: Policy instance to destroy
+            policy_name: Name for logging purposes
+        """
+        logging.info(f"  Destroying {policy_name} and freeing memory...")
+
+        try:
+            # Step 1: Move policy to CPU to free GPU/MPS memory
+            policy.cpu()
+
+            # Step 2: Delete the policy object
+            del policy
+
+            # Step 3: Force garbage collection to reclaim memory immediately
+            gc.collect()
+
+            # Step 4: Clear device-specific caches
+            if torch.cuda.is_available():
+                torch.cuda.empty_cache()
+                torch.cuda.synchronize()  # Ensure all operations complete
+
+            if torch.backends.mps.is_available():
+                torch.mps.empty_cache()
+
+            logging.info(f"  ✓ {policy_name} destroyed and memory freed")
+
+        except Exception as e:
+            logging.warning(f"  Warning: Error during {policy_name} cleanup: {e}")
+
+    def run_evaluation(self):
+        """Run evaluation on two random dataset samples using three separate policies.
+
+        Note: Policies are deinitalized after each step to free memory. Large models
+        (e.g., VLA models with billions of parameters) cannot fit three instances in
+        memory simultaneously. By deleting and garbage collecting after each step,
+        we ensure only one policy is loaded at a time.
+        """
+        # Create output directory
+        os.makedirs(self.cfg.output_dir, exist_ok=True)
+        logging.info(f"Output directory: {self.cfg.output_dir}")
+
+        logging.info("=" * 80)
+        logging.info("Starting RTC evaluation")
+        logging.info(f"Inference delay: {self.cfg.inference_delay}")
+        logging.info("=" * 80)
+
+        # Load two random samples from dataset
+        data_loader = torch.utils.data.DataLoader(self.dataset, batch_size=1, shuffle=True)
+        loader_iter = iter(data_loader)
+        first_sample = next(loader_iter)
+        second_sample = next(loader_iter)
+
+        preprocessed_first_sample = self.preprocessor(first_sample)
+        preprocessed_second_sample = self.preprocessor(second_sample)
+
+        # ============================================================================
+        # Step 1: Generate previous chunk using policy_prev_chunk
+        # ============================================================================
+        # This policy is only used to generate the reference chunk and then freed
+        logging.info("=" * 80)
+        logging.info("Step 1: Generating previous chunk with policy_prev_chunk")
+        logging.info("=" * 80)
+
+        # Initialize policy 1
+        policy_prev_chunk_policy = self._init_policy(
+            name="policy_prev_chunk",
+            rtc_enabled=False,
+            rtc_debug=False,
+        )
+        with torch.no_grad():
+            prev_chunk_left_over = policy_prev_chunk_policy.predict_action_chunk(
+                preprocessed_first_sample,
+            )[:, :25, :].squeeze(0)
+        logging.info(f"  Generated prev_chunk shape: {prev_chunk_left_over.shape}")
+
+        # Destroy policy_prev_chunk to free memory for large models
+        self._destroy_policy(policy_prev_chunk_policy, "policy_prev_chunk")
+
+        # ============================================================================
+        # Step 2: Generate actions WITHOUT RTC using policy_no_rtc
+        # ============================================================================
+        logging.info("=" * 80)
+        logging.info("Step 2: Generating actions WITHOUT RTC with policy_no_rtc")
+        logging.info("=" * 80)
+
+        set_seed(self.cfg.seed)
+
+        # Initialize policy 2
+        policy_no_rtc_policy = self._init_policy(
+            name="policy_no_rtc",
+            rtc_enabled=False,
+            rtc_debug=True,
+        )
+
+        # Sample noise (use same noise for both RTC and non-RTC for fair comparison)
+        noise_size = (1, policy_no_rtc_policy.config.chunk_size, policy_no_rtc_policy.config.max_action_dim)
+        noise = policy_no_rtc_policy.model.sample_noise(noise_size, self.device)
+        noise_clone = noise.clone()
+        policy_no_rtc_policy.rtc_processor.reset_tracker()
+        with torch.no_grad():
+            no_rtc_actions = policy_no_rtc_policy.predict_action_chunk(
+                preprocessed_second_sample,
+                noise=noise,
+            )
+        no_rtc_tracked_steps = policy_no_rtc_policy.rtc_processor.tracker.get_all_steps()
+        logging.info(f"  Tracked {len(no_rtc_tracked_steps)} steps without RTC")
+        logging.info(f"  Generated no_rtc_actions shape: {no_rtc_actions.shape}")
+
+        # Destroy policy_no_rtc to free memory before loading policy_rtc
+        self._destroy_policy(policy_no_rtc_policy, "policy_no_rtc")
+
+        # ============================================================================
+        # Step 3: Generate actions WITH RTC using policy_rtc
+        # ============================================================================
+        logging.info("=" * 80)
+        logging.info("Step 3: Generating actions WITH RTC with policy_rtc")
+        logging.info("=" * 80)
+
+        set_seed(self.cfg.seed)
+
+        # Initialize policy 3
+        policy_rtc_policy = self._init_policy(
+            name="policy_rtc",
+            rtc_enabled=True,
+            rtc_debug=True,
+        )
+        policy_rtc_policy.rtc_processor.reset_tracker()
+        with torch.no_grad():
+            rtc_actions = policy_rtc_policy.predict_action_chunk(
+                preprocessed_second_sample,
+                noise=noise_clone,
+                inference_delay=self.cfg.inference_delay,
+                prev_chunk_left_over=prev_chunk_left_over,
+                execution_horizon=self.cfg.rtc.execution_horizon,
+            )
+        rtc_tracked_steps = policy_rtc_policy.rtc_processor.get_all_debug_steps()
+        logging.info(f"  Tracked {len(rtc_tracked_steps)} steps with RTC")
+        logging.info(f"  Generated rtc_actions shape: {rtc_actions.shape}")
+
+        # Save num_steps before destroying policy (needed for plotting)
+        try:
+            num_steps = policy_rtc_policy.config.num_steps
+        except Exception as e:
+            logging.error(f"  Error getting num_steps: {e}")
+            num_steps = policy_rtc_policy.config.num_inference_steps
+            logging.warning(f"  Using num_inference_steps: {num_steps} instead of num_steps")
+
+        # Destroy policy_rtc after final use
+        self._destroy_policy(policy_rtc_policy, "policy_rtc")
+
+        # Plot and save results
+        logging.info("=" * 80)
+        logging.info("Plotting results...")
+        self.plot_tracked_data(rtc_tracked_steps, no_rtc_tracked_steps, prev_chunk_left_over, num_steps)
+
+        # Plot final actions comparison
+        logging.info("=" * 80)
+        logging.info("Plotting final actions comparison...")
+        self.plot_final_actions_comparison(rtc_actions, no_rtc_actions, prev_chunk_left_over)
+
+        logging.info("=" * 80)
+        logging.info("Evaluation completed successfully")
+
+    def plot_final_actions_comparison(self, rtc_actions, no_rtc_actions, prev_chunk_left_over):
+        """Plot final action predictions comparison on a single chart.
+
+        Args:
+            rtc_actions: Final actions from RTC policy
+            no_rtc_actions: Final actions from non-RTC policy
+            prev_chunk_left_over: Previous chunk used as ground truth
+        """
+        _check_matplotlib_available()
+
+        # Remove batch dimension if present
+        rtc_actions_plot = rtc_actions.squeeze(0).cpu() if len(rtc_actions.shape) == 3 else rtc_actions.cpu()
+        no_rtc_actions_plot = (
+            no_rtc_actions.squeeze(0).cpu() if len(no_rtc_actions.shape) == 3 else no_rtc_actions.cpu()
+        )
+        prev_chunk_plot = prev_chunk_left_over.cpu()
+
+        # Create figure with 6 subplots (one per action dimension)
+        fig, axes = plt.subplots(6, 1, figsize=(16, 12))
+        fig.suptitle("Final Action Predictions Comparison (Raw)", fontsize=16)
+
+        # Plot each action dimension
+        for dim_idx, ax in enumerate(axes):
+            # Plot previous chunk (ground truth) in red
+            RTCDebugVisualizer.plot_waypoints(
+                [ax],
+                prev_chunk_plot[:, dim_idx : dim_idx + 1],
+                start_from=0,
+                color="red",
+                label="Previous Chunk (Ground Truth)",
+                linewidth=2.5,
+                alpha=0.8,
+            )
+
+            # Plot no-RTC actions in blue
+            RTCDebugVisualizer.plot_waypoints(
+                [ax],
+                no_rtc_actions_plot[:, dim_idx : dim_idx + 1],
+                start_from=0,
+                color="blue",
+                label="No RTC",
+                linewidth=2,
+                alpha=0.7,
+            )
+
+            # Plot RTC actions in green
+            RTCDebugVisualizer.plot_waypoints(
+                [ax],
+                rtc_actions_plot[:, dim_idx : dim_idx + 1],
+                start_from=0,
+                color="green",
+                label="RTC",
+                linewidth=2,
+                alpha=0.7,
+            )
+
+            # Add vertical lines for inference delay and execution horizon
+            inference_delay = self.cfg.inference_delay
+            execution_horizon = self.cfg.rtc.execution_horizon
+
+            if inference_delay > 0:
+                ax.axvline(
+                    x=inference_delay - 1,
+                    color="orange",
+                    linestyle="--",
+                    alpha=0.5,
+                    label=f"Inference Delay ({inference_delay})",
+                )
+
+            if execution_horizon > 0:
+                ax.axvline(
+                    x=execution_horizon,
+                    color="purple",
+                    linestyle="--",
+                    alpha=0.5,
+                    label=f"Execution Horizon ({execution_horizon})",
+                )
+
+            ax.set_ylabel(f"Dim {dim_idx}", fontsize=10)
+            ax.grid(True, alpha=0.3)
+
+            # Set x-axis ticks to show all integer values
+            max_len = max(rtc_actions_plot.shape[0], no_rtc_actions_plot.shape[0], prev_chunk_plot.shape[0])
+            ax.set_xticks(range(0, max_len, max(1, max_len // 20)))  # Show ~20 ticks
+            ax.set_xlim(-0.5, max_len - 0.5)
+
+        axes[-1].set_xlabel("Step", fontsize=10)
+
+        # Collect legend handles and labels from first subplot
+        handles, labels = axes[0].get_legend_handles_labels()
+        # Remove duplicates while preserving order
+        seen = set()
+        unique_handles = []
+        unique_labels = []
+        for handle, label in zip(handles, labels, strict=True):
+            if label not in seen:
+                seen.add(label)
+                unique_handles.append(handle)
+                unique_labels.append(label)
+
+        # Add legend outside the plot area (to the right)
+        fig.legend(
+            unique_handles,
+            unique_labels,
+            loc="center right",
+            fontsize=9,
+            bbox_to_anchor=(1.0, 0.5),
+            framealpha=0.9,
+        )
+
+        # Save figure
+        output_path = os.path.join(self.cfg.output_dir, "final_actions_comparison.png")
+        fig.tight_layout(rect=[0, 0, 0.85, 1])  # Leave space for legend on right
+        fig.savefig(output_path, dpi=150, bbox_inches="tight")
+        logging.info(f"Saved final actions comparison to {output_path}")
+        plt.close(fig)
+
+    def plot_tracked_data(self, rtc_tracked_steps, no_rtc_tracked_steps, prev_chunk_left_over, num_steps):
+        _check_matplotlib_available()
+
+        # Create side-by-side figures for denoising visualization
+        fig_xt, axs_xt = self._create_figure("x_t Denoising: No RTC (left) vs RTC (right)")
+        fig_vt, axs_vt = self._create_figure("v_t Denoising: No RTC (left) vs RTC (right)")
+        fig_corr, axs_corr = self._create_figure("Correction: No RTC (left) vs RTC (right)")
+        fig_x1t, axs_x1t = self._create_figure(
+            "x1_t Predicted State & Error: No RTC (left - empty) vs RTC (right)"
+        )
+        self._plot_denoising_steps_from_tracker(
+            rtc_tracked_steps,
+            axs_xt[:, 1],  # Right column for x_t
+            axs_vt[:, 1],  # Right column for v_t
+            axs_corr[:, 1],  # Right column for correction
+            axs_x1t[:, 1],  # Right column for x1_t
+            num_steps,
+            add_labels=True,  # Add labels for RTC (right column)
+        )
+
+        self._plot_denoising_steps_from_tracker(
+            no_rtc_tracked_steps,
+            axs_xt[:, 0],  # Left column for x_t
+            axs_vt[:, 0],  # Left column for v_t
+            axs_corr[:, 0],  # Left column for correction
+            axs_x1t[:, 0],  # Left column for x1_t
+            num_steps,
+            add_labels=False,  # No labels for No RTC (left column)
+        )
+
+        # Plot no-RTC x_t data on right chart as orange dashed line for comparison
+        self._plot_no_rtc_xt_reference(no_rtc_tracked_steps, axs_xt[:, 1], num_steps)
+
+        # Plot ground truth on x_t axes
+        RTCDebugVisualizer.plot_waypoints(
+            axs_xt[:, 1], prev_chunk_left_over, start_from=0, color="red", label="Ground truth"
+        )
+
+        # Plot ground truth on x1_t axes
+        RTCDebugVisualizer.plot_waypoints(
+            axs_x1t[:, 1], prev_chunk_left_over, start_from=0, color="red", label="Ground truth"
+        )
+
+        # Plot ground truth on x_t axes (no labels for left column)
+        RTCDebugVisualizer.plot_waypoints(
+            axs_xt[:, 0], prev_chunk_left_over, start_from=0, color="red", label=None
+        )
+
+        RTCDebugVisualizer.plot_waypoints(
+            axs_x1t[:, 0], prev_chunk_left_over, start_from=0, color="red", label=None
+        )
+
+        # Add legends outside the plot area for each figure
+        self._add_figure_legend(fig_xt, axs_xt)
+        self._add_figure_legend(fig_vt, axs_vt)
+        self._add_figure_legend(fig_corr, axs_corr)
+        self._add_figure_legend(fig_x1t, axs_x1t)
+
+        # Save denoising plots
+        self._save_figure(fig_xt, os.path.join(self.cfg.output_dir, "denoising_xt_comparison.png"))
+        self._save_figure(fig_vt, os.path.join(self.cfg.output_dir, "denoising_vt_comparison.png"))
+        self._save_figure(fig_corr, os.path.join(self.cfg.output_dir, "denoising_correction_comparison.png"))
+        self._save_figure(fig_x1t, os.path.join(self.cfg.output_dir, "denoising_x1t_comparison.png"))
+
+    def _create_figure(self, title):
+        fig, axs = plt.subplots(6, 2, figsize=(24, 12))
+        fig.suptitle(title, fontsize=16)
+
+        for ax in axs[:, 0]:
+            ax.set_title("No RTC (N/A)" if ax == axs[0, 0] else "", fontsize=12)
+        for ax in axs[:, 1]:
+            ax.set_title("RTC" if ax == axs[0, 1] else "", fontsize=12)
+
+        return fig, axs
+
+    def _add_figure_legend(self, fig, axs):
+        """Add a legend outside the plot area on the right side.
+
+        Args:
+            fig: Matplotlib figure to add legend to
+            axs: Array of axes to collect legend handles from
+        """
+        # Collect all handles and labels from the first row of axes (right column)
+        handles, labels = axs[0, 1].get_legend_handles_labels()
+
+        # Remove duplicates while preserving order
+        seen = set()
+        unique_handles = []
+        unique_labels = []
+        for handle, label in zip(handles, labels, strict=True):
+            if label not in seen:
+                seen.add(label)
+                unique_handles.append(handle)
+                unique_labels.append(label)
+
+        # Add legend outside the plot area (to the right, close to charts)
+        if unique_handles:
+            fig.legend(
+                unique_handles,
+                unique_labels,
+                loc="center left",
+                fontsize=8,
+                bbox_to_anchor=(0.87, 0.5),
+                framealpha=0.9,
+                ncol=1,
+            )
+
+    def _save_figure(self, fig, path):
+        fig.tight_layout(rect=[0, 0, 0.85, 1])  # Leave space for legend/colorbar on right
+        fig.savefig(path, dpi=150, bbox_inches="tight")
+        logging.info(f"Saved figure to {path}")
+        plt.close(fig)
+
+    def _plot_denoising_steps_from_tracker(
+        self, tracked_steps, xt_axs, vt_axs, corr_axs, x1t_axs, num_steps, add_labels=True
+    ):
+        """Plot denoising steps from tracker data.
+
+        Args:
+            tracked_steps: List of DebugStep objects containing debug steps
+            xt_axs: Matplotlib axes for x_t plots (array of 6 axes)
+            vt_axs: Matplotlib axes for v_t plots (array of 6 axes)
+            corr_axs: Matplotlib axes for correction plots (array of 6 axes)
+            x1t_axs: Matplotlib axes for x1_t plots (array of 6 axes)
+            num_steps: Total number of denoising steps for colormap
+            add_labels: Whether to add legend labels for the plots
+        """
+
+        logging.info("=" * 80)
+        logging.info(f"Plotting {len(tracked_steps)} steps")
+
+        debug_steps = tracked_steps
+        if not debug_steps:
+            return
+
+        # Define colors for different denoise steps (using a colormap)
+        colors = plt.cm.viridis(np.linspace(0, 1, num_steps))
+
+        for step_idx, debug_step in enumerate(debug_steps):
+            color = colors[step_idx % len(colors)]
+            label = f"Step {step_idx}" if add_labels else None
+
+            # Plot x_t
+            if debug_step.x_t is not None:
+                RTCDebugVisualizer.plot_waypoints(
+                    xt_axs, debug_step.x_t, start_from=0, color=color, label=label
+                )
+
+            # Plot v_t
+            if debug_step.v_t is not None:
+                RTCDebugVisualizer.plot_waypoints(
+                    vt_axs, debug_step.v_t, start_from=0, color=color, label=label
+                )
+
+            # Plot correction on separate axes
+            if debug_step.correction is not None:
+                RTCDebugVisualizer.plot_waypoints(
+                    corr_axs,
+                    debug_step.correction,
+                    start_from=0,
+                    color=color,
+                    label=label,
+                )
+
+            # Plot x1_t (predicted state)
+            if x1t_axs is not None and debug_step.x1_t is not None:
+                x1t_label = f"x1_t Step {step_idx}" if add_labels else None
+                RTCDebugVisualizer.plot_waypoints(
+                    x1t_axs,
+                    debug_step.x1_t,
+                    start_from=0,
+                    color=color,
+                    label=x1t_label,
+                )
+
+            # Plot error in orange dashed
+            if x1t_axs is not None and debug_step.err is not None:
+                error_chunk = (
+                    debug_step.err[0].cpu().numpy()
+                    if len(debug_step.err.shape) == 3
+                    else debug_step.err.cpu().numpy()
+                )
+
+                num_dims = min(error_chunk.shape[-1], 6)
+                error_label = f"error Step {step_idx}" if add_labels else None
+                for j in range(num_dims):
+                    x1t_axs[j].plot(
+                        np.arange(0, error_chunk.shape[0]),
+                        error_chunk[:, j],
+                        color="orange",
+                        linestyle="--",
+                        alpha=0.7,
+                        label=error_label,
+                    )
+
+        # Recalculate axis limits after plotting to ensure proper scaling
+        self._rescale_axes(xt_axs)
+        self._rescale_axes(vt_axs)
+        self._rescale_axes(corr_axs)
+        self._rescale_axes(x1t_axs)
+
+    def _plot_no_rtc_xt_reference(self, no_rtc_tracked_steps, xt_axs, num_steps):
+        """Plot final no-RTC x_t data as orange dashed line on the RTC chart for comparison.
+
+        Args:
+            no_rtc_tracked_steps: List of DebugStep objects containing no-RTC debug steps
+            xt_axs: Matplotlib axes for x_t plots (array of 6 axes, right column)
+            num_steps: Total number of denoising steps for colormap
+        """
+        debug_steps = no_rtc_tracked_steps
+        if not debug_steps:
+            return
+
+        # Plot only the final x_t step as orange dashed line
+        final_step = debug_steps[-1]
+        logging.info("Plotting final no-RTC x_t step as orange dashed reference")
+
+        if final_step.x_t is not None:
+            x_t_chunk = (
+                final_step.x_t[0].cpu().numpy()
+                if len(final_step.x_t.shape) == 3
+                else final_step.x_t.cpu().numpy()
+            )
+
+            num_dims = min(x_t_chunk.shape[-1], 6)
+            for j in range(num_dims):
+                xt_axs[j].plot(
+                    np.arange(0, x_t_chunk.shape[0]),
+                    x_t_chunk[:, j],
+                    color="orange",
+                    linestyle="--",
+                    alpha=0.7,
+                    linewidth=2,
+                    label="No RTC (final)" if j == 0 else "",
+                )
+
+    def _rescale_axes(self, axes):
+        """Rescale axes to show all data with proper margins.
+
+        Args:
+            axes: Array of matplotlib axes to rescale
+        """
+        for ax in axes:
+            ax.relim()
+            ax.autoscale_view()
+
+            # Add 10% margin to y-axis for better visualization
+            ylim = ax.get_ylim()
+            y_range = ylim[1] - ylim[0]
+            if y_range > 0:  # Avoid division by zero
+                margin = y_range * 0.1
+                ax.set_ylim(ylim[0] - margin, ylim[1] + margin)
+
+            # Set x-axis ticks to show all integer values
+            xlim = ax.get_xlim()
+            max_len = int(xlim[1]) + 1
+            if max_len > 0:
+                ax.set_xticks(range(0, max_len, max(1, max_len // 20)))  # Show ~20 ticks
+                ax.set_xlim(-0.5, max_len - 0.5)
+
+
+@parser.wrap()
+def main(cfg: RTCEvalConfig):
+    """Main entry point for RTC evaluation."""
+    # Set random seed for reproducibility
+    set_seed(cfg.seed)
+
+    init_logging()
+
+    logging.info("=" * 80)
+    logging.info("RTC Dataset Evaluation")
+    logging.info(f"Config: {cfg}")
+    logging.info("=" * 80)
+
+    evaluator = RTCEvaluator(cfg)
+    evaluator.run_evaluation()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/rtc/eval_with_real_robot.py b/lerobot/examples/rtc/eval_with_real_robot.py
new file mode 100644
index 0000000000000000000000000000000000000000..36da88e1b1aaf0f4a7b42c7e577a8849b3f409d5
--- /dev/null
+++ b/lerobot/examples/rtc/eval_with_real_robot.py
@@ -0,0 +1,562 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Demo script showing how to use Real-Time Chunking (RTC) with action chunking policies on real robots.
+
+This script demonstrates:
+1. Creating a robot and policy (SmolVLA, Pi0, etc.) with RTC
+2. Consuming actions from the policy while the robot executes
+3. Periodically requesting new action chunks in the background using threads
+4. Managing action buffers and timing for real-time operation
+
+For simulation environments, see eval_with_simulation.py
+
+Usage:
+    # Run RTC with Real robot with RTC
+    uv run examples/rtc/eval_with_real_robot.py \
+        --policy.path=<USER>/smolvla_check_rtc_last3 \
+        --policy.device=mps \
+        --rtc.enabled=true \
+        --rtc.execution_horizon=20 \
+        --robot.type=so100_follower \
+        --robot.port=/dev/tty.usbmodem58FA0834591 \
+        --robot.id=so100_follower \
+        --robot.cameras="{ gripper: {type: opencv, index_or_path: 1, width: 640, height: 480, fps: 30}, front: {type: opencv, index_or_path: 0, width: 640, height: 480, fps: 30}}" \
+        --task="Move green small object into the purple platform" \
+        --duration=120
+
+    # Run RTC with Real robot without RTC
+    uv run examples/rtc/eval_with_real_robot.py \
+        --policy.path=<USER>/smolvla_check_rtc_last3 \
+        --policy.device=mps \
+        --rtc.enabled=false \
+        --robot.type=so100_follower \
+        --robot.port=/dev/tty.usbmodem58FA0834591 \
+        --robot.id=so100_follower \
+        --robot.cameras="{ gripper: {type: opencv, index_or_path: 1, width: 640, height: 480, fps: 30}, front: {type: opencv, index_or_path: 0, width: 640, height: 480, fps: 30}}" \
+        --task="Move green small object into the purple platform" \
+        --duration=120
+
+    # Run RTC with Real robot with pi0.5 policy
+    uv run examples/rtc/eval_with_real_robot.py \
+        --policy.path=<USER>/pi05_check_rtc \
+        --policy.device=mps \
+        --rtc.enabled=true \
+        --rtc.execution_horizon=20 \
+        --robot.type=so100_follower \
+        --robot.port=/dev/tty.usbmodem58FA0834591 \
+        --robot.id=so100_follower \
+        --robot.cameras="{ gripper: {type: opencv, index_or_path: 0, width: 640, height: 480, fps: 30}, front: {type: opencv, index_or_path: 1, width: 640, height: 480, fps: 30}}" \
+        --task="Move green small object into the purple platform" \
+        --duration=120
+"""
+
+import logging
+import math
+import sys
+import time
+import traceback
+from dataclasses import dataclass, field
+from threading import Event, Lock, Thread
+
+import torch
+from torch import Tensor
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig  # noqa: F401
+from lerobot.cameras.realsense.configuration_realsense import RealSenseCameraConfig  # noqa: F401
+from lerobot.cameras.zmq.configuration_zmq import ZMQCameraConfig  # noqa: F401
+from lerobot.configs import parser
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import RTCAttentionSchedule
+from lerobot.datasets.feature_utils import build_dataset_frame, hw_to_dataset_features
+from lerobot.policies.factory import get_policy_class, make_pre_post_processors
+from lerobot.policies.rtc.action_queue import ActionQueue
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+from lerobot.policies.rtc.latency_tracker import LatencyTracker
+from lerobot.processor.factory import (
+    make_default_robot_action_processor,
+    make_default_robot_observation_processor,
+)
+from lerobot.rl.process import ProcessSignalHandler
+from lerobot.robots import (  # noqa: F401
+    Robot,
+    RobotConfig,
+    bi_so_follower,
+    koch_follower,
+    so_follower,
+    unitree_g1,
+)
+from lerobot.robots.utils import make_robot_from_config
+from lerobot.utils.constants import OBS_IMAGES
+from lerobot.utils.hub import HubMixin
+from lerobot.utils.utils import init_logging
+
+logging.basicConfig(level=logging.INFO)
+logger = logging.getLogger(__name__)
+
+
+class RobotWrapper:
+    def __init__(self, robot: Robot):
+        self.robot = robot
+        self.lock = Lock()
+
+    def get_observation(self) -> dict[str, Tensor]:
+        with self.lock:
+            return self.robot.get_observation()
+
+    def send_action(self, action: Tensor):
+        with self.lock:
+            self.robot.send_action(action)
+
+    def observation_features(self) -> list[str]:
+        with self.lock:
+            return self.robot.observation_features
+
+    def action_features(self) -> list[str]:
+        with self.lock:
+            return self.robot.action_features
+
+
+@dataclass
+class RTCDemoConfig(HubMixin):
+    """Configuration for RTC demo with action chunking policies and real robots."""
+
+    # Policy configuration
+    policy: PreTrainedConfig | None = None
+
+    # Robot configuration
+    robot: RobotConfig | None = None
+
+    # RTC configuration
+    rtc: RTCConfig = field(
+        default_factory=lambda: RTCConfig(
+            execution_horizon=10,
+            max_guidance_weight=1.0,
+            prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        )
+    )
+
+    # Demo parameters
+    duration: float = 30.0  # Duration to run the demo (seconds)
+    fps: float = 10.0  # Action execution frequency (Hz)
+
+    # Compute device
+    device: str | None = None  # Device to run on (cuda, cpu, auto)
+
+    # Get new actions horizon. The amount of executed steps after which will be requested new actions.
+    # It should be higher than inference delay + execution horizon.
+    action_queue_size_to_get_new_actions: int = 30
+
+    # Task to execute
+    task: str = field(default="", metadata={"help": "Task to execute"})
+
+    # Torch compile configuration
+    use_torch_compile: bool = field(
+        default=False,
+        metadata={"help": "Use torch.compile for faster inference (PyTorch 2.0+)"},
+    )
+
+    torch_compile_backend: str = field(
+        default="inductor",
+        metadata={"help": "Backend for torch.compile (inductor, aot_eager, cudagraphs)"},
+    )
+
+    torch_compile_mode: str = field(
+        default="default",
+        metadata={"help": "Compilation mode (default, reduce-overhead, max-autotune)"},
+    )
+
+    torch_compile_disable_cudagraphs: bool = field(
+        default=True,
+        metadata={
+            "help": "Disable CUDA graphs in torch.compile. Required due to in-place tensor "
+            "operations in denoising loop (x_t += dt * v_t) which cause tensor aliasing issues."
+        },
+    )
+
+    def __post_init__(self):
+        # HACK: We parse again the cli args here to get the pretrained path if there was one.
+        policy_path = parser.get_path_arg("policy")
+        if policy_path:
+            cli_overrides = parser.get_cli_overrides("policy")
+            self.policy = PreTrainedConfig.from_pretrained(policy_path, cli_overrides=cli_overrides)
+            self.policy.pretrained_path = policy_path
+        else:
+            raise ValueError("Policy path is required")
+
+        # Validate that robot configuration is provided
+        if self.robot is None:
+            raise ValueError("Robot configuration must be provided")
+
+    @classmethod
+    def __get_path_fields__(cls) -> list[str]:
+        """This enables the parser to load config from the policy using `--policy.path=local/dir`"""
+        return ["policy"]
+
+
+def is_image_key(k: str) -> bool:
+    return k.startswith(OBS_IMAGES)
+
+
+def get_actions(
+    policy,
+    robot: RobotWrapper,
+    robot_observation_processor,
+    action_queue: ActionQueue,
+    shutdown_event: Event,
+    cfg: RTCDemoConfig,
+):
+    """Thread function to request action chunks from the policy.
+
+    Args:
+        policy: The policy instance (SmolVLA, Pi0, etc.)
+        robot: The robot instance for getting observations
+        robot_observation_processor: Processor for raw robot observations
+        action_queue: Queue to put new action chunks
+        shutdown_event: Event to signal shutdown
+        cfg: Demo configuration
+    """
+    try:
+        logger.info("[GET_ACTIONS] Starting get actions thread")
+
+        latency_tracker = LatencyTracker()  # Track latency of action chunks
+        fps = cfg.fps
+        time_per_chunk = 1.0 / fps
+
+        dataset_features = hw_to_dataset_features(robot.observation_features(), "observation")
+        policy_device = policy.config.device
+
+        # Load preprocessor and postprocessor from pretrained files
+        # The stats are embedded in the processor .safetensors files
+        logger.info(f"[GET_ACTIONS] Loading preprocessor/postprocessor from {cfg.policy.pretrained_path}")
+
+        preprocessor, postprocessor = make_pre_post_processors(
+            policy_cfg=cfg.policy,
+            pretrained_path=cfg.policy.pretrained_path,
+            dataset_stats=None,  # Will load from pretrained processor files
+            preprocessor_overrides={
+                "device_processor": {"device": cfg.policy.device},
+            },
+        )
+
+        logger.info("[GET_ACTIONS] Preprocessor/postprocessor loaded successfully with embedded stats")
+
+        get_actions_threshold = cfg.action_queue_size_to_get_new_actions
+
+        if not cfg.rtc.enabled:
+            get_actions_threshold = 0
+
+        while not shutdown_event.is_set():
+            if action_queue.qsize() <= get_actions_threshold:
+                current_time = time.perf_counter()
+                action_index_before_inference = action_queue.get_action_index()
+                prev_actions = action_queue.get_left_over()
+
+                inference_latency = latency_tracker.max()
+                inference_delay = math.ceil(inference_latency / time_per_chunk)
+
+                obs = robot.get_observation()
+
+                # Apply robot observation processor
+                obs_processed = robot_observation_processor(obs)
+
+                obs_with_policy_features = build_dataset_frame(
+                    dataset_features, obs_processed, prefix="observation"
+                )
+
+                for name in obs_with_policy_features:
+                    obs_with_policy_features[name] = torch.from_numpy(obs_with_policy_features[name])
+                    if "image" in name:
+                        obs_with_policy_features[name] = (
+                            obs_with_policy_features[name].type(torch.float32) / 255
+                        )
+                        obs_with_policy_features[name] = (
+                            obs_with_policy_features[name].permute(2, 0, 1).contiguous()
+                        )
+                    obs_with_policy_features[name] = obs_with_policy_features[name].unsqueeze(0)
+                    obs_with_policy_features[name] = obs_with_policy_features[name].to(policy_device)
+
+                obs_with_policy_features["task"] = [cfg.task]  # Task should be a list, not a string!
+                obs_with_policy_features["robot_type"] = (
+                    robot.robot.name if hasattr(robot.robot, "name") else ""
+                )
+
+                preproceseded_obs = preprocessor(obs_with_policy_features)
+
+                # Generate actions WITH RTC
+                actions = policy.predict_action_chunk(
+                    preproceseded_obs,
+                    inference_delay=inference_delay,
+                    prev_chunk_left_over=prev_actions,
+                )
+
+                # Store original actions (before postprocessing) for RTC
+                original_actions = actions.squeeze(0).clone()
+
+                postprocessed_actions = postprocessor(actions)
+
+                postprocessed_actions = postprocessed_actions.squeeze(0)
+
+                new_latency = time.perf_counter() - current_time
+                new_delay = math.ceil(new_latency / time_per_chunk)
+                latency_tracker.add(new_latency)
+
+                if cfg.action_queue_size_to_get_new_actions < cfg.rtc.execution_horizon + new_delay:
+                    logger.warning(
+                        "[GET_ACTIONS] cfg.action_queue_size_to_get_new_actions Too small, It should be higher than inference delay + execution horizon."
+                    )
+
+                action_queue.merge(
+                    original_actions, postprocessed_actions, new_delay, action_index_before_inference
+                )
+            else:
+                # Small sleep to prevent busy waiting
+                time.sleep(0.1)
+
+        logger.info("[GET_ACTIONS] get actions thread shutting down")
+    except Exception as e:
+        logger.error(f"[GET_ACTIONS] Fatal exception in get_actions thread: {e}")
+        logger.error(traceback.format_exc())
+        sys.exit(1)
+
+
+def actor_control(
+    robot: RobotWrapper,
+    robot_action_processor,
+    action_queue: ActionQueue,
+    shutdown_event: Event,
+    cfg: RTCDemoConfig,
+):
+    """Thread function to execute actions on the robot.
+
+    Args:
+        robot: The robot instance
+        action_queue: Queue to get actions from
+        shutdown_event: Event to signal shutdown
+        cfg: Demo configuration
+    """
+    try:
+        logger.info("[ACTOR] Starting actor thread")
+
+        action_count = 0
+        action_interval = 1.0 / cfg.fps
+
+        while not shutdown_event.is_set():
+            start_time = time.perf_counter()
+
+            # Try to get an action from the queue with timeout
+            action = action_queue.get()
+
+            if action is not None:
+                action = action.cpu()
+                action_dict = {key: action[i].item() for i, key in enumerate(robot.action_features())}
+                action_processed = robot_action_processor((action_dict, None))
+                robot.send_action(action_processed)
+
+                action_count += 1
+
+            dt_s = time.perf_counter() - start_time
+            time.sleep(max(0, (action_interval - dt_s) - 0.001))
+
+        logger.info(f"[ACTOR] Actor thread shutting down. Total actions executed: {action_count}")
+    except Exception as e:
+        logger.error(f"[ACTOR] Fatal exception in actor_control thread: {e}")
+        logger.error(traceback.format_exc())
+        sys.exit(1)
+
+
+def _apply_torch_compile(policy, cfg: RTCDemoConfig):
+    """Apply torch.compile to the policy's predict_action_chunk method.
+
+    Args:
+        policy: Policy instance to compile
+        cfg: Configuration containing torch compile settings
+
+    Returns:
+        Policy with compiled predict_action_chunk method
+    """
+
+    # PI models handle their own compilation
+    if policy.type == "pi05" or policy.type == "pi0":
+        return policy
+
+    try:
+        # Check if torch.compile is available (PyTorch 2.0+)
+        if not hasattr(torch, "compile"):
+            logger.warning(
+                f"torch.compile is not available. Requires PyTorch 2.0+. "
+                f"Current version: {torch.__version__}. Skipping compilation."
+            )
+            return policy
+
+        logger.info("Applying torch.compile to predict_action_chunk...")
+        logger.info(f"  Backend: {cfg.torch_compile_backend}")
+        logger.info(f"  Mode: {cfg.torch_compile_mode}")
+        logger.info(f"  Disable CUDA graphs: {cfg.torch_compile_disable_cudagraphs}")
+
+        # Compile the predict_action_chunk method
+        # - CUDA graphs disabled to prevent tensor aliasing from in-place ops (x_t += dt * v_t)
+        compile_kwargs = {
+            "backend": cfg.torch_compile_backend,
+            "mode": cfg.torch_compile_mode,
+        }
+
+        # Disable CUDA graphs if requested (prevents tensor aliasing issues)
+        if cfg.torch_compile_disable_cudagraphs:
+            compile_kwargs["options"] = {"triton.cudagraphs": False}
+
+        original_method = policy.predict_action_chunk
+        compiled_method = torch.compile(original_method, **compile_kwargs)
+        policy.predict_action_chunk = compiled_method
+        logger.info("✓ Successfully compiled predict_action_chunk")
+
+    except Exception as e:
+        logger.error(f"Failed to apply torch.compile: {e}")
+        logger.warning("Continuing without torch.compile")
+
+    return policy
+
+
+@parser.wrap()
+def demo_cli(cfg: RTCDemoConfig):
+    """Main entry point for RTC demo with draccus configuration."""
+
+    # Initialize logging
+    init_logging()
+
+    logger.info(f"Using device: {cfg.device}")
+
+    # Setup signal handler for graceful shutdown
+    signal_handler = ProcessSignalHandler(use_threads=True, display_pid=False)
+    shutdown_event = signal_handler.shutdown_event
+
+    policy = None
+    robot = None
+    get_actions_thread = None
+    actor_thread = None
+
+    policy_class = get_policy_class(cfg.policy.type)
+
+    # Load config and set compile_model for pi0/pi05 models
+    config = PreTrainedConfig.from_pretrained(cfg.policy.pretrained_path)
+
+    if cfg.policy.type == "pi05" or cfg.policy.type == "pi0":
+        config.compile_model = cfg.use_torch_compile
+
+    if config.use_peft:
+        from peft import PeftConfig, PeftModel
+
+        peft_pretrained_path = cfg.policy.pretrained_path
+        peft_config = PeftConfig.from_pretrained(peft_pretrained_path)
+
+        policy = policy_class.from_pretrained(
+            pretrained_name_or_path=peft_config.base_model_name_or_path, config=config
+        )
+        policy = PeftModel.from_pretrained(policy, peft_pretrained_path, config=peft_config)
+    else:
+        policy = policy_class.from_pretrained(cfg.policy.pretrained_path, config=config)
+
+    # Turn on RTC
+    policy.config.rtc_config = cfg.rtc
+
+    # Init RTC processort, as by default if RTC disabled in the config
+    # The processor won't be created
+    policy.init_rtc_processor()
+
+    assert policy.name in ["smolvla", "pi05", "pi0"], "Only smolvla, pi05, and pi0 are supported for RTC"
+
+    policy = policy.to(cfg.device)
+    policy.eval()
+
+    # Apply torch.compile to predict_action_chunk method if enabled
+    if cfg.use_torch_compile:
+        policy = _apply_torch_compile(policy, cfg)
+
+    # Create robot
+    logger.info(f"Initializing robot: {cfg.robot.type}")
+    robot = make_robot_from_config(cfg.robot)
+    robot.connect()
+    robot_wrapper = RobotWrapper(robot)
+
+    # Create robot observation processor
+    robot_observation_processor = make_default_robot_observation_processor()
+    robot_action_processor = make_default_robot_action_processor()
+
+    # Create action queue for communication between threads
+    action_queue = ActionQueue(cfg.rtc)
+
+    # Start chunk requester thread
+    get_actions_thread = Thread(
+        target=get_actions,
+        args=(policy, robot_wrapper, robot_observation_processor, action_queue, shutdown_event, cfg),
+        daemon=True,
+        name="GetActions",
+    )
+    get_actions_thread.start()
+    logger.info("Started get actions thread")
+
+    # Start action executor thread
+    actor_thread = Thread(
+        target=actor_control,
+        args=(robot_wrapper, robot_action_processor, action_queue, shutdown_event, cfg),
+        daemon=True,
+        name="Actor",
+    )
+    actor_thread.start()
+    logger.info("Started actor thread")
+
+    logger.info("Started stop by duration thread")
+
+    # Main thread monitors for duration or shutdown
+    logger.info(f"Running demo for {cfg.duration} seconds...")
+    start_time = time.time()
+
+    while not shutdown_event.is_set() and (time.time() - start_time) < cfg.duration:
+        time.sleep(10)
+
+        # Log queue status periodically
+        if int(time.time() - start_time) % 5 == 0:
+            logger.info(f"[MAIN] Action queue size: {action_queue.qsize()}")
+
+        if time.time() - start_time > cfg.duration:
+            break
+
+    logger.info("Demo duration reached or shutdown requested")
+
+    # Signal shutdown
+    shutdown_event.set()
+
+    # Wait for threads to finish
+    if get_actions_thread and get_actions_thread.is_alive():
+        logger.info("Waiting for chunk requester thread to finish...")
+        get_actions_thread.join()
+
+    if actor_thread and actor_thread.is_alive():
+        logger.info("Waiting for action executor thread to finish...")
+        actor_thread.join()
+
+    # Cleanup robot
+    if robot:
+        robot.disconnect()
+        logger.info("Robot disconnected")
+
+    logger.info("Cleanup completed")
+
+
+if __name__ == "__main__":
+    demo_cli()
+    logging.info("RTC demo finished")
diff --git a/lerobot/examples/so100_to_so100_EE/evaluate.py b/lerobot/examples/so100_to_so100_EE/evaluate.py
new file mode 100644
index 0000000000000000000000000000000000000000..6385910213b045ac917e08fbccb82567355bfee0
--- /dev/null
+++ b/lerobot/examples/so100_to_so100_EE/evaluate.py
@@ -0,0 +1,208 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.datasets.feature_utils import combine_feature_dicts
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.policies.act.modeling_act import ACTPolicy
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.processor import (
+    RobotProcessorPipeline,
+    make_default_teleop_action_processor,
+)
+from lerobot.processor.converters import (
+    observation_to_transition,
+    robot_action_observation_to_transition,
+    transition_to_observation,
+    transition_to_robot_action,
+)
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+from lerobot.robots.so_follower.robot_kinematic_processor import (
+    ForwardKinematicsJointsToEE,
+    InverseKinematicsEEToJoints,
+)
+from lerobot.scripts.lerobot_record import record_loop
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.control_utils import init_keyboard_listener
+from lerobot.utils.utils import log_say
+from lerobot.utils.visualization_utils import init_rerun
+
+NUM_EPISODES = 5
+FPS = 30
+EPISODE_TIME_SEC = 60
+TASK_DESCRIPTION = "My task description"
+HF_MODEL_ID = "<hf_username>/<model_repo_id>"
+HF_DATASET_ID = "<hf_username>/<dataset_repo_id>"
+
+
+def main():
+    # Create the robot configuration & robot
+    camera_config = {"front": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=FPS)}
+    robot_config = SO100FollowerConfig(
+        port="/dev/tty.usbmodem5A460814411",
+        id="my_awesome_follower_arm",
+        cameras=camera_config,
+        use_degrees=True,
+    )
+
+    robot = SO100Follower(robot_config)
+
+    # Create policy
+    policy = ACTPolicy.from_pretrained(HF_MODEL_ID)
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(robot.bus.motors.keys()),
+    )
+
+    # Build pipeline to convert EE action to joints action
+    robot_ee_to_joints_processor = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        steps=[
+            InverseKinematicsEEToJoints(
+                kinematics=kinematics_solver,
+                motor_names=list(robot.bus.motors.keys()),
+                initial_guess_current_joints=True,
+            ),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Build pipeline to convert joints observation to EE observation
+    robot_joints_to_ee_pose_processor = RobotProcessorPipeline[RobotObservation, RobotObservation](
+        steps=[
+            ForwardKinematicsJointsToEE(
+                kinematics=kinematics_solver, motor_names=list(robot.bus.motors.keys())
+            )
+        ],
+        to_transition=observation_to_transition,
+        to_output=transition_to_observation,
+    )
+
+    # Create the dataset
+    dataset = LeRobotDataset.create(
+        repo_id=HF_DATASET_ID,
+        fps=FPS,
+        features=combine_feature_dicts(
+            aggregate_pipeline_dataset_features(
+                pipeline=robot_joints_to_ee_pose_processor,
+                initial_features=create_initial_features(observation=robot.observation_features),
+                use_videos=True,
+            ),
+            # User for now should be explicit on the feature keys that were used for record
+            # Alternatively, the user can pass the processor step that has the right features
+            aggregate_pipeline_dataset_features(
+                pipeline=make_default_teleop_action_processor(),
+                initial_features=create_initial_features(
+                    action={
+                        f"ee.{k}": PolicyFeature(type=FeatureType.ACTION, shape=(1,))
+                        for k in ["x", "y", "z", "wx", "wy", "wz", "gripper_pos"]
+                    }
+                ),
+                use_videos=True,
+            ),
+        ),
+        robot_type=robot.name,
+        use_videos=True,
+        image_writer_threads=4,
+    )
+
+    # Build Policy Processors
+    preprocessor, postprocessor = make_pre_post_processors(
+        policy_cfg=policy,
+        pretrained_path=HF_MODEL_ID,
+        dataset_stats=dataset.meta.stats,
+        # The inference device is automatically set to match the detected hardware, overriding any previous device settings from training to ensure compatibility.
+        preprocessor_overrides={"device_processor": {"device": str(policy.config.device)}},
+    )
+
+    # Connect the robot and teleoperator
+    robot.connect()
+
+    # Initialize the keyboard listener and rerun visualization
+    listener, events = init_keyboard_listener()
+    init_rerun(session_name="so100_so100_evaluate")
+
+    try:
+        if not robot.is_connected:
+            raise ValueError("Robot is not connected!")
+
+        print("Starting evaluate loop...")
+        episode_idx = 0
+        for episode_idx in range(NUM_EPISODES):
+            log_say(f"Running inference, recording eval episode {episode_idx + 1} of {NUM_EPISODES}")
+
+            # Main record loop
+            record_loop(
+                robot=robot,
+                events=events,
+                fps=FPS,
+                policy=policy,
+                preprocessor=preprocessor,  # Pass the pre and post policy processors
+                postprocessor=postprocessor,
+                dataset=dataset,
+                control_time_s=EPISODE_TIME_SEC,
+                single_task=TASK_DESCRIPTION,
+                display_data=True,
+                teleop_action_processor=make_default_teleop_action_processor(),
+                robot_action_processor=robot_ee_to_joints_processor,
+                robot_observation_processor=robot_joints_to_ee_pose_processor,
+            )
+
+            # Reset the environment if not stopping or re-recording
+            if not events["stop_recording"] and (
+                (episode_idx < NUM_EPISODES - 1) or events["rerecord_episode"]
+            ):
+                log_say("Reset the environment")
+                record_loop(
+                    robot=robot,
+                    events=events,
+                    fps=FPS,
+                    control_time_s=EPISODE_TIME_SEC,
+                    single_task=TASK_DESCRIPTION,
+                    display_data=True,
+                    teleop_action_processor=make_default_teleop_action_processor(),
+                    robot_action_processor=robot_ee_to_joints_processor,
+                    robot_observation_processor=robot_joints_to_ee_pose_processor,
+                )
+
+            if events["rerecord_episode"]:
+                log_say("Re-record episode")
+                events["rerecord_episode"] = False
+                events["exit_early"] = False
+                dataset.clear_episode_buffer()
+                continue
+
+            # Save episode
+            dataset.save_episode()
+            episode_idx += 1
+    finally:
+        # Clean up
+        log_say("Stop recording")
+        robot.disconnect()
+        listener.stop()
+
+        dataset.finalize()
+        dataset.push_to_hub()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/so100_to_so100_EE/record.py b/lerobot/examples/so100_to_so100_EE/record.py
new file mode 100644
index 0000000000000000000000000000000000000000..634bd891a81d9b5eed9a93d15c776bd55b319000
--- /dev/null
+++ b/lerobot/examples/so100_to_so100_EE/record.py
@@ -0,0 +1,215 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.datasets.feature_utils import combine_feature_dicts
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor.converters import (
+    observation_to_transition,
+    robot_action_observation_to_transition,
+    transition_to_observation,
+    transition_to_robot_action,
+)
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+from lerobot.robots.so_follower.robot_kinematic_processor import (
+    EEBoundsAndSafety,
+    ForwardKinematicsJointsToEE,
+    InverseKinematicsEEToJoints,
+)
+from lerobot.scripts.lerobot_record import record_loop
+from lerobot.teleoperators.so_leader import SO100Leader, SO100LeaderConfig
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.control_utils import init_keyboard_listener
+from lerobot.utils.utils import log_say
+from lerobot.utils.visualization_utils import init_rerun
+
+NUM_EPISODES = 2
+FPS = 30
+EPISODE_TIME_SEC = 60
+RESET_TIME_SEC = 30
+TASK_DESCRIPTION = "My task description"
+HF_REPO_ID = "<hf_username>/<dataset_repo_id>"
+
+
+def main():
+    # Create the robot and teleoperator configurations
+    camera_config = {"front": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=FPS)}
+    follower_config = SO100FollowerConfig(
+        port="/dev/tty.usbmodem5A460814411",
+        id="my_awesome_follower_arm",
+        cameras=camera_config,
+        use_degrees=True,
+    )
+    leader_config = SO100LeaderConfig(port="/dev/tty.usbmodem5A460819811", id="my_awesome_leader_arm")
+
+    # Initialize the robot and teleoperator
+    follower = SO100Follower(follower_config)
+    leader = SO100Leader(leader_config)
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    follower_kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(follower.bus.motors.keys()),
+    )
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    leader_kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(leader.bus.motors.keys()),
+    )
+
+    # Build pipeline to convert follower joints to EE observation
+    follower_joints_to_ee = RobotProcessorPipeline[RobotObservation, RobotObservation](
+        steps=[
+            ForwardKinematicsJointsToEE(
+                kinematics=follower_kinematics_solver, motor_names=list(follower.bus.motors.keys())
+            ),
+        ],
+        to_transition=observation_to_transition,
+        to_output=transition_to_observation,
+    )
+
+    # Build pipeline to convert leader joints to EE action
+    leader_joints_to_ee = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        steps=[
+            ForwardKinematicsJointsToEE(
+                kinematics=leader_kinematics_solver, motor_names=list(leader.bus.motors.keys())
+            ),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Build pipeline to convert EE action to follower joints
+    ee_to_follower_joints = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        [
+            EEBoundsAndSafety(
+                end_effector_bounds={"min": [-1.0, -1.0, -1.0], "max": [1.0, 1.0, 1.0]},
+                max_ee_step_m=0.10,
+            ),
+            InverseKinematicsEEToJoints(
+                kinematics=follower_kinematics_solver,
+                motor_names=list(follower.bus.motors.keys()),
+                initial_guess_current_joints=True,
+            ),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Create the dataset
+    dataset = LeRobotDataset.create(
+        repo_id=HF_REPO_ID,
+        fps=FPS,
+        features=combine_feature_dicts(
+            # Run the feature contract of the pipelines
+            # This tells you how the features would look like after the pipeline steps
+            aggregate_pipeline_dataset_features(
+                pipeline=leader_joints_to_ee,
+                initial_features=create_initial_features(action=leader.action_features),
+                use_videos=True,
+            ),
+            aggregate_pipeline_dataset_features(
+                pipeline=follower_joints_to_ee,
+                initial_features=create_initial_features(observation=follower.observation_features),
+                use_videos=True,
+            ),
+        ),
+        robot_type=follower.name,
+        use_videos=True,
+        image_writer_threads=4,
+    )
+
+    # Connect the robot and teleoperator
+    leader.connect()
+    follower.connect()
+
+    # Initialize the keyboard listener and rerun visualization
+    listener, events = init_keyboard_listener()
+    init_rerun(session_name="recording_phone")
+
+    try:
+        if not leader.is_connected or not follower.is_connected:
+            raise ValueError("Robot or teleop is not connected!")
+
+        print("Starting record loop...")
+        episode_idx = 0
+        while episode_idx < NUM_EPISODES and not events["stop_recording"]:
+            log_say(f"Recording episode {episode_idx + 1} of {NUM_EPISODES}")
+
+            # Main record loop
+            record_loop(
+                robot=follower,
+                events=events,
+                fps=FPS,
+                teleop=leader,
+                dataset=dataset,
+                control_time_s=EPISODE_TIME_SEC,
+                single_task=TASK_DESCRIPTION,
+                display_data=True,
+                teleop_action_processor=leader_joints_to_ee,
+                robot_action_processor=ee_to_follower_joints,
+                robot_observation_processor=follower_joints_to_ee,
+            )
+
+            # Reset the environment if not stopping or re-recording
+            if not events["stop_recording"] and (
+                episode_idx < NUM_EPISODES - 1 or events["rerecord_episode"]
+            ):
+                log_say("Reset the environment")
+                record_loop(
+                    robot=follower,
+                    events=events,
+                    fps=FPS,
+                    teleop=leader,
+                    control_time_s=RESET_TIME_SEC,
+                    single_task=TASK_DESCRIPTION,
+                    display_data=True,
+                    teleop_action_processor=leader_joints_to_ee,
+                    robot_action_processor=ee_to_follower_joints,
+                    robot_observation_processor=follower_joints_to_ee,
+                )
+
+            if events["rerecord_episode"]:
+                log_say("Re-recording episode")
+                events["rerecord_episode"] = False
+                events["exit_early"] = False
+                dataset.clear_episode_buffer()
+                continue
+
+            # Save episode
+            dataset.save_episode()
+            episode_idx += 1
+
+    finally:
+        # Clean up
+        log_say("Stop recording")
+        leader.disconnect()
+        follower.disconnect()
+        listener.stop()
+
+        dataset.finalize()
+        dataset.push_to_hub()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/so100_to_so100_EE/replay.py b/lerobot/examples/so100_to_so100_EE/replay.py
new file mode 100644
index 0000000000000000000000000000000000000000..b042e02dddb51fd43119d5ab7fa805ebb0604164
--- /dev/null
+++ b/lerobot/examples/so100_to_so100_EE/replay.py
@@ -0,0 +1,110 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+
+import time
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor.converters import (
+    robot_action_observation_to_transition,
+    transition_to_robot_action,
+)
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+from lerobot.robots.so_follower.robot_kinematic_processor import (
+    InverseKinematicsEEToJoints,
+)
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.constants import ACTION
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import log_say
+
+EPISODE_IDX = 0
+HF_REPO_ID = "<hf_username>/<dataset_repo_id>"
+
+
+def main():
+    # Initialize the robot config
+    robot_config = SO100FollowerConfig(
+        port="/dev/tty.usbmodem5A460814411", id="my_awesome_follower_arm", use_degrees=True
+    )
+
+    # Initialize the robot
+    robot = SO100Follower(robot_config)
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(robot.bus.motors.keys()),
+    )
+
+    # Build pipeline to convert EE action to joints action
+    robot_ee_to_joints_processor = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        steps=[
+            InverseKinematicsEEToJoints(
+                kinematics=kinematics_solver,
+                motor_names=list(robot.bus.motors.keys()),
+                initial_guess_current_joints=False,  # Because replay is open loop
+            ),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Fetch the dataset to replay
+    dataset = LeRobotDataset(HF_REPO_ID, episodes=[EPISODE_IDX])
+    # Filter dataset to only include frames from the specified episode since episodes are chunked in dataset V3.0
+    episode_frames = dataset.hf_dataset.filter(lambda x: x["episode_index"] == EPISODE_IDX)
+    actions = episode_frames.select_columns(ACTION)
+
+    # Connect to the robot
+    robot.connect()
+
+    try:
+        if not robot.is_connected:
+            raise ValueError("Robot is not connected!")
+
+        print("Starting replay loop...")
+        log_say(f"Replaying episode {EPISODE_IDX}")
+        for idx in range(len(episode_frames)):
+            t0 = time.perf_counter()
+
+            # Get recorded action from dataset
+            ee_action = {
+                name: float(actions[idx][ACTION][i])
+                for i, name in enumerate(dataset.features[ACTION]["names"])
+            }
+
+            # Get robot observation
+            robot_obs = robot.get_observation()
+
+            # Dataset EE -> robot joints
+            joint_action = robot_ee_to_joints_processor((ee_action, robot_obs))
+
+            # Send action to robot
+            _ = robot.send_action(joint_action)
+
+            precise_sleep(max(1.0 / dataset.fps - (time.perf_counter() - t0), 0.0))
+
+    finally:
+        # Clean up
+        robot.disconnect()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/so100_to_so100_EE/teleoperate.py b/lerobot/examples/so100_to_so100_EE/teleoperate.py
new file mode 100644
index 0000000000000000000000000000000000000000..af21f079bb5d817c46be530581c198fc452e1a0b
--- /dev/null
+++ b/lerobot/examples/so100_to_so100_EE/teleoperate.py
@@ -0,0 +1,126 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import time
+
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.processor import RobotProcessorPipeline
+from lerobot.processor.converters import (
+    robot_action_observation_to_transition,
+    robot_action_to_transition,
+    transition_to_robot_action,
+)
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+from lerobot.robots.so_follower.robot_kinematic_processor import (
+    EEBoundsAndSafety,
+    ForwardKinematicsJointsToEE,
+    InverseKinematicsEEToJoints,
+)
+from lerobot.teleoperators.so_leader import SO100Leader, SO100LeaderConfig
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.visualization_utils import init_rerun, log_rerun_data
+
+FPS = 30
+
+
+def main():
+    # Initialize the robot and teleoperator config
+    follower_config = SO100FollowerConfig(
+        port="/dev/tty.usbmodem5A460814411", id="my_awesome_follower_arm", use_degrees=True
+    )
+    leader_config = SO100LeaderConfig(port="/dev/tty.usbmodem5A460819811", id="my_awesome_leader_arm")
+
+    # Initialize the robot and teleoperator
+    follower = SO100Follower(follower_config)
+    leader = SO100Leader(leader_config)
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    follower_kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(follower.bus.motors.keys()),
+    )
+
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo: https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    leader_kinematics_solver = RobotKinematics(
+        urdf_path="./SO101/so101_new_calib.urdf",
+        target_frame_name="gripper_frame_link",
+        joint_names=list(leader.bus.motors.keys()),
+    )
+
+    # Build pipeline to convert teleop joints to EE action
+    leader_to_ee = RobotProcessorPipeline[RobotAction, RobotAction](
+        steps=[
+            ForwardKinematicsJointsToEE(
+                kinematics=leader_kinematics_solver, motor_names=list(leader.bus.motors.keys())
+            ),
+        ],
+        to_transition=robot_action_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # build pipeline to convert EE action to robot joints
+    ee_to_follower_joints = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        [
+            EEBoundsAndSafety(
+                end_effector_bounds={"min": [-1.0, -1.0, -1.0], "max": [1.0, 1.0, 1.0]},
+                max_ee_step_m=0.10,
+            ),
+            InverseKinematicsEEToJoints(
+                kinematics=follower_kinematics_solver,
+                motor_names=list(follower.bus.motors.keys()),
+                initial_guess_current_joints=False,
+            ),
+        ],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+
+    # Connect to the robot and teleoperator
+    follower.connect()
+    leader.connect()
+
+    # Init rerun viewer
+    init_rerun(session_name="so100_so100_EE_teleop")
+
+    print("Starting teleop loop...")
+    while True:
+        t0 = time.perf_counter()
+
+        # Get robot observation
+        robot_obs = follower.get_observation()
+
+        # Get teleop observation
+        leader_joints_obs = leader.get_action()
+
+        # teleop joints -> teleop EE action
+        leader_ee_act = leader_to_ee(leader_joints_obs)
+
+        # teleop EE -> robot joints
+        follower_joints_act = ee_to_follower_joints((leader_ee_act, robot_obs))
+
+        # Send action to robot
+        _ = follower.send_action(follower_joints_act)
+
+        # Visualize
+        log_rerun_data(observation=leader_ee_act, action=follower_joints_act)
+
+        precise_sleep(max(1.0 / FPS - (time.perf_counter() - t0), 0.0))
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/training/train_policy.py b/lerobot/examples/training/train_policy.py
new file mode 100644
index 0000000000000000000000000000000000000000..07ec10c921711e6ba6b9ef8ff375acf07790b219
--- /dev/null
+++ b/lerobot/examples/training/train_policy.py
@@ -0,0 +1,121 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""This script demonstrates how to train Diffusion Policy on the PushT environment."""
+
+from pathlib import Path
+
+import torch
+
+from lerobot.configs.types import FeatureType
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.feature_utils import dataset_to_policy_features
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.diffusion.configuration_diffusion import DiffusionConfig
+from lerobot.policies.diffusion.modeling_diffusion import DiffusionPolicy
+from lerobot.policies.factory import make_pre_post_processors
+
+
+def main():
+    # Create a directory to store the training checkpoint.
+    output_directory = Path("outputs/train/example_pusht_diffusion")
+    output_directory.mkdir(parents=True, exist_ok=True)
+
+    # # Select your device
+    device = torch.device("cuda")
+
+    # Number of offline training steps (we'll only do offline training for this example.)
+    # Adjust as you prefer. 5000 steps are needed to get something worth evaluating.
+    training_steps = 5000
+    log_freq = 1
+
+    # When starting from scratch (i.e. not from a pretrained policy), we need to specify 2 things before
+    # creating the policy:
+    #   - input/output shapes: to properly size the policy
+    #   - dataset stats: for normalization and denormalization of input/outputs
+    dataset_metadata = LeRobotDatasetMetadata("lerobot/pusht")
+    features = dataset_to_policy_features(dataset_metadata.features)
+    output_features = {key: ft for key, ft in features.items() if ft.type is FeatureType.ACTION}
+    input_features = {key: ft for key, ft in features.items() if key not in output_features}
+
+    # Policies are initialized with a configuration class, in this case `DiffusionConfig`. For this example,
+    # we'll just use the defaults and so no arguments other than input/output features need to be passed.
+    cfg = DiffusionConfig(input_features=input_features, output_features=output_features)
+
+    # We can now instantiate our policy with this config and the dataset stats.
+    policy = DiffusionPolicy(cfg)
+    policy.train()
+    policy.to(device)
+    preprocessor, postprocessor = make_pre_post_processors(cfg, dataset_stats=dataset_metadata.stats)
+
+    # Another policy-dataset interaction is with the delta_timestamps. Each policy expects a given number frames
+    # which can differ for inputs, outputs and rewards (if there are some).
+    delta_timestamps = {
+        "observation.image": [i / dataset_metadata.fps for i in cfg.observation_delta_indices],
+        "observation.state": [i / dataset_metadata.fps for i in cfg.observation_delta_indices],
+        "action": [i / dataset_metadata.fps for i in cfg.action_delta_indices],
+    }
+
+    # In this case with the standard configuration for Diffusion Policy, it is equivalent to this:
+    delta_timestamps = {
+        # Load the previous image and state at -0.1 seconds before current frame,
+        # then load current image and state corresponding to 0.0 second.
+        "observation.image": [-0.1, 0.0],
+        "observation.state": [-0.1, 0.0],
+        # Load the previous action (-0.1), the next action to be executed (0.0),
+        # and 14 future actions with a 0.1 seconds spacing. All these actions will be
+        # used to supervise the policy.
+        "action": [-0.1, 0.0, 0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9, 1.0, 1.1, 1.2, 1.3, 1.4],
+    }
+
+    # We can then instantiate the dataset with these delta_timestamps configuration.
+    dataset = LeRobotDataset("lerobot/pusht", delta_timestamps=delta_timestamps)
+
+    # Then we create our optimizer and dataloader for offline training.
+    optimizer = torch.optim.Adam(policy.parameters(), lr=1e-4)
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        num_workers=4,
+        batch_size=64,
+        shuffle=True,
+        pin_memory=device.type != "cpu",
+        drop_last=True,
+    )
+
+    # Run training loop.
+    step = 0
+    done = False
+    while not done:
+        for batch in dataloader:
+            batch = preprocessor(batch)
+            loss, _ = policy.forward(batch)
+            loss.backward()
+            optimizer.step()
+            optimizer.zero_grad()
+
+            if step % log_freq == 0:
+                print(f"step: {step} loss: {loss.item():.3f}")
+            step += 1
+            if step >= training_steps:
+                done = True
+                break
+
+    # Save a policy checkpoint.
+    policy.save_pretrained(output_directory)
+    preprocessor.save_pretrained(output_directory)
+    postprocessor.save_pretrained(output_directory)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/training/train_with_streaming.py b/lerobot/examples/training/train_with_streaming.py
new file mode 100644
index 0000000000000000000000000000000000000000..973698e74b8499a105bd250bb52dd8cdeb3614a7
--- /dev/null
+++ b/lerobot/examples/training/train_with_streaming.py
@@ -0,0 +1,108 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""This script demonstrates how to train a Diffusion Policy on the PushT environment,
+using a dataset processed in streaming mode."""
+
+from pathlib import Path
+
+import torch
+
+from lerobot.configs.types import FeatureType
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.feature_utils import dataset_to_policy_features
+from lerobot.datasets.streaming_dataset import StreamingLeRobotDataset
+from lerobot.policies.act.configuration_act import ACTConfig
+from lerobot.policies.act.modeling_act import ACTPolicy
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.utils.constants import ACTION
+
+
+def main():
+    # Create a directory to store the training checkpoint.
+    output_directory = Path("outputs/train/example_streaming_dataset")
+    output_directory.mkdir(parents=True, exist_ok=True)
+
+    # Selects the "best" device available
+    device = (
+        torch.device("cuda")
+        if torch.cuda.is_available()
+        else torch.device("mps")
+        if torch.backends.mps.is_available()
+        else torch.device("cpu")
+    )
+    print(f"Using device: {device}")
+
+    training_steps = 10
+    log_freq = 1
+
+    dataset_id = "lerobot/droid_1.0.1"  # 26M frames! Would require 4TB of disk space if installed locally (:
+    dataset_metadata = LeRobotDatasetMetadata(dataset_id)
+    features = dataset_to_policy_features(dataset_metadata.features)
+    output_features = {key: ft for key, ft in features.items() if ft.type is FeatureType.ACTION}
+    input_features = {key: ft for key, ft in features.items() if key not in output_features}
+
+    # We can now instantiate our policy with this config and the dataset stats.
+    cfg = ACTConfig(input_features=input_features, output_features=output_features)
+    policy = ACTPolicy(cfg)
+    policy.train()
+    policy.to(device)
+    preprocessor, postprocessor = make_pre_post_processors(cfg, dataset_stats=dataset_metadata.stats)
+
+    # Delta timestamps are used to (1) augment frames used during training and (2) supervise the policy.
+    # Here, we use delta-timestamps to only provide ground truth actions for supervision
+    delta_timestamps = {
+        ACTION: [t / dataset_metadata.fps for t in range(cfg.n_action_steps)],
+    }
+
+    # Instantiating the training dataset in streaming mode allows to not consume up memory as the data is fetched
+    # iteratively rather than being load into memory all at once. Retrieved frames are shuffled across epochs
+    dataset = StreamingLeRobotDataset(dataset_id, delta_timestamps=delta_timestamps, tolerance_s=1e-3)
+
+    optimizer = torch.optim.Adam(policy.parameters(), lr=1e-4)
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        num_workers=4,
+        batch_size=16,
+        pin_memory=device.type != "cpu",
+        drop_last=True,
+        prefetch_factor=2,  # loads batches with multiprocessing while policy trains
+    )
+
+    # Run training loop.
+    step = 0
+    done = False
+    while not done:
+        for batch in dataloader:
+            batch = preprocessor(batch)
+            loss, _ = policy.forward(batch)
+            loss.backward()
+            optimizer.step()
+            optimizer.zero_grad()
+
+            if step % log_freq == 0:
+                print(f"step: {step} loss: {loss.item():.3f}")
+            step += 1
+            if step >= training_steps:
+                done = True
+                break
+
+    # Save a policy checkpoint.
+    policy.save_pretrained(output_directory)
+    preprocessor.save_pretrained(output_directory)
+    postprocessor.save_pretrained(output_directory)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/act/act_training_example.py b/lerobot/examples/tutorial/act/act_training_example.py
new file mode 100644
index 0000000000000000000000000000000000000000..b62c49cac1a1d96be637c1acaf4229de42cb615b
--- /dev/null
+++ b/lerobot/examples/tutorial/act/act_training_example.py
@@ -0,0 +1,105 @@
+"""This script demonstrates how to train ACT Policy on a real-world dataset."""
+
+from pathlib import Path
+
+import torch
+
+from lerobot.configs.types import FeatureType
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.feature_utils import dataset_to_policy_features
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.act.configuration_act import ACTConfig
+from lerobot.policies.act.modeling_act import ACTPolicy
+from lerobot.policies.factory import make_pre_post_processors
+
+
+def make_delta_timestamps(delta_indices: list[int] | None, fps: int) -> list[float]:
+    if delta_indices is None:
+        return [0]
+
+    return [i / fps for i in delta_indices]
+
+
+def main():
+    output_directory = Path("outputs/robot_learning_tutorial/act")
+    output_directory.mkdir(parents=True, exist_ok=True)
+
+    # Select your device
+    device = torch.device("mps")  # or "cuda" or "cpu"
+
+    dataset_id = "lerobot/svla_so101_pickplace"
+
+    # This specifies the inputs the model will be expecting and the outputs it will produce
+    dataset_metadata = LeRobotDatasetMetadata(dataset_id)
+    features = dataset_to_policy_features(dataset_metadata.features)
+
+    output_features = {key: ft for key, ft in features.items() if ft.type is FeatureType.ACTION}
+    input_features = {key: ft for key, ft in features.items() if key not in output_features}
+
+    cfg = ACTConfig(input_features=input_features, output_features=output_features)
+    policy = ACTPolicy(cfg)
+    preprocessor, postprocessor = make_pre_post_processors(cfg, dataset_stats=dataset_metadata.stats)
+
+    policy.train()
+    policy.to(device)
+
+    # To perform action chunking, ACT expects a given number of actions as targets
+    delta_timestamps = {
+        "action": make_delta_timestamps(cfg.action_delta_indices, dataset_metadata.fps),
+    }
+
+    # add image features if they are present
+    delta_timestamps |= {
+        k: make_delta_timestamps(cfg.observation_delta_indices, dataset_metadata.fps)
+        for k in cfg.image_features
+    }
+
+    # Instantiate the dataset
+    dataset = LeRobotDataset(dataset_id, delta_timestamps=delta_timestamps)
+
+    # Create the optimizer and dataloader for offline training
+    optimizer = cfg.get_optimizer_preset().build(policy.parameters())
+    batch_size = 32
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        batch_size=batch_size,
+        shuffle=True,
+        pin_memory=device.type != "cpu",
+        drop_last=True,
+    )
+
+    # Number of training steps and logging frequency
+    training_steps = 1
+    log_freq = 1
+
+    # Run training loop
+    step = 0
+    done = False
+    while not done:
+        for batch in dataloader:
+            batch = preprocessor(batch)
+            loss, _ = policy.forward(batch)
+            loss.backward()
+            optimizer.step()
+            optimizer.zero_grad()
+
+            if step % log_freq == 0:
+                print(f"step: {step} loss: {loss.item():.3f}")
+            step += 1
+            if step >= training_steps:
+                done = True
+                break
+
+    # Save the policy checkpoint, alongside the pre/post processors
+    policy.save_pretrained(output_directory)
+    preprocessor.save_pretrained(output_directory)
+    postprocessor.save_pretrained(output_directory)
+
+    # Save all assets to the Hub
+    policy.push_to_hub("<user>/robot_learning_tutorial_act")
+    preprocessor.push_to_hub("<user>/robot_learning_tutorial_act")
+    postprocessor.push_to_hub("<user>/robot_learning_tutorial_act")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/act/act_using_example.py b/lerobot/examples/tutorial/act/act_using_example.py
new file mode 100644
index 0000000000000000000000000000000000000000..15254d8eb66effed7f776985f501ed8f591daf5e
--- /dev/null
+++ b/lerobot/examples/tutorial/act/act_using_example.py
@@ -0,0 +1,62 @@
+import torch
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.policies.act.modeling_act import ACTPolicy
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.policies.utils import build_inference_frame, make_robot_action
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+
+MAX_EPISODES = 5
+MAX_STEPS_PER_EPISODE = 20
+
+
+def main():
+    device = torch.device("mps")  # or "cuda" or "cpu"
+    model_id = "<user>/robot_learning_tutorial_act"
+    model = ACTPolicy.from_pretrained(model_id)
+
+    dataset_id = "lerobot/svla_so101_pickplace"
+    # This only downloads the metadata for the dataset, ~10s of MB even for large-scale datasets
+    dataset_metadata = LeRobotDatasetMetadata(dataset_id)
+    preprocess, postprocess = make_pre_post_processors(model.config, dataset_stats=dataset_metadata.stats)
+
+    # # find ports using lerobot-find-port
+    follower_port = ...  # something like "/dev/tty.usbmodem58760431631"
+
+    # # the robot ids are used the load the right calibration files
+    follower_id = ...  # something like "follower_so100"
+
+    # Robot and environment configuration
+    # Camera keys must match the name and resolutions of the ones used for training!
+    # You can check the camera keys expected by a model in the info.json card on the model card on the Hub
+    camera_config = {
+        "side": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=30),
+        "up": OpenCVCameraConfig(index_or_path=1, width=640, height=480, fps=30),
+    }
+
+    robot_cfg = SO100FollowerConfig(port=follower_port, id=follower_id, cameras=camera_config)
+    robot = SO100Follower(robot_cfg)
+    robot.connect()
+
+    for _ in range(MAX_EPISODES):
+        for _ in range(MAX_STEPS_PER_EPISODE):
+            obs = robot.get_observation()
+            obs_frame = build_inference_frame(
+                observation=obs, ds_features=dataset_metadata.features, device=device
+            )
+
+            obs = preprocess(obs_frame)
+
+            action = model.select_action(obs)
+            action = postprocess(action)
+
+            action = make_robot_action(action, dataset_metadata.features)
+
+            robot.send_action(action)
+
+        print("Episode finished! Starting new episode...")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/async-inf/policy_server.py b/lerobot/examples/tutorial/async-inf/policy_server.py
new file mode 100644
index 0000000000000000000000000000000000000000..244205bcfc6cab046678baeca6d0d5b742c016dd
--- /dev/null
+++ b/lerobot/examples/tutorial/async-inf/policy_server.py
@@ -0,0 +1,17 @@
+from lerobot.async_inference.configs import PolicyServerConfig
+from lerobot.async_inference.policy_server import serve
+
+
+def main():
+    host = ...  # something like "127.0.0.1" if you're exposing to localhost
+    port = ...  # something like 8080
+
+    config = PolicyServerConfig(
+        host=host,
+        port=port,
+    )
+    serve(config)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/async-inf/robot_client.py b/lerobot/examples/tutorial/async-inf/robot_client.py
new file mode 100644
index 0000000000000000000000000000000000000000..db6ead3fecdbf375f9672bf91384c1e760faed5c
--- /dev/null
+++ b/lerobot/examples/tutorial/async-inf/robot_client.py
@@ -0,0 +1,62 @@
+import threading
+
+from lerobot.async_inference.configs import RobotClientConfig
+from lerobot.async_inference.helpers import visualize_action_queue_size
+from lerobot.async_inference.robot_client import RobotClient
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.robots.so_follower import SO100FollowerConfig
+
+
+def main():
+    # these cameras must match the ones expected by the policy - find your cameras with lerobot-find-cameras
+    # check the config.json on the Hub for the policy you are using to see the expected camera specs
+    camera_cfg = {
+        "up": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=30),
+        "side": OpenCVCameraConfig(index_or_path=1, width=640, height=480, fps=30),
+    }
+
+    # # find ports using lerobot-find-port
+    follower_port = ...  # something like "/dev/tty.usbmodem58760431631"
+
+    # # the robot ids are used the load the right calibration files
+    follower_id = ...  # something like "follower_so100"
+
+    robot_cfg = SO100FollowerConfig(port=follower_port, id=follower_id, cameras=camera_cfg)
+
+    server_address = ...  # something like "127.0.0.1:8080" if using localhost
+
+    # 3. Create client configuration
+    client_cfg = RobotClientConfig(
+        robot=robot_cfg,
+        server_address=server_address,
+        policy_device="mps",
+        client_device="cpu",
+        policy_type="act",
+        pretrained_name_or_path="<user>/robot_learning_tutorial_act",
+        chunk_size_threshold=0.5,  # g
+        actions_per_chunk=50,  # make sure this is less than the max actions of the policy
+    )
+
+    # 4. Create and start client
+    client = RobotClient(client_cfg)
+
+    # 5. Provide a textual description of the task
+    task = ...
+
+    if client.start():
+        # Start action receiver thread
+        action_receiver_thread = threading.Thread(target=client.receive_actions, daemon=True)
+        action_receiver_thread.start()
+
+        try:
+            # Run the control loop
+            client.control_loop(task)
+        except KeyboardInterrupt:
+            client.stop()
+            action_receiver_thread.join()
+            # (Optionally) plot the action queue size
+            visualize_action_queue_size(client.action_queue_size)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/diffusion/diffusion_training_example.py b/lerobot/examples/tutorial/diffusion/diffusion_training_example.py
new file mode 100644
index 0000000000000000000000000000000000000000..dc6ca68a31640ac76682e3652a8fe044956599a4
--- /dev/null
+++ b/lerobot/examples/tutorial/diffusion/diffusion_training_example.py
@@ -0,0 +1,106 @@
+"""This script demonstrates how to train Diffusion Policy on a real-world dataset."""
+
+from pathlib import Path
+
+import torch
+
+from lerobot.configs.types import FeatureType
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.feature_utils import dataset_to_policy_features
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.diffusion.configuration_diffusion import DiffusionConfig
+from lerobot.policies.diffusion.modeling_diffusion import DiffusionPolicy
+from lerobot.policies.factory import make_pre_post_processors
+
+
+def make_delta_timestamps(delta_indices: list[int] | None, fps: int) -> list[float]:
+    if delta_indices is None:
+        return [0]
+
+    return [i / fps for i in delta_indices]
+
+
+def main():
+    output_directory = Path("outputs/robot_learning_tutorial/diffusion")
+    output_directory.mkdir(parents=True, exist_ok=True)
+
+    # Select your device
+    device = torch.device("mps")  # or "cuda" or "cpu"
+
+    dataset_id = "lerobot/svla_so101_pickplace"
+
+    # This specifies the inputs the model will be expecting and the outputs it will produce
+    dataset_metadata = LeRobotDatasetMetadata(dataset_id)
+    features = dataset_to_policy_features(dataset_metadata.features)
+
+    output_features = {key: ft for key, ft in features.items() if ft.type is FeatureType.ACTION}
+    input_features = {key: ft for key, ft in features.items() if key not in output_features}
+
+    cfg = DiffusionConfig(input_features=input_features, output_features=output_features)
+    policy = DiffusionPolicy(cfg)
+    preprocessor, postprocessor = make_pre_post_processors(cfg, dataset_stats=dataset_metadata.stats)
+
+    policy.train()
+    policy.to(device)
+
+    # To perform action chunking, ACT expects a given number of actions as targets
+    delta_timestamps = {
+        "observation.state": make_delta_timestamps(cfg.observation_delta_indices, dataset_metadata.fps),
+        "action": make_delta_timestamps(cfg.action_delta_indices, dataset_metadata.fps),
+    }
+
+    # add image features if they are present
+    delta_timestamps |= {
+        k: make_delta_timestamps(cfg.observation_delta_indices, dataset_metadata.fps)
+        for k in cfg.image_features
+    }
+
+    # Instantiate the dataset
+    dataset = LeRobotDataset(dataset_id, delta_timestamps=delta_timestamps)
+
+    # Create the optimizer and dataloader for offline training
+    optimizer = cfg.get_optimizer_preset().build(policy.parameters())
+    batch_size = 32
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        batch_size=batch_size,
+        shuffle=True,
+        pin_memory=device.type != "cpu",
+        drop_last=True,
+    )
+
+    # Number of training steps and logging frequency
+    training_steps = 1
+    log_freq = 1
+
+    # Run training loop
+    step = 0
+    done = False
+    while not done:
+        for batch in dataloader:
+            batch = preprocessor(batch)
+            loss, _ = policy.forward(batch)
+            loss.backward()
+            optimizer.step()
+            optimizer.zero_grad()
+
+            if step % log_freq == 0:
+                print(f"step: {step} loss: {loss.item():.3f}")
+            step += 1
+            if step >= training_steps:
+                done = True
+                break
+
+    # Save the policy checkpoint, alongside the pre/post processors
+    policy.save_pretrained(output_directory)
+    preprocessor.save_pretrained(output_directory)
+    postprocessor.save_pretrained(output_directory)
+
+    # Save all assets to the Hub
+    policy.push_to_hub("<user>/robot_learning_tutorial_diffusion")
+    preprocessor.push_to_hub("<user>/robot_learning_tutorial_diffusion")
+    postprocessor.push_to_hub("<user>/robot_learning_tutorial_diffusion")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/diffusion/diffusion_using_example.py b/lerobot/examples/tutorial/diffusion/diffusion_using_example.py
new file mode 100644
index 0000000000000000000000000000000000000000..9b31cf359ac3c6f59395fc5b56edae223d179a75
--- /dev/null
+++ b/lerobot/examples/tutorial/diffusion/diffusion_using_example.py
@@ -0,0 +1,63 @@
+import torch
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.policies.diffusion.modeling_diffusion import DiffusionPolicy
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.policies.utils import build_inference_frame, make_robot_action
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+
+MAX_EPISODES = 5
+MAX_STEPS_PER_EPISODE = 20
+
+
+def main():
+    device = torch.device("mps")  # or "cuda" or "cpu"
+    model_id = "<user>/robot_learning_tutorial_diffusion"
+
+    model = DiffusionPolicy.from_pretrained(model_id)
+
+    dataset_id = "lerobot/svla_so101_pickplace"
+    # This only downloads the metadata for the dataset, ~10s of MB even for large-scale datasets
+    dataset_metadata = LeRobotDatasetMetadata(dataset_id)
+    preprocess, postprocess = make_pre_post_processors(
+        model.config, model_id, dataset_stats=dataset_metadata.stats
+    )
+
+    # # find ports using lerobot-find-port
+    follower_port = ...  # something like "/dev/tty.usbmodem58760431631"
+
+    # # the robot ids are used the load the right calibration files
+    follower_id = ...  # something like "follower_so100"
+
+    # Robot and environment configuration
+    # Camera keys must match the name and resolutions of the ones used for training!
+    # You can check the camera keys expected by a model in the info.json card on the model card on the Hub
+    camera_config = {
+        "side": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=30),
+        "up": OpenCVCameraConfig(index_or_path=1, width=640, height=480, fps=30),
+    }
+
+    robot_cfg = SO100FollowerConfig(port=follower_port, id=follower_id, cameras=camera_config)
+    robot = SO100Follower(robot_cfg)
+    robot.connect()
+
+    for _ in range(MAX_EPISODES):
+        for _ in range(MAX_STEPS_PER_EPISODE):
+            obs = robot.get_observation()
+            obs_frame = build_inference_frame(
+                observation=obs, ds_features=dataset_metadata.features, device=device
+            )
+
+            obs = preprocess(obs_frame)
+
+            action = model.select_action(obs)
+            action = postprocess(action)
+            action = make_robot_action(action, dataset_metadata.features)
+            robot.send_action(action)
+
+        print("Episode finished! Starting new episode...")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/pi0/using_pi0_example.py b/lerobot/examples/tutorial/pi0/using_pi0_example.py
new file mode 100644
index 0000000000000000000000000000000000000000..d8cf9dbffd09d28263dac8a82521e3cf64a948eb
--- /dev/null
+++ b/lerobot/examples/tutorial/pi0/using_pi0_example.py
@@ -0,0 +1,72 @@
+import torch
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.datasets.feature_utils import hw_to_dataset_features
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.policies.pi0.modeling_pi0 import PI0Policy
+from lerobot.policies.utils import build_inference_frame, make_robot_action
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+
+MAX_EPISODES = 5
+MAX_STEPS_PER_EPISODE = 20
+
+
+def main():
+    device = torch.device("mps")  # or "cuda" or "cpu"
+    model_id = "lerobot/pi0_base"
+
+    model = PI0Policy.from_pretrained(model_id)
+
+    preprocess, postprocess = make_pre_post_processors(
+        model.config,
+        model_id,
+        # This overrides allows to run on MPS, otherwise defaults to CUDA (if available)
+        preprocessor_overrides={"device_processor": {"device": str(device)}},
+    )
+
+    # find ports using lerobot-find-port
+    follower_port = ...  # something like "/dev/tty.usbmodem58760431631"
+
+    # the robot ids are used the load the right calibration files
+    follower_id = ...  # something like "follower_so100"
+
+    # Robot and environment configuration
+    # Camera keys must match the name and resolutions of the ones used for training!
+    # You can check the camera keys expected by a model in the info.json card on the model card on the Hub
+    camera_config = {
+        "base_0_rgb": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=30),
+        "left_wrist_0_rgb": OpenCVCameraConfig(index_or_path=1, width=640, height=480, fps=30),
+        "right_wrist_0_rgb": OpenCVCameraConfig(index_or_path=2, width=640, height=480, fps=30),
+    }
+
+    robot_cfg = SO100FollowerConfig(port=follower_port, id=follower_id, cameras=camera_config)
+    robot = SO100Follower(robot_cfg)
+    robot.connect()
+
+    task = ""  # something like "pick the red block"
+    robot_type = ""  # something like "so100_follower" for multi-embodiment datasets
+
+    # This is used to match the raw observation keys to the keys expected by the policy
+    action_features = hw_to_dataset_features(robot.action_features, "action")
+    obs_features = hw_to_dataset_features(robot.observation_features, "observation")
+    dataset_features = {**action_features, **obs_features}
+
+    for _ in range(MAX_EPISODES):
+        for _ in range(MAX_STEPS_PER_EPISODE):
+            obs = robot.get_observation()
+            obs_frame = build_inference_frame(
+                observation=obs, ds_features=dataset_features, device=device, task=task, robot_type=robot_type
+            )
+
+            obs = preprocess(obs_frame)
+
+            action = model.select_action(obs)
+            action = postprocess(action)
+            action = make_robot_action(action, dataset_features)
+            robot.send_action(action)
+
+        print("Episode finished! Starting new episode...")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/rl/hilserl_example.py b/lerobot/examples/tutorial/rl/hilserl_example.py
new file mode 100644
index 0000000000000000000000000000000000000000..d367a01cebdb1f5c6dd73abc24a23b4dcbf8e0d3
--- /dev/null
+++ b/lerobot/examples/tutorial/rl/hilserl_example.py
@@ -0,0 +1,347 @@
+import multiprocessing as mp
+import signal
+from pathlib import Path
+from queue import Empty, Full
+
+import torch
+import torch.optim as optim
+
+from lerobot.datasets.feature_utils import hw_to_dataset_features
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.envs.configs import HILSerlProcessorConfig, HILSerlRobotEnvConfig
+from lerobot.policies.sac.configuration_sac import SACConfig
+from lerobot.policies.sac.modeling_sac import SACPolicy
+from lerobot.policies.sac.reward_model.modeling_classifier import Classifier
+from lerobot.rl.buffer import ReplayBuffer
+from lerobot.rl.gym_manipulator import make_robot_env
+from lerobot.robots.so_follower import SO100FollowerConfig
+from lerobot.teleoperators.so_leader import SO100LeaderConfig
+from lerobot.teleoperators.utils import TeleopEvents
+
+LOG_EVERY = 10
+SEND_EVERY = 10
+MAX_EPISODES = 5
+MAX_STEPS_PER_EPISODE = 20
+
+
+def run_learner(
+    transitions_queue: mp.Queue,
+    parameters_queue: mp.Queue,
+    shutdown_event: mp.Event,
+    policy_learner: SACPolicy,
+    online_buffer: ReplayBuffer,
+    offline_buffer: ReplayBuffer,
+    lr: float = 3e-4,
+    batch_size: int = 32,
+    device: torch.device = "mps",
+):
+    """The learner process - trains SAC policy on transitions streamed from the actor, updating parameters
+    for the actor to adopt."""
+    policy_learner.train()
+    policy_learner.to(device)
+
+    # Create Adam optimizer from scratch - simple and clean
+    optimizer = optim.Adam(policy_learner.parameters(), lr=lr)
+
+    print(f"[LEARNER] Online buffer capacity: {online_buffer.capacity}")
+    print(f"[LEARNER] Offline buffer capacity: {offline_buffer.capacity}")
+
+    training_step = 0
+
+    while not shutdown_event.is_set():
+        # retrieve incoming transitions from the actor process
+        try:
+            transitions = transitions_queue.get(timeout=0.1)
+            for transition in transitions:
+                # HIL-SERL: Add ALL transitions to online buffer
+                online_buffer.add(**transition)
+
+                # HIL-SERL: Add ONLY human intervention transitions to offline buffer
+                is_intervention = transition.get("complementary_info", {}).get("is_intervention", False)
+                if is_intervention:
+                    offline_buffer.add(**transition)
+                    print(
+                        f"[LEARNER] Human intervention detected! Added to offline buffer (now {len(offline_buffer)} transitions)"
+                    )
+
+        except Empty:
+            pass  # No transitions available, continue
+
+        # Train if we have enough data
+        if len(online_buffer) >= policy_learner.config.online_step_before_learning:
+            # Sample from online buffer (autonomous + human data)
+            online_batch = online_buffer.sample(batch_size // 2)
+
+            # Sample from offline buffer (human demonstrations only, either precollected or at runtime)
+            offline_batch = offline_buffer.sample(batch_size // 2)
+
+            # Combine batches - this is the key HIL-SERL mechanism!
+            batch = {}
+            for key in online_batch:
+                if key in offline_batch:
+                    batch[key] = torch.cat([online_batch[key], offline_batch[key]], dim=0)
+                else:
+                    batch[key] = online_batch[key]
+
+            loss, _ = policy_learner.forward(batch)
+
+            optimizer.zero_grad()
+            loss.backward()
+            optimizer.step()
+            training_step += 1
+
+            if training_step % LOG_EVERY == 0:
+                print(
+                    f"[LEARNER] Training step {training_step}, Loss: {loss.item():.4f}, "
+                    f"Buffers: Online={len(online_buffer)}, Offline={len(offline_buffer)}"
+                )
+
+            # Send updated parameters to actor every 10 training steps
+            if training_step % SEND_EVERY == 0:
+                try:
+                    state_dict = {k: v.cpu() for k, v in policy_learner.state_dict().items()}
+                    parameters_queue.put_nowait(state_dict)
+                    print("[LEARNER] Sent updated parameters to actor")
+                except Full:
+                    # Missing write due to queue not being consumed (should happen rarely)
+                    pass
+
+    print("[LEARNER] Learner process finished")
+
+
+def run_actor(
+    transitions_queue: mp.Queue,
+    parameters_queue: mp.Queue,
+    shutdown_event: mp.Event,
+    policy_actor: SACPolicy,
+    reward_classifier: Classifier,
+    env_cfg: HILSerlRobotEnvConfig,
+    device: torch.device = "mps",
+    output_directory: Path | None = None,
+):
+    """The actor process - interacts with environment and collects data.
+    The policy is frozen and only the parameters are updated, popping the most recent ones from a queue."""
+    policy_actor.eval()
+    policy_actor.to(device)
+
+    reward_classifier.eval()
+    reward_classifier.to(device)
+
+    # Create robot environment inside the actor process
+    env, teleop_device = make_robot_env(env_cfg)
+
+    try:
+        for episode in range(MAX_EPISODES):
+            if shutdown_event.is_set():
+                break
+
+            obs, _info = env.reset()
+            episode_reward = 0.0
+            step = 0
+            episode_transitions = []
+
+            print(f"[ACTOR] Starting episode {episode + 1}")
+
+            while step < MAX_STEPS_PER_EPISODE and not shutdown_event.is_set():
+                try:
+                    new_params = parameters_queue.get_nowait()
+                    policy_actor.load_state_dict(new_params)
+                    print("[ACTOR] Updated policy parameters from learner")
+                except Empty:  # No new updated parameters available from learner, waiting
+                    pass
+
+                # Get action from policy
+                policy_obs = make_policy_obs(obs, device=device)
+                action_tensor = policy_actor.select_action(policy_obs)  # predicts a single action
+                action = action_tensor.squeeze(0).cpu().numpy()
+
+                # Step environment
+                next_obs, _env_reward, terminated, truncated, _info = env.step(action)
+                done = terminated or truncated
+
+                # Predict reward
+                policy_next_obs = make_policy_obs(next_obs, device=device)
+                reward = reward_classifier.predict_reward(policy_next_obs)
+
+                if reward >= 1.0 and not done:  # success detected! halt episode
+                    terminated = True
+                    done = True
+
+                # In HIL-SERL, human interventions come from the teleop device
+                is_intervention = False
+                if hasattr(teleop_device, "get_teleop_events"):
+                    # Real intervention detection from teleop device
+                    teleop_events = teleop_device.get_teleop_events()
+                    is_intervention = teleop_events.get(TeleopEvents.IS_INTERVENTION, False)
+
+                # Store transition with intervention metadata
+                transition = {
+                    "state": policy_obs,
+                    "action": action,
+                    "reward": float(reward) if hasattr(reward, "item") else reward,
+                    "next_state": policy_next_obs,
+                    "done": done,
+                    "truncated": truncated,
+                    "complementary_info": {
+                        "is_intervention": is_intervention,
+                    },
+                }
+
+                episode_transitions.append(transition)
+
+                episode_reward += reward
+                step += 1
+
+                obs = next_obs
+
+                if done:
+                    break
+
+            # Send episode transitions to learner
+            transitions_queue.put_nowait(episode_transitions)
+
+    except KeyboardInterrupt:
+        print("[ACTOR] Interrupted by user")
+    finally:
+        # Clean up
+        if hasattr(env, "robot") and env.robot.is_connected:
+            env.robot.disconnect()
+        if teleop_device and hasattr(teleop_device, "disconnect"):
+            teleop_device.disconnect()
+        if output_directory is not None:
+            policy_actor.save_pretrained(output_directory)
+            print(f"[ACTOR] Latest actor policy saved at: {output_directory}")
+
+        print("[ACTOR] Actor process finished")
+
+
+def make_policy_obs(obs, device: torch.device = "cpu"):
+    return {
+        "observation.state": torch.from_numpy(obs["agent_pos"]).float().unsqueeze(0).to(device),
+        **{
+            f"observation.image.{k}": torch.from_numpy(obs["pixels"][k]).float().unsqueeze(0).to(device)
+            for k in obs["pixels"]
+        },
+    }
+
+
+def main():
+    """Main function - coordinates actor and learner processes."""
+
+    device = "mps"  # or "cuda" or "cpu"
+    output_directory = Path("outputs/robot_learning_tutorial/hil_serl")
+    output_directory.mkdir(parents=True, exist_ok=True)
+
+    # find ports using lerobot-find-port
+    follower_port = ...
+    leader_port = ...
+
+    # the robot ids are used the load the right calibration files
+    follower_id = ...
+    leader_id = ...
+
+    # A pretrained model (to be used in-distribution!)
+    reward_classifier_id = "<user>/reward_classifier_hil_serl_example"
+    reward_classifier = Classifier.from_pretrained(reward_classifier_id)
+
+    reward_classifier.to(device)
+    reward_classifier.eval()
+
+    # Robot and environment configuration
+    robot_cfg = SO100FollowerConfig(port=follower_port, id=follower_id)
+    teleop_cfg = SO100LeaderConfig(port=leader_port, id=leader_id)
+    processor_cfg = HILSerlProcessorConfig(control_mode="leader")
+
+    env_cfg = HILSerlRobotEnvConfig(robot=robot_cfg, teleop=teleop_cfg, processor=processor_cfg)
+
+    # Create robot environment
+    env, teleop_device = make_robot_env(env_cfg)
+
+    obs_features = hw_to_dataset_features(env.robot.observation_features, "observation")
+    action_features = hw_to_dataset_features(env.robot.action_features, "action")
+
+    # Create SAC policy for action selection
+    policy_cfg = SACConfig(
+        device=device,
+        input_features=obs_features,
+        output_features=action_features,
+    )
+
+    policy_actor = SACPolicy(policy_cfg)
+    policy_learner = SACPolicy(policy_cfg)
+
+    demonstrations_repo_id = "lerobot/example_hil_serl_dataset"
+    offline_dataset = LeRobotDataset(repo_id=demonstrations_repo_id)
+
+    # Online buffer: initialized from scratch
+    online_replay_buffer = ReplayBuffer(device=device, state_keys=list(obs_features.keys()))
+    # Offline buffer: Created from dataset (pre-populated it with demonstrations)
+    offline_replay_buffer = ReplayBuffer.from_lerobot_dataset(
+        lerobot_dataset=offline_dataset, device=device, state_keys=list(obs_features.keys())
+    )
+
+    # Create communication channels between learner and actor processes
+    transitions_queue = mp.Queue(maxsize=10)
+    parameters_queue = mp.Queue(maxsize=2)
+    shutdown_event = mp.Event()
+
+    # Signal handler for graceful shutdown
+    def signal_handler(sig):
+        print(f"\nSignal {sig} received, shutting down...")
+        shutdown_event.set()
+
+    signal.signal(signal.SIGINT, signal_handler)
+    signal.signal(signal.SIGTERM, signal_handler)
+
+    # Create processes
+    learner_process = mp.Process(
+        target=run_learner,
+        args=(
+            transitions_queue,
+            parameters_queue,
+            shutdown_event,
+            policy_learner,
+            online_replay_buffer,
+            offline_replay_buffer,
+        ),
+        kwargs={"device": device},  # can run on accelerated hardware for training
+    )
+
+    actor_process = mp.Process(
+        target=run_actor,
+        args=(
+            transitions_queue,
+            parameters_queue,
+            shutdown_event,
+            policy_actor,
+            reward_classifier,
+            env_cfg,
+            output_directory,
+        ),
+        kwargs={"device": "cpu"},  # actor is frozen, can run on CPU or accelerate for inference
+    )
+
+    learner_process.start()
+    actor_process.start()
+
+    try:
+        # Wait for actor to finish (it controls the episode loop)
+        actor_process.join()
+        shutdown_event.set()
+        learner_process.join(timeout=10)
+
+    except KeyboardInterrupt:
+        print("Main process interrupted")
+        shutdown_event.set()
+        actor_process.join(timeout=5)
+        learner_process.join(timeout=10)
+
+    finally:
+        if learner_process.is_alive():
+            learner_process.terminate()
+        if actor_process.is_alive():
+            actor_process.terminate()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/rl/reward_classifier_example.py b/lerobot/examples/tutorial/rl/reward_classifier_example.py
new file mode 100644
index 0000000000000000000000000000000000000000..4af6b899c2a962f6a0a8b9ac55c1c10cb20ad4eb
--- /dev/null
+++ b/lerobot/examples/tutorial/rl/reward_classifier_example.py
@@ -0,0 +1,67 @@
+import torch
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.factory import make_policy, make_pre_post_processors
+from lerobot.policies.sac.reward_model.configuration_classifier import RewardClassifierConfig
+
+
+def main():
+    # Device to use for training
+    device = "mps"  # or "cuda", or "cpu"
+
+    # Load the dataset used for training
+    repo_id = "lerobot/example_hil_serl_dataset"
+    dataset = LeRobotDataset(repo_id)
+
+    # Configure the policy to extract features from the image frames
+    camera_keys = dataset.meta.camera_keys
+
+    config = RewardClassifierConfig(
+        num_cameras=len(camera_keys),
+        device=device,
+        # backbone model to extract features from the image frames
+        model_name="microsoft/resnet-18",
+    )
+
+    # Make policy, preprocessor, and optimizer
+    policy = make_policy(config, ds_meta=dataset.meta)
+    optimizer = config.get_optimizer_preset().build(policy.parameters())
+    preprocessor, _ = make_pre_post_processors(policy_cfg=config, dataset_stats=dataset.meta.stats)
+
+    classifier_id = "<user>/reward_classifier_hil_serl_example"
+
+    # Instantiate a dataloader
+    dataloader = torch.utils.data.DataLoader(dataset, batch_size=16, shuffle=True)
+
+    # Training loop
+    num_epochs = 5
+    for epoch in range(num_epochs):
+        total_loss = 0
+        total_accuracy = 0
+        for batch in dataloader:
+            # Preprocess the batch and move it to the correct device.
+            batch = preprocessor(batch)
+
+            # Forward pass
+            loss, output_dict = policy.forward(batch)
+
+            # Backward pass and optimization
+            optimizer.zero_grad()
+            loss.backward()
+            optimizer.step()
+
+            total_loss += loss.item()
+            total_accuracy += output_dict["accuracy"]
+
+        avg_loss = total_loss / len(dataloader)
+        avg_accuracy = total_accuracy / len(dataloader)
+        print(f"Epoch {epoch + 1}/{num_epochs}, Loss: {avg_loss:.4f}, Accuracy: {avg_accuracy:.2f}%")
+
+    print("Training finished!")
+
+    # You can now save the trained policy.
+    policy.push_to_hub(classifier_id)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/examples/tutorial/smolvla/using_smolvla_example.py b/lerobot/examples/tutorial/smolvla/using_smolvla_example.py
new file mode 100644
index 0000000000000000000000000000000000000000..b99126efa77b947f95e2f0b0ebf1a974ab162d84
--- /dev/null
+++ b/lerobot/examples/tutorial/smolvla/using_smolvla_example.py
@@ -0,0 +1,71 @@
+import torch
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.datasets.feature_utils import hw_to_dataset_features
+from lerobot.policies.factory import make_pre_post_processors
+from lerobot.policies.smolvla.modeling_smolvla import SmolVLAPolicy
+from lerobot.policies.utils import build_inference_frame, make_robot_action
+from lerobot.robots.so_follower import SO100Follower, SO100FollowerConfig
+
+MAX_EPISODES = 5
+MAX_STEPS_PER_EPISODE = 20
+
+
+def main():
+    device = torch.device("mps")  # or "cuda" or "cpu"
+    model_id = "lerobot/smolvla_base"
+
+    model = SmolVLAPolicy.from_pretrained(model_id)
+
+    preprocess, postprocess = make_pre_post_processors(
+        model.config,
+        model_id,
+        # This overrides allows to run on MPS, otherwise defaults to CUDA (if available)
+        preprocessor_overrides={"device_processor": {"device": str(device)}},
+    )
+
+    # find ports using lerobot-find-port
+    follower_port = ...  # something like "/dev/tty.usbmodem58760431631"
+
+    # the robot ids are used the load the right calibration files
+    follower_id = ...  # something like "follower_so100"
+
+    # Robot and environment configuration
+    # Camera keys must match the name and resolutions of the ones used for training!
+    # You can check the camera keys expected by a model in the info.json card on the model card on the Hub
+    camera_config = {
+        "camera1": OpenCVCameraConfig(index_or_path=0, width=640, height=480, fps=30),
+        "camera2": OpenCVCameraConfig(index_or_path=1, width=640, height=480, fps=30),
+    }
+
+    robot_cfg = SO100FollowerConfig(port=follower_port, id=follower_id, cameras=camera_config)
+    robot = SO100Follower(robot_cfg)
+    robot.connect()
+
+    task = ""  # something like "pick the red block"
+    robot_type = ""  # something like "so100_follower" for multi-embodiment datasets
+
+    # This is used to match the raw observation keys to the keys expected by the policy
+    action_features = hw_to_dataset_features(robot.action_features, "action")
+    obs_features = hw_to_dataset_features(robot.observation_features, "observation")
+    dataset_features = {**action_features, **obs_features}
+
+    for _ in range(MAX_EPISODES):
+        for _ in range(MAX_STEPS_PER_EPISODE):
+            obs = robot.get_observation()
+            obs_frame = build_inference_frame(
+                observation=obs, ds_features=dataset_features, device=device, task=task, robot_type=robot_type
+            )
+
+            obs = preprocess(obs_frame)
+
+            action = model.select_action(obs)
+            action = postprocess(action)
+            action = make_robot_action(action, dataset_features)
+            robot.send_action(action)
+
+        print("Episode finished! Starting new episode...")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/media/readme/VLA_architecture.jpg b/lerobot/media/readme/VLA_architecture.jpg
new file mode 100644
index 0000000000000000000000000000000000000000..fdfbd1ded2ba02c75847246cb9c284423f83e1f4
--- /dev/null
+++ b/lerobot/media/readme/VLA_architecture.jpg
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a54f82e318e0a9c0c6bbd1ef099685518f001d5bf3a5f341d6d2090994e7cb43
+size 792640
diff --git a/lerobot/media/readme/lerobot-logo-thumbnail.png b/lerobot/media/readme/lerobot-logo-thumbnail.png
new file mode 100644
index 0000000000000000000000000000000000000000..d14daf6ada759f06fadc9684c4b5f5e01756d5ab
--- /dev/null
+++ b/lerobot/media/readme/lerobot-logo-thumbnail.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:72ee48061c2528eb9f6a1f163622d4805476a52c813bd03f2f32e32d89afd63e
+size 164066
diff --git a/lerobot/media/readme/robots_control_video.webp b/lerobot/media/readme/robots_control_video.webp
new file mode 100644
index 0000000000000000000000000000000000000000..c0b8aadb393ce127a51960f12bb95f022fdb6d40
--- /dev/null
+++ b/lerobot/media/readme/robots_control_video.webp
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:977b535598044cefbae8f5e9aef976dd98377e45575f95f3900d99a5d7ba8939
+size 2425850
diff --git a/lerobot/media/readme/so100_video.webp b/lerobot/media/readme/so100_video.webp
new file mode 100644
index 0000000000000000000000000000000000000000..46b1a28ce03dba429cfa8465ecc46fd3c3a6775b
--- /dev/null
+++ b/lerobot/media/readme/so100_video.webp
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:11857a0729814afa529571c30ad4acb9f8cf8f4fa743163d462cb31a10f9782b
+size 492230
diff --git a/lerobot/pyproject.toml b/lerobot/pyproject.toml
new file mode 100644
index 0000000000000000000000000000000000000000..5f45626c011992fc745f963cac99d1277c46ac67
--- /dev/null
+++ b/lerobot/pyproject.toml
@@ -0,0 +1,414 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+[build-system]
+requires = ["setuptools"]
+build-backend = "setuptools.build_meta"
+
+[project.urls]
+homepage = "https://huggingface.co/lerobot"
+documentation = "https://huggingface.co/docs/lerobot/index"
+source = "https://github.com/huggingface/lerobot"
+issues = "https://github.com/huggingface/lerobot/issues"
+discord = "https://discord.gg/s3KuuzsPFb"
+
+[project]
+name = "lerobot"
+version = "0.5.1"
+description = "🤗 LeRobot: State-of-the-art Machine Learning for Real-World Robotics in Pytorch"
+dynamic = ["readme"]
+license = { text = "Apache-2.0" }
+requires-python = ">=3.12"
+authors = [
+    { name = "Rémi Cadène", email = "re.cadene@gmail.com" },
+    { name = "Simon Alibert", email = "alibert.sim@gmail.com" },
+    { name = "Alexander Soare", email = "alexander.soare159@gmail.com" },
+    { name = "Quentin Gallouédec", email = "quentin.gallouedec@ec-lyon.fr" },
+    { name = "Steven Palma", email = "imstevenpmwork@ieee.org" },
+    { name = "Pepijn Kooijmans", email = "pepijnkooijmans@outlook.com"},
+    { name = "Michel Aractingi", email = "michel.aractingi@gmail.com"},
+    { name = "Adil Zouitine", email = "adilzouitinegm@gmail.com" },
+    { name = "Dana Aubakirova", email = "danaaubakirova17@gmail.com"},
+    { name = "Caroline Pascal", email = "caroline8.pascal@gmail.com"},
+    { name = "Martino Russi", email = "nopyeps@gmail.com"},
+    { name = "Thomas Wolf", email = "thomaswolfcontact@gmail.com" },
+]
+classifiers = [
+    "Development Status :: 3 - Alpha",
+    "Intended Audience :: Developers",
+    "Intended Audience :: Education",
+    "Intended Audience :: Science/Research",
+    "License :: OSI Approved :: Apache Software License",
+    "Programming Language :: Python :: 3.12",
+    "Programming Language :: Python :: 3.13",
+    "Topic :: Software Development :: Build Tools",
+    "Topic :: Scientific/Engineering :: Artificial Intelligence",
+]
+keywords = ["lerobot", "huggingface", "robotics",  "machine learning", "artificial intelligence"]
+
+dependencies = [
+
+    # Hugging Face dependencies
+    "datasets>=4.0.0,<5.0.0",
+    "diffusers>=0.27.2,<0.36.0",
+    "huggingface-hub>=1.0.0,<2.0.0",
+    "accelerate>=1.10.0,<2.0.0",
+
+    # Core dependencies
+    "numpy>=2.0.0,<2.3.0", # NOTE: Explicitly listing numpy helps the resolver converge faster. Upper bound imposed by opencv-python-headless.
+    "setuptools>=71.0.0,<81.0.0",
+    "cmake>=3.29.0.1,<4.2.0",
+    "packaging>=24.2,<26.0",
+
+    "torch>=2.2.1,<2.11.0",
+    "torchcodec>=0.2.1,<0.11.0; sys_platform != 'win32' and (sys_platform != 'linux' or (platform_machine != 'aarch64' and platform_machine != 'arm64' and platform_machine != 'armv7l')) and (sys_platform != 'darwin' or platform_machine != 'x86_64')",
+    "torchvision>=0.21.0,<0.26.0",
+
+    "einops>=0.8.0,<0.9.0",
+    "opencv-python-headless>=4.9.0,<4.14.0",
+    "av>=15.0.0,<16.0.0",
+    "jsonlines>=4.0.0,<5.0.0",
+    "pynput>=1.7.8,<1.9.0",
+    "pyserial>=3.5,<4.0",
+
+    "wandb>=0.24.0,<0.25.0",
+    "draccus==0.10.0", # TODO: Relax version constraint
+    "gymnasium>=1.1.1,<2.0.0",
+    "rerun-sdk>=0.24.0,<0.27.0",
+
+    # Support dependencies
+    "deepdiff>=7.0.1,<9.0.0",
+    "imageio[ffmpeg]>=2.34.0,<3.0.0",
+    "termcolor>=2.4.0,<4.0.0",
+]
+
+# Optional dependencies
+[project.optional-dependencies]
+
+# Common
+pygame-dep = ["pygame>=2.5.1,<2.7.0"]
+placo-dep = ["placo>=0.9.6,<0.9.17"]
+transformers-dep = ["transformers>=5.3.0,<6.0.0"]
+grpcio-dep = ["grpcio==1.73.1", "protobuf>=6.31.1,<6.32.0"]
+can-dep = ["python-can>=4.2.0,<5.0.0"]
+peft-dep = ["peft>=0.18.0,<1.0.0"]
+scipy-dep = ["scipy>=1.14.0,<2.0.0"]
+qwen-vl-utils-dep = ["qwen-vl-utils>=0.0.11,<0.1.0"]
+matplotlib-dep = ["matplotlib>=3.10.3,<4.0.0", "contourpy>=1.3.0,<2.0.0"] # NOTE: Explicitly listing contourpy helps the resolver converge faster.
+
+# Motors
+feetech = ["feetech-servo-sdk>=1.0.0,<2.0.0"]
+dynamixel = ["dynamixel-sdk>=3.7.31,<3.9.0"]
+damiao = ["lerobot[can-dep]"]
+robstride = ["lerobot[can-dep]"]
+
+# Robots
+openarms = ["lerobot[damiao]"]
+gamepad = ["lerobot[pygame-dep]", "hidapi>=0.14.0,<0.15.0"]
+hopejr = ["lerobot[feetech]", "lerobot[pygame-dep]"]
+lekiwi = ["lerobot[feetech]", "pyzmq>=26.2.1,<28.0.0"]
+unitree_g1 = [
+    # "unitree-sdk2==1.0.1",
+    "pyzmq>=26.2.1,<28.0.0",
+    "onnxruntime>=1.16.0,<2.0.0",
+    "onnx>=1.16.0,<2.0.0",
+    "meshcat>=0.3.0,<0.4.0",
+    "lerobot[matplotlib-dep]",
+    "lerobot[pygame-dep]",
+]
+reachy2 = ["reachy2_sdk>=1.0.15,<1.1.0"]
+kinematics = ["lerobot[placo-dep]"]
+intelrealsense = [
+    "pyrealsense2>=2.55.1.6486,<2.57.0 ; sys_platform != 'darwin'",
+    "pyrealsense2-macosx>=2.54,<2.57.0 ; sys_platform == 'darwin'",
+]
+phone = ["hebi-py>=2.8.0,<2.12.0", "teleop>=0.1.0,<0.2.0", "fastapi<1.0", "lerobot[scipy-dep]"]
+
+# Policies
+wallx = [
+    "lerobot[transformers-dep]",
+    "lerobot[peft]",
+    "lerobot[scipy-dep]",
+    "torchdiffeq>=0.2.4,<0.3.0",
+    "lerobot[qwen-vl-utils-dep]",
+]
+pi = ["lerobot[transformers-dep]", "lerobot[scipy-dep]"]
+smolvla = ["lerobot[transformers-dep]", "num2words>=0.5.14,<0.6.0", "accelerate>=1.7.0,<2.0.0", "safetensors>=0.4.3,<1.0.0"]
+groot = [
+    "lerobot[transformers-dep]",
+    "lerobot[peft]",
+    "dm-tree>=0.1.8,<1.0.0",
+    "timm>=1.0.0,<1.1.0",
+    "safetensors>=0.4.3,<1.0.0",
+    "Pillow>=10.0.0,<13.0.0",
+    "decord>=0.6.0,<1.0.0; (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
+    "ninja>=1.11.1,<2.0.0",
+    "flash-attn>=2.5.9,<3.0.0 ; sys_platform != 'darwin'"
+]
+sarm = ["lerobot[transformers-dep]", "faker>=33.0.0,<35.0.0", "lerobot[matplotlib-dep]", "lerobot[qwen-vl-utils-dep]"]
+xvla = ["lerobot[transformers-dep]"]
+hilserl = ["lerobot[transformers-dep]", "gym-hil>=0.1.13,<0.2.0", "lerobot[grpcio-dep]", "lerobot[placo-dep]"]
+
+# Features
+async = ["lerobot[grpcio-dep]", "lerobot[matplotlib-dep]"]
+peft = ["lerobot[transformers-dep]", "lerobot[peft-dep]"]
+
+# Development
+dev = ["pre-commit>=3.7.0,<5.0.0", "debugpy>=1.8.1,<1.9.0", "lerobot[grpcio-dep]", "grpcio-tools==1.73.1", "mypy>=1.19.1"]
+test = ["pytest>=8.1.0,<9.0.0", "pytest-timeout>=2.4.0,<3.0.0", "pytest-cov>=5.0.0,<8.0.0", "mock-serial>=0.0.1,<0.1.0 ; sys_platform != 'win32'"]
+video_benchmark = ["scikit-image>=0.23.2,<0.26.0", "pandas>=2.2.2,<2.4.0"]
+
+# Simulation
+# NOTE: Explicitly listing scipy helps flatten the dependecy tree.
+aloha = ["gym-aloha>=0.1.2,<0.2.0", "lerobot[scipy-dep]"]
+pusht = ["gym-pusht>=0.1.5,<0.2.0", "pymunk>=6.6.0,<7.0.0"] # TODO: Fix pymunk version in gym-pusht instead
+libero = ["lerobot[transformers-dep]", "hf-libero>=0.1.3,<0.2.0; sys_platform == 'linux'", "lerobot[scipy-dep]"]
+metaworld = ["metaworld==3.0.0", "lerobot[scipy-dep]"]
+
+# All
+all = [
+    # NOTE(resolver hint): scipy is pulled in transitively via lerobot[scipy-dep] through
+    # multiple extras (aloha, metaworld, pi, wallx, phone). Listing it explicitly
+    # helps pip's resolver converge by constraining scipy early, before it encounters
+    # the loose scipy requirements from transitive deps like dm-control and metaworld.
+    "scipy>=1.14.0,<2.0.0",
+    "lerobot[dynamixel]",
+    "lerobot[gamepad]",
+    "lerobot[hopejr]",
+    "lerobot[lekiwi]",
+    "lerobot[reachy2]",
+    "lerobot[kinematics]",
+    "lerobot[intelrealsense]",
+    "lerobot[wallx]",
+    "lerobot[pi]",
+    "lerobot[smolvla]",
+    # "lerobot[groot]", TODO(Steven): Gr00t requires specific installation instructions for flash-attn
+    "lerobot[xvla]",
+    "lerobot[hilserl]",
+    "lerobot[async]",
+    "lerobot[dev]",
+    "lerobot[test]",
+    "lerobot[video_benchmark]",
+    "lerobot[aloha]",
+    "lerobot[pusht]",
+    "lerobot[phone]",
+    "lerobot[libero]; sys_platform == 'linux'",
+    "lerobot[metaworld]",
+    "lerobot[sarm]",
+    "lerobot[peft]",
+    # "lerobot[unitree_g1]", TODO: Unitree requires specific installation instructions for unitree_sdk2
+]
+
+[project.scripts]
+lerobot-calibrate="lerobot.scripts.lerobot_calibrate:main"
+lerobot-find-cameras="lerobot.scripts.lerobot_find_cameras:main"
+lerobot-find-port="lerobot.scripts.lerobot_find_port:main"
+lerobot-record="lerobot.scripts.lerobot_record:main"
+lerobot-replay="lerobot.scripts.lerobot_replay:main"
+lerobot-setup-motors="lerobot.scripts.lerobot_setup_motors:main"
+lerobot-teleoperate="lerobot.scripts.lerobot_teleoperate:main"
+lerobot-eval="lerobot.scripts.lerobot_eval:main"
+lerobot-train="lerobot.scripts.lerobot_train:main"
+lerobot-train-tokenizer="lerobot.scripts.lerobot_train_tokenizer:main"
+lerobot-dataset-viz="lerobot.scripts.lerobot_dataset_viz:main"
+lerobot-info="lerobot.scripts.lerobot_info:main"
+lerobot-find-joint-limits="lerobot.scripts.lerobot_find_joint_limits:main"
+lerobot-imgtransform-viz="lerobot.scripts.lerobot_imgtransform_viz:main"
+lerobot-edit-dataset="lerobot.scripts.lerobot_edit_dataset:main"
+lerobot-setup-can="lerobot.scripts.lerobot_setup_can:main"
+
+# ---------------- Tool Configurations ----------------
+[tool.setuptools.package-data]
+lerobot = ["envs/*.json"]
+
+[tool.setuptools.packages.find]
+where = ["src"]
+
+[tool.ruff]
+target-version = "py312"
+line-length = 110
+exclude = ["tests/artifacts/**/*.safetensors", "*_pb2.py", "*_pb2_grpc.py"]
+
+[tool.ruff.lint]
+# E, W: pycodestyle errors and warnings
+# F: PyFlakes
+# I: isort
+# UP: pyupgrade
+# B: flake8-bugbear (good practices, potential bugs)
+# C4: flake8-comprehensions (more concise comprehensions)
+# A: flake8-builtins (shadowing builtins)
+# SIM: flake8-simplify
+# RUF: Ruff-specific rules
+# D: pydocstyle (for docstring style/formatting)
+# S: flake8-bandit (some security checks, complements Bandit)
+# T20: flake8-print (discourage print statements in production code)
+# N: pep8-naming
+# TODO: Uncomment rules when ready to use
+select = [
+    "E", "W", "F", "I", "B", "C4", "T20", "N", "UP", "SIM" #, "A", "S", "D", "RUF"
+]
+ignore = [
+    "E501", # Line too long
+    "T201", # Print statement found
+    "T203", # Pprint statement found
+    "B008", # Perform function call in argument defaults
+]
+
+[tool.ruff.lint.per-file-ignores]
+"__init__.py" = ["F401", "F403"]
+"src/lerobot/policies/wall_x/**" = ["N801", "N812", "SIM102", "SIM108", "SIM210", "SIM211", "B006", "B007", "SIM118"] # Supprese these as they are coming from original Qwen2_5_vl code TODO(pepijn): refactor original
+
+[tool.ruff.lint.isort]
+combine-as-imports = true
+known-first-party = ["lerobot"]
+
+[tool.ruff.lint.pydocstyle]
+convention = "google"
+
+[tool.ruff.format]
+quote-style = "double"
+indent-style = "space"
+skip-magic-trailing-comma = false
+line-ending = "auto"
+docstring-code-format = true
+
+[tool.bandit]
+exclude_dirs = [
+    "tests",
+    "benchmarks",
+    "src/lerobot/datasets/push_dataset_to_hub",
+]
+skips = ["B101", "B311", "B404", "B603", "B615"]
+
+[tool.typos]
+default.extend-ignore-re = [
+    "(?Rm)^.*(#|//)\\s*spellchecker:disable-line$",                      # spellchecker:disable-line
+    "(?s)(#|//)\\s*spellchecker:off.*?\\n\\s*(#|//)\\s*spellchecker:on", # spellchecker:<on|off>
+]
+default.extend-ignore-identifiers-re = [
+    # Add individual words here to ignore them
+    "2nd",
+    "pn",
+    "ser",
+    "ein",
+    "thw",
+    "inpt",
+    "ROBOTIS",
+    "OT_VALUE"
+]
+
+# TODO: Uncomment when ready to use
+# [tool.interrogate]
+# ignore-init-module = true
+# ignore-init-method = true
+# ignore-nested-functions = false
+# ignore-magic = false
+# ignore-semiprivate = false
+# ignore-private = false
+# ignore-property-decorators = false
+# ignore-module = false
+# ignore-setters = false
+# fail-under = 80
+# output-format = "term-missing"
+# color = true
+# paths = ["src/lerobot"]
+
+# TODO: Enable mypy gradually module by module across multiple PRs
+# Uncomment [tool.mypy] first, then uncomment individual module overrides as they get proper type annotations
+
+[tool.mypy]
+python_version = "3.12"
+ignore_missing_imports = true
+follow_imports = "skip"
+# warn_return_any = true
+# warn_unused_configs = true
+# strict = true
+# disallow_untyped_defs = true
+# disallow_incomplete_defs = true
+# check_untyped_defs = true
+
+[[tool.mypy.overrides]]
+module = "lerobot.*"
+ignore_errors = true
+
+[[tool.mypy.overrides]]
+module = "lerobot.envs.*"
+ignore_errors = false
+
+
+# [[tool.mypy.overrides]]
+# module = "lerobot.utils.*"
+# ignore_errors = false
+
+[[tool.mypy.overrides]]
+module = "lerobot.configs.*"
+ignore_errors = false
+
+# extra strictness for configs
+disallow_untyped_defs = true
+disallow_incomplete_defs = true
+check_untyped_defs = true
+
+[[tool.mypy.overrides]]
+module = "lerobot.optim.*"
+ignore_errors = false
+
+[[tool.mypy.overrides]]
+module = "lerobot.model.*"
+ignore_errors = false
+
+# [[tool.mypy.overrides]]
+# module = "lerobot.processor.*"
+# ignore_errors = false
+
+# [[tool.mypy.overrides]]
+# module = "lerobot.datasets.*"
+# ignore_errors = false
+
+[[tool.mypy.overrides]]
+module = "lerobot.cameras.*"
+ignore_errors = false
+
+[[tool.mypy.overrides]]
+module = "lerobot.motors.*"
+ignore_errors = false
+
+# [[tool.mypy.overrides]]
+# module = "lerobot.robots.*"
+# ignore_errors = false
+
+# [[tool.mypy.overrides]]
+# module = "lerobot.teleoperators.*"
+# ignore_errors = false
+
+# [[tool.mypy.overrides]]
+# module = "lerobot.policies.*"
+# ignore_errors = false
+
+# [[tool.mypy.overrides]]
+# module = "lerobot.rl.*"
+# ignore_errors = false
+
+
+# [[tool.mypy.overrides]]
+# module = "lerobot.async_inference.*"
+# ignore_errors = false
+
+[[tool.mypy.overrides]]
+module = "lerobot.transport.*"
+ignore_errors = false
+
+# [[tool.mypy.overrides]]
+# module = "lerobot.scripts.*"
+# ignore_errors = false
diff --git a/lerobot/requirements-macos.txt b/lerobot/requirements-macos.txt
new file mode 100644
index 0000000000000000000000000000000000000000..c5bbe1c8aff611964f63b28f56e26efd194aeb46
--- /dev/null
+++ b/lerobot/requirements-macos.txt
@@ -0,0 +1,729 @@
+#
+# This file is autogenerated by pip-compile with Python 3.12
+# by the following command:
+#
+#    pip-compile --output-file=requirements-macos.txt requirements.in
+#
+-e .[all]
+    # via -[all]
+absl-py==2.4.0
+    # via
+    #   dm-control
+    #   dm-env
+    #   dm-tree
+    #   labmaze
+    #   mujoco
+accelerate==1.13.0
+    # via
+    #   lerobot
+    #   peft
+aiohappyeyeballs==2.6.1
+    # via aiohttp
+aiohttp==3.13.3
+    # via fsspec
+aiosignal==1.4.0
+    # via aiohttp
+annotated-doc==0.0.4
+    # via
+    #   fastapi
+    #   typer
+annotated-types==0.7.0
+    # via pydantic
+anyio==4.12.1
+    # via
+    #   httpx
+    #   starlette
+    #   watchfiles
+asttokens==3.0.1
+    # via stack-data
+attrs==25.4.0
+    # via
+    #   aiohttp
+    #   dm-tree
+    #   jsonlines
+    #   rerun-sdk
+av==15.1.0
+    # via
+    #   lerobot
+    #   qwen-vl-utils
+certifi==2026.2.25
+    # via
+    #   httpcore
+    #   httpx
+    #   requests
+    #   sentry-sdk
+cffi==2.0.0
+    # via pymunk
+cfgv==3.5.0
+    # via pre-commit
+charset-normalizer==3.4.5
+    # via requests
+click==8.3.1
+    # via
+    #   typer
+    #   uvicorn
+    #   wandb
+cloudpickle==3.1.2
+    # via gymnasium
+cmake==4.1.3
+    # via lerobot
+cmeel==0.59.0
+    # via
+    #   cmeel-assimp
+    #   cmeel-boost
+    #   cmeel-console-bridge
+    #   cmeel-octomap
+    #   cmeel-qhull
+    #   cmeel-tinyxml2
+    #   cmeel-urdfdom
+    #   cmeel-zlib
+    #   coal-library
+    #   eigenpy
+    #   eiquadprog
+    #   pin
+    #   placo
+    #   rhoban-cmeel-jsoncpp
+cmeel-assimp==5.4.3.1
+    # via coal-library
+cmeel-boost==1.87.0.1
+    # via
+    #   coal-library
+    #   eigenpy
+    #   eiquadprog
+    #   pin
+cmeel-console-bridge==1.0.2.3
+    # via cmeel-urdfdom
+cmeel-octomap==1.10.0
+    # via coal-library
+cmeel-qhull==8.0.2.1
+    # via coal-library
+cmeel-tinyxml2==10.0.0
+    # via cmeel-urdfdom
+cmeel-urdfdom==4.0.1
+    # via pin
+cmeel-zlib==1.3.1
+    # via cmeel-assimp
+coal-library==3.0.1
+    # via pin
+contourpy==1.3.3
+    # via
+    #   lerobot
+    #   matplotlib
+coverage[toml]==7.13.4
+    # via pytest-cov
+cycler==0.12.1
+    # via matplotlib
+datasets==4.6.1
+    # via lerobot
+debugpy==1.8.20
+    # via lerobot
+decorator==5.2.1
+    # via ipython
+deepdiff==8.6.1
+    # via lerobot
+diffusers==0.35.2
+    # via lerobot
+dill==0.4.0
+    # via
+    #   datasets
+    #   multiprocess
+distlib==0.4.0
+    # via virtualenv
+dm-control==1.0.37
+    # via gym-aloha
+dm-env==1.6
+    # via dm-control
+dm-tree==0.1.9
+    # via
+    #   dm-control
+    #   dm-env
+docopt==0.6.2
+    # via num2words
+draccus==0.10.0
+    # via lerobot
+dynamixel-sdk==3.8.4
+    # via lerobot
+eigenpy==3.10.3
+    # via coal-library
+einops==0.8.2
+    # via lerobot
+eiquadprog==1.2.9
+    # via placo
+etils[epath,epy]==1.14.0
+    # via mujoco
+executing==2.2.1
+    # via stack-data
+faker==34.0.2
+    # via lerobot
+farama-notifications==0.0.4
+    # via gymnasium
+fastapi==0.135.1
+    # via
+    #   lerobot
+    #   teleop
+feetech-servo-sdk==1.0.0
+    # via lerobot
+filelock==3.25.0
+    # via
+    #   datasets
+    #   diffusers
+    #   huggingface-hub
+    #   python-discovery
+    #   torch
+    #   virtualenv
+fonttools==4.61.1
+    # via matplotlib
+frozenlist==1.8.0
+    # via
+    #   aiohttp
+    #   aiosignal
+fsspec[http]==2026.2.0
+    # via
+    #   datasets
+    #   etils
+    #   huggingface-hub
+    #   torch
+gitdb==4.0.12
+    # via gitpython
+gitpython==3.1.46
+    # via wandb
+glfw==2.10.0
+    # via
+    #   dm-control
+    #   mujoco
+grpcio==1.73.1
+    # via
+    #   grpcio-tools
+    #   lerobot
+    #   reachy2-sdk
+    #   reachy2-sdk-api
+grpcio-tools==1.73.1
+    # via
+    #   lerobot
+    #   reachy2-sdk-api
+gym-aloha==0.1.3
+    # via lerobot
+gym-hil==0.1.13
+    # via lerobot
+gym-pusht==0.1.6
+    # via lerobot
+gymnasium==1.2.3
+    # via
+    #   gym-aloha
+    #   gym-hil
+    #   gym-pusht
+    #   lerobot
+    #   metaworld
+h11==0.16.0
+    # via
+    #   httpcore
+    #   uvicorn
+hebi-py==2.11.0
+    # via lerobot
+hf-xet==1.3.2
+    # via huggingface-hub
+hidapi==0.14.0.post4
+    # via
+    #   gym-hil
+    #   lerobot
+httpcore==1.0.9
+    # via httpx
+httptools==0.7.1
+    # via uvicorn
+httpx==0.28.1
+    # via
+    #   datasets
+    #   huggingface-hub
+huggingface-hub==1.6.0
+    # via
+    #   accelerate
+    #   datasets
+    #   diffusers
+    #   lerobot
+    #   peft
+    #   tokenizers
+    #   transformers
+identify==2.6.17
+    # via pre-commit
+idna==3.11
+    # via
+    #   anyio
+    #   httpx
+    #   requests
+    #   yarl
+imageio[ffmpeg]==2.37.2
+    # via
+    #   gym-aloha
+    #   gym-hil
+    #   lerobot
+    #   metaworld
+    #   scikit-image
+imageio-ffmpeg==0.6.0
+    # via imageio
+importlib-metadata==8.7.1
+    # via diffusers
+iniconfig==2.3.0
+    # via pytest
+ipython==9.11.0
+    # via meshcat
+ipython-pygments-lexers==1.1.1
+    # via ipython
+ischedule==1.2.7
+    # via placo
+jedi==0.19.2
+    # via ipython
+jinja2==3.1.6
+    # via torch
+jsonlines==4.0.0
+    # via lerobot
+kiwisolver==1.4.9
+    # via matplotlib
+labmaze==1.0.6
+    # via dm-control
+lazy-loader==0.5
+    # via scikit-image
+librt==0.8.1
+    # via mypy
+lxml==6.0.2
+    # via dm-control
+markdown-it-py==4.0.0
+    # via rich
+markupsafe==3.0.3
+    # via jinja2
+matplotlib==3.10.8
+    # via lerobot
+matplotlib-inline==0.2.1
+    # via ipython
+mdurl==0.1.2
+    # via markdown-it-py
+mergedeep==1.3.4
+    # via draccus
+meshcat==0.3.2
+    # via placo
+metaworld==3.0.0
+    # via lerobot
+mock-serial==0.0.1
+    # via lerobot
+mpmath==1.3.0
+    # via sympy
+mujoco==3.5.0
+    # via
+    #   dm-control
+    #   gym-aloha
+    #   gym-hil
+    #   metaworld
+multidict==6.7.1
+    # via
+    #   aiohttp
+    #   yarl
+multiprocess==0.70.18
+    # via datasets
+mypy==1.19.1
+    # via lerobot
+mypy-extensions==1.1.0
+    # via
+    #   mypy
+    #   typing-inspect
+networkx==3.6.1
+    # via
+    #   scikit-image
+    #   torch
+nodeenv==1.10.0
+    # via pre-commit
+num2words==0.5.14
+    # via lerobot
+numpy==2.2.6
+    # via
+    #   accelerate
+    #   cmeel-boost
+    #   contourpy
+    #   datasets
+    #   diffusers
+    #   dm-control
+    #   dm-env
+    #   dm-tree
+    #   gymnasium
+    #   hebi-py
+    #   imageio
+    #   labmaze
+    #   lerobot
+    #   matplotlib
+    #   meshcat
+    #   metaworld
+    #   mujoco
+    #   opencv-python
+    #   opencv-python-headless
+    #   pandas
+    #   peft
+    #   pyquaternion
+    #   reachy2-sdk
+    #   rerun-sdk
+    #   scikit-image
+    #   scipy
+    #   shapely
+    #   teleop
+    #   tifffile
+    #   torchvision
+    #   transformers
+    #   transforms3d
+opencv-python==4.13.0.92
+    # via
+    #   gym-pusht
+    #   reachy2-sdk
+opencv-python-headless==4.12.0.88
+    # via lerobot
+orderly-set==5.5.0
+    # via deepdiff
+packaging==25.0
+    # via
+    #   accelerate
+    #   datasets
+    #   huggingface-hub
+    #   lazy-loader
+    #   lerobot
+    #   matplotlib
+    #   peft
+    #   pytest
+    #   qwen-vl-utils
+    #   reachy2-sdk
+    #   scikit-image
+    #   transformers
+    #   wandb
+pandas==2.3.3
+    # via
+    #   datasets
+    #   lerobot
+parso==0.8.6
+    # via jedi
+pathspec==1.0.4
+    # via mypy
+peft==0.18.1
+    # via lerobot
+pexpect==4.9.0
+    # via ipython
+pillow==12.1.1
+    # via
+    #   diffusers
+    #   imageio
+    #   matplotlib
+    #   meshcat
+    #   qwen-vl-utils
+    #   rerun-sdk
+    #   scikit-image
+    #   torchvision
+pin==3.4.0
+    # via placo
+placo==0.9.16
+    # via lerobot
+platformdirs==4.9.4
+    # via
+    #   python-discovery
+    #   virtualenv
+    #   wandb
+pluggy==1.6.0
+    # via
+    #   pytest
+    #   pytest-cov
+pre-commit==4.5.1
+    # via lerobot
+prompt-toolkit==3.0.52
+    # via ipython
+propcache==0.4.1
+    # via
+    #   aiohttp
+    #   yarl
+protobuf==6.31.1
+    # via
+    #   dm-control
+    #   grpcio-tools
+    #   lerobot
+    #   reachy2-sdk
+    #   reachy2-sdk-api
+    #   wandb
+psutil==7.2.2
+    # via
+    #   accelerate
+    #   imageio
+    #   peft
+ptyprocess==0.7.0
+    # via pexpect
+pure-eval==0.2.3
+    # via stack-data
+pyarrow==23.0.1
+    # via
+    #   datasets
+    #   rerun-sdk
+pycparser==3.0
+    # via cffi
+pydantic==2.12.5
+    # via
+    #   fastapi
+    #   wandb
+pydantic-core==2.41.5
+    # via pydantic
+pygame==2.6.1
+    # via
+    #   gym-hil
+    #   gym-pusht
+    #   lerobot
+pygments==2.19.2
+    # via
+    #   ipython
+    #   ipython-pygments-lexers
+    #   pytest
+    #   rich
+pymunk==6.11.1
+    # via
+    #   gym-pusht
+    #   lerobot
+pyngrok==7.5.1
+    # via meshcat
+pynput==1.8.1
+    # via
+    #   gym-hil
+    #   lerobot
+pyobjc-core==12.1
+    # via
+    #   pyobjc-framework-applicationservices
+    #   pyobjc-framework-cocoa
+    #   pyobjc-framework-coretext
+    #   pyobjc-framework-quartz
+pyobjc-framework-applicationservices==12.1
+    # via pynput
+pyobjc-framework-cocoa==12.1
+    # via
+    #   pyobjc-framework-applicationservices
+    #   pyobjc-framework-coretext
+    #   pyobjc-framework-quartz
+pyobjc-framework-coretext==12.1
+    # via pyobjc-framework-applicationservices
+pyobjc-framework-quartz==12.1
+    # via
+    #   pynput
+    #   pyobjc-framework-applicationservices
+    #   pyobjc-framework-coretext
+pyopengl==3.1.10
+    # via
+    #   dm-control
+    #   mujoco
+pyparsing==3.3.2
+    # via
+    #   dm-control
+    #   matplotlib
+pyquaternion==0.9.9
+    # via reachy2-sdk
+pyrealsense2-macosx==2.56.5
+    # via lerobot
+pyserial==3.5
+    # via
+    #   dynamixel-sdk
+    #   feetech-servo-sdk
+    #   lerobot
+pytest==8.4.2
+    # via
+    #   lerobot
+    #   pytest-cov
+    #   pytest-timeout
+    #   teleop
+pytest-cov==7.0.0
+    # via lerobot
+pytest-timeout==2.4.0
+    # via lerobot
+python-dateutil==2.9.0.post0
+    # via
+    #   faker
+    #   matplotlib
+    #   pandas
+python-discovery==1.1.1
+    # via virtualenv
+python-dotenv==1.2.2
+    # via uvicorn
+pytz==2026.1.post1
+    # via pandas
+pyyaml==6.0.3
+    # via
+    #   accelerate
+    #   datasets
+    #   draccus
+    #   hebi-py
+    #   huggingface-hub
+    #   peft
+    #   pre-commit
+    #   pyngrok
+    #   pyyaml-include
+    #   transformers
+    #   uvicorn
+    #   wandb
+pyyaml-include==1.4.1
+    # via draccus
+pyzmq==27.1.0
+    # via
+    #   lerobot
+    #   meshcat
+qwen-vl-utils==0.0.14
+    # via lerobot
+reachy2-sdk==1.0.15
+    # via lerobot
+reachy2-sdk-api==1.0.21
+    # via reachy2-sdk
+regex==2026.2.28
+    # via
+    #   diffusers
+    #   transformers
+requests==2.32.5
+    # via
+    #   datasets
+    #   diffusers
+    #   dm-control
+    #   qwen-vl-utils
+    #   teleop
+    #   wandb
+rerun-sdk==0.26.2
+    # via lerobot
+rhoban-cmeel-jsoncpp==1.9.4.9
+    # via placo
+rich==14.3.3
+    # via typer
+safetensors==0.7.0
+    # via
+    #   accelerate
+    #   diffusers
+    #   lerobot
+    #   peft
+    #   transformers
+scikit-image==0.25.2
+    # via
+    #   gym-pusht
+    #   lerobot
+scipy==1.17.1
+    # via
+    #   dm-control
+    #   lerobot
+    #   metaworld
+    #   scikit-image
+    #   torchdiffeq
+sentry-sdk==2.54.0
+    # via wandb
+shapely==2.1.2
+    # via gym-pusht
+shellingham==1.5.4
+    # via typer
+six==1.17.0
+    # via
+    #   pynput
+    #   python-dateutil
+smmap==5.0.3
+    # via gitdb
+stack-data==0.6.3
+    # via ipython
+starlette==0.52.1
+    # via fastapi
+sympy==1.14.0
+    # via torch
+teleop==0.1.4
+    # via lerobot
+termcolor==3.3.0
+    # via lerobot
+tifffile==2026.3.3
+    # via scikit-image
+tokenizers==0.22.2
+    # via transformers
+toml==0.10.2
+    # via draccus
+torch==2.10.0
+    # via
+    #   accelerate
+    #   lerobot
+    #   peft
+    #   torchdiffeq
+    #   torchvision
+torchcodec==0.10.0
+    # via lerobot
+torchdiffeq==0.2.5
+    # via lerobot
+torchvision==0.25.0
+    # via lerobot
+tornado==6.5.4
+    # via meshcat
+tqdm==4.67.3
+    # via
+    #   datasets
+    #   dm-control
+    #   huggingface-hub
+    #   peft
+    #   transformers
+traitlets==5.14.3
+    # via
+    #   ipython
+    #   matplotlib-inline
+transformers==5.3.0
+    # via
+    #   lerobot
+    #   peft
+transforms3d==0.4.2
+    # via teleop
+typer==0.24.1
+    # via
+    #   huggingface-hub
+    #   transformers
+typing-extensions==4.15.0
+    # via
+    #   aiosignal
+    #   anyio
+    #   etils
+    #   faker
+    #   fastapi
+    #   gymnasium
+    #   huggingface-hub
+    #   mypy
+    #   pydantic
+    #   pydantic-core
+    #   rerun-sdk
+    #   starlette
+    #   torch
+    #   typing-inspect
+    #   typing-inspection
+    #   wandb
+typing-inspect==0.9.0
+    # via draccus
+typing-inspection==0.4.2
+    # via
+    #   fastapi
+    #   pydantic
+tzdata==2025.3
+    # via pandas
+u-msgpack-python==2.8.0
+    # via meshcat
+urllib3==2.6.3
+    # via
+    #   requests
+    #   sentry-sdk
+uvicorn[standard]==0.41.0
+    # via teleop
+uvloop==0.22.1
+    # via uvicorn
+virtualenv==21.1.0
+    # via pre-commit
+wandb==0.24.2
+    # via lerobot
+watchfiles==1.1.1
+    # via uvicorn
+wcwidth==0.6.0
+    # via prompt-toolkit
+websocket-client==1.9.0
+    # via teleop
+websockets==16.0
+    # via uvicorn
+wrapt==2.1.2
+    # via dm-tree
+xxhash==3.6.0
+    # via datasets
+yarl==1.23.0
+    # via aiohttp
+zipp==3.23.0
+    # via
+    #   etils
+    #   importlib-metadata
+
+# The following packages are considered to be unsafe in a requirements file:
+# setuptools
diff --git a/lerobot/requirements-ubuntu.txt b/lerobot/requirements-ubuntu.txt
new file mode 100644
index 0000000000000000000000000000000000000000..0cdc541908931a68c69c0bf3076a8fffe4681fe9
--- /dev/null
+++ b/lerobot/requirements-ubuntu.txt
@@ -0,0 +1,882 @@
+#
+# This file is autogenerated by pip-compile with Python 3.12
+# by the following command:
+#
+#    pip-compile --output-file=requirements-ubuntu.txt requirements.in
+#
+-e .[all]
+    # via -[all]
+absl-py==2.4.0
+    # via
+    #   dm-control
+    #   dm-env
+    #   dm-tree
+    #   labmaze
+    #   mujoco
+    #   tensorboard
+accelerate==1.13.0
+    # via
+    #   lerobot
+    #   peft
+aiohappyeyeballs==2.6.1
+    # via aiohttp
+aiohttp==3.13.3
+    # via fsspec
+aiosignal==1.4.0
+    # via aiohttp
+annotated-doc==0.0.4
+    # via
+    #   fastapi
+    #   typer
+annotated-types==0.7.0
+    # via pydantic
+antlr4-python3-runtime==4.9.3
+    # via
+    #   hydra-core
+    #   omegaconf
+anyio==4.12.1
+    # via
+    #   httpx
+    #   starlette
+    #   watchfiles
+asttokens==3.0.1
+    # via stack-data
+attrs==25.4.0
+    # via
+    #   aiohttp
+    #   dm-tree
+    #   jsonlines
+    #   jsonschema
+    #   referencing
+    #   rerun-sdk
+av==15.1.0
+    # via
+    #   lerobot
+    #   qwen-vl-utils
+bddl==1.0.1
+    # via hf-libero
+certifi==2026.2.25
+    # via
+    #   httpcore
+    #   httpx
+    #   requests
+    #   sentry-sdk
+cffi==2.0.0
+    # via pymunk
+cfgv==3.5.0
+    # via pre-commit
+charset-normalizer==3.4.5
+    # via requests
+click==8.3.1
+    # via
+    #   typer
+    #   uvicorn
+    #   wandb
+cloudpickle==3.1.2
+    # via
+    #   gymnasium
+    #   hf-libero
+cmake==4.1.3
+    # via lerobot
+cmeel==0.59.0
+    # via
+    #   cmeel-assimp
+    #   cmeel-boost
+    #   cmeel-console-bridge
+    #   cmeel-octomap
+    #   cmeel-qhull
+    #   cmeel-tinyxml2
+    #   cmeel-urdfdom
+    #   cmeel-zlib
+    #   coal-library
+    #   eigenpy
+    #   eiquadprog
+    #   pin
+    #   placo
+    #   rhoban-cmeel-jsoncpp
+cmeel-assimp==5.4.3.1
+    # via coal-library
+cmeel-boost==1.87.0.1
+    # via
+    #   coal-library
+    #   eigenpy
+    #   eiquadprog
+    #   pin
+cmeel-console-bridge==1.0.2.3
+    # via cmeel-urdfdom
+cmeel-octomap==1.10.0
+    # via coal-library
+cmeel-qhull==8.0.2.1
+    # via coal-library
+cmeel-tinyxml2==10.0.0
+    # via cmeel-urdfdom
+cmeel-urdfdom==4.0.1
+    # via pin
+cmeel-zlib==1.3.1
+    # via cmeel-assimp
+coal-library==3.0.1
+    # via pin
+contourpy==1.3.3
+    # via
+    #   lerobot
+    #   matplotlib
+coverage[toml]==7.13.4
+    # via pytest-cov
+cuda-bindings==12.9.4
+    # via torch
+cuda-pathfinder==1.4.1
+    # via cuda-bindings
+cycler==0.12.1
+    # via matplotlib
+datasets==4.6.1
+    # via lerobot
+debugpy==1.8.20
+    # via lerobot
+decorator==5.2.1
+    # via ipython
+deepdiff==8.6.1
+    # via lerobot
+diffusers==0.35.2
+    # via lerobot
+dill==0.4.0
+    # via
+    #   datasets
+    #   multiprocess
+distlib==0.4.0
+    # via virtualenv
+dm-control==1.0.37
+    # via gym-aloha
+dm-env==1.6
+    # via dm-control
+dm-tree==0.1.9
+    # via
+    #   dm-control
+    #   dm-env
+docopt==0.6.2
+    # via num2words
+draccus==0.10.0
+    # via lerobot
+dynamixel-sdk==3.8.4
+    # via lerobot
+easydict==1.13
+    # via hf-libero
+egl-probe==1.0.2
+    # via robomimic
+eigenpy==3.10.3
+    # via coal-library
+einops==0.8.2
+    # via
+    #   hf-libero
+    #   lerobot
+eiquadprog==1.2.9
+    # via placo
+etils[epath,epy]==1.14.0
+    # via mujoco
+evdev==1.9.3
+    # via pynput
+executing==2.2.1
+    # via stack-data
+faker==34.0.2
+    # via lerobot
+farama-notifications==0.0.4
+    # via gymnasium
+fastapi==0.135.1
+    # via
+    #   lerobot
+    #   teleop
+fastjsonschema==2.21.2
+    # via nbformat
+feetech-servo-sdk==1.0.0
+    # via lerobot
+filelock==3.25.0
+    # via
+    #   datasets
+    #   diffusers
+    #   huggingface-hub
+    #   python-discovery
+    #   torch
+    #   virtualenv
+fonttools==4.61.1
+    # via matplotlib
+frozenlist==1.8.0
+    # via
+    #   aiohttp
+    #   aiosignal
+fsspec[http]==2026.2.0
+    # via
+    #   datasets
+    #   etils
+    #   huggingface-hub
+    #   torch
+future==1.0.0
+    # via hf-libero
+gitdb==4.0.12
+    # via gitpython
+gitpython==3.1.46
+    # via wandb
+glfw==2.10.0
+    # via
+    #   dm-control
+    #   mujoco
+grpcio==1.73.1
+    # via
+    #   grpcio-tools
+    #   lerobot
+    #   reachy2-sdk
+    #   reachy2-sdk-api
+    #   tensorboard
+grpcio-tools==1.73.1
+    # via
+    #   lerobot
+    #   reachy2-sdk-api
+gym-aloha==0.1.3
+    # via lerobot
+gym-hil==0.1.13
+    # via lerobot
+gym-pusht==0.1.6
+    # via lerobot
+gymnasium==1.2.3
+    # via
+    #   gym-aloha
+    #   gym-hil
+    #   gym-pusht
+    #   hf-libero
+    #   lerobot
+    #   metaworld
+h11==0.16.0
+    # via
+    #   httpcore
+    #   uvicorn
+h5py==3.16.0
+    # via robomimic
+hebi-py==2.11.0
+    # via lerobot
+hf-egl-probe==1.0.2
+    # via hf-libero
+hf-libero==0.1.3
+    # via lerobot
+hf-xet==1.3.2
+    # via huggingface-hub
+hidapi==0.14.0.post4
+    # via
+    #   gym-hil
+    #   lerobot
+httpcore==1.0.9
+    # via httpx
+httptools==0.7.1
+    # via uvicorn
+httpx==0.28.1
+    # via
+    #   datasets
+    #   huggingface-hub
+huggingface-hub==1.6.0
+    # via
+    #   accelerate
+    #   datasets
+    #   diffusers
+    #   lerobot
+    #   peft
+    #   tokenizers
+    #   transformers
+hydra-core==1.3.2
+    # via hf-libero
+identify==2.6.17
+    # via pre-commit
+idna==3.11
+    # via
+    #   anyio
+    #   httpx
+    #   requests
+    #   yarl
+imageio[ffmpeg]==2.37.2
+    # via
+    #   gym-aloha
+    #   gym-hil
+    #   lerobot
+    #   metaworld
+    #   robomimic
+    #   scikit-image
+imageio-ffmpeg==0.6.0
+    # via
+    #   imageio
+    #   robomimic
+importlib-metadata==8.7.1
+    # via diffusers
+iniconfig==2.3.0
+    # via pytest
+ipython==9.11.0
+    # via meshcat
+ipython-pygments-lexers==1.1.1
+    # via ipython
+ischedule==1.2.7
+    # via placo
+jedi==0.19.2
+    # via ipython
+jinja2==3.1.6
+    # via torch
+jsonlines==4.0.0
+    # via lerobot
+jsonschema==4.26.0
+    # via nbformat
+jsonschema-specifications==2025.9.1
+    # via jsonschema
+jupyter-core==5.9.1
+    # via nbformat
+jupytext==1.19.1
+    # via bddl
+kiwisolver==1.4.9
+    # via matplotlib
+labmaze==1.0.6
+    # via dm-control
+lazy-loader==0.5
+    # via scikit-image
+librt==0.8.1
+    # via mypy
+llvmlite==0.46.0
+    # via numba
+lxml==6.0.2
+    # via dm-control
+markdown==3.10.2
+    # via tensorboard
+markdown-it-py==4.0.0
+    # via
+    #   jupytext
+    #   mdit-py-plugins
+    #   rich
+markupsafe==3.0.3
+    # via
+    #   jinja2
+    #   werkzeug
+matplotlib==3.10.8
+    # via
+    #   hf-libero
+    #   lerobot
+matplotlib-inline==0.2.1
+    # via ipython
+mdit-py-plugins==0.5.0
+    # via jupytext
+mdurl==0.1.2
+    # via markdown-it-py
+mergedeep==1.3.4
+    # via draccus
+meshcat==0.3.2
+    # via placo
+metaworld==3.0.0
+    # via lerobot
+mock-serial==0.0.1
+    # via lerobot
+mpmath==1.3.0
+    # via sympy
+mujoco==3.5.0
+    # via
+    #   dm-control
+    #   gym-aloha
+    #   gym-hil
+    #   hf-libero
+    #   metaworld
+    #   robosuite
+multidict==6.7.1
+    # via
+    #   aiohttp
+    #   yarl
+multiprocess==0.70.18
+    # via datasets
+mypy==1.19.1
+    # via lerobot
+mypy-extensions==1.1.0
+    # via
+    #   mypy
+    #   typing-inspect
+nbformat==5.10.4
+    # via jupytext
+networkx==3.6.1
+    # via
+    #   bddl
+    #   scikit-image
+    #   torch
+nodeenv==1.10.0
+    # via pre-commit
+num2words==0.5.14
+    # via lerobot
+numba==0.64.0
+    # via robosuite
+numpy==2.2.6
+    # via
+    #   accelerate
+    #   bddl
+    #   cmeel-boost
+    #   contourpy
+    #   datasets
+    #   diffusers
+    #   dm-control
+    #   dm-env
+    #   dm-tree
+    #   gymnasium
+    #   h5py
+    #   hebi-py
+    #   hf-libero
+    #   imageio
+    #   labmaze
+    #   lerobot
+    #   matplotlib
+    #   meshcat
+    #   metaworld
+    #   mujoco
+    #   numba
+    #   opencv-python
+    #   opencv-python-headless
+    #   pandas
+    #   peft
+    #   pyquaternion
+    #   reachy2-sdk
+    #   rerun-sdk
+    #   robomimic
+    #   robosuite
+    #   scikit-image
+    #   scipy
+    #   shapely
+    #   teleop
+    #   tensorboard
+    #   tensorboardx
+    #   tifffile
+    #   torchvision
+    #   transformers
+    #   transforms3d
+nvidia-cublas-cu12==12.8.4.1
+    # via
+    #   nvidia-cudnn-cu12
+    #   nvidia-cusolver-cu12
+    #   torch
+nvidia-cuda-cupti-cu12==12.8.90
+    # via torch
+nvidia-cuda-nvrtc-cu12==12.8.93
+    # via torch
+nvidia-cuda-runtime-cu12==12.8.90
+    # via torch
+nvidia-cudnn-cu12==9.10.2.21
+    # via torch
+nvidia-cufft-cu12==11.3.3.83
+    # via torch
+nvidia-cufile-cu12==1.13.1.3
+    # via torch
+nvidia-curand-cu12==10.3.9.90
+    # via torch
+nvidia-cusolver-cu12==11.7.3.90
+    # via torch
+nvidia-cusparse-cu12==12.5.8.93
+    # via
+    #   nvidia-cusolver-cu12
+    #   torch
+nvidia-cusparselt-cu12==0.7.1
+    # via torch
+nvidia-nccl-cu12==2.27.5
+    # via torch
+nvidia-nvjitlink-cu12==12.8.93
+    # via
+    #   nvidia-cufft-cu12
+    #   nvidia-cusolver-cu12
+    #   nvidia-cusparse-cu12
+    #   torch
+nvidia-nvshmem-cu12==3.4.5
+    # via torch
+nvidia-nvtx-cu12==12.8.90
+    # via torch
+omegaconf==2.3.0
+    # via hydra-core
+opencv-python==4.13.0.92
+    # via
+    #   gym-pusht
+    #   hf-libero
+    #   reachy2-sdk
+    #   robosuite
+opencv-python-headless==4.12.0.88
+    # via lerobot
+orderly-set==5.5.0
+    # via deepdiff
+packaging==25.0
+    # via
+    #   accelerate
+    #   datasets
+    #   huggingface-hub
+    #   hydra-core
+    #   jupytext
+    #   lazy-loader
+    #   lerobot
+    #   matplotlib
+    #   peft
+    #   pytest
+    #   qwen-vl-utils
+    #   reachy2-sdk
+    #   scikit-image
+    #   tensorboard
+    #   tensorboardx
+    #   transformers
+    #   wandb
+pandas==2.3.3
+    # via
+    #   datasets
+    #   lerobot
+parso==0.8.6
+    # via jedi
+pathspec==1.0.4
+    # via mypy
+peft==0.18.1
+    # via lerobot
+pexpect==4.9.0
+    # via ipython
+pillow==12.1.1
+    # via
+    #   diffusers
+    #   imageio
+    #   matplotlib
+    #   meshcat
+    #   qwen-vl-utils
+    #   rerun-sdk
+    #   robosuite
+    #   scikit-image
+    #   tensorboard
+    #   torchvision
+pin==3.4.0
+    # via placo
+placo==0.9.16
+    # via lerobot
+platformdirs==4.9.4
+    # via
+    #   jupyter-core
+    #   python-discovery
+    #   virtualenv
+    #   wandb
+pluggy==1.6.0
+    # via
+    #   pytest
+    #   pytest-cov
+pre-commit==4.5.1
+    # via lerobot
+prompt-toolkit==3.0.52
+    # via ipython
+propcache==0.4.1
+    # via
+    #   aiohttp
+    #   yarl
+protobuf==6.31.1
+    # via
+    #   dm-control
+    #   grpcio-tools
+    #   lerobot
+    #   reachy2-sdk
+    #   reachy2-sdk-api
+    #   tensorboard
+    #   tensorboardx
+    #   wandb
+psutil==7.2.2
+    # via
+    #   accelerate
+    #   imageio
+    #   peft
+    #   robomimic
+ptyprocess==0.7.0
+    # via pexpect
+pure-eval==0.2.3
+    # via stack-data
+pyarrow==23.0.1
+    # via
+    #   datasets
+    #   rerun-sdk
+pycparser==3.0
+    # via cffi
+pydantic==2.12.5
+    # via
+    #   fastapi
+    #   wandb
+pydantic-core==2.41.5
+    # via pydantic
+pygame==2.6.1
+    # via
+    #   gym-hil
+    #   gym-pusht
+    #   lerobot
+pygments==2.19.2
+    # via
+    #   ipython
+    #   ipython-pygments-lexers
+    #   pytest
+    #   rich
+pymunk==6.11.1
+    # via
+    #   gym-pusht
+    #   lerobot
+pyngrok==7.5.1
+    # via meshcat
+pynput==1.8.1
+    # via
+    #   gym-hil
+    #   lerobot
+pyopengl==3.1.10
+    # via
+    #   dm-control
+    #   mujoco
+pyparsing==3.3.2
+    # via
+    #   dm-control
+    #   matplotlib
+pyquaternion==0.9.9
+    # via reachy2-sdk
+pyrealsense2==2.56.5.9235
+    # via lerobot
+pyserial==3.5
+    # via
+    #   dynamixel-sdk
+    #   feetech-servo-sdk
+    #   lerobot
+pytest==8.4.2
+    # via
+    #   bddl
+    #   lerobot
+    #   pytest-cov
+    #   pytest-timeout
+    #   teleop
+pytest-cov==7.0.0
+    # via lerobot
+pytest-timeout==2.4.0
+    # via lerobot
+python-dateutil==2.9.0.post0
+    # via
+    #   faker
+    #   matplotlib
+    #   pandas
+python-discovery==1.1.1
+    # via virtualenv
+python-dotenv==1.2.2
+    # via uvicorn
+python-xlib==0.33
+    # via pynput
+pytz==2026.1.post1
+    # via pandas
+pyyaml==6.0.3
+    # via
+    #   accelerate
+    #   datasets
+    #   draccus
+    #   hebi-py
+    #   huggingface-hub
+    #   jupytext
+    #   omegaconf
+    #   peft
+    #   pre-commit
+    #   pyngrok
+    #   pyyaml-include
+    #   transformers
+    #   uvicorn
+    #   wandb
+pyyaml-include==1.4.1
+    # via draccus
+pyzmq==27.1.0
+    # via
+    #   lerobot
+    #   meshcat
+qwen-vl-utils==0.0.14
+    # via lerobot
+reachy2-sdk==1.0.15
+    # via lerobot
+reachy2-sdk-api==1.0.21
+    # via reachy2-sdk
+referencing==0.37.0
+    # via
+    #   jsonschema
+    #   jsonschema-specifications
+regex==2026.2.28
+    # via
+    #   diffusers
+    #   transformers
+requests==2.32.5
+    # via
+    #   datasets
+    #   diffusers
+    #   dm-control
+    #   qwen-vl-utils
+    #   teleop
+    #   wandb
+rerun-sdk==0.26.2
+    # via lerobot
+rhoban-cmeel-jsoncpp==1.9.4.9
+    # via placo
+rich==14.3.3
+    # via typer
+robomimic==0.2.0
+    # via hf-libero
+robosuite==1.4.0
+    # via hf-libero
+rpds-py==0.30.0
+    # via
+    #   jsonschema
+    #   referencing
+safetensors==0.7.0
+    # via
+    #   accelerate
+    #   diffusers
+    #   lerobot
+    #   peft
+    #   transformers
+scikit-image==0.25.2
+    # via
+    #   gym-pusht
+    #   lerobot
+scipy==1.17.1
+    # via
+    #   dm-control
+    #   lerobot
+    #   metaworld
+    #   robosuite
+    #   scikit-image
+    #   torchdiffeq
+sentry-sdk==2.54.0
+    # via wandb
+shapely==2.1.2
+    # via gym-pusht
+shellingham==1.5.4
+    # via typer
+six==1.17.0
+    # via
+    #   pynput
+    #   python-dateutil
+    #   python-xlib
+smmap==5.0.3
+    # via gitdb
+stack-data==0.6.3
+    # via ipython
+starlette==0.52.1
+    # via fastapi
+sympy==1.14.0
+    # via torch
+teleop==0.1.4
+    # via lerobot
+tensorboard==2.20.0
+    # via robomimic
+tensorboard-data-server==0.7.2
+    # via tensorboard
+tensorboardx==2.6.4
+    # via robomimic
+termcolor==3.3.0
+    # via
+    #   lerobot
+    #   robomimic
+thop==0.1.1.post2209072238
+    # via hf-libero
+tifffile==2026.3.3
+    # via scikit-image
+tokenizers==0.22.2
+    # via transformers
+toml==0.10.2
+    # via draccus
+torch==2.10.0
+    # via
+    #   accelerate
+    #   lerobot
+    #   peft
+    #   robomimic
+    #   thop
+    #   torchdiffeq
+    #   torchvision
+torchcodec==0.10.0
+    # via lerobot
+torchdiffeq==0.2.5
+    # via lerobot
+torchvision==0.25.0
+    # via
+    #   lerobot
+    #   robomimic
+tornado==6.5.4
+    # via meshcat
+tqdm==4.67.3
+    # via
+    #   datasets
+    #   dm-control
+    #   huggingface-hub
+    #   peft
+    #   robomimic
+    #   transformers
+traitlets==5.14.3
+    # via
+    #   ipython
+    #   jupyter-core
+    #   matplotlib-inline
+    #   nbformat
+transformers==5.3.0
+    # via
+    #   hf-libero
+    #   lerobot
+    #   peft
+transforms3d==0.4.2
+    # via teleop
+triton==3.6.0
+    # via torch
+typer==0.24.1
+    # via
+    #   huggingface-hub
+    #   transformers
+typing-extensions==4.15.0
+    # via
+    #   aiosignal
+    #   anyio
+    #   etils
+    #   faker
+    #   fastapi
+    #   gymnasium
+    #   huggingface-hub
+    #   mypy
+    #   pydantic
+    #   pydantic-core
+    #   referencing
+    #   rerun-sdk
+    #   starlette
+    #   torch
+    #   typing-inspect
+    #   typing-inspection
+    #   wandb
+typing-inspect==0.9.0
+    # via draccus
+typing-inspection==0.4.2
+    # via
+    #   fastapi
+    #   pydantic
+tzdata==2025.3
+    # via pandas
+u-msgpack-python==2.8.0
+    # via meshcat
+urllib3==2.6.3
+    # via
+    #   requests
+    #   sentry-sdk
+uvicorn[standard]==0.41.0
+    # via teleop
+uvloop==0.22.1
+    # via uvicorn
+virtualenv==21.1.0
+    # via pre-commit
+wandb==0.24.2
+    # via
+    #   hf-libero
+    #   lerobot
+watchfiles==1.1.1
+    # via uvicorn
+wcwidth==0.6.0
+    # via prompt-toolkit
+websocket-client==1.9.0
+    # via teleop
+websockets==16.0
+    # via uvicorn
+werkzeug==3.1.6
+    # via tensorboard
+wrapt==2.1.2
+    # via dm-tree
+xxhash==3.6.0
+    # via datasets
+yarl==1.23.0
+    # via aiohttp
+zipp==3.23.0
+    # via
+    #   etils
+    #   importlib-metadata
+
+# The following packages are considered to be unsafe in a requirements file:
+# setuptools
diff --git a/lerobot/requirements.in b/lerobot/requirements.in
new file mode 100644
index 0000000000000000000000000000000000000000..b39632f7192095f1e386b04869594706dd9eefa9
--- /dev/null
+++ b/lerobot/requirements.in
@@ -0,0 +1,9 @@
+# requirements.in
+
+# requirements-macos.txt was generated on macOS and is platform-specific (macOS 26.3.1 25D2128 arm64).
+# Darwin MacBook-Pro.local 25.3.0 Darwin Kernel Version 25.3.0: Wed Jan 28 20:54:55 PST 2026; root:xnu-12377.91.3~2/RELEASE_ARM64_T8132 arm64
+
+# requirements-ubuntu.txt was generated on Linux and is platform-specific (Ubuntu 24.04.4 LTS x86_64).
+# Linux lerobot-linux 6.17.0-14-generic #14~24.04.1-Ubuntu SMP PREEMPT_DYNAMIC Thu Jan 15 15:52:10 UTC 2 x86_64 x86_64 x86_64 GNU/Linux
+
+-e .[all]
diff --git a/lerobot/setup.py b/lerobot/setup.py
new file mode 100644
index 0000000000000000000000000000000000000000..d97b6f835661d533054d2b05de1c38304e17459c
--- /dev/null
+++ b/lerobot/setup.py
@@ -0,0 +1,72 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+
+from setuptools import setup
+
+
+def get_version_from_toml() -> str:
+    """Return the project's version string parsed from `pyproject.toml`.
+
+    The function scans `pyproject.toml` line-by-line looking for a line
+    that starts with ``version`` (for example: ``version = "1.2.3"``)
+    and returns the value without surrounding quotes. If no such line is
+    found a :class:`ValueError` is raised.
+
+    Returns:
+        The version string from `pyproject.toml` (e.g. ``"1.2.3"`` ->
+        ``1.2.3``).
+    """
+
+    version = None
+    with open("pyproject.toml", encoding="utf-8") as f:
+        for line in f:
+            if line.strip().startswith("version"):
+                version = line.split("=")[1].strip().strip('"')
+                break
+    if version is None:
+        raise ValueError("Version not found in pyproject.toml")
+    return version
+
+
+def read_long_description() -> str:
+    """Read and return the project's long description for setup.
+
+    This function reads `README.md` and replaces image links that point
+    to the local `./media/` directory with absolute raw GitHub URLs that
+    reference the release tag corresponding to the version parsed from
+    `pyproject.toml` (for example, ``v1.2.3``). The modified README
+    content is returned as a string suitable for passing to
+    ``setuptools.setup(long_description=...)``.
+
+    Returns:
+        The README content with rewritten media links.
+    """
+
+    with open("README.md", encoding="utf-8") as f:
+        content = f.read()
+
+    version = get_version_from_toml()
+    git_tag = f"v{version}"
+
+    base_raw_url = f"https://raw.githubusercontent.com/huggingface/lerobot/{git_tag}/"
+    content = content.replace('src="./media/', f'src="{base_raw_url}media/')
+
+    return content
+
+
+setup(
+    long_description=read_long_description(),
+    long_description_content_type="text/markdown",
+)
diff --git a/lerobot/src/lerobot/__init__.py b/lerobot/src/lerobot/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..eec574296c3cbddd35da578fc96eea00bed995d9
--- /dev/null
+++ b/lerobot/src/lerobot/__init__.py
@@ -0,0 +1,200 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""
+This file contains lists of available environments, dataset and policies to reflect the current state of LeRobot library.
+We do not want to import all the dependencies, but instead we keep it lightweight to ensure fast access to these variables.
+
+Example:
+    ```python
+        import lerobot
+        print(lerobot.available_envs)
+        print(lerobot.available_tasks_per_env)
+        print(lerobot.available_datasets)
+        print(lerobot.available_datasets_per_env)
+        print(lerobot.available_real_world_datasets)
+        print(lerobot.available_policies)
+        print(lerobot.available_policies_per_env)
+        print(lerobot.available_robots)
+        print(lerobot.available_cameras)
+        print(lerobot.available_motors)
+    ```
+
+When implementing a new dataset loadable with LeRobotDataset follow these steps:
+- Update `available_datasets_per_env` in `lerobot/__init__.py`
+
+When implementing a new environment (e.g. `gym_aloha`), follow these steps:
+- Update `available_tasks_per_env` and `available_datasets_per_env` in `lerobot/__init__.py`
+
+When implementing a new policy class (e.g. `DiffusionPolicy`) follow these steps:
+- Update `available_policies` and `available_policies_per_env`, in `lerobot/__init__.py`
+- Set the required `name` class attribute.
+- Update variables in `tests/test_available.py` by importing your new Policy class
+"""
+
+import itertools
+
+from lerobot.__version__ import __version__  # noqa: F401
+
+# TODO(rcadene): Improve policies and envs. As of now, an item in `available_policies`
+# refers to a yaml file AND a modeling name. Same for `available_envs` which refers to
+# a yaml file AND a environment name. The difference should be more obvious.
+available_tasks_per_env = {
+    "aloha": [
+        "AlohaInsertion-v0",
+        "AlohaTransferCube-v0",
+    ],
+    "pusht": ["PushT-v0"],
+}
+available_envs = list(available_tasks_per_env.keys())
+
+available_datasets_per_env = {
+    "aloha": [
+        "lerobot/aloha_sim_insertion_human",
+        "lerobot/aloha_sim_insertion_scripted",
+        "lerobot/aloha_sim_transfer_cube_human",
+        "lerobot/aloha_sim_transfer_cube_scripted",
+        "lerobot/aloha_sim_insertion_human_image",
+        "lerobot/aloha_sim_insertion_scripted_image",
+        "lerobot/aloha_sim_transfer_cube_human_image",
+        "lerobot/aloha_sim_transfer_cube_scripted_image",
+    ],
+    # TODO(alexander-soare): Add "lerobot/pusht_keypoints". Right now we can't because this is too tightly
+    # coupled with tests.
+    "pusht": ["lerobot/pusht", "lerobot/pusht_image"],
+}
+
+available_real_world_datasets = [
+    "lerobot/aloha_mobile_cabinet",
+    "lerobot/aloha_mobile_chair",
+    "lerobot/aloha_mobile_elevator",
+    "lerobot/aloha_mobile_shrimp",
+    "lerobot/aloha_mobile_wash_pan",
+    "lerobot/aloha_mobile_wipe_wine",
+    "lerobot/aloha_static_battery",
+    "lerobot/aloha_static_candy",
+    "lerobot/aloha_static_coffee",
+    "lerobot/aloha_static_coffee_new",
+    "lerobot/aloha_static_cups_open",
+    "lerobot/aloha_static_fork_pick_up",
+    "lerobot/aloha_static_pingpong_test",
+    "lerobot/aloha_static_pro_pencil",
+    "lerobot/aloha_static_screw_driver",
+    "lerobot/aloha_static_tape",
+    "lerobot/aloha_static_thread_velcro",
+    "lerobot/aloha_static_towel",
+    "lerobot/aloha_static_vinh_cup",
+    "lerobot/aloha_static_vinh_cup_left",
+    "lerobot/aloha_static_ziploc_slide",
+    "lerobot/umi_cup_in_the_wild",
+    "lerobot/unitreeh1_fold_clothes",
+    "lerobot/unitreeh1_rearrange_objects",
+    "lerobot/unitreeh1_two_robot_greeting",
+    "lerobot/unitreeh1_warehouse",
+    "lerobot/nyu_rot_dataset",
+    "lerobot/utokyo_saytap",
+    "lerobot/imperialcollege_sawyer_wrist_cam",
+    "lerobot/utokyo_xarm_bimanual",
+    "lerobot/tokyo_u_lsmo",
+    "lerobot/utokyo_pr2_opening_fridge",
+    "lerobot/cmu_franka_exploration_dataset",
+    "lerobot/cmu_stretch",
+    "lerobot/asu_table_top",
+    "lerobot/utokyo_pr2_tabletop_manipulation",
+    "lerobot/utokyo_xarm_pick_and_place",
+    "lerobot/ucsd_kitchen_dataset",
+    "lerobot/austin_buds_dataset",
+    "lerobot/dlr_sara_grid_clamp",
+    "lerobot/conq_hose_manipulation",
+    "lerobot/columbia_cairlab_pusht_real",
+    "lerobot/dlr_sara_pour",
+    "lerobot/dlr_edan_shared_control",
+    "lerobot/ucsd_pick_and_place_dataset",
+    "lerobot/berkeley_cable_routing",
+    "lerobot/nyu_franka_play_dataset",
+    "lerobot/austin_sirius_dataset",
+    "lerobot/cmu_play_fusion",
+    "lerobot/berkeley_gnm_sac_son",
+    "lerobot/nyu_door_opening_surprising_effectiveness",
+    "lerobot/berkeley_fanuc_manipulation",
+    "lerobot/jaco_play",
+    "lerobot/viola",
+    "lerobot/kaist_nonprehensile",
+    "lerobot/berkeley_mvp",
+    "lerobot/uiuc_d3field",
+    "lerobot/berkeley_gnm_recon",
+    "lerobot/austin_sailor_dataset",
+    "lerobot/utaustin_mutex",
+    "lerobot/roboturk",
+    "lerobot/stanford_hydra_dataset",
+    "lerobot/berkeley_autolab_ur5",
+    "lerobot/stanford_robocook",
+    "lerobot/toto",
+    "lerobot/fmb",
+    "lerobot/droid_100",
+    "lerobot/berkeley_rpt",
+    "lerobot/stanford_kuka_multimodal_dataset",
+    "lerobot/iamlab_cmu_pickup_insert",
+    "lerobot/taco_play",
+    "lerobot/berkeley_gnm_cory_hall",
+    "lerobot/usc_cloth_sim",
+]
+
+available_datasets = sorted(
+    set(itertools.chain(*available_datasets_per_env.values(), available_real_world_datasets))
+)
+
+# lists all available policies from `lerobot/policies`
+available_policies = ["act", "diffusion", "tdmpc", "vqbet"]
+
+# lists all available robots from `lerobot/robots`
+available_robots = [
+    "koch",
+    "koch_bimanual",
+    "aloha",
+    "so100",
+    "so101",
+]
+
+# lists all available cameras from `lerobot/cameras`
+available_cameras = [
+    "opencv",
+    "intelrealsense",
+]
+
+# lists all available motors from `lerobot/motors`
+available_motors = [
+    "dynamixel",
+    "feetech",
+]
+
+# keys and values refer to yaml files
+available_policies_per_env = {
+    "aloha": ["act"],
+    "pusht": ["diffusion", "vqbet"],
+    "koch_real": ["act_koch_real"],
+    "aloha_real": ["act_aloha_real"],
+}
+
+env_task_pairs = [(env, task) for env, tasks in available_tasks_per_env.items() for task in tasks]
+env_dataset_pairs = [
+    (env, dataset) for env, datasets in available_datasets_per_env.items() for dataset in datasets
+]
+env_dataset_policy_triplets = [
+    (env, dataset, policy)
+    for env, datasets in available_datasets_per_env.items()
+    for dataset in datasets
+    for policy in available_policies_per_env[env]
+]
diff --git a/lerobot/src/lerobot/__version__.py b/lerobot/src/lerobot/__version__.py
new file mode 100644
index 0000000000000000000000000000000000000000..d12aafaa9e573f408ce0d06654b519ba97832738
--- /dev/null
+++ b/lerobot/src/lerobot/__version__.py
@@ -0,0 +1,23 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""To enable `lerobot.__version__`"""
+
+from importlib.metadata import PackageNotFoundError, version
+
+try:
+    __version__ = version("lerobot")
+except PackageNotFoundError:
+    __version__ = "unknown"
diff --git a/lerobot/src/lerobot/async_inference/configs.py b/lerobot/src/lerobot/async_inference/configs.py
new file mode 100644
index 0000000000000000000000000000000000000000..2e3fe576db3bbf34c4dfad2e33bdcdbd76b64b65
--- /dev/null
+++ b/lerobot/src/lerobot/async_inference/configs.py
@@ -0,0 +1,203 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from collections.abc import Callable
+from dataclasses import dataclass, field
+
+import torch
+
+from lerobot.robots.config import RobotConfig
+
+from .constants import (
+    DEFAULT_FPS,
+    DEFAULT_INFERENCE_LATENCY,
+    DEFAULT_OBS_QUEUE_TIMEOUT,
+)
+
+# Aggregate function registry for CLI usage
+AGGREGATE_FUNCTIONS = {
+    "weighted_average": lambda old, new: 0.3 * old + 0.7 * new,
+    "latest_only": lambda old, new: new,
+    "average": lambda old, new: 0.5 * old + 0.5 * new,
+    "conservative": lambda old, new: 0.7 * old + 0.3 * new,
+}
+
+
+def get_aggregate_function(name: str) -> Callable[[torch.Tensor, torch.Tensor], torch.Tensor]:
+    """Get aggregate function by name from registry."""
+    if name not in AGGREGATE_FUNCTIONS:
+        available = list(AGGREGATE_FUNCTIONS.keys())
+        raise ValueError(f"Unknown aggregate function '{name}'. Available: {available}")
+    return AGGREGATE_FUNCTIONS[name]
+
+
+@dataclass
+class PolicyServerConfig:
+    """Configuration for PolicyServer.
+
+    This class defines all configurable parameters for the PolicyServer,
+    including networking settings and action chunking specifications.
+    """
+
+    # Networking configuration
+    host: str = field(default="localhost", metadata={"help": "Host address to bind the server to"})
+    port: int = field(default=8080, metadata={"help": "Port number to bind the server to"})
+
+    # Timing configuration
+    fps: int = field(default=DEFAULT_FPS, metadata={"help": "Frames per second"})
+    inference_latency: float = field(
+        default=DEFAULT_INFERENCE_LATENCY, metadata={"help": "Target inference latency in seconds"}
+    )
+
+    obs_queue_timeout: float = field(
+        default=DEFAULT_OBS_QUEUE_TIMEOUT, metadata={"help": "Timeout for observation queue in seconds"}
+    )
+
+    def __post_init__(self):
+        """Validate configuration after initialization."""
+        if self.port < 1 or self.port > 65535:
+            raise ValueError(f"Port must be between 1 and 65535, got {self.port}")
+
+        if self.environment_dt <= 0:
+            raise ValueError(f"environment_dt must be positive, got {self.environment_dt}")
+
+        if self.inference_latency < 0:
+            raise ValueError(f"inference_latency must be non-negative, got {self.inference_latency}")
+
+        if self.obs_queue_timeout < 0:
+            raise ValueError(f"obs_queue_timeout must be non-negative, got {self.obs_queue_timeout}")
+
+    @classmethod
+    def from_dict(cls, config_dict: dict) -> "PolicyServerConfig":
+        """Create a PolicyServerConfig from a dictionary."""
+        return cls(**config_dict)
+
+    @property
+    def environment_dt(self) -> float:
+        """Environment time step, in seconds"""
+        return 1 / self.fps
+
+    def to_dict(self) -> dict:
+        """Convert the configuration to a dictionary."""
+        return {
+            "host": self.host,
+            "port": self.port,
+            "fps": self.fps,
+            "environment_dt": self.environment_dt,
+            "inference_latency": self.inference_latency,
+        }
+
+
+@dataclass
+class RobotClientConfig:
+    """Configuration for RobotClient.
+
+    This class defines all configurable parameters for the RobotClient,
+    including network connection, policy settings, and control behavior.
+    """
+
+    # Policy configuration
+    policy_type: str = field(metadata={"help": "Type of policy to use"})
+    pretrained_name_or_path: str = field(metadata={"help": "Pretrained model name or path"})
+
+    # Robot configuration (for CLI usage - robot instance will be created from this)
+    robot: RobotConfig = field(metadata={"help": "Robot configuration"})
+
+    # Policies typically output K actions at max, but we can use less to avoid wasting bandwidth (as actions
+    # would be aggregated on the client side anyway, depending on the value of `chunk_size_threshold`)
+    actions_per_chunk: int = field(metadata={"help": "Number of actions per chunk"})
+
+    # Task instruction for the robot to execute (e.g., 'fold my tshirt')
+    task: str = field(default="", metadata={"help": "Task instruction for the robot to execute"})
+
+    # Network configuration
+    server_address: str = field(default="localhost:8080", metadata={"help": "Server address to connect to"})
+
+    # Device configuration
+    policy_device: str = field(default="cpu", metadata={"help": "Device for policy inference"})
+    client_device: str = field(
+        default="cpu",
+        metadata={
+            "help": "Device to move actions to after receiving from server (e.g., for downstream planners)"
+        },
+    )
+
+    # Control behavior configuration
+    chunk_size_threshold: float = field(default=0.5, metadata={"help": "Threshold for chunk size control"})
+    fps: int = field(default=DEFAULT_FPS, metadata={"help": "Frames per second"})
+
+    # Aggregate function configuration (CLI-compatible)
+    aggregate_fn_name: str = field(
+        default="weighted_average",
+        metadata={"help": f"Name of aggregate function to use. Options: {list(AGGREGATE_FUNCTIONS.keys())}"},
+    )
+
+    # Debug configuration
+    debug_visualize_queue_size: bool = field(
+        default=False, metadata={"help": "Visualize the action queue size"}
+    )
+
+    @property
+    def environment_dt(self) -> float:
+        """Environment time step, in seconds"""
+        return 1 / self.fps
+
+    def __post_init__(self):
+        """Validate configuration after initialization."""
+        if not self.server_address:
+            raise ValueError("server_address cannot be empty")
+
+        if not self.policy_type:
+            raise ValueError("policy_type cannot be empty")
+
+        if not self.pretrained_name_or_path:
+            raise ValueError("pretrained_name_or_path cannot be empty")
+
+        if not self.policy_device:
+            raise ValueError("policy_device cannot be empty")
+
+        if not self.client_device:
+            raise ValueError("client_device cannot be empty")
+
+        if self.chunk_size_threshold < 0 or self.chunk_size_threshold > 1:
+            raise ValueError(f"chunk_size_threshold must be between 0 and 1, got {self.chunk_size_threshold}")
+
+        if self.fps <= 0:
+            raise ValueError(f"fps must be positive, got {self.fps}")
+
+        if self.actions_per_chunk <= 0:
+            raise ValueError(f"actions_per_chunk must be positive, got {self.actions_per_chunk}")
+
+        self.aggregate_fn = get_aggregate_function(self.aggregate_fn_name)
+
+    @classmethod
+    def from_dict(cls, config_dict: dict) -> "RobotClientConfig":
+        """Create a RobotClientConfig from a dictionary."""
+        return cls(**config_dict)
+
+    def to_dict(self) -> dict:
+        """Convert the configuration to a dictionary."""
+        return {
+            "server_address": self.server_address,
+            "policy_type": self.policy_type,
+            "pretrained_name_or_path": self.pretrained_name_or_path,
+            "policy_device": self.policy_device,
+            "client_device": self.client_device,
+            "chunk_size_threshold": self.chunk_size_threshold,
+            "fps": self.fps,
+            "actions_per_chunk": self.actions_per_chunk,
+            "task": self.task,
+            "debug_visualize_queue_size": self.debug_visualize_queue_size,
+            "aggregate_fn_name": self.aggregate_fn_name,
+        }
diff --git a/lerobot/src/lerobot/async_inference/constants.py b/lerobot/src/lerobot/async_inference/constants.py
new file mode 100644
index 0000000000000000000000000000000000000000..56910e67f17f7dd62b65ed5ff3e39b2abb00132a
--- /dev/null
+++ b/lerobot/src/lerobot/async_inference/constants.py
@@ -0,0 +1,29 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Client side: The environment evolves with a time resolution equal to 1/fps"""
+
+DEFAULT_FPS = 30
+
+"""Server side: Running inference on (at most) 1/fps"""
+DEFAULT_INFERENCE_LATENCY = 1 / DEFAULT_FPS
+
+"""Server side: Timeout for observation queue in seconds"""
+DEFAULT_OBS_QUEUE_TIMEOUT = 2
+
+# All action chunking policies
+SUPPORTED_POLICIES = ["act", "smolvla", "diffusion", "tdmpc", "vqbet", "pi0", "pi05", "groot"]
+
+# TODO: Add all other robots
+SUPPORTED_ROBOTS = ["so100_follower", "so101_follower", "bi_so_follower", "omx_follower"]
diff --git a/lerobot/src/lerobot/async_inference/helpers.py b/lerobot/src/lerobot/async_inference/helpers.py
new file mode 100644
index 0000000000000000000000000000000000000000..9dd44eb44726016b3144a364668314d1dafceb74
--- /dev/null
+++ b/lerobot/src/lerobot/async_inference/helpers.py
@@ -0,0 +1,297 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import logging.handlers
+import os
+import time
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import Any
+
+import torch
+
+from lerobot.configs.types import PolicyFeature
+from lerobot.datasets.feature_utils import build_dataset_frame, hw_to_dataset_features
+
+# NOTE: Configs need to be loaded for the client to be able to instantiate the policy config
+from lerobot.policies import (  # noqa: F401
+    ACTConfig,
+    DiffusionConfig,
+    PI0Config,
+    PI05Config,
+    SmolVLAConfig,
+    VQBeTConfig,
+)
+from lerobot.robots.robot import Robot
+from lerobot.utils.constants import OBS_IMAGES, OBS_STATE, OBS_STR
+from lerobot.utils.utils import init_logging
+
+Action = torch.Tensor
+
+# observation as received from the robot (can be numpy arrays, floats, etc.)
+RawObservation = dict[str, Any]
+
+# observation as those recorded in LeRobot dataset (keys are different)
+LeRobotObservation = dict[str, torch.Tensor]
+
+# observation, ready for policy inference (image keys resized)
+Observation = dict[str, torch.Tensor]
+
+
+def visualize_action_queue_size(action_queue_size: list[int]) -> None:
+    import matplotlib.pyplot as plt
+
+    _, ax = plt.subplots()
+    ax.set_title("Action Queue Size Over Time")
+    ax.set_xlabel("Environment steps")
+    ax.set_ylabel("Action Queue Size")
+    ax.set_ylim(0, max(action_queue_size) * 1.1)
+    ax.grid(True, alpha=0.3)
+    ax.plot(range(len(action_queue_size)), action_queue_size)
+    plt.show()
+
+
+def map_robot_keys_to_lerobot_features(robot: Robot) -> dict[str, dict]:
+    return hw_to_dataset_features(robot.observation_features, OBS_STR, use_video=False)
+
+
+def is_image_key(k: str) -> bool:
+    return k.startswith(OBS_IMAGES)
+
+
+def resize_robot_observation_image(image: torch.tensor, resize_dims: tuple[int, int, int]) -> torch.tensor:
+    assert image.ndim == 3, f"Image must be (C, H, W)! Received {image.shape}"
+    # (H, W, C) -> (C, H, W) for resizing from robot obsevation resolution to policy image resolution
+    image = image.permute(2, 0, 1)
+    dims = (resize_dims[1], resize_dims[2])
+    # Add batch dimension for interpolate: (C, H, W) -> (1, C, H, W)
+    image_batched = image.unsqueeze(0)
+    # Interpolate and remove batch dimension: (1, C, H, W) -> (C, H, W)
+    resized = torch.nn.functional.interpolate(image_batched, size=dims, mode="bilinear", align_corners=False)
+
+    return resized.squeeze(0)
+
+
+# TODO(Steven): Consider implementing a pipeline step for this
+def raw_observation_to_observation(
+    raw_observation: RawObservation,
+    lerobot_features: dict[str, dict],
+    policy_image_features: dict[str, PolicyFeature],
+) -> Observation:
+    observation = {}
+
+    observation = prepare_raw_observation(raw_observation, lerobot_features, policy_image_features)
+    for k, v in observation.items():
+        if isinstance(v, torch.Tensor):  # VLAs present natural-language instructions in observations
+            if "image" in k:
+                # Policy expects images in shape (B, C, H, W)
+                observation[k] = prepare_image(v).unsqueeze(0)
+        else:
+            observation[k] = v
+
+    return observation
+
+
+def prepare_image(image: torch.Tensor) -> torch.Tensor:
+    """Minimal preprocessing to turn int8 images to float32 in [0, 1], and create a memory-contiguous tensor"""
+    image = image.type(torch.float32) / 255
+    image = image.contiguous()
+
+    return image
+
+
+def extract_state_from_raw_observation(
+    lerobot_obs: RawObservation,
+) -> torch.Tensor:
+    """Extract the state from a raw observation."""
+    state = torch.tensor(lerobot_obs[OBS_STATE])
+
+    if state.ndim == 1:
+        state = state.unsqueeze(0)
+
+    return state
+
+
+def extract_images_from_raw_observation(
+    lerobot_obs: RawObservation,
+    camera_key: str,
+) -> dict[str, torch.Tensor]:
+    """Extract the images from a raw observation."""
+    return torch.tensor(lerobot_obs[camera_key])
+
+
+def make_lerobot_observation(
+    robot_obs: RawObservation,
+    lerobot_features: dict[str, dict],
+) -> LeRobotObservation:
+    """Make a lerobot observation from a raw observation."""
+    return build_dataset_frame(lerobot_features, robot_obs, prefix=OBS_STR)
+
+
+def prepare_raw_observation(
+    robot_obs: RawObservation,
+    lerobot_features: dict[str, dict],
+    policy_image_features: dict[str, PolicyFeature],
+) -> Observation:
+    """Matches keys from the raw robot_obs dict to the keys expected by a given policy (passed as
+    policy_image_features)."""
+    # 1. {motor.pos1:value1, motor.pos2:value2, ..., laptop:np.ndarray} ->
+    # -> {observation.state:[value1,value2,...], observation.images.laptop:np.ndarray}
+    lerobot_obs = make_lerobot_observation(robot_obs, lerobot_features)
+
+    # 2. Greps all observation.images.<> keys
+    image_keys = list(filter(is_image_key, lerobot_obs))
+    # state's shape is expected as (B, state_dim)
+    state_dict = {OBS_STATE: extract_state_from_raw_observation(lerobot_obs)}
+    image_dict = {
+        image_k: extract_images_from_raw_observation(lerobot_obs, image_k) for image_k in image_keys
+    }
+
+    # Turns the image features to (C, H, W) with H, W matching the policy image features.
+    # This reduces the resolution of the images
+    image_dict = {
+        key: resize_robot_observation_image(torch.tensor(lerobot_obs[key]), policy_image_features[key].shape)
+        for key in image_keys
+    }
+
+    if "task" in robot_obs:
+        state_dict["task"] = robot_obs["task"]
+
+    return {**state_dict, **image_dict}
+
+
+def get_logger(name: str, log_to_file: bool = True) -> logging.Logger:
+    """
+    Get a logger using the standardized logging setup from utils.py.
+
+    Args:
+        name: Logger name (e.g., 'policy_server', 'robot_client')
+        log_to_file: Whether to also log to a file
+
+    Returns:
+        Configured logger instance
+    """
+    # Create logs directory if logging to file
+    if log_to_file:
+        os.makedirs("logs", exist_ok=True)
+        log_file = Path(f"logs/{name}_{int(time.time())}.log")
+    else:
+        log_file = None
+
+    # Initialize the standardized logging
+    init_logging(log_file=log_file, display_pid=False)
+
+    # Return a named logger
+    return logging.getLogger(name)
+
+
+@dataclass
+class TimedData:
+    """A data object with timestamp and timestep information.
+
+    Args:
+        timestamp: Unix timestamp relative to data's creation.
+        data: The actual data to wrap a timestamp around.
+        timestep: The timestep of the data.
+    """
+
+    timestamp: float
+    timestep: int
+
+    def get_timestamp(self):
+        return self.timestamp
+
+    def get_timestep(self):
+        return self.timestep
+
+
+@dataclass
+class TimedAction(TimedData):
+    action: Action
+
+    def get_action(self):
+        return self.action
+
+
+@dataclass
+class TimedObservation(TimedData):
+    observation: RawObservation
+    must_go: bool = False
+
+    def get_observation(self):
+        return self.observation
+
+
+@dataclass
+class FPSTracker:
+    """Utility class to track FPS metrics over time."""
+
+    target_fps: float
+    first_timestamp: float = None
+    total_obs_count: int = 0
+
+    def calculate_fps_metrics(self, current_timestamp: float) -> dict[str, float]:
+        """Calculate average FPS vs target"""
+        self.total_obs_count += 1
+
+        # Initialize first observation time
+        if self.first_timestamp is None:
+            self.first_timestamp = current_timestamp
+
+        # Calculate overall average FPS (since start)
+        total_duration = current_timestamp - self.first_timestamp
+        avg_fps = (self.total_obs_count - 1) / total_duration if total_duration > 1e-6 else 0.0
+
+        return {"avg_fps": avg_fps, "target_fps": self.target_fps}
+
+    def reset(self):
+        """Reset the FPS tracker state"""
+        self.first_timestamp = None
+        self.total_obs_count = 0
+
+
+@dataclass
+class RemotePolicyConfig:
+    policy_type: str
+    pretrained_name_or_path: str
+    lerobot_features: dict[str, PolicyFeature]
+    actions_per_chunk: int
+    device: str = "cpu"
+    rename_map: dict[str, str] = field(default_factory=dict)
+
+
+def _compare_observation_states(obs1_state: torch.Tensor, obs2_state: torch.Tensor, atol: float) -> bool:
+    """Check if two observation states are similar, under a tolerance threshold"""
+    return bool(torch.linalg.norm(obs1_state - obs2_state) < atol)
+
+
+def observations_similar(
+    obs1: TimedObservation, obs2: TimedObservation, lerobot_features: dict[str, dict], atol: float = 1
+) -> bool:
+    """Check if two observations are similar, under a tolerance threshold. Measures distance between
+    observations as the difference in joint-space between the two observations.
+
+    NOTE(fracapuano): This is a very simple check, and it is enough for the current use case.
+    An immediate next step is to use (fast) perceptual difference metrics comparing some camera views,
+    to surpass this joint-space similarity check.
+    """
+    obs1_state = extract_state_from_raw_observation(
+        make_lerobot_observation(obs1.get_observation(), lerobot_features)
+    )
+    obs2_state = extract_state_from_raw_observation(
+        make_lerobot_observation(obs2.get_observation(), lerobot_features)
+    )
+
+    return _compare_observation_states(obs1_state, obs2_state, atol=atol)
diff --git a/lerobot/src/lerobot/async_inference/policy_server.py b/lerobot/src/lerobot/async_inference/policy_server.py
new file mode 100644
index 0000000000000000000000000000000000000000..3f63929dfa3d0208f6db1d7b946f284b8950d493
--- /dev/null
+++ b/lerobot/src/lerobot/async_inference/policy_server.py
@@ -0,0 +1,439 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Example:
+```shell
+python -m lerobot.async_inference.policy_server \
+     --host=127.0.0.1 \
+     --port=8080 \
+     --fps=30 \
+     --inference_latency=0.033 \
+     --obs_queue_timeout=1
+```
+"""
+
+import logging
+import pickle  # nosec
+import threading
+import time
+from concurrent import futures
+from dataclasses import asdict
+from pprint import pformat
+from queue import Empty, Queue
+from typing import Any
+
+import draccus
+import grpc
+import torch
+
+from lerobot.policies.factory import get_policy_class, make_pre_post_processors
+from lerobot.processor import PolicyProcessorPipeline
+from lerobot.transport import (
+    services_pb2,  # type: ignore
+    services_pb2_grpc,  # type: ignore
+)
+from lerobot.transport.utils import receive_bytes_in_chunks
+from lerobot.types import PolicyAction
+
+from .configs import PolicyServerConfig
+from .constants import SUPPORTED_POLICIES
+from .helpers import (
+    FPSTracker,
+    Observation,
+    RemotePolicyConfig,
+    TimedAction,
+    TimedObservation,
+    get_logger,
+    observations_similar,
+    raw_observation_to_observation,
+)
+
+
+class PolicyServer(services_pb2_grpc.AsyncInferenceServicer):
+    prefix = "policy_server"
+    logger = get_logger(prefix)
+
+    def __init__(self, config: PolicyServerConfig):
+        self.config = config
+        self.shutdown_event = threading.Event()
+
+        # FPS measurement
+        self.fps_tracker = FPSTracker(target_fps=config.fps)
+
+        self.observation_queue = Queue(maxsize=1)
+
+        self._predicted_timesteps_lock = threading.Lock()
+        self._predicted_timesteps = set()
+
+        self.last_processed_obs = None
+
+        # Attributes will be set by SendPolicyInstructions
+        self.device = None
+        self.policy_type = None
+        self.lerobot_features = None
+        self.actions_per_chunk = None
+        self.policy = None
+        self.preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]] | None = None
+        self.postprocessor: PolicyProcessorPipeline[PolicyAction, PolicyAction] | None = None
+
+    @property
+    def running(self):
+        return not self.shutdown_event.is_set()
+
+    @property
+    def policy_image_features(self):
+        return self.policy.config.image_features
+
+    def _reset_server(self) -> None:
+        """Flushes server state when new client connects."""
+        # only running inference on the latest observation received by the server
+        self.shutdown_event.set()
+        self.observation_queue = Queue(maxsize=1)
+
+        with self._predicted_timesteps_lock:
+            self._predicted_timesteps = set()
+
+    def Ready(self, request, context):  # noqa: N802
+        client_id = context.peer()
+        self.logger.info(f"Client {client_id} connected and ready")
+        self._reset_server()
+        self.shutdown_event.clear()
+
+        return services_pb2.Empty()
+
+    def SendPolicyInstructions(self, request, context):  # noqa: N802
+        """Receive policy instructions from the robot client"""
+
+        if not self.running:
+            self.logger.warning("Server is not running. Ignoring policy instructions.")
+            return services_pb2.Empty()
+
+        client_id = context.peer()
+
+        policy_specs = pickle.loads(request.data)  # nosec
+
+        if not isinstance(policy_specs, RemotePolicyConfig):
+            raise TypeError(f"Policy specs must be a RemotePolicyConfig. Got {type(policy_specs)}")
+
+        if policy_specs.policy_type not in SUPPORTED_POLICIES:
+            raise ValueError(
+                f"Policy type {policy_specs.policy_type} not supported. "
+                f"Supported policies: {SUPPORTED_POLICIES}"
+            )
+
+        self.logger.info(
+            f"Receiving policy instructions from {client_id} | "
+            f"Policy type: {policy_specs.policy_type} | "
+            f"Pretrained name or path: {policy_specs.pretrained_name_or_path} | "
+            f"Actions per chunk: {policy_specs.actions_per_chunk} | "
+            f"Device: {policy_specs.device}"
+        )
+
+        self.device = policy_specs.device
+        self.policy_type = policy_specs.policy_type  # act, pi0, etc.
+        self.lerobot_features = policy_specs.lerobot_features
+        self.actions_per_chunk = policy_specs.actions_per_chunk
+
+        policy_class = get_policy_class(self.policy_type)
+
+        start = time.perf_counter()
+        self.policy = policy_class.from_pretrained(policy_specs.pretrained_name_or_path)
+        self.policy.to(self.device)
+
+        # Load preprocessor and postprocessor, overriding device to match requested device
+        device_override = {"device": self.device}
+        self.preprocessor, self.postprocessor = make_pre_post_processors(
+            self.policy.config,
+            pretrained_path=policy_specs.pretrained_name_or_path,
+            preprocessor_overrides={
+                "device_processor": device_override,
+                "rename_observations_processor": {"rename_map": policy_specs.rename_map},
+            },
+            postprocessor_overrides={"device_processor": device_override},
+        )
+
+        end = time.perf_counter()
+
+        self.logger.info(f"Time taken to put policy on {self.device}: {end - start:.4f} seconds")
+
+        return services_pb2.Empty()
+
+    def SendObservations(self, request_iterator, context):  # noqa: N802
+        """Receive observations from the robot client"""
+        client_id = context.peer()
+        self.logger.debug(f"Receiving observations from {client_id}")
+
+        receive_time = time.time()  # comparing timestamps so need time.time()
+        start_deserialize = time.perf_counter()
+        received_bytes = receive_bytes_in_chunks(
+            request_iterator, None, self.shutdown_event, self.logger
+        )  # blocking call while looping over request_iterator
+        timed_observation = pickle.loads(received_bytes)  # nosec
+        deserialize_time = time.perf_counter() - start_deserialize
+
+        self.logger.debug(f"Received observation #{timed_observation.get_timestep()}")
+
+        obs_timestep = timed_observation.get_timestep()
+        obs_timestamp = timed_observation.get_timestamp()
+
+        # Calculate FPS metrics
+        fps_metrics = self.fps_tracker.calculate_fps_metrics(obs_timestamp)
+
+        self.logger.debug(
+            f"Received observation #{obs_timestep} | "
+            f"Avg FPS: {fps_metrics['avg_fps']:.2f} | "  # fps at which observations are received from client
+            f"Target: {fps_metrics['target_fps']:.2f} | "
+            f"One-way latency: {(receive_time - obs_timestamp) * 1000:.2f}ms"
+        )
+
+        self.logger.debug(
+            f"Server timestamp: {receive_time:.6f} | "
+            f"Client timestamp: {obs_timestamp:.6f} | "
+            f"Deserialization time: {deserialize_time:.6f}s"
+        )
+
+        if not self._enqueue_observation(
+            timed_observation  # wrapping a RawObservation
+        ):
+            self.logger.debug(f"Observation #{obs_timestep} has been filtered out")
+
+        return services_pb2.Empty()
+
+    def GetActions(self, request, context):  # noqa: N802
+        """Returns actions to the robot client. Actions are sent as a single
+        chunk, containing multiple actions."""
+        client_id = context.peer()
+        self.logger.debug(f"Client {client_id} connected for action streaming")
+
+        # Generate action based on the most recent observation and its timestep
+        try:
+            getactions_starts = time.perf_counter()
+            obs = self.observation_queue.get(timeout=self.config.obs_queue_timeout)
+            self.logger.info(
+                f"Running inference for observation #{obs.get_timestep()} (must_go: {obs.must_go})"
+            )
+
+            with self._predicted_timesteps_lock:
+                self._predicted_timesteps.add(obs.get_timestep())
+
+            start_time = time.perf_counter()
+            action_chunk = self._predict_action_chunk(obs)
+            inference_time = time.perf_counter() - start_time
+
+            start_time = time.perf_counter()
+            actions_bytes = pickle.dumps(action_chunk)  # nosec
+            serialize_time = time.perf_counter() - start_time
+
+            # Create and return the action chunk
+            actions = services_pb2.Actions(data=actions_bytes)
+
+            self.logger.info(
+                f"Action chunk #{obs.get_timestep()} generated | "
+                f"Total time: {(inference_time + serialize_time) * 1000:.2f}ms"
+            )
+
+            self.logger.debug(
+                f"Action chunk #{obs.get_timestep()} generated | "
+                f"Inference time: {inference_time:.2f}s |"
+                f"Serialize time: {serialize_time:.2f}s |"
+                f"Total time: {inference_time + serialize_time:.2f}s"
+            )
+
+            time.sleep(
+                max(0, self.config.inference_latency - max(0, time.perf_counter() - getactions_starts))
+            )  # sleep controls inference latency
+
+            return actions
+
+        except Empty:  # no observation added to queue in obs_queue_timeout
+            return services_pb2.Empty()
+
+        except Exception as e:
+            self.logger.error(f"Error in StreamActions: {e}")
+
+            return services_pb2.Empty()
+
+    def _obs_sanity_checks(self, obs: TimedObservation, previous_obs: TimedObservation) -> bool:
+        """Check if the observation is valid to be processed by the policy"""
+        with self._predicted_timesteps_lock:
+            predicted_timesteps = self._predicted_timesteps
+
+        if obs.get_timestep() in predicted_timesteps:
+            self.logger.debug(f"Skipping observation #{obs.get_timestep()} - Timestep predicted already!")
+            return False
+
+        elif observations_similar(obs, previous_obs, lerobot_features=self.lerobot_features):
+            self.logger.debug(
+                f"Skipping observation #{obs.get_timestep()} - Observation too similar to last obs predicted!"
+            )
+            return False
+
+        else:
+            return True
+
+    def _enqueue_observation(self, obs: TimedObservation) -> bool:
+        """Enqueue an observation if it must go through processing, otherwise skip it.
+        Observations not in queue are never run through the policy network"""
+
+        if (
+            obs.must_go
+            or self.last_processed_obs is None
+            or self._obs_sanity_checks(obs, self.last_processed_obs)
+        ):
+            last_obs = self.last_processed_obs.get_timestep() if self.last_processed_obs else "None"
+            self.logger.debug(
+                f"Enqueuing observation. Must go: {obs.must_go} | Last processed obs: {last_obs}"
+            )
+
+            # If queue is full, get the old observation to make room
+            if self.observation_queue.full():
+                # pops from queue
+                _ = self.observation_queue.get_nowait()
+                self.logger.debug("Observation queue was full, removed oldest observation")
+
+            # Now put the new observation (never blocks as queue is non-full here)
+            self.observation_queue.put(obs)
+            return True
+
+        return False
+
+    def _time_action_chunk(self, t_0: float, action_chunk: list[torch.Tensor], i_0: int) -> list[TimedAction]:
+        """Turn a chunk of actions into a list of TimedAction instances,
+        with the first action corresponding to t_0 and the rest corresponding to
+        t_0 + i*environment_dt for i in range(len(action_chunk))
+        """
+        return [
+            TimedAction(timestamp=t_0 + i * self.config.environment_dt, timestep=i_0 + i, action=action)
+            for i, action in enumerate(action_chunk)
+        ]
+
+    def _get_action_chunk(self, observation: dict[str, torch.Tensor]) -> torch.Tensor:
+        """Get an action chunk from the policy. The chunk contains only"""
+        chunk = self.policy.predict_action_chunk(observation)
+        if chunk.ndim != 3:
+            chunk = chunk.unsqueeze(0)  # adding batch dimension, now shape is (B, chunk_size, action_dim)
+
+        return chunk[:, : self.actions_per_chunk, :]
+
+    def _predict_action_chunk(self, observation_t: TimedObservation) -> list[TimedAction]:
+        """Predict an action chunk based on an observation.
+
+        Pipeline:
+        1. Convert raw observation to LeRobot format
+        2. Apply preprocessor (tokenization, normalization, batching, device placement)
+        3. Run policy inference to get action chunk
+        4. Apply postprocessor (unnormalization, device movement)
+        5. Convert to TimedAction list
+        """
+        """1. Prepare observation"""
+        start_prepare = time.perf_counter()
+        observation: Observation = raw_observation_to_observation(
+            observation_t.get_observation(),
+            self.lerobot_features,
+            self.policy_image_features,
+        )
+        prepare_time = time.perf_counter() - start_prepare
+
+        """2. Apply preprocessor"""
+        start_preprocess = time.perf_counter()
+        observation = self.preprocessor(observation)
+        self.last_processed_obs: TimedObservation = observation_t
+        preprocessing_time = time.perf_counter() - start_preprocess
+
+        """3. Get action chunk"""
+        start_inference = time.perf_counter()
+        action_tensor = self._get_action_chunk(observation)
+        inference_time = time.perf_counter() - start_inference
+        self.logger.info(
+            f"Preprocessing and inference took {inference_time:.4f}s, action shape: {action_tensor.shape}"
+        )
+
+        """4. Apply postprocessor"""
+        # Apply postprocessor (handles unnormalization and device movement)
+        # Postprocessor expects (B, action_dim) per action, but we have (B, chunk_size, action_dim)
+        # So we process each action in the chunk individually
+        start_postprocess = time.perf_counter()
+        _, chunk_size, _ = action_tensor.shape
+
+        # Process each action in the chunk
+        processed_actions = []
+        for i in range(chunk_size):
+            # Extract action at timestep i: (B, action_dim)
+            single_action = action_tensor[:, i, :]
+            processed_action = self.postprocessor(single_action)
+            processed_actions.append(processed_action)
+
+        # Stack back to (B, chunk_size, action_dim), then remove batch dim
+        action_tensor = torch.stack(processed_actions, dim=1).squeeze(0)
+        self.logger.debug(f"Postprocessed action shape: {action_tensor.shape}")
+
+        action_tensor = action_tensor.detach().cpu()
+
+        """5. Convert to TimedAction list"""
+        action_chunk = self._time_action_chunk(
+            observation_t.get_timestamp(), list(action_tensor), observation_t.get_timestep()
+        )
+        postprocess_stops = time.perf_counter()
+        postprocessing_time = postprocess_stops - start_postprocess
+
+        self.logger.info(
+            f"Observation {observation_t.get_timestep()} | "
+            f"Total time: {1000 * (postprocess_stops - start_prepare):.2f}ms"
+        )
+
+        self.logger.debug(
+            f"Observation {observation_t.get_timestep()} | "
+            f"Prepare time: {1000 * prepare_time:.2f}ms | "
+            f"Preprocessing time: {1000 * preprocessing_time:.2f}ms | "
+            f"Inference time: {1000 * inference_time:.2f}ms | "
+            f"Postprocessing time: {1000 * postprocessing_time:.2f}ms | "
+            f"Total time: {1000 * (postprocess_stops - start_prepare):.2f}ms"
+        )
+
+        return action_chunk
+
+    def stop(self):
+        """Stop the server"""
+        self._reset_server()
+        self.logger.info("Server stopping...")
+
+
+@draccus.wrap()
+def serve(cfg: PolicyServerConfig):
+    """Start the PolicyServer with the given configuration.
+
+    Args:
+        config: PolicyServerConfig instance. If None, uses default configuration.
+    """
+    logging.info(pformat(asdict(cfg)))
+
+    # Create the server instance first
+    policy_server = PolicyServer(cfg)
+
+    # Setup and start gRPC server
+    server = grpc.server(futures.ThreadPoolExecutor(max_workers=4))
+    services_pb2_grpc.add_AsyncInferenceServicer_to_server(policy_server, server)
+    server.add_insecure_port(f"{cfg.host}:{cfg.port}")
+
+    policy_server.logger.info(f"PolicyServer started on {cfg.host}:{cfg.port}")
+    server.start()
+
+    server.wait_for_termination()
+
+    policy_server.logger.info("Server terminated")
+
+
+if __name__ == "__main__":
+    serve()
diff --git a/lerobot/src/lerobot/async_inference/robot_client.py b/lerobot/src/lerobot/async_inference/robot_client.py
new file mode 100644
index 0000000000000000000000000000000000000000..0ee70a0e629496886c965b6035c9c83fff9e6390
--- /dev/null
+++ b/lerobot/src/lerobot/async_inference/robot_client.py
@@ -0,0 +1,517 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Example command:
+```shell
+python src/lerobot/async_inference/robot_client.py \
+    --robot.type=so100_follower \
+    --robot.port=/dev/tty.usbmodem58760431541 \
+    --robot.cameras="{ front: {type: opencv, index_or_path: 0, width: 1920, height: 1080, fps: 30}}" \
+    --robot.id=black \
+    --task="dummy" \
+    --server_address=127.0.0.1:8080 \
+    --policy_type=act \
+    --pretrained_name_or_path=user/model \
+    --policy_device=mps \
+    --client_device=cpu \
+    --actions_per_chunk=50 \
+    --chunk_size_threshold=0.5 \
+    --aggregate_fn_name=weighted_average \
+    --debug_visualize_queue_size=True
+```
+"""
+
+import logging
+import pickle  # nosec
+import threading
+import time
+from collections.abc import Callable
+from dataclasses import asdict
+from pprint import pformat
+from queue import Queue
+from typing import Any
+
+import draccus
+import grpc
+import torch
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig  # noqa: F401
+from lerobot.cameras.realsense.configuration_realsense import RealSenseCameraConfig  # noqa: F401
+from lerobot.robots import (  # noqa: F401
+    Robot,
+    RobotConfig,
+    bi_so_follower,
+    koch_follower,
+    make_robot_from_config,
+    omx_follower,
+    so_follower,
+)
+from lerobot.transport import (
+    services_pb2,  # type: ignore
+    services_pb2_grpc,  # type: ignore
+)
+from lerobot.transport.utils import grpc_channel_options, send_bytes_in_chunks
+from lerobot.utils.import_utils import register_third_party_plugins
+
+from .configs import RobotClientConfig
+from .helpers import (
+    Action,
+    FPSTracker,
+    Observation,
+    RawObservation,
+    RemotePolicyConfig,
+    TimedAction,
+    TimedObservation,
+    get_logger,
+    map_robot_keys_to_lerobot_features,
+    visualize_action_queue_size,
+)
+
+
+class RobotClient:
+    prefix = "robot_client"
+    logger = get_logger(prefix)
+
+    def __init__(self, config: RobotClientConfig):
+        """Initialize RobotClient with unified configuration.
+
+        Args:
+            config: RobotClientConfig containing all configuration parameters
+        """
+        # Store configuration
+        self.config = config
+        self.robot = make_robot_from_config(config.robot)
+        self.robot.connect()
+
+        lerobot_features = map_robot_keys_to_lerobot_features(self.robot)
+
+        # Use environment variable if server_address is not provided in config
+        self.server_address = config.server_address
+
+        self.policy_config = RemotePolicyConfig(
+            config.policy_type,
+            config.pretrained_name_or_path,
+            lerobot_features,
+            config.actions_per_chunk,
+            config.policy_device,
+        )
+        self.channel = grpc.insecure_channel(
+            self.server_address, grpc_channel_options(initial_backoff=f"{config.environment_dt:.4f}s")
+        )
+        self.stub = services_pb2_grpc.AsyncInferenceStub(self.channel)
+        self.logger.info(f"Initializing client to connect to server at {self.server_address}")
+
+        self.shutdown_event = threading.Event()
+
+        # Initialize client side variables
+        self.latest_action_lock = threading.Lock()
+        self.latest_action = -1
+        self.action_chunk_size = -1
+
+        self._chunk_size_threshold = config.chunk_size_threshold
+
+        self.action_queue = Queue()
+        self.action_queue_lock = threading.Lock()  # Protect queue operations
+        self.action_queue_size = []
+        self.start_barrier = threading.Barrier(2)  # 2 threads: action receiver, control loop
+
+        # FPS measurement
+        self.fps_tracker = FPSTracker(target_fps=self.config.fps)
+
+        self.logger.info("Robot connected and ready")
+
+        # Use an event for thread-safe coordination
+        self.must_go = threading.Event()
+        self.must_go.set()  # Initially set - observations qualify for direct processing
+
+    @property
+    def running(self):
+        return not self.shutdown_event.is_set()
+
+    def start(self):
+        """Start the robot client and connect to the policy server"""
+        try:
+            # client-server handshake
+            start_time = time.perf_counter()
+            self.stub.Ready(services_pb2.Empty())
+            end_time = time.perf_counter()
+            self.logger.debug(f"Connected to policy server in {end_time - start_time:.4f}s")
+
+            # send policy instructions
+            policy_config_bytes = pickle.dumps(self.policy_config)
+            policy_setup = services_pb2.PolicySetup(data=policy_config_bytes)
+
+            self.logger.info("Sending policy instructions to policy server")
+            self.logger.debug(
+                f"Policy type: {self.policy_config.policy_type} | "
+                f"Pretrained name or path: {self.policy_config.pretrained_name_or_path} | "
+                f"Device: {self.policy_config.device}"
+            )
+
+            self.stub.SendPolicyInstructions(policy_setup)
+
+            self.shutdown_event.clear()
+
+            return True
+
+        except grpc.RpcError as e:
+            self.logger.error(f"Failed to connect to policy server: {e}")
+            return False
+
+    def stop(self):
+        """Stop the robot client"""
+        self.shutdown_event.set()
+
+        self.robot.disconnect()
+        self.logger.debug("Robot disconnected")
+
+        self.channel.close()
+        self.logger.debug("Client stopped, channel closed")
+
+    def send_observation(
+        self,
+        obs: TimedObservation,
+    ) -> bool:
+        """Send observation to the policy server.
+        Returns True if the observation was sent successfully, False otherwise."""
+        if not self.running:
+            raise RuntimeError("Client not running. Run RobotClient.start() before sending observations.")
+
+        if not isinstance(obs, TimedObservation):
+            raise ValueError("Input observation needs to be a TimedObservation!")
+
+        start_time = time.perf_counter()
+        observation_bytes = pickle.dumps(obs)
+        serialize_time = time.perf_counter() - start_time
+        self.logger.debug(f"Observation serialization time: {serialize_time:.6f}s")
+
+        try:
+            observation_iterator = send_bytes_in_chunks(
+                observation_bytes,
+                services_pb2.Observation,
+                log_prefix="[CLIENT] Observation",
+                silent=True,
+            )
+            _ = self.stub.SendObservations(observation_iterator)
+            obs_timestep = obs.get_timestep()
+            self.logger.debug(f"Sent observation #{obs_timestep} | ")
+
+            return True
+
+        except grpc.RpcError as e:
+            self.logger.error(f"Error sending observation #{obs.get_timestep()}: {e}")
+            return False
+
+    def _inspect_action_queue(self):
+        with self.action_queue_lock:
+            queue_size = self.action_queue.qsize()
+            timestamps = sorted([action.get_timestep() for action in self.action_queue.queue])
+        self.logger.debug(f"Queue size: {queue_size}, Queue contents: {timestamps}")
+        return queue_size, timestamps
+
+    def _aggregate_action_queues(
+        self,
+        incoming_actions: list[TimedAction],
+        aggregate_fn: Callable[[torch.Tensor, torch.Tensor], torch.Tensor] | None = None,
+    ):
+        """Finds the same timestep actions in the queue and aggregates them using the aggregate_fn"""
+        if aggregate_fn is None:
+            # default aggregate function: take the latest action
+            def aggregate_fn(x1, x2):
+                return x2
+
+        future_action_queue = Queue()
+        with self.action_queue_lock:
+            internal_queue = self.action_queue.queue
+
+        current_action_queue = {action.get_timestep(): action.get_action() for action in internal_queue}
+
+        for new_action in incoming_actions:
+            with self.latest_action_lock:
+                latest_action = self.latest_action
+
+            # New action is older than the latest action in the queue, skip it
+            if new_action.get_timestep() <= latest_action:
+                continue
+
+            # If the new action's timestep is not in the current action queue, add it directly
+            elif new_action.get_timestep() not in current_action_queue:
+                future_action_queue.put(new_action)
+                continue
+
+            # If the new action's timestep is in the current action queue, aggregate it
+            # TODO: There is probably a way to do this with broadcasting of the two action tensors
+            future_action_queue.put(
+                TimedAction(
+                    timestamp=new_action.get_timestamp(),
+                    timestep=new_action.get_timestep(),
+                    action=aggregate_fn(
+                        current_action_queue[new_action.get_timestep()], new_action.get_action()
+                    ),
+                )
+            )
+
+        with self.action_queue_lock:
+            self.action_queue = future_action_queue
+
+    def receive_actions(self, verbose: bool = False):
+        """Receive actions from the policy server"""
+        # Wait at barrier for synchronized start
+        self.start_barrier.wait()
+        self.logger.info("Action receiving thread starting")
+
+        while self.running:
+            try:
+                # Use StreamActions to get a stream of actions from the server
+                actions_chunk = self.stub.GetActions(services_pb2.Empty())
+                if len(actions_chunk.data) == 0:
+                    continue  # received `Empty` from server, wait for next call
+
+                receive_time = time.time()
+
+                # Deserialize bytes back into list[TimedAction]
+                deserialize_start = time.perf_counter()
+                timed_actions = pickle.loads(actions_chunk.data)  # nosec
+                deserialize_time = time.perf_counter() - deserialize_start
+
+                # Log device type of received actions
+                if len(timed_actions) > 0:
+                    received_device = timed_actions[0].get_action().device.type
+                    self.logger.debug(f"Received actions on device: {received_device}")
+
+                # Move actions to client_device (e.g., for downstream planners that need GPU)
+                client_device = self.config.client_device
+                if client_device != "cpu":
+                    for timed_action in timed_actions:
+                        if timed_action.get_action().device.type != client_device:
+                            timed_action.action = timed_action.get_action().to(client_device)
+                    self.logger.debug(f"Converted actions to device: {client_device}")
+                else:
+                    self.logger.debug(f"Actions kept on device: {client_device}")
+
+                self.action_chunk_size = max(self.action_chunk_size, len(timed_actions))
+
+                # Calculate network latency if we have matching observations
+                if len(timed_actions) > 0 and verbose:
+                    with self.latest_action_lock:
+                        latest_action = self.latest_action
+
+                    self.logger.debug(f"Current latest action: {latest_action}")
+
+                    # Get queue state before changes
+                    old_size, old_timesteps = self._inspect_action_queue()
+                    if not old_timesteps:
+                        old_timesteps = [latest_action]  # queue was empty
+
+                    # Log incoming actions
+                    incoming_timesteps = [a.get_timestep() for a in timed_actions]
+
+                    first_action_timestep = timed_actions[0].get_timestep()
+                    server_to_client_latency = (receive_time - timed_actions[0].get_timestamp()) * 1000
+
+                    self.logger.info(
+                        f"Received action chunk for step #{first_action_timestep} | "
+                        f"Latest action: #{latest_action} | "
+                        f"Incoming actions: {incoming_timesteps[0]}:{incoming_timesteps[-1]} | "
+                        f"Network latency (server->client): {server_to_client_latency:.2f}ms | "
+                        f"Deserialization time: {deserialize_time * 1000:.2f}ms"
+                    )
+
+                # Update action queue
+                start_time = time.perf_counter()
+                self._aggregate_action_queues(timed_actions, self.config.aggregate_fn)
+                queue_update_time = time.perf_counter() - start_time
+
+                self.must_go.set()  # after receiving actions, next empty queue triggers must-go processing!
+
+                if verbose:
+                    # Get queue state after changes
+                    new_size, new_timesteps = self._inspect_action_queue()
+
+                    with self.latest_action_lock:
+                        latest_action = self.latest_action
+
+                    self.logger.info(
+                        f"Latest action: {latest_action} | "
+                        f"Old action steps: {old_timesteps[0]}:{old_timesteps[-1]} | "
+                        f"Incoming action steps: {incoming_timesteps[0]}:{incoming_timesteps[-1]} | "
+                        f"Updated action steps: {new_timesteps[0]}:{new_timesteps[-1]}"
+                    )
+                    self.logger.debug(
+                        f"Queue update complete ({queue_update_time:.6f}s) | "
+                        f"Before: {old_size} items | "
+                        f"After: {new_size} items | "
+                    )
+
+            except grpc.RpcError as e:
+                self.logger.error(f"Error receiving actions: {e}")
+
+    def actions_available(self):
+        """Check if there are actions available in the queue"""
+        with self.action_queue_lock:
+            return not self.action_queue.empty()
+
+    def _action_tensor_to_action_dict(self, action_tensor: torch.Tensor) -> dict[str, float]:
+        action = {key: action_tensor[i].item() for i, key in enumerate(self.robot.action_features)}
+        return action
+
+    def control_loop_action(self, verbose: bool = False) -> dict[str, Any]:
+        """Reading and performing actions in local queue"""
+
+        # Lock only for queue operations
+        get_start = time.perf_counter()
+        with self.action_queue_lock:
+            self.action_queue_size.append(self.action_queue.qsize())
+            # Get action from queue
+            timed_action = self.action_queue.get_nowait()
+        get_end = time.perf_counter() - get_start
+
+        _performed_action = self.robot.send_action(
+            self._action_tensor_to_action_dict(timed_action.get_action())
+        )
+        with self.latest_action_lock:
+            self.latest_action = timed_action.get_timestep()
+
+        if verbose:
+            with self.action_queue_lock:
+                current_queue_size = self.action_queue.qsize()
+
+            self.logger.debug(
+                f"Ts={timed_action.get_timestamp()} | "
+                f"Action #{timed_action.get_timestep()} performed | "
+                f"Queue size: {current_queue_size}"
+            )
+
+            self.logger.debug(
+                f"Popping action from queue to perform took {get_end:.6f}s | Queue size: {current_queue_size}"
+            )
+
+        return _performed_action
+
+    def _ready_to_send_observation(self):
+        """Flags when the client is ready to send an observation"""
+        with self.action_queue_lock:
+            return self.action_queue.qsize() / self.action_chunk_size <= self._chunk_size_threshold
+
+    def control_loop_observation(self, task: str, verbose: bool = False) -> RawObservation:
+        try:
+            # Get serialized observation bytes from the function
+            start_time = time.perf_counter()
+
+            raw_observation: RawObservation = self.robot.get_observation()
+            raw_observation["task"] = task
+
+            with self.latest_action_lock:
+                latest_action = self.latest_action
+
+            observation = TimedObservation(
+                timestamp=time.time(),  # need time.time() to compare timestamps across client and server
+                observation=raw_observation,
+                timestep=max(latest_action, 0),
+            )
+
+            obs_capture_time = time.perf_counter() - start_time
+
+            # If there are no actions left in the queue, the observation must go through processing!
+            with self.action_queue_lock:
+                observation.must_go = self.must_go.is_set() and self.action_queue.empty()
+                current_queue_size = self.action_queue.qsize()
+
+            _ = self.send_observation(observation)
+
+            self.logger.debug(f"QUEUE SIZE: {current_queue_size} (Must go: {observation.must_go})")
+            if observation.must_go:
+                # must-go event will be set again after receiving actions
+                self.must_go.clear()
+
+            if verbose:
+                # Calculate comprehensive FPS metrics
+                fps_metrics = self.fps_tracker.calculate_fps_metrics(observation.get_timestamp())
+
+                self.logger.info(
+                    f"Obs #{observation.get_timestep()} | "
+                    f"Avg FPS: {fps_metrics['avg_fps']:.2f} | "
+                    f"Target: {fps_metrics['target_fps']:.2f}"
+                )
+
+                self.logger.debug(
+                    f"Ts={observation.get_timestamp():.6f} | Capturing observation took {obs_capture_time:.6f}s"
+                )
+
+            return raw_observation
+
+        except Exception as e:
+            self.logger.error(f"Error in observation sender: {e}")
+
+    def control_loop(self, task: str, verbose: bool = False) -> tuple[Observation, Action]:
+        """Combined function for executing actions and streaming observations"""
+        # Wait at barrier for synchronized start
+        self.start_barrier.wait()
+        self.logger.info("Control loop thread starting")
+
+        _performed_action = None
+        _captured_observation = None
+
+        while self.running:
+            control_loop_start = time.perf_counter()
+            """Control loop: (1) Performing actions, when available"""
+            if self.actions_available():
+                _performed_action = self.control_loop_action(verbose)
+
+            """Control loop: (2) Streaming observations to the remote policy server"""
+            if self._ready_to_send_observation():
+                _captured_observation = self.control_loop_observation(task, verbose)
+
+            self.logger.debug(f"Control loop (ms): {(time.perf_counter() - control_loop_start) * 1000:.2f}")
+            # Dynamically adjust sleep time to maintain the desired control frequency
+            time.sleep(max(0, self.config.environment_dt - (time.perf_counter() - control_loop_start)))
+
+        return _captured_observation, _performed_action
+
+
+@draccus.wrap()
+def async_client(cfg: RobotClientConfig):
+    logging.info(pformat(asdict(cfg)))
+
+    # TODO: Assert if checking robot support is still needed with the plugin system
+    # if cfg.robot.type not in SUPPORTED_ROBOTS:
+    #     raise ValueError(f"Robot {cfg.robot.type} not yet supported!")
+
+    client = RobotClient(cfg)
+
+    if client.start():
+        client.logger.info("Starting action receiver thread...")
+
+        # Create and start action receiver thread
+        action_receiver_thread = threading.Thread(target=client.receive_actions, daemon=True)
+
+        # Start action receiver thread
+        action_receiver_thread.start()
+
+        try:
+            # The main thread runs the control loop
+            client.control_loop(task=cfg.task)
+
+        finally:
+            client.stop()
+            action_receiver_thread.join()
+            if cfg.debug_visualize_queue_size:
+                visualize_action_queue_size(client.action_queue_size)
+            client.logger.info("Client stopped")
+
+
+if __name__ == "__main__":
+    register_third_party_plugins()
+    async_client()  # run the client
diff --git a/lerobot/src/lerobot/cameras/__init__.py b/lerobot/src/lerobot/cameras/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..cbf1f11bff181a51d8a1a603e96d8c95a7bca00d
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/__init__.py
@@ -0,0 +1,17 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .camera import Camera
+from .configs import CameraConfig, ColorMode, Cv2Backends, Cv2Rotation
+from .utils import make_cameras_from_configs
diff --git a/lerobot/src/lerobot/cameras/camera.py b/lerobot/src/lerobot/cameras/camera.py
new file mode 100644
index 0000000000000000000000000000000000000000..2a53d25446734cf2411a7aadcb8fcb474cf00125
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/camera.py
@@ -0,0 +1,184 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import abc
+import warnings
+from typing import Any
+
+from numpy.typing import NDArray  # type: ignore  # TODO: add type stubs for numpy.typing
+
+from .configs import CameraConfig
+
+
+class Camera(abc.ABC):
+    """Base class for camera implementations.
+
+    Defines a standard interface for camera operations across different backends.
+    Subclasses must implement all abstract methods.
+
+    Manages basic camera properties (FPS, resolution) and core operations:
+    - Connection/disconnection
+    - Frame capture (sync/async/latest)
+
+    Attributes:
+        fps (int | None): Configured frames per second
+        width (int | None): Frame width in pixels
+        height (int | None): Frame height in pixels
+    """
+
+    def __init__(self, config: CameraConfig):
+        """Initialize the camera with the given configuration.
+
+        Args:
+            config: Camera configuration containing FPS and resolution.
+        """
+        self.fps: int | None = config.fps
+        self.width: int | None = config.width
+        self.height: int | None = config.height
+
+    def __enter__(self):
+        """
+        Context manager entry.
+        Automatically connects to the camera.
+        """
+        self.connect()
+        return self
+
+    def __exit__(self, exc_type, exc_value, traceback) -> None:
+        """
+        Context manager exit.
+        Automatically disconnects, ensuring resources are released even on error.
+        """
+        self.disconnect()
+
+    def __del__(self) -> None:
+        """
+        Destructor safety net.
+        Attempts to disconnect if the object is garbage collected without cleanup.
+        """
+        try:
+            if self.is_connected:
+                self.disconnect()
+        except Exception:  # nosec B110
+            pass
+
+    @property
+    @abc.abstractmethod
+    def is_connected(self) -> bool:
+        """Check if the camera is currently connected.
+
+        Returns:
+            bool: True if the camera is connected and ready to capture frames,
+                  False otherwise.
+        """
+        pass
+
+    @staticmethod
+    @abc.abstractmethod
+    def find_cameras() -> list[dict[str, Any]]:
+        """Detects available cameras connected to the system.
+        Returns:
+            List[Dict[str, Any]]: A list of dictionaries,
+            where each dictionary contains information about a detected camera.
+        """
+        pass
+
+    @abc.abstractmethod
+    def connect(self, warmup: bool = True) -> None:
+        """Establish connection to the camera.
+
+        Args:
+            warmup: If True (default), captures a warmup frame before returning. Useful
+                   for cameras that require time to adjust capture settings.
+                   If False, skips the warmup frame.
+        """
+        pass
+
+    @abc.abstractmethod
+    def read(self) -> NDArray[Any]:
+        """Capture and return a single frame from the camera synchronously.
+
+        This is a blocking call that will wait for the hardware and its SDK.
+
+        Returns:
+            np.ndarray: Captured frame as a numpy array.
+        """
+        pass
+
+    @abc.abstractmethod
+    def async_read(self, timeout_ms: float = ...) -> NDArray[Any]:
+        """Return the most recent new frame.
+
+        This method retrieves the latest frame captured by the background thread.
+        If a new frame is already available in the buffer (captured since the last call),
+        it returns it immediately.
+
+        It blocks up to `timeout_ms` only if the buffer is empty or if the latest frame
+        was already consumed by a previous `async_read` call.
+
+        Essentially, this method return the latest unconsumed frame, waiting if necessary
+        for a new one to arrive within the specified timeout.
+
+        Usage:
+            - Ideal for control loops where you want to ensure every processed frame
+            is fresh, effectively synchronizing your loop to the camera's FPS.
+            - Causes of a timeout usually include: very low camera FPS, heavy processing load,
+            or if the camera is disconnected.
+
+        Args:
+            timeout_ms: Maximum time to wait for a new frame in milliseconds.
+                        Defaults to 200ms (0.2s).
+
+        Returns:
+            np.ndarray: Captured frame as a numpy array.
+
+        Raises:
+            TimeoutError: If no new frame arrives within `timeout_ms`.
+        """
+        pass
+
+    def read_latest(self, max_age_ms: int = 500) -> NDArray[Any]:
+        """Return the most recent frame captured immediately (Peeking).
+
+        This method is non-blocking and returns whatever is currently in the
+        memory buffer. The frame may be stale,
+        meaning it could have been captured a while ago (hanging camera scenario e.g.).
+
+        Usage:
+            Ideal for scenarios requiring zero latency or decoupled frequencies & when
+            we want a guaranteed frame, such as UI visualization, logging, or
+            non-critical monitoring.
+
+        Returns:
+            NDArray[Any]: The frame image (numpy array).
+
+        Raises:
+            TimeoutError: If the latest frame is older than `max_age_ms`.
+            NotConnectedError: If the camera is not connected.
+            RuntimeError: If the camera is connected but has not captured any frames yet.
+        """
+        warnings.warn(
+            f"{self.__class__.__name__}.read_latest() is not implemented. "
+            "Please override read_latest(); it will be required in future releases.",
+            FutureWarning,
+            stacklevel=2,
+        )
+        return self.async_read()
+
+    @abc.abstractmethod
+    def disconnect(self) -> None:
+        """Disconnect from the camera and release resources."""
+        pass
diff --git a/lerobot/src/lerobot/cameras/configs.py b/lerobot/src/lerobot/cameras/configs.py
new file mode 100644
index 0000000000000000000000000000000000000000..987b7477510d85edb336899cf544fd9cc8567c86
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/configs.py
@@ -0,0 +1,67 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import abc
+from dataclasses import dataclass
+from enum import Enum
+
+import draccus  # type: ignore  # TODO: add type stubs for draccus
+
+
+class ColorMode(str, Enum):
+    RGB = "rgb"
+    BGR = "bgr"
+
+    @classmethod
+    def _missing_(cls, value: object) -> None:
+        raise ValueError(f"`color_mode` is expected to be in {list(cls)}, but {value} is provided.")
+
+
+class Cv2Rotation(int, Enum):
+    NO_ROTATION = 0
+    ROTATE_90 = 90
+    ROTATE_180 = 180
+    ROTATE_270 = -90
+
+    @classmethod
+    def _missing_(cls, value: object) -> None:
+        raise ValueError(f"`rotation` is expected to be in {list(cls)}, but {value} is provided.")
+
+
+# Subset from https://docs.opencv.org/3.4/d4/d15/group__videoio__flags__base.html
+class Cv2Backends(int, Enum):
+    ANY = 0
+    V4L2 = 200
+    DSHOW = 700
+    PVAPI = 800
+    ANDROID = 1000
+    AVFOUNDATION = 1200
+    MSMF = 1400
+
+    @classmethod
+    def _missing_(cls, value: object) -> None:
+        raise ValueError(f"`backend` is expected to be in {list(cls)}, but {value} is provided.")
+
+
+@dataclass(kw_only=True)
+class CameraConfig(draccus.ChoiceRegistry, abc.ABC):  # type: ignore  # TODO: add type stubs for draccus
+    fps: int | None = None
+    width: int | None = None
+    height: int | None = None
+
+    @property
+    def type(self) -> str:
+        return str(self.get_choice_name(self.__class__))
diff --git a/lerobot/src/lerobot/cameras/opencv/__init__.py b/lerobot/src/lerobot/cameras/opencv/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..377dafc05f2475c9238f1e73a6bc4dc2c80d3de4
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/opencv/__init__.py
@@ -0,0 +1,18 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .camera_opencv import OpenCVCamera
+from .configuration_opencv import OpenCVCameraConfig
+
+__all__ = ["OpenCVCamera", "OpenCVCameraConfig"]
diff --git a/lerobot/src/lerobot/cameras/opencv/camera_opencv.py b/lerobot/src/lerobot/cameras/opencv/camera_opencv.py
new file mode 100644
index 0000000000000000000000000000000000000000..f3289ddc7e38f3d418b53eba1d7268abb32d1264
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/opencv/camera_opencv.py
@@ -0,0 +1,592 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Provides the OpenCVCamera class for capturing frames from cameras using OpenCV.
+"""
+
+import logging
+import math
+import os
+import platform
+import time
+from pathlib import Path
+from threading import Event, Lock, Thread
+from typing import Any
+
+from numpy.typing import NDArray  # type: ignore  # TODO: add type stubs for numpy.typing
+
+# Fix MSMF hardware transform compatibility for Windows before importing cv2
+if platform.system() == "Windows" and "OPENCV_VIDEOIO_MSMF_ENABLE_HW_TRANSFORMS" not in os.environ:
+    os.environ["OPENCV_VIDEOIO_MSMF_ENABLE_HW_TRANSFORMS"] = "0"
+import cv2  # type: ignore  # TODO: add type stubs for OpenCV
+
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.errors import DeviceNotConnectedError
+
+from ..camera import Camera
+from ..utils import get_cv2_rotation
+from .configuration_opencv import ColorMode, OpenCVCameraConfig
+
+# NOTE(Steven): The maximum opencv device index depends on your operating system. For instance,
+# if you have 3 cameras, they should be associated to index 0, 1, and 2. This is the case
+# on MacOS. However, on Ubuntu, the indices are different like 6, 16, 23.
+# When you change the USB port or reboot the computer, the operating system might
+# treat the same cameras as new devices. Thus we select a higher bound to search indices.
+MAX_OPENCV_INDEX = 60
+
+logger = logging.getLogger(__name__)
+
+
+class OpenCVCamera(Camera):
+    """
+    Manages camera interactions using OpenCV for efficient frame recording.
+
+    This class provides a high-level interface to connect to, configure, and read
+    frames from cameras compatible with OpenCV's VideoCapture. It supports both
+    synchronous and asynchronous frame reading.
+
+    An OpenCVCamera instance requires a camera index (e.g., 0) or a device path
+    (e.g., '/dev/video0' on Linux). Camera indices can be unstable across reboots
+    or port changes, especially on Linux. Use the provided utility script to find
+    available camera indices or paths:
+    ```bash
+    lerobot-find-cameras opencv
+    ```
+
+    The camera's default settings (FPS, resolution, color mode) are used unless
+    overridden in the configuration.
+
+    Example:
+        ```python
+        from lerobot.cameras.opencv import OpenCVCamera
+        from lerobot.cameras.configuration_opencv import OpenCVCameraConfig
+
+        # Basic usage with camera index 0
+        config = OpenCVCameraConfig(index_or_path=0)
+        camera = OpenCVCamera(config)
+        camera.connect()
+
+        # Read 1 frame synchronously (blocking)
+        color_image = camera.read()
+
+        # Read 1 frame asynchronously (waits for new frame with a timeout)
+        async_image = camera.async_read()
+
+        # Get the latest frame immediately (no wait, returns timestamp)
+        latest_image, timestamp = camera.read_latest()
+
+        # When done, properly disconnect the camera using
+        camera.disconnect()
+        ```
+    """
+
+    def __init__(self, config: OpenCVCameraConfig):
+        """
+        Initializes the OpenCVCamera instance.
+
+        Args:
+            config: The configuration settings for the camera.
+        """
+        super().__init__(config)
+
+        self.config = config
+        self.index_or_path = config.index_or_path
+
+        self.fps = config.fps
+        self.color_mode = config.color_mode
+        self.warmup_s = config.warmup_s
+
+        self.videocapture: cv2.VideoCapture | None = None
+
+        self.thread: Thread | None = None
+        self.stop_event: Event | None = None
+        self.frame_lock: Lock = Lock()
+        self.latest_frame: NDArray[Any] | None = None
+        self.latest_timestamp: float | None = None
+        self.new_frame_event: Event = Event()
+
+        self.rotation: int | None = get_cv2_rotation(config.rotation)
+        self.backend: int = config.backend
+
+        if self.height and self.width:
+            self.capture_width, self.capture_height = self.width, self.height
+            if self.rotation in [cv2.ROTATE_90_CLOCKWISE, cv2.ROTATE_90_COUNTERCLOCKWISE]:
+                self.capture_width, self.capture_height = self.height, self.width
+
+    def __str__(self) -> str:
+        return f"{self.__class__.__name__}({self.index_or_path})"
+
+    @property
+    def is_connected(self) -> bool:
+        """Checks if the camera is currently connected and opened."""
+        return isinstance(self.videocapture, cv2.VideoCapture) and self.videocapture.isOpened()
+
+    @check_if_already_connected
+    def connect(self, warmup: bool = True) -> None:
+        """
+        Connects to the OpenCV camera specified in the configuration.
+
+        Initializes the OpenCV VideoCapture object, sets desired camera properties
+        (FPS, width, height), starts the background reading thread and performs initial checks.
+
+        Args:
+            warmup (bool): If True, waits at connect() time until at least one valid frame
+                           has been captured by the background thread. Defaults to True.
+
+        Raises:
+            DeviceAlreadyConnectedError: If the camera is already connected.
+            ConnectionError: If the specified camera index/path is not found or fails to open.
+            RuntimeError: If the camera opens but fails to apply requested settings.
+        """
+
+        # Use 1 thread for OpenCV operations to avoid potential conflicts or
+        # blocking in multi-threaded applications, especially during data collection.
+        cv2.setNumThreads(1)
+
+        self.videocapture = cv2.VideoCapture(self.index_or_path, self.backend)
+
+        if not self.videocapture.isOpened():
+            self.videocapture.release()
+            self.videocapture = None
+            raise ConnectionError(
+                f"Failed to open {self}.Run `lerobot-find-cameras opencv` to find available cameras."
+            )
+
+        self._configure_capture_settings()
+        self._start_read_thread()
+
+        if warmup and self.warmup_s > 0:
+            start_time = time.time()
+            while time.time() - start_time < self.warmup_s:
+                self.async_read(timeout_ms=self.warmup_s * 1000)
+                time.sleep(0.1)
+            with self.frame_lock:
+                if self.latest_frame is None:
+                    raise ConnectionError(f"{self} failed to capture frames during warmup.")
+
+        logger.info(f"{self} connected.")
+
+    @check_if_not_connected
+    def _configure_capture_settings(self) -> None:
+        """
+        Applies the specified FOURCC, FPS, width, and height settings to the connected camera.
+
+        This method attempts to set the camera properties via OpenCV. It checks if
+        the camera successfully applied the settings and raises an error if not.
+        FOURCC is set first (if specified) as it can affect the available FPS and resolution options.
+
+        Args:
+            fourcc: The desired FOURCC code (e.g., "MJPG", "YUYV"). If None, auto-detect.
+            fps: The desired frames per second. If None, the setting is skipped.
+            width: The desired capture width. If None, the setting is skipped.
+            height: The desired capture height. If None, the setting is skipped.
+
+        Raises:
+            RuntimeError: If the camera fails to set any of the specified properties
+                          to the requested value.
+            DeviceNotConnectedError: If the camera is not connected.
+        """
+
+        # Set FOURCC first (if specified) as it can affect available FPS/resolution options
+        if self.config.fourcc is not None:
+            self._validate_fourcc()
+        if self.videocapture is None:
+            raise DeviceNotConnectedError(f"{self} videocapture is not initialized")
+
+        default_width = int(round(self.videocapture.get(cv2.CAP_PROP_FRAME_WIDTH)))
+        default_height = int(round(self.videocapture.get(cv2.CAP_PROP_FRAME_HEIGHT)))
+
+        if self.width is None or self.height is None:
+            self.width, self.height = default_width, default_height
+            self.capture_width, self.capture_height = default_width, default_height
+            if self.rotation in [cv2.ROTATE_90_CLOCKWISE, cv2.ROTATE_90_COUNTERCLOCKWISE]:
+                self.width, self.height = default_height, default_width
+                self.capture_width, self.capture_height = default_width, default_height
+        else:
+            self._validate_width_and_height()
+
+        if self.fps is None:
+            self.fps = self.videocapture.get(cv2.CAP_PROP_FPS)
+        else:
+            self._validate_fps()
+
+    def _validate_fps(self) -> None:
+        """Validates and sets the camera's frames per second (FPS)."""
+
+        if self.videocapture is None:
+            raise DeviceNotConnectedError(f"{self} videocapture is not initialized")
+
+        if self.fps is None:
+            raise ValueError(f"{self} FPS is not set")
+
+        success = self.videocapture.set(cv2.CAP_PROP_FPS, float(self.fps))
+        actual_fps = self.videocapture.get(cv2.CAP_PROP_FPS)
+        # Use math.isclose for robust float comparison
+        if not success or not math.isclose(self.fps, actual_fps, rel_tol=1e-3):
+            raise RuntimeError(f"{self} failed to set fps={self.fps} ({actual_fps=}).")
+
+    def _validate_fourcc(self) -> None:
+        """Validates and sets the camera's FOURCC code."""
+
+        fourcc_code = cv2.VideoWriter_fourcc(*self.config.fourcc)
+
+        if self.videocapture is None:
+            raise DeviceNotConnectedError(f"{self} videocapture is not initialized")
+
+        success = self.videocapture.set(cv2.CAP_PROP_FOURCC, fourcc_code)
+        actual_fourcc_code = self.videocapture.get(cv2.CAP_PROP_FOURCC)
+
+        # Convert actual FOURCC code back to string for comparison
+        actual_fourcc_code_int = int(actual_fourcc_code)
+        actual_fourcc = "".join([chr((actual_fourcc_code_int >> 8 * i) & 0xFF) for i in range(4)])
+
+        if not success or actual_fourcc != self.config.fourcc:
+            logger.warning(
+                f"{self} failed to set fourcc={self.config.fourcc} (actual={actual_fourcc}, success={success}). "
+                f"Continuing with default format."
+            )
+
+    def _validate_width_and_height(self) -> None:
+        """Validates and sets the camera's frame capture width and height."""
+
+        if self.videocapture is None:
+            raise DeviceNotConnectedError(f"{self} videocapture is not initialized")
+
+        if self.capture_width is None or self.capture_height is None:
+            raise ValueError(f"{self} capture_width or capture_height is not set")
+
+        width_success = self.videocapture.set(cv2.CAP_PROP_FRAME_WIDTH, float(self.capture_width))
+        height_success = self.videocapture.set(cv2.CAP_PROP_FRAME_HEIGHT, float(self.capture_height))
+
+        actual_width = int(round(self.videocapture.get(cv2.CAP_PROP_FRAME_WIDTH)))
+        if not width_success or self.capture_width != actual_width:
+            raise RuntimeError(
+                f"{self} failed to set capture_width={self.capture_width} ({actual_width=}, {width_success=})."
+            )
+
+        actual_height = int(round(self.videocapture.get(cv2.CAP_PROP_FRAME_HEIGHT)))
+        if not height_success or self.capture_height != actual_height:
+            raise RuntimeError(
+                f"{self} failed to set capture_height={self.capture_height} ({actual_height=}, {height_success=})."
+            )
+
+    @staticmethod
+    def find_cameras() -> list[dict[str, Any]]:
+        """
+        Detects available OpenCV cameras connected to the system.
+
+        On Linux, it scans '/dev/video*' paths. On other systems (like macOS, Windows),
+        it checks indices from 0 up to `MAX_OPENCV_INDEX`.
+
+        Returns:
+            List[Dict[str, Any]]: A list of dictionaries,
+            where each dictionary contains 'type', 'id' (port index or path),
+            and the default profile properties (width, height, fps, format).
+        """
+        found_cameras_info = []
+
+        targets_to_scan: list[str | int]
+        if platform.system() == "Linux":
+            possible_paths = sorted(Path("/dev").glob("video*"), key=lambda p: p.name)
+            targets_to_scan = [str(p) for p in possible_paths]
+        else:
+            targets_to_scan = [int(i) for i in range(MAX_OPENCV_INDEX)]
+
+        for target in targets_to_scan:
+            camera = cv2.VideoCapture(target)
+            if camera.isOpened():
+                default_width = int(camera.get(cv2.CAP_PROP_FRAME_WIDTH))
+                default_height = int(camera.get(cv2.CAP_PROP_FRAME_HEIGHT))
+                default_fps = camera.get(cv2.CAP_PROP_FPS)
+                default_format = camera.get(cv2.CAP_PROP_FORMAT)
+
+                # Get FOURCC code and convert to string
+                default_fourcc_code = camera.get(cv2.CAP_PROP_FOURCC)
+                default_fourcc_code_int = int(default_fourcc_code)
+                default_fourcc = "".join([chr((default_fourcc_code_int >> 8 * i) & 0xFF) for i in range(4)])
+
+                camera_info = {
+                    "name": f"OpenCV Camera @ {target}",
+                    "type": "OpenCV",
+                    "id": target,
+                    "backend_api": camera.getBackendName(),
+                    "default_stream_profile": {
+                        "format": default_format,
+                        "fourcc": default_fourcc,
+                        "width": default_width,
+                        "height": default_height,
+                        "fps": default_fps,
+                    },
+                }
+
+                found_cameras_info.append(camera_info)
+                camera.release()
+
+        return found_cameras_info
+
+    def _read_from_hardware(self) -> NDArray[Any]:
+        if self.videocapture is None:
+            raise DeviceNotConnectedError(f"{self} videocapture is not initialized")
+
+        ret, frame = self.videocapture.read()
+
+        if not ret:
+            raise RuntimeError(f"{self} read failed (status={ret}).")
+
+        return frame
+
+    @check_if_not_connected
+    def read(self, color_mode: ColorMode | None = None) -> NDArray[Any]:
+        """
+        Reads a single frame synchronously from the camera.
+
+        This is a blocking call. It waits for the next available frame from the
+        camera hardware via OpenCV.
+
+        Returns:
+            np.ndarray: The captured frame as a NumPy array in the format
+                       (height, width, channels), using the specified or default
+                       color mode and applying any configured rotation.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is not connected.
+            RuntimeError: If reading the frame from the camera fails or if the
+                          received frame dimensions don't match expectations before rotation.
+            ValueError: If an invalid `color_mode` is requested.
+        """
+
+        start_time = time.perf_counter()
+
+        if color_mode is not None:
+            logger.warning(
+                f"{self} read() color_mode parameter is deprecated and will be removed in future versions."
+            )
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        self.new_frame_event.clear()
+        frame = self.async_read(timeout_ms=10000)
+
+        read_duration_ms = (time.perf_counter() - start_time) * 1e3
+        logger.debug(f"{self} read took: {read_duration_ms:.1f}ms")
+
+        return frame
+
+    def _postprocess_image(self, image: NDArray[Any]) -> NDArray[Any]:
+        """
+        Applies color conversion, dimension validation, and rotation to a raw frame.
+
+        Args:
+            image (np.ndarray): The raw image frame (expected BGR format from OpenCV).
+
+        Returns:
+            np.ndarray: The processed image frame.
+
+        Raises:
+            ValueError: If the requested `color_mode` is invalid.
+            RuntimeError: If the raw frame dimensions do not match the configured
+                          `width` and `height`.
+        """
+
+        if self.color_mode not in (ColorMode.RGB, ColorMode.BGR):
+            raise ValueError(
+                f"Invalid color mode '{self.color_mode}'. Expected {ColorMode.RGB} or {ColorMode.BGR}."
+            )
+
+        h, w, c = image.shape
+
+        if h != self.capture_height or w != self.capture_width:
+            raise RuntimeError(
+                f"{self} frame width={w} or height={h} do not match configured width={self.capture_width} or height={self.capture_height}."
+            )
+
+        if c != 3:
+            raise RuntimeError(f"{self} frame channels={c} do not match expected 3 channels (RGB/BGR).")
+
+        processed_image = image
+        if self.color_mode == ColorMode.RGB:
+            processed_image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)
+
+        if self.rotation in [cv2.ROTATE_90_CLOCKWISE, cv2.ROTATE_90_COUNTERCLOCKWISE, cv2.ROTATE_180]:
+            processed_image = cv2.rotate(processed_image, self.rotation)
+
+        return processed_image
+
+    def _read_loop(self) -> None:
+        """
+        Internal loop run by the background thread for asynchronous reading.
+
+        On each iteration:
+        1. Reads a color frame
+        2. Stores result in latest_frame and updates timestamp (thread-safe)
+        3. Sets new_frame_event to notify listeners
+
+        Stops on DeviceNotConnectedError, logs other errors and continues.
+        """
+        if self.stop_event is None:
+            raise RuntimeError(f"{self}: stop_event is not initialized before starting read loop.")
+
+        failure_count = 0
+        while not self.stop_event.is_set():
+            try:
+                raw_frame = self._read_from_hardware()
+                processed_frame = self._postprocess_image(raw_frame)
+                capture_time = time.perf_counter()
+
+                with self.frame_lock:
+                    self.latest_frame = processed_frame
+                    self.latest_timestamp = capture_time
+                self.new_frame_event.set()
+                failure_count = 0
+
+            except DeviceNotConnectedError:
+                break
+            except Exception as e:
+                if failure_count <= 10:
+                    failure_count += 1
+                    logger.warning(f"Error reading frame in background thread for {self}: {e}")
+                else:
+                    raise RuntimeError(f"{self} exceeded maximum consecutive read failures.") from e
+
+    def _start_read_thread(self) -> None:
+        """Starts or restarts the background read thread if it's not running."""
+        self._stop_read_thread()
+
+        self.stop_event = Event()
+        self.thread = Thread(target=self._read_loop, args=(), name=f"{self}_read_loop")
+        self.thread.daemon = True
+        self.thread.start()
+        time.sleep(0.1)
+
+    def _stop_read_thread(self) -> None:
+        """Signals the background read thread to stop and waits for it to join."""
+        if self.stop_event is not None:
+            self.stop_event.set()
+
+        if self.thread is not None and self.thread.is_alive():
+            self.thread.join(timeout=2.0)
+
+        self.thread = None
+        self.stop_event = None
+
+        with self.frame_lock:
+            self.latest_frame = None
+            self.latest_timestamp = None
+            self.new_frame_event.clear()
+
+    @check_if_not_connected
+    def async_read(self, timeout_ms: float = 200) -> NDArray[Any]:
+        """
+        Reads the latest available frame asynchronously.
+
+        This method retrieves the most recent frame captured by the background
+        read thread. It does not block waiting for the camera hardware directly,
+        but may wait up to timeout_ms for the background thread to provide a frame.
+        It is “best effort” under high FPS.
+
+        Args:
+            timeout_ms (float): Maximum time in milliseconds to wait for a frame
+                to become available. Defaults to 200ms (0.2 seconds).
+
+        Returns:
+            np.ndarray: The latest captured frame as a NumPy array in the format
+                       (height, width, channels), processed according to configuration.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is not connected.
+            TimeoutError: If no frame becomes available within the specified timeout.
+            RuntimeError: If an unexpected error occurs.
+        """
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        if not self.new_frame_event.wait(timeout=timeout_ms / 1000.0):
+            raise TimeoutError(
+                f"Timed out waiting for frame from camera {self} after {timeout_ms} ms. "
+                f"Read thread alive: {self.thread.is_alive()}."
+            )
+
+        with self.frame_lock:
+            frame = self.latest_frame
+            self.new_frame_event.clear()
+
+        if frame is None:
+            raise RuntimeError(f"Internal error: Event set but no frame available for {self}.")
+
+        return frame
+
+    @check_if_not_connected
+    def read_latest(self, max_age_ms: int = 500) -> NDArray[Any]:
+        """Return the most recent frame captured immediately (Peeking).
+
+        This method is non-blocking and returns whatever is currently in the
+        memory buffer. The frame may be stale,
+        meaning it could have been captured a while ago (hanging camera scenario e.g.).
+
+        Returns:
+            NDArray[Any]: The frame image (numpy array).
+
+        Raises:
+            TimeoutError: If the latest frame is older than `max_age_ms`.
+            DeviceNotConnectedError: If the camera is not connected.
+            RuntimeError: If the camera is connected but has not captured any frames yet.
+        """
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        with self.frame_lock:
+            frame = self.latest_frame
+            timestamp = self.latest_timestamp
+
+        if frame is None or timestamp is None:
+            raise RuntimeError(f"{self} has not captured any frames yet.")
+
+        age_ms = (time.perf_counter() - timestamp) * 1e3
+        if age_ms > max_age_ms:
+            raise TimeoutError(
+                f"{self} latest frame is too old: {age_ms:.1f} ms (max allowed: {max_age_ms} ms)."
+            )
+
+        return frame
+
+    def disconnect(self) -> None:
+        """
+        Disconnects from the camera and cleans up resources.
+
+        Stops the background read thread (if running) and releases the OpenCV
+        VideoCapture object.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is already disconnected.
+        """
+        if not self.is_connected and self.thread is None:
+            raise DeviceNotConnectedError(f"{self} not connected.")
+
+        if self.thread is not None:
+            self._stop_read_thread()
+
+        if self.videocapture is not None:
+            self.videocapture.release()
+            self.videocapture = None
+
+        with self.frame_lock:
+            self.latest_frame = None
+            self.latest_timestamp = None
+            self.new_frame_event.clear()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/cameras/opencv/configuration_opencv.py b/lerobot/src/lerobot/cameras/opencv/configuration_opencv.py
new file mode 100644
index 0000000000000000000000000000000000000000..8ae57fe3c8e6f435a0ffc4e8fe1e812f64564a5b
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/opencv/configuration_opencv.py
@@ -0,0 +1,76 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+from pathlib import Path
+
+from ..configs import CameraConfig, ColorMode, Cv2Backends, Cv2Rotation
+
+__all__ = ["OpenCVCameraConfig", "ColorMode", "Cv2Rotation", "Cv2Backends"]
+
+
+@CameraConfig.register_subclass("opencv")
+@dataclass
+class OpenCVCameraConfig(CameraConfig):
+    """Configuration class for OpenCV-based camera devices or video files.
+
+    This class provides configuration options for cameras accessed through OpenCV,
+    supporting both physical camera devices and video files. It includes settings
+    for resolution, frame rate, color mode, and image rotation.
+
+    Example configurations:
+    ```python
+    # Basic configurations
+    OpenCVCameraConfig(0, 30, 1280, 720)   # 1280x720 @ 30FPS
+    OpenCVCameraConfig(/dev/video4, 60, 640, 480)   # 640x480 @ 60FPS
+
+    # Advanced configurations with FOURCC format
+    OpenCVCameraConfig(128422271347, 30, 640, 480, rotation=Cv2Rotation.ROTATE_90, fourcc="MJPG")     # With 90° rotation and MJPG format
+    OpenCVCameraConfig(0, 30, 1280, 720, fourcc="YUYV")     # With YUYV format
+    ```
+
+    Attributes:
+        index_or_path: Either an integer representing the camera device index,
+                      or a Path object pointing to a video file.
+        fps: Requested frames per second for the color stream.
+        width: Requested frame width in pixels for the color stream.
+        height: Requested frame height in pixels for the color stream.
+        color_mode: Color mode for image output (RGB or BGR). Defaults to RGB.
+        rotation: Image rotation setting (0°, 90°, 180°, or 270°). Defaults to no rotation.
+        warmup_s: Time reading frames before returning from connect (in seconds)
+        fourcc: FOURCC code for video format (e.g., "MJPG", "YUYV", "I420"). Defaults to None (auto-detect).
+        backend: OpenCV backend identifier (https://docs.opencv.org/3.4/d4/d15/group__videoio__flags__base.html). Defaults to ANY.
+
+    Note:
+        - Only 3-channel color output (RGB/BGR) is currently supported.
+        - FOURCC codes must be 4-character strings (e.g., "MJPG", "YUYV"). Some common FOUCC codes: https://learn.microsoft.com/en-us/windows/win32/medfound/video-fourccs#fourcc-constants
+        - Setting FOURCC can help achieve higher frame rates on some cameras.
+    """
+
+    index_or_path: int | Path
+    color_mode: ColorMode = ColorMode.RGB
+    rotation: Cv2Rotation = Cv2Rotation.NO_ROTATION
+    warmup_s: int = 1
+    fourcc: str | None = None
+    backend: Cv2Backends = Cv2Backends.ANY
+
+    def __post_init__(self) -> None:
+        self.color_mode = ColorMode(self.color_mode)
+        self.rotation = Cv2Rotation(self.rotation)
+        self.backend = Cv2Backends(self.backend)
+
+        if self.fourcc is not None and (not isinstance(self.fourcc, str) or len(self.fourcc) != 4):
+            raise ValueError(
+                f"`fourcc` must be a 4-character string (e.g., 'MJPG', 'YUYV'), but '{self.fourcc}' is provided."
+            )
diff --git a/lerobot/src/lerobot/cameras/reachy2_camera/__init__.py b/lerobot/src/lerobot/cameras/reachy2_camera/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..72e45f32a043c7dc1bde52ded35e4700a1a73176
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/reachy2_camera/__init__.py
@@ -0,0 +1,16 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configuration_reachy2_camera import Reachy2CameraConfig
+from .reachy2_camera import Reachy2Camera
diff --git a/lerobot/src/lerobot/cameras/reachy2_camera/configuration_reachy2_camera.py b/lerobot/src/lerobot/cameras/reachy2_camera/configuration_reachy2_camera.py
new file mode 100644
index 0000000000000000000000000000000000000000..b40bfe71bb396de212419f383c1b9c58bac19eed
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/reachy2_camera/configuration_reachy2_camera.py
@@ -0,0 +1,77 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..configs import CameraConfig, ColorMode
+
+__all__ = ["CameraConfig", "ColorMode", "Reachy2CameraConfig"]
+
+
+@CameraConfig.register_subclass("reachy2_camera")
+@dataclass
+class Reachy2CameraConfig(CameraConfig):
+    """Configuration class for Reachy 2 camera devices.
+
+    This class provides configuration options for Reachy 2 cameras,
+    supporting both the teleop and depth cameras. It includes settings
+    for resolution, frame rate, color mode, and the selection of the cameras.
+
+    Example configurations:
+    ```python
+    # Basic configurations
+    Reachy2CameraConfig(
+        name="teleop",
+        image_type="left",
+        ip_address="192.168.0.200",  # IP address of the robot
+        port=50065,  # Port of the camera server
+        width=640,
+        height=480,
+        fps=30,  # Not configurable for Reachy 2 cameras
+        color_mode=ColorMode.RGB,
+    )  # Left teleop camera, 640x480 @ 30FPS
+    ```
+
+    Attributes:
+        name: Name of the camera device. Can be "teleop" or "depth".
+        image_type: Type of image stream. For "teleop" camera, can be "left" or "right".
+                    For "depth" camera, can be "rgb" or "depth". (depth is not supported yet)
+        fps: Requested frames per second for the color stream. Not configurable for Reachy 2 cameras.
+        width: Requested frame width in pixels for the color stream.
+        height: Requested frame height in pixels for the color stream.
+        color_mode: Color mode for image output (RGB or BGR). Defaults to RGB.
+        ip_address: IP address of the robot. Defaults to "localhost".
+        port: Port number for the camera server. Defaults to 50065.
+
+    Note:
+        - Only 3-channel color output (RGB/BGR) is currently supported.
+    """
+
+    name: str
+    image_type: str
+    color_mode: ColorMode = ColorMode.RGB
+    ip_address: str | None = "localhost"
+    port: int = 50065
+
+    def __post_init__(self) -> None:
+        if self.name not in ["teleop", "depth"]:
+            raise ValueError(f"`name` is expected to be 'teleop' or 'depth', but {self.name} is provided.")
+        if (self.name == "teleop" and self.image_type not in ["left", "right"]) or (
+            self.name == "depth" and self.image_type not in ["rgb", "depth"]
+        ):
+            raise ValueError(
+                f"`image_type` is expected to be 'left' or 'right' for teleop camera, and 'rgb' or 'depth' for depth camera, but {self.image_type} is provided."
+            )
+
+        self.color_mode = ColorMode(self.color_mode)
diff --git a/lerobot/src/lerobot/cameras/reachy2_camera/reachy2_camera.py b/lerobot/src/lerobot/cameras/reachy2_camera/reachy2_camera.py
new file mode 100644
index 0000000000000000000000000000000000000000..9bef957bc66b1af2c4d3a855cc098bec2ea8bbd8
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/reachy2_camera/reachy2_camera.py
@@ -0,0 +1,245 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Provides the Reachy2Camera class for capturing frames from Reachy 2 cameras using Reachy 2's CameraManager.
+"""
+
+from __future__ import annotations
+
+import logging
+import os
+import platform
+import time
+from typing import TYPE_CHECKING, Any
+
+from numpy.typing import NDArray  # type: ignore  # TODO: add type stubs for numpy.typing
+
+# Fix MSMF hardware transform compatibility for Windows before importing cv2
+if platform.system() == "Windows" and "OPENCV_VIDEOIO_MSMF_ENABLE_HW_TRANSFORMS" not in os.environ:
+    os.environ["OPENCV_VIDEOIO_MSMF_ENABLE_HW_TRANSFORMS"] = "0"
+import cv2  # type: ignore  # TODO: add type stubs for OpenCV
+import numpy as np  # type: ignore  # TODO: add type stubs for numpy
+
+from lerobot.utils.decorators import check_if_not_connected
+from lerobot.utils.import_utils import _reachy2_sdk_available
+
+if TYPE_CHECKING or _reachy2_sdk_available:
+    from reachy2_sdk.media.camera import CameraView
+    from reachy2_sdk.media.camera_manager import CameraManager
+else:
+    CameraManager = None
+
+    class CameraView:
+        LEFT = 0
+        RIGHT = 1
+
+
+from lerobot.utils.errors import DeviceNotConnectedError
+
+from ..camera import Camera
+from .configuration_reachy2_camera import ColorMode, Reachy2CameraConfig
+
+logger = logging.getLogger(__name__)
+
+
+class Reachy2Camera(Camera):
+    """
+    Manages Reachy 2 camera using Reachy 2 CameraManager.
+
+    This class provides a high-level interface to connect to, configure, and read
+    frames from Reachy 2 cameras. It supports both synchronous and asynchronous
+    frame reading.
+
+    An Reachy2Camera instance requires a camera name (e.g., "teleop") and an image
+    type (e.g., "left") to be specified in the configuration.
+
+    The camera's default settings (FPS, resolution, color mode) are used unless
+    overridden in the configuration.
+    """
+
+    def __init__(self, config: Reachy2CameraConfig):
+        """
+        Initializes the Reachy2Camera instance.
+
+        Args:
+            config: The configuration settings for the camera.
+        """
+        super().__init__(config)
+
+        self.config = config
+
+        self.color_mode = config.color_mode
+        self.latest_frame: NDArray[Any] | None = None
+        self.latest_timestamp: float | None = None
+
+        self.cam_manager: CameraManager | None = None
+
+    def __str__(self) -> str:
+        return f"{self.__class__.__name__}({self.config.name}, {self.config.image_type})"
+
+    @property
+    def is_connected(self) -> bool:
+        """Checks if the camera is currently connected and opened."""
+        if self.config.name == "teleop":
+            return bool(
+                self.cam_manager._grpc_connected and self.cam_manager.teleop if self.cam_manager else False
+            )
+        elif self.config.name == "depth":
+            return bool(
+                self.cam_manager._grpc_connected and self.cam_manager.depth if self.cam_manager else False
+            )
+        else:
+            raise ValueError(f"Invalid camera name '{self.config.name}'. Expected 'teleop' or 'depth'.")
+
+    def connect(self, warmup: bool = True) -> None:
+        """
+        Connects to the Reachy2 CameraManager as specified in the configuration.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is not connected.
+        """
+        self.cam_manager = CameraManager(host=self.config.ip_address, port=self.config.port)
+        if self.cam_manager is None:
+            raise DeviceNotConnectedError(f"Could not connect to {self}.")
+        self.cam_manager.initialize_cameras()
+
+        logger.info(f"{self} connected.")
+
+    @staticmethod
+    def find_cameras() -> list[dict[str, Any]]:
+        """
+        Detection not implemented for Reachy2 cameras.
+        """
+        raise NotImplementedError("Camera detection is not implemented for Reachy2 cameras.")
+
+    @check_if_not_connected
+    def read(self, color_mode: ColorMode | None = None) -> NDArray[Any]:
+        """
+        Reads a single frame synchronously from the camera.
+
+        This method retrieves the most recent frame available in Reachy 2's low-level software.
+
+        Returns:
+            np.ndarray: The captured frame as a NumPy array in the format
+                       (height, width, channels), using the specified or default
+                       color mode and applying any configured rotation.
+        """
+        start_time = time.perf_counter()
+
+        if self.cam_manager is None:
+            raise DeviceNotConnectedError(f"{self} is not connected.")
+
+        if color_mode is not None:
+            logger.warning(
+                f"{self} read() color_mode parameter is deprecated and will be removed in future versions."
+            )
+
+        frame: NDArray[Any] = np.empty((0, 0, 3), dtype=np.uint8)
+
+        if self.config.name == "teleop" and hasattr(self.cam_manager, "teleop"):
+            if self.config.image_type == "left":
+                frame = self.cam_manager.teleop.get_frame(
+                    CameraView.LEFT, size=(self.config.width, self.config.height)
+                )[0]
+            elif self.config.image_type == "right":
+                frame = self.cam_manager.teleop.get_frame(
+                    CameraView.RIGHT, size=(self.config.width, self.config.height)
+                )[0]
+        elif self.config.name == "depth" and hasattr(self.cam_manager, "depth"):
+            if self.config.image_type == "depth":
+                frame = self.cam_manager.depth.get_depth_frame()[0]
+            elif self.config.image_type == "rgb":
+                frame = self.cam_manager.depth.get_frame(size=(self.config.width, self.config.height))[0]
+        else:
+            raise ValueError(f"Invalid camera name '{self.config.name}'. Expected 'teleop' or 'depth'.")
+
+        if frame is None:
+            raise RuntimeError(f"Internal error: No frame available for {self}.")
+
+        if self.color_mode not in (ColorMode.RGB, ColorMode.BGR):
+            raise ValueError(
+                f"Invalid color mode '{self.color_mode}'. Expected {ColorMode.RGB} or {ColorMode.BGR}."
+            )
+        if self.color_mode == ColorMode.RGB:
+            frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
+
+        self.latest_frame = frame
+        self.latest_timestamp = time.perf_counter()
+
+        read_duration_ms = (time.perf_counter() - start_time) * 1e3
+        logger.debug(f"{self} read took: {read_duration_ms:.1f}ms")
+
+        return frame
+
+    @check_if_not_connected
+    def async_read(self, timeout_ms: float = 200) -> NDArray[Any]:
+        """
+        Same as read()
+
+        Returns:
+            np.ndarray: The latest captured frame as a NumPy array in the format
+                       (height, width, channels), processed according to configuration.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is not connected.
+            TimeoutError: If no frame becomes available within the specified timeout.
+            RuntimeError: If an unexpected error occurs.
+        """
+
+        return self.read()
+
+    @check_if_not_connected
+    def read_latest(self, max_age_ms: int = 500) -> NDArray[Any]:
+        """Return the most recent frame captured immediately (Peeking).
+
+        This method is non-blocking and returns whatever is currently in the
+        memory buffer. The frame may be stale,
+        meaning it could have been captured a while ago (hanging camera scenario e.g.).
+
+        Returns:
+            tuple[NDArray, float]:
+                - The frame image (numpy array).
+                - The timestamp (time.perf_counter) when this frame was captured.
+
+        Raises:
+            TimeoutError: If the latest frame is older than `max_age_ms`.
+            DeviceNotConnectedError: If the camera is not connected.
+            RuntimeError: If the camera is connected but has not captured any frames yet.
+        """
+
+        if self.latest_frame is None or self.latest_timestamp is None:
+            raise RuntimeError(f"{self} has not captured any frames yet.")
+
+        age_ms = (time.perf_counter() - self.latest_timestamp) * 1e3
+        if age_ms > max_age_ms:
+            raise TimeoutError(
+                f"{self} latest frame is too old: {age_ms:.1f} ms (max allowed: {max_age_ms} ms)."
+            )
+
+        return self.latest_frame
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        """
+        Stops the background read thread (if running).
+
+        Raises:
+            DeviceNotConnectedError: If the camera is already disconnected.
+        """
+
+        if self.cam_manager is not None:
+            self.cam_manager.disconnect()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/cameras/realsense/__init__.py b/lerobot/src/lerobot/cameras/realsense/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..67f2f4000d61397e15271b2fa6a6fff089fc5a1a
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/realsense/__init__.py
@@ -0,0 +1,16 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .camera_realsense import RealSenseCamera
+from .configuration_realsense import RealSenseCameraConfig
diff --git a/lerobot/src/lerobot/cameras/realsense/camera_realsense.py b/lerobot/src/lerobot/cameras/realsense/camera_realsense.py
new file mode 100644
index 0000000000000000000000000000000000000000..d80ec8093801765dafb6b400dcce317fde3b49a7
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/realsense/camera_realsense.py
@@ -0,0 +1,639 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Provides the RealSenseCamera class for capturing frames from Intel RealSense cameras.
+"""
+
+import logging
+import time
+from threading import Event, Lock, Thread
+from typing import Any
+
+import cv2  # type: ignore  # TODO: add type stubs for OpenCV
+import numpy as np  # type: ignore  # TODO: add type stubs for numpy
+from numpy.typing import NDArray  # type: ignore  # TODO: add type stubs for numpy.typing
+
+try:
+    import pyrealsense2 as rs  # type: ignore  # TODO: add type stubs for pyrealsense2
+except Exception as e:
+    logging.info(f"Could not import realsense: {e}")
+
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.errors import DeviceNotConnectedError
+
+from ..camera import Camera
+from ..configs import ColorMode
+from ..utils import get_cv2_rotation
+from .configuration_realsense import RealSenseCameraConfig
+
+logger = logging.getLogger(__name__)
+
+
+class RealSenseCamera(Camera):
+    """
+    Manages interactions with Intel RealSense cameras for frame and depth recording.
+
+    This class provides an interface similar to `OpenCVCamera` but tailored for
+    RealSense devices, leveraging the `pyrealsense2` library. It uses the camera's
+    unique serial number for identification, offering more stability than device
+    indices, especially on Linux. It also supports capturing depth maps alongside
+    color frames.
+
+    Use the provided utility script to find available camera indices and default profiles:
+    ```bash
+    lerobot-find-cameras realsense
+    ```
+
+    A `RealSenseCamera` instance requires a configuration object specifying the
+    camera's serial number or a unique device name. If using the name, ensure only
+    one camera with that name is connected.
+
+    The camera's default settings (FPS, resolution, color mode) from the stream
+    profile are used unless overridden in the configuration.
+
+    Example:
+        ```python
+        from lerobot.cameras.realsense import RealSenseCamera, RealSenseCameraConfig
+        from lerobot.cameras import ColorMode, Cv2Rotation
+
+        # Basic usage with serial number
+        config = RealSenseCameraConfig(serial_number_or_name="0123456789") # Replace with actual SN
+        camera = RealSenseCamera(config)
+        camera.connect()
+
+        # Read 1 frame synchronously (blocking)
+        color_image = camera.read()
+
+        # Read 1 frame asynchronously (waits for new frame with a timeout)
+        async_image = camera.async_read()
+
+        # Get the latest frame immediately (no wait, returns timestamp)
+        latest_image, timestamp = camera.read_latest()
+
+        # Example with depth capture and custom settings
+        custom_config = RealSenseCameraConfig(
+            serial_number_or_name="0123456789", # Replace with actual SN
+            fps=30,
+            width=1280,
+            height=720,
+            color_mode=ColorMode.BGR, # Request BGR output
+            rotation=Cv2Rotation.NO_ROTATION,
+            use_depth=True
+        )
+        depth_camera = RealSenseCamera(custom_config)
+        depth_camera.connect()
+
+        # Read 1 depth frame
+        depth_map = depth_camera.read_depth()
+
+        # Example using a unique camera name
+        name_config = RealSenseCameraConfig(serial_number_or_name="Intel RealSense D435") # If unique
+        name_camera = RealSenseCamera(name_config)
+        # ... connect, read, disconnect ...
+        ```
+    """
+
+    def __init__(self, config: RealSenseCameraConfig):
+        """
+        Initializes the RealSenseCamera instance.
+
+        Args:
+            config: The configuration settings for the camera.
+        """
+
+        super().__init__(config)
+
+        self.config = config
+
+        if config.serial_number_or_name.isdigit():
+            self.serial_number = config.serial_number_or_name
+        else:
+            self.serial_number = self._find_serial_number_from_name(config.serial_number_or_name)
+
+        self.fps = config.fps
+        self.color_mode = config.color_mode
+        self.use_depth = config.use_depth
+        self.warmup_s = config.warmup_s
+
+        self.rs_pipeline: rs.pipeline | None = None
+        self.rs_profile: rs.pipeline_profile | None = None
+
+        self.thread: Thread | None = None
+        self.stop_event: Event | None = None
+        self.frame_lock: Lock = Lock()
+        self.latest_color_frame: NDArray[Any] | None = None
+        self.latest_depth_frame: NDArray[Any] | None = None
+        self.latest_timestamp: float | None = None
+        self.new_frame_event: Event = Event()
+
+        self.rotation: int | None = get_cv2_rotation(config.rotation)
+
+        if self.height and self.width:
+            self.capture_width, self.capture_height = self.width, self.height
+            if self.rotation in [cv2.ROTATE_90_CLOCKWISE, cv2.ROTATE_90_COUNTERCLOCKWISE]:
+                self.capture_width, self.capture_height = self.height, self.width
+
+    def __str__(self) -> str:
+        return f"{self.__class__.__name__}({self.serial_number})"
+
+    @property
+    def is_connected(self) -> bool:
+        """Checks if the camera pipeline is started and streams are active."""
+        return self.rs_pipeline is not None and self.rs_profile is not None
+
+    @check_if_already_connected
+    def connect(self, warmup: bool = True) -> None:
+        """
+        Connects to the RealSense camera specified in the configuration.
+
+        Initializes the RealSense pipeline, configures the required streams (color
+        and optionally depth), starts the pipeline, and validates the actual stream settings.
+
+        Args:
+            warmup (bool): If True, waits at connect() time until at least one valid frame
+                           has been captured by the background thread. Defaults to True.
+
+        Raises:
+            DeviceAlreadyConnectedError: If the camera is already connected.
+            ValueError: If the configuration is invalid (e.g., missing serial/name, name not unique).
+            ConnectionError: If the camera is found but fails to start the pipeline or no RealSense devices are detected at all.
+            RuntimeError: If the pipeline starts but fails to apply requested settings.
+        """
+
+        self.rs_pipeline = rs.pipeline()
+        rs_config = rs.config()
+        self._configure_rs_pipeline_config(rs_config)
+
+        try:
+            self.rs_profile = self.rs_pipeline.start(rs_config)
+        except RuntimeError as e:
+            self.rs_profile = None
+            self.rs_pipeline = None
+            raise ConnectionError(
+                f"Failed to open {self}.Run `lerobot-find-cameras realsense` to find available cameras."
+            ) from e
+
+        self._configure_capture_settings()
+        self._start_read_thread()
+
+        # NOTE(Steven/Caroline): Enforcing at least one second of warmup as RS cameras need a bit of time before the first read. If we don't wait, the first read from the warmup will raise.
+        self.warmup_s = max(self.warmup_s, 1)
+
+        start_time = time.time()
+        while time.time() - start_time < self.warmup_s:
+            self.async_read(timeout_ms=self.warmup_s * 1000)
+            time.sleep(0.1)
+        with self.frame_lock:
+            if self.latest_color_frame is None or self.use_depth and self.latest_depth_frame is None:
+                raise ConnectionError(f"{self} failed to capture frames during warmup.")
+
+        logger.info(f"{self} connected.")
+
+    @staticmethod
+    def find_cameras() -> list[dict[str, Any]]:
+        """
+        Detects available Intel RealSense cameras connected to the system.
+
+        Returns:
+            List[Dict[str, Any]]: A list of dictionaries,
+            where each dictionary contains 'type', 'id' (serial number), 'name',
+            firmware version, USB type, and other available specs, and the default profile properties (width, height, fps, format).
+
+        Raises:
+            OSError: If pyrealsense2 is not installed.
+            ImportError: If pyrealsense2 is not installed.
+        """
+        found_cameras_info = []
+        context = rs.context()
+        devices = context.query_devices()
+
+        for device in devices:
+            camera_info = {
+                "name": device.get_info(rs.camera_info.name),
+                "type": "RealSense",
+                "id": device.get_info(rs.camera_info.serial_number),
+                "firmware_version": device.get_info(rs.camera_info.firmware_version),
+                "usb_type_descriptor": device.get_info(rs.camera_info.usb_type_descriptor),
+                "physical_port": device.get_info(rs.camera_info.physical_port),
+                "product_id": device.get_info(rs.camera_info.product_id),
+                "product_line": device.get_info(rs.camera_info.product_line),
+            }
+
+            # Get stream profiles for each sensor
+            sensors = device.query_sensors()
+            for sensor in sensors:
+                profiles = sensor.get_stream_profiles()
+
+                for profile in profiles:
+                    if profile.is_video_stream_profile() and profile.is_default():
+                        vprofile = profile.as_video_stream_profile()
+                        stream_info = {
+                            "stream_type": vprofile.stream_name(),
+                            "format": vprofile.format().name,
+                            "width": vprofile.width(),
+                            "height": vprofile.height(),
+                            "fps": vprofile.fps(),
+                        }
+                        camera_info["default_stream_profile"] = stream_info
+
+            found_cameras_info.append(camera_info)
+
+        return found_cameras_info
+
+    def _find_serial_number_from_name(self, name: str) -> str:
+        """Finds the serial number for a given unique camera name."""
+        camera_infos = self.find_cameras()
+        found_devices = [cam for cam in camera_infos if str(cam["name"]) == name]
+
+        if not found_devices:
+            available_names = [cam["name"] for cam in camera_infos]
+            raise ValueError(
+                f"No RealSense camera found with name '{name}'. Available camera names: {available_names}"
+            )
+
+        if len(found_devices) > 1:
+            serial_numbers = [dev["serial_number"] for dev in found_devices]
+            raise ValueError(
+                f"Multiple RealSense cameras found with name '{name}'. "
+                f"Please use a unique serial number instead. Found SNs: {serial_numbers}"
+            )
+
+        serial_number = str(found_devices[0]["serial_number"])
+        return serial_number
+
+    def _configure_rs_pipeline_config(self, rs_config: Any) -> None:
+        """Creates and configures the RealSense pipeline configuration object."""
+        rs.config.enable_device(rs_config, self.serial_number)
+
+        if self.width and self.height and self.fps:
+            rs_config.enable_stream(
+                rs.stream.color, self.capture_width, self.capture_height, rs.format.rgb8, self.fps
+            )
+            if self.use_depth:
+                rs_config.enable_stream(
+                    rs.stream.depth, self.capture_width, self.capture_height, rs.format.z16, self.fps
+                )
+        else:
+            rs_config.enable_stream(rs.stream.color)
+            if self.use_depth:
+                rs_config.enable_stream(rs.stream.depth)
+
+    @check_if_not_connected
+    def _configure_capture_settings(self) -> None:
+        """Sets fps, width, and height from device stream if not already configured.
+
+        Uses the color stream profile to update unset attributes. Handles rotation by
+        swapping width/height when needed. Original capture dimensions are always stored.
+
+        Raises:
+            DeviceNotConnectedError: If device is not connected.
+        """
+
+        if self.rs_profile is None:
+            raise RuntimeError(f"{self}: rs_profile must be initialized before use.")
+
+        stream = self.rs_profile.get_stream(rs.stream.color).as_video_stream_profile()
+
+        if self.fps is None:
+            self.fps = stream.fps()
+
+        if self.width is None or self.height is None:
+            actual_width = int(round(stream.width()))
+            actual_height = int(round(stream.height()))
+            if self.rotation in [cv2.ROTATE_90_CLOCKWISE, cv2.ROTATE_90_COUNTERCLOCKWISE]:
+                self.width, self.height = actual_height, actual_width
+                self.capture_width, self.capture_height = actual_width, actual_height
+            else:
+                self.width, self.height = actual_width, actual_height
+                self.capture_width, self.capture_height = actual_width, actual_height
+
+    @check_if_not_connected
+    def read_depth(self, timeout_ms: int = 200) -> NDArray[Any]:
+        """
+        Reads a single frame (depth) synchronously from the camera.
+
+        This is a blocking call. It waits for a coherent set of frames (depth)
+        from the camera hardware via the RealSense pipeline.
+
+        Returns:
+            np.ndarray: The depth map as a NumPy array (height, width)
+                  of type `np.uint16` (raw depth values in millimeters) and rotation.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is not connected.
+            RuntimeError: If reading frames from the pipeline fails or frames are invalid.
+        """
+        if timeout_ms:
+            logger.warning(
+                f"{self} read() timeout_ms parameter is deprecated and will be removed in future versions."
+            )
+
+        if not self.use_depth:
+            raise RuntimeError(
+                f"Failed to capture depth frame '.read_depth()'. Depth stream is not enabled for {self}."
+            )
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        self.new_frame_event.clear()
+
+        _ = self.async_read(timeout_ms=10000)
+
+        with self.frame_lock:
+            depth_map = self.latest_depth_frame
+
+        if depth_map is None:
+            raise RuntimeError("No depth frame available. Ensure camera is streaming.")
+
+        return depth_map
+
+    def _read_from_hardware(self):
+        if self.rs_pipeline is None:
+            raise RuntimeError(f"{self}: rs_pipeline must be initialized before use.")
+
+        ret, frame = self.rs_pipeline.try_wait_for_frames(timeout_ms=10000)
+
+        if not ret or frame is None:
+            raise RuntimeError(f"{self} read failed (status={ret}).")
+
+        return frame
+
+    @check_if_not_connected
+    def read(self, color_mode: ColorMode | None = None, timeout_ms: int = 0) -> NDArray[Any]:
+        """
+        Reads a single frame (color) synchronously from the camera.
+
+        This is a blocking call. It waits for a coherent set of frames (color)
+        from the camera hardware via the RealSense pipeline.
+
+        Returns:
+            np.ndarray: The captured color frame as a NumPy array
+              (height, width, channels), processed according to `color_mode` and rotation.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is not connected.
+            RuntimeError: If reading frames from the pipeline fails or frames are invalid.
+            ValueError: If an invalid `color_mode` is requested.
+        """
+
+        start_time = time.perf_counter()
+
+        if color_mode is not None:
+            logger.warning(
+                f"{self} read() color_mode parameter is deprecated and will be removed in future versions."
+            )
+
+        if timeout_ms:
+            logger.warning(
+                f"{self} read() timeout_ms parameter is deprecated and will be removed in future versions."
+            )
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        self.new_frame_event.clear()
+
+        frame = self.async_read(timeout_ms=10000)
+
+        read_duration_ms = (time.perf_counter() - start_time) * 1e3
+        logger.debug(f"{self} read took: {read_duration_ms:.1f}ms")
+
+        return frame
+
+    def _postprocess_image(self, image: NDArray[Any], depth_frame: bool = False) -> NDArray[Any]:
+        """
+        Applies color conversion, dimension validation, and rotation to a raw color frame.
+
+        Args:
+            image (np.ndarray): The raw image frame (expected RGB format from RealSense).
+
+        Returns:
+            np.ndarray: The processed image frame according to `self.color_mode` and `self.rotation`.
+
+        Raises:
+            ValueError: If the requested `color_mode` is invalid.
+            RuntimeError: If the raw frame dimensions do not match the configured
+                          `width` and `height`.
+        """
+
+        if self.color_mode and self.color_mode not in (ColorMode.RGB, ColorMode.BGR):
+            raise ValueError(
+                f"Invalid requested color mode '{self.color_mode}'. Expected {ColorMode.RGB} or {ColorMode.BGR}."
+            )
+
+        if depth_frame:
+            h, w = image.shape
+        else:
+            h, w, c = image.shape
+
+            if c != 3:
+                raise RuntimeError(f"{self} frame channels={c} do not match expected 3 channels (RGB/BGR).")
+
+        if h != self.capture_height or w != self.capture_width:
+            raise RuntimeError(
+                f"{self} frame width={w} or height={h} do not match configured width={self.capture_width} or height={self.capture_height}."
+            )
+
+        processed_image = image
+        if self.color_mode == ColorMode.BGR:
+            processed_image = cv2.cvtColor(image, cv2.COLOR_RGB2BGR)
+
+        if self.rotation in [cv2.ROTATE_90_CLOCKWISE, cv2.ROTATE_90_COUNTERCLOCKWISE, cv2.ROTATE_180]:
+            processed_image = cv2.rotate(processed_image, self.rotation)
+
+        return processed_image
+
+    def _read_loop(self) -> None:
+        """
+        Internal loop run by the background thread for asynchronous reading.
+
+        On each iteration:
+        1. Reads a color frame with 500ms timeout
+        2. Stores result in latest_frame and updates timestamp (thread-safe)
+        3. Sets new_frame_event to notify listeners
+
+        Stops on DeviceNotConnectedError, logs other errors and continues.
+        """
+        if self.stop_event is None:
+            raise RuntimeError(f"{self}: stop_event is not initialized before starting read loop.")
+
+        failure_count = 0
+        while not self.stop_event.is_set():
+            try:
+                frame = self._read_from_hardware()
+                color_frame_raw = frame.get_color_frame()
+                color_frame = np.asanyarray(color_frame_raw.get_data())
+                processed_color_frame = self._postprocess_image(color_frame)
+
+                if self.use_depth:
+                    depth_frame_raw = frame.get_depth_frame()
+                    depth_frame = np.asanyarray(depth_frame_raw.get_data())
+                    processed_depth_frame = self._postprocess_image(depth_frame, depth_frame=True)
+
+                capture_time = time.perf_counter()
+
+                with self.frame_lock:
+                    self.latest_color_frame = processed_color_frame
+                    if self.use_depth:
+                        self.latest_depth_frame = processed_depth_frame
+                    self.latest_timestamp = capture_time
+                self.new_frame_event.set()
+                failure_count = 0
+
+            except DeviceNotConnectedError:
+                break
+            except Exception as e:
+                if failure_count <= 10:
+                    failure_count += 1
+                    logger.warning(f"Error reading frame in background thread for {self}: {e}")
+                else:
+                    raise RuntimeError(f"{self} exceeded maximum consecutive read failures.") from e
+
+    def _start_read_thread(self) -> None:
+        """Starts or restarts the background read thread if it's not running."""
+        self._stop_read_thread()
+
+        self.stop_event = Event()
+        self.thread = Thread(target=self._read_loop, args=(), name=f"{self}_read_loop")
+        self.thread.daemon = True
+        self.thread.start()
+
+    def _stop_read_thread(self) -> None:
+        """Signals the background read thread to stop and waits for it to join."""
+        if self.stop_event is not None:
+            self.stop_event.set()
+
+        if self.thread is not None and self.thread.is_alive():
+            self.thread.join(timeout=2.0)
+
+        self.thread = None
+        self.stop_event = None
+
+        with self.frame_lock:
+            self.latest_color_frame = None
+            self.latest_depth_frame = None
+            self.latest_timestamp = None
+            self.new_frame_event.clear()
+
+    # NOTE(Steven): Missing implementation for depth for now
+    @check_if_not_connected
+    def async_read(self, timeout_ms: float = 200) -> NDArray[Any]:
+        """
+        Reads the latest available frame data (color) asynchronously.
+
+        This method retrieves the most recent color frame captured by the background
+        read thread. It does not block waiting for the camera hardware directly,
+        but may wait up to timeout_ms for the background thread to provide a frame.
+        It is “best effort” under high FPS.
+
+        Args:
+            timeout_ms (float): Maximum time in milliseconds to wait for a frame
+                to become available. Defaults to 200ms (0.2 seconds).
+
+        Returns:
+            np.ndarray:
+            The latest captured frame data (color image), processed according to configuration.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is not connected.
+            TimeoutError: If no frame data becomes available within the specified timeout.
+            RuntimeError: If the background thread died unexpectedly or another error occurs.
+        """
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        if not self.new_frame_event.wait(timeout=timeout_ms / 1000.0):
+            raise TimeoutError(
+                f"Timed out waiting for frame from camera {self} after {timeout_ms} ms. "
+                f"Read thread alive: {self.thread.is_alive()}."
+            )
+
+        with self.frame_lock:
+            frame = self.latest_color_frame
+            self.new_frame_event.clear()
+
+        if frame is None:
+            raise RuntimeError(f"Internal error: Event set but no frame available for {self}.")
+
+        return frame
+
+    # NOTE(Steven): Missing implementation for depth for now
+    @check_if_not_connected
+    def read_latest(self, max_age_ms: int = 500) -> NDArray[Any]:
+        """Return the most recent (color) frame captured immediately (Peeking).
+
+        This method is non-blocking and returns whatever is currently in the
+        memory buffer. The frame may be stale,
+        meaning it could have been captured a while ago (hanging camera scenario e.g.).
+
+        Returns:
+            NDArray[Any]: The frame image (numpy array).
+
+        Raises:
+            TimeoutError: If the latest frame is older than `max_age_ms`.
+            DeviceNotConnectedError: If the camera is not connected.
+            RuntimeError: If the camera is connected but has not captured any frames yet.
+        """
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        with self.frame_lock:
+            frame = self.latest_color_frame
+            timestamp = self.latest_timestamp
+
+        if frame is None or timestamp is None:
+            raise RuntimeError(f"{self} has not captured any frames yet.")
+
+        age_ms = (time.perf_counter() - timestamp) * 1e3
+        if age_ms > max_age_ms:
+            raise TimeoutError(
+                f"{self} latest frame is too old: {age_ms:.1f} ms (max allowed: {max_age_ms} ms)."
+            )
+
+        return frame
+
+    def disconnect(self) -> None:
+        """
+        Disconnects from the camera, stops the pipeline, and cleans up resources.
+
+        Stops the background read thread (if running) and stops the RealSense pipeline.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is already disconnected (pipeline not running).
+        """
+
+        if not self.is_connected and self.thread is None:
+            raise DeviceNotConnectedError(
+                f"Attempted to disconnect {self}, but it appears already disconnected."
+            )
+
+        if self.thread is not None:
+            self._stop_read_thread()
+
+        if self.rs_pipeline is not None:
+            self.rs_pipeline.stop()
+            self.rs_pipeline = None
+            self.rs_profile = None
+
+        with self.frame_lock:
+            self.latest_color_frame = None
+            self.latest_depth_frame = None
+            self.latest_timestamp = None
+            self.new_frame_event.clear()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/cameras/realsense/configuration_realsense.py b/lerobot/src/lerobot/cameras/realsense/configuration_realsense.py
new file mode 100644
index 0000000000000000000000000000000000000000..71b083b0072c786f48d09ecd25d56b7637a60973
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/realsense/configuration_realsense.py
@@ -0,0 +1,70 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..configs import CameraConfig, ColorMode, Cv2Rotation
+
+
+@CameraConfig.register_subclass("intelrealsense")
+@dataclass
+class RealSenseCameraConfig(CameraConfig):
+    """Configuration class for Intel RealSense cameras.
+
+    This class provides specialized configuration options for Intel RealSense cameras,
+    including support for depth sensing and device identification via serial number or name.
+
+    Example configurations for Intel RealSense D405:
+    ```python
+    # Basic configurations
+    RealSenseCameraConfig("0123456789", 30, 1280, 720)  # 1280x720 @ 30FPS
+    RealSenseCameraConfig("0123456789", 60, 640, 480)  # 640x480 @ 60FPS
+
+    # Advanced configurations
+    RealSenseCameraConfig("0123456789", 30, 640, 480, use_depth=True)  # With depth sensing
+    RealSenseCameraConfig("0123456789", 30, 640, 480, rotation=Cv2Rotation.ROTATE_90)  # With 90° rotation
+    ```
+
+    Attributes:
+        fps: Requested frames per second for the color stream.
+        width: Requested frame width in pixels for the color stream.
+        height: Requested frame height in pixels for the color stream.
+        serial_number_or_name: Unique serial number or human-readable name to identify the camera.
+        color_mode: Color mode for image output (RGB or BGR). Defaults to RGB.
+        use_depth: Whether to enable depth stream. Defaults to False.
+        rotation: Image rotation setting (0°, 90°, 180°, or 270°). Defaults to no rotation.
+        warmup_s: Time reading frames before returning from connect (in seconds)
+
+    Note:
+        - Either name or serial_number must be specified.
+        - Depth stream configuration (if enabled) will use the same FPS as the color stream.
+        - The actual resolution and FPS may be adjusted by the camera to the nearest supported mode.
+        - For `fps`, `width` and `height`, either all of them need to be set, or none of them.
+    """
+
+    serial_number_or_name: str
+    color_mode: ColorMode = ColorMode.RGB
+    use_depth: bool = False
+    rotation: Cv2Rotation = Cv2Rotation.NO_ROTATION
+    warmup_s: int = 1
+
+    def __post_init__(self) -> None:
+        self.color_mode = ColorMode(self.color_mode)
+        self.rotation = Cv2Rotation(self.rotation)
+
+        values = (self.fps, self.width, self.height)
+        if any(v is not None for v in values) and any(v is None for v in values):
+            raise ValueError(
+                "For `fps`, `width` and `height`, either all of them need to be set, or none of them."
+            )
diff --git a/lerobot/src/lerobot/cameras/utils.py b/lerobot/src/lerobot/cameras/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..7fb2c3bb1844f1fe032dc257cad50e9df0b070d4
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/utils.py
@@ -0,0 +1,69 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from typing import cast
+
+from lerobot.utils.import_utils import make_device_from_device_class
+
+from .camera import Camera
+from .configs import CameraConfig, Cv2Rotation
+
+
+def make_cameras_from_configs(camera_configs: dict[str, CameraConfig]) -> dict[str, Camera]:
+    cameras: dict[str, Camera] = {}
+
+    for key, cfg in camera_configs.items():
+        # TODO(Steven): Consider just using the make_device_from_device_class for all types
+        if cfg.type == "opencv":
+            from .opencv import OpenCVCamera
+
+            cameras[key] = OpenCVCamera(cfg)
+
+        elif cfg.type == "intelrealsense":
+            from .realsense.camera_realsense import RealSenseCamera
+
+            cameras[key] = RealSenseCamera(cfg)
+
+        elif cfg.type == "reachy2_camera":
+            from .reachy2_camera.reachy2_camera import Reachy2Camera
+
+            cameras[key] = Reachy2Camera(cfg)
+
+        elif cfg.type == "zmq":
+            from .zmq.camera_zmq import ZMQCamera
+
+            cameras[key] = ZMQCamera(cfg)
+
+        else:
+            try:
+                cameras[key] = cast(Camera, make_device_from_device_class(cfg))
+            except Exception as e:
+                raise ValueError(f"Error creating camera {key} with config {cfg}: {e}") from e
+
+    return cameras
+
+
+def get_cv2_rotation(rotation: Cv2Rotation) -> int | None:
+    import cv2  # type: ignore  # TODO: add type stubs for OpenCV
+
+    if rotation == Cv2Rotation.ROTATE_90:
+        return int(cv2.ROTATE_90_CLOCKWISE)
+    elif rotation == Cv2Rotation.ROTATE_180:
+        return int(cv2.ROTATE_180)
+    elif rotation == Cv2Rotation.ROTATE_270:
+        return int(cv2.ROTATE_90_COUNTERCLOCKWISE)
+    else:
+        return None
diff --git a/lerobot/src/lerobot/cameras/zmq/__init__.py b/lerobot/src/lerobot/cameras/zmq/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..d760c5325f6bc1f9ae2d38ba49504bb5406b4992
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/zmq/__init__.py
@@ -0,0 +1,20 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .camera_zmq import ZMQCamera
+from .configuration_zmq import ZMQCameraConfig
+
+__all__ = ["ZMQCamera", "ZMQCameraConfig"]
diff --git a/lerobot/src/lerobot/cameras/zmq/camera_zmq.py b/lerobot/src/lerobot/cameras/zmq/camera_zmq.py
new file mode 100644
index 0000000000000000000000000000000000000000..2fbe50d8be2a1a78d22cef74c52017018caab6c1
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/zmq/camera_zmq.py
@@ -0,0 +1,385 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+ZMQCamera - Captures frames from remote cameras via ZeroMQ using JSON protocol in the
+following format:
+    {
+        "timestamps": {"camera_name": float},
+        "images": {"camera_name": "<base64-jpeg>"}
+    }
+"""
+
+import base64
+import json
+import logging
+import time
+from threading import Event, Lock, Thread
+from typing import Any
+
+import cv2
+import numpy as np
+from numpy.typing import NDArray
+
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.errors import DeviceNotConnectedError
+
+from ..camera import Camera
+from ..configs import ColorMode
+from .configuration_zmq import ZMQCameraConfig
+
+logger = logging.getLogger(__name__)
+
+
+class ZMQCamera(Camera):
+    """
+    Manages camera interactions via ZeroMQ for receiving frames from a remote server.
+
+    This class connects to a ZMQ Publisher, subscribes to frame topics, and decodes
+    incoming JSON messages containing Base64 encoded images. It supports both
+    synchronous and asynchronous frame reading patterns.
+
+    Example usage:
+        ```python
+        from lerobot.cameras.zmq import ZMQCamera, ZMQCameraConfig
+
+        config = ZMQCameraConfig(server_address="192.168.123.164", port=5555, camera_name="head_camera")
+        camera = ZMQCamera(config)
+        camera.connect()
+
+        # Read 1 frame synchronously (blocking)
+        color_image = camera.read()
+
+        # Read 1 frame asynchronously (waits for new frame with a timeout)
+        async_image = camera.async_read()
+
+        # Get the latest frame immediately (no wait, returns timestamp)
+        latest_image, timestamp = camera.read_latest()
+
+        camera.disconnect()
+        ```
+    """
+
+    def __init__(self, config: ZMQCameraConfig):
+        super().__init__(config)
+        import zmq
+
+        self.config = config
+        self.server_address = config.server_address
+        self.port = config.port
+        self.camera_name = config.camera_name
+        self.color_mode = config.color_mode
+        self.timeout_ms = config.timeout_ms
+
+        # ZMQ Context and Socket
+        self.context: zmq.Context | None = None
+        self.socket: zmq.Socket | None = None
+        self._connected = False
+
+        # Threading resources
+        self.thread: Thread | None = None
+        self.stop_event: Event | None = None
+        self.frame_lock: Lock = Lock()
+        self.latest_frame: NDArray[Any] | None = None
+        self.latest_timestamp: float | None = None
+        self.new_frame_event: Event = Event()
+
+    def __str__(self) -> str:
+        return f"ZMQCamera({self.camera_name}@{self.server_address}:{self.port})"
+
+    @property
+    def is_connected(self) -> bool:
+        """Checks if the ZMQ socket is initialized and connected."""
+        return self._connected and self.context is not None and self.socket is not None
+
+    @check_if_already_connected
+    def connect(self, warmup: bool = True) -> None:
+        """Connect to ZMQ camera server.
+
+        Args:
+            warmup (bool): If True, waits for the camera to provide at least one
+                           valid frame before returning. Defaults to True.
+        """
+
+        logger.info(f"Connecting to {self}...")
+
+        try:
+            import zmq
+
+            self.context = zmq.Context()
+            self.socket = self.context.socket(zmq.SUB)
+            self.socket.setsockopt_string(zmq.SUBSCRIBE, "")
+            self.socket.setsockopt(zmq.RCVTIMEO, self.timeout_ms)
+            self.socket.setsockopt(zmq.CONFLATE, True)
+            self.socket.connect(f"tcp://{self.server_address}:{self.port}")
+            self._connected = True
+
+            # Auto-detect resolution if not provided
+            if self.width is None or self.height is None:
+                # Read directly from hardware because the thread isn't running yet
+                temp_frame = self._read_from_hardware()
+                h, w = temp_frame.shape[:2]
+                self.height = h
+                self.width = w
+                logger.info(f"{self} resolution detected: {w}x{h}")
+
+            self._start_read_thread()
+            logger.info(f"{self} connected.")
+
+            if warmup:
+                # Ensure we have captured at least one frame via the thread
+                start_time = time.time()
+                while time.time() - start_time < (self.config.warmup_s):  # Wait a bit more than timeout
+                    self.async_read(timeout_ms=self.config.warmup_s * 1000)
+                    time.sleep(0.1)
+
+                with self.frame_lock:
+                    if self.latest_frame is None:
+                        raise ConnectionError(f"{self} failed to capture frames during warmup.")
+
+        except Exception as e:
+            self._cleanup()
+            raise RuntimeError(f"Failed to connect to {self}: {e}") from e
+
+    def _cleanup(self):
+        """Clean up ZMQ resources."""
+        self._connected = False
+        if self.socket:
+            self.socket.close()
+            self.socket = None
+        if self.context:
+            self.context.term()
+            self.context = None
+
+    @staticmethod
+    def find_cameras() -> list[dict[str, Any]]:
+        """
+        Detection not implemented for ZMQ cameras. These cameras require manual configuration (server address/port).
+        """
+        raise NotImplementedError("Camera detection is not implemented for ZMQ cameras.")
+
+    def _read_from_hardware(self) -> NDArray[Any]:
+        """
+        Reads a single frame directly from the ZMQ socket.
+        """
+        if not self.is_connected or self.socket is None:
+            raise DeviceNotConnectedError(f"{self} is not connected.")
+
+        try:
+            message = self.socket.recv_string()
+        except Exception as e:
+            # zmq is lazy-imported in connect(), so check by name to avoid a top-level import
+            if type(e).__name__ == "Again":
+                raise TimeoutError(f"{self} timeout after {self.timeout_ms}ms") from e
+            raise
+
+        # Decode JSON message
+        data = json.loads(message)
+
+        if "images" not in data:
+            raise RuntimeError(f"{self} invalid message: missing 'images' key")
+
+        images = data["images"]
+
+        # Get image by camera name or first available
+        if self.camera_name in images:
+            img_b64 = images[self.camera_name]
+        elif images:
+            img_b64 = next(iter(images.values()))
+        else:
+            raise RuntimeError(f"{self} no images in message")
+
+        # Decode base64 JPEG
+        img_bytes = base64.b64decode(img_b64)
+        frame = cv2.imdecode(np.frombuffer(img_bytes, np.uint8), cv2.IMREAD_COLOR)
+
+        if frame is None:
+            raise RuntimeError(f"{self} failed to decode image")
+
+        return frame
+
+    @check_if_not_connected
+    def read(self, color_mode: ColorMode | None = None) -> NDArray[Any]:
+        """
+        Reads a single frame synchronously from the camera.
+
+        This is a blocking call. It waits for the next available frame from the
+        camera background thread.
+
+        Returns:
+            np.ndarray: Decoded frame (height, width, 3)
+        """
+        start_time = time.perf_counter()
+
+        if color_mode is not None:
+            logger.warning(
+                f"{self} read() color_mode parameter is deprecated and will be removed in future versions."
+            )
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        self.new_frame_event.clear()
+        frame = self.async_read(timeout_ms=10000)
+
+        read_duration_ms = (time.perf_counter() - start_time) * 1e3
+        logger.debug(f"{self} read took: {read_duration_ms:.1f}ms")
+
+        return frame
+
+    def _read_loop(self) -> None:
+        """
+        Internal loop run by the background thread for asynchronous reading.
+        """
+        if self.stop_event is None:
+            raise RuntimeError(f"{self}: stop_event is not initialized.")
+
+        failure_count = 0
+        while not self.stop_event.is_set():
+            try:
+                frame = self._read_from_hardware()
+                capture_time = time.perf_counter()
+
+                with self.frame_lock:
+                    self.latest_frame = frame
+                    self.latest_timestamp = capture_time
+                self.new_frame_event.set()
+                failure_count = 0
+
+            except DeviceNotConnectedError:
+                break
+            except (TimeoutError, Exception) as e:
+                if failure_count <= 10:
+                    failure_count += 1
+                    logger.warning(f"Read error: {e}")
+                else:
+                    raise RuntimeError(f"{self} exceeded maximum consecutive read failures.") from e
+
+    def _start_read_thread(self) -> None:
+        if self.stop_event is not None:
+            self.stop_event.set()
+        if self.thread is not None and self.thread.is_alive():
+            self.thread.join(timeout=2.0)
+
+        with self.frame_lock:
+            self.latest_frame = None
+            self.latest_timestamp = None
+            self.new_frame_event.clear()
+
+        self.stop_event = Event()
+        self.thread = Thread(target=self._read_loop, daemon=True, name=f"{self}_read_loop")
+        self.thread.start()
+        time.sleep(0.1)
+
+    def _stop_read_thread(self) -> None:
+        if self.stop_event is not None:
+            self.stop_event.set()
+
+        if self.thread is not None and self.thread.is_alive():
+            self.thread.join(timeout=2.0)
+
+        self.thread = None
+        self.stop_event = None
+
+        with self.frame_lock:
+            self.latest_frame = None
+            self.latest_timestamp = None
+            self.new_frame_event.clear()
+
+    @check_if_not_connected
+    def async_read(self, timeout_ms: float = 200) -> NDArray[Any]:
+        """
+        Reads the latest available frame asynchronously.
+
+        Args:
+            timeout_ms (float): Maximum time in milliseconds to wait for a frame
+                to become available. Defaults to 200ms.
+
+        Returns:
+            np.ndarray: The latest captured frame.
+
+        Raises:
+            DeviceNotConnectedError: If the camera is not connected.
+            TimeoutError: If no frame data becomes available within the specified timeout.
+            RuntimeError: If the background thread is not running.
+        """
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        if not self.new_frame_event.wait(timeout=timeout_ms / 1000.0):
+            raise TimeoutError(f"{self} async_read timeout after {timeout_ms}ms")
+
+        with self.frame_lock:
+            frame = self.latest_frame
+            self.new_frame_event.clear()
+
+        if frame is None:
+            raise RuntimeError(f"{self} no frame available")
+
+        return frame
+
+    @check_if_not_connected
+    def read_latest(self, max_age_ms: int = 1000) -> NDArray[Any]:
+        """Return the most recent frame captured immediately (Peeking).
+
+        This method is non-blocking and returns whatever is currently in the
+        memory buffer. The frame may be stale,
+        meaning it could have been captured a while ago (hanging camera scenario e.g.).
+
+        Returns:
+            NDArray[Any]: The frame image (numpy array).
+
+        Raises:
+            TimeoutError: If the latest frame is older than `max_age_ms`.
+            DeviceNotConnectedError: If the camera is not connected.
+            RuntimeError: If the camera is connected but has not captured any frames yet.
+        """
+
+        if self.thread is None or not self.thread.is_alive():
+            raise RuntimeError(f"{self} read thread is not running.")
+
+        with self.frame_lock:
+            frame = self.latest_frame
+            timestamp = self.latest_timestamp
+
+        if frame is None or timestamp is None:
+            raise RuntimeError(f"{self} has not captured any frames yet.")
+
+        age_ms = (time.perf_counter() - timestamp) * 1e3
+        if age_ms > max_age_ms:
+            raise TimeoutError(
+                f"{self} latest frame is too old: {age_ms:.1f} ms (max allowed: {max_age_ms} ms)."
+            )
+
+        return frame
+
+    def disconnect(self) -> None:
+        """Disconnect from ZMQ camera."""
+        if not self.is_connected and self.thread is None:
+            raise DeviceNotConnectedError(f"{self} not connected.")
+
+        if self.thread is not None:
+            self._stop_read_thread()
+
+        self._cleanup()
+
+        with self.frame_lock:
+            self.latest_frame = None
+            self.latest_timestamp = None
+            self.new_frame_event.clear()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/cameras/zmq/configuration_zmq.py b/lerobot/src/lerobot/cameras/zmq/configuration_zmq.py
new file mode 100644
index 0000000000000000000000000000000000000000..13690e14c834d8e8fe7e5a1754f74b8c9b6245b8
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/zmq/configuration_zmq.py
@@ -0,0 +1,44 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..configs import CameraConfig, ColorMode
+
+__all__ = ["ZMQCameraConfig", "ColorMode"]
+
+
+@CameraConfig.register_subclass("zmq")
+@dataclass
+class ZMQCameraConfig(CameraConfig):
+    server_address: str
+    port: int = 5555
+    camera_name: str = "zmq_camera"
+    color_mode: ColorMode = ColorMode.RGB
+    timeout_ms: int = 5000
+    warmup_s: int = 1
+
+    def __post_init__(self) -> None:
+        self.color_mode = ColorMode(self.color_mode)
+
+        if self.timeout_ms <= 0:
+            raise ValueError(f"`timeout_ms` must be positive, but {self.timeout_ms} is provided.")
+
+        if not self.server_address:
+            raise ValueError("`server_address` cannot be empty.")
+
+        if self.port <= 0 or self.port > 65535:
+            raise ValueError(f"`port` must be between 1 and 65535, but {self.port} is provided.")
diff --git a/lerobot/src/lerobot/cameras/zmq/image_server.py b/lerobot/src/lerobot/cameras/zmq/image_server.py
new file mode 100644
index 0000000000000000000000000000000000000000..8222b9feecfe0041be94b3f435f214ccff4ac28f
--- /dev/null
+++ b/lerobot/src/lerobot/cameras/zmq/image_server.py
@@ -0,0 +1,182 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Streams camera images over ZMQ.
+Uses lerobot's OpenCVCamera for capture, encodes images to base64 and sends them over ZMQ.
+"""
+
+import base64
+import contextlib
+import json
+import logging
+import threading
+import time
+from collections import deque
+
+import cv2
+import numpy as np
+import zmq
+
+from lerobot.cameras.configs import ColorMode
+from lerobot.cameras.opencv import OpenCVCamera, OpenCVCameraConfig
+
+logger = logging.getLogger(__name__)
+
+
+def encode_image(image: np.ndarray, quality: int = 80) -> str:
+    """Encode RGB image to base64 JPEG string."""
+    _, buffer = cv2.imencode(".jpg", image, [int(cv2.IMWRITE_JPEG_QUALITY), quality])
+    return base64.b64encode(buffer).decode("utf-8")
+
+
+class CameraCaptureThread:
+    """Background thread that continuously captures and encodes frames from a camera."""
+
+    def __init__(self, camera: OpenCVCamera, name: str):
+        self.camera = camera
+        self.name = name
+        self.latest_encoded: str | None = None  # Pre-encoded JPEG as base64
+        self.latest_timestamp: float = 0.0
+        self.frame_lock = threading.Lock()
+        self.running = False
+        self.thread: threading.Thread | None = None
+
+    def start(self):
+        """Start the capture thread."""
+        self.running = True
+        self.thread = threading.Thread(target=self._capture_loop, daemon=True)
+        self.thread.start()
+
+    def stop(self):
+        """Stop the capture thread."""
+        self.running = False
+        if self.thread:
+            self.thread.join(timeout=1.0)
+
+    def _capture_loop(self):
+        """Continuously capture and encode frames at the camera's native rate."""
+        while self.running:
+            try:
+                frame = self.camera.read()  # Blocks at camera's native rate
+                timestamp = time.time()
+                # Encode immediately in capture thread (this is the slow part)
+                encoded = encode_image(frame)
+                with self.frame_lock:
+                    self.latest_encoded = encoded
+                    self.latest_timestamp = timestamp
+            except Exception as e:
+                logger.warning(f"Camera {self.name} capture error: {e}")
+                time.sleep(0.01)
+
+    def get_latest(self) -> tuple[str | None, float]:
+        """Get the latest encoded frame and its timestamp."""
+        with self.frame_lock:
+            return self.latest_encoded, self.latest_timestamp
+
+
+class ImageServer:
+    def __init__(self, config: dict, port: int = 5555):
+        # fps controls the publish loop rate (how often frames are sent over ZMQ), not the camera capture rate
+        self.fps = config.get("fps", 30)
+        self.cameras: dict[str, OpenCVCamera] = {}
+        self.capture_threads: dict[str, CameraCaptureThread] = {}
+
+        for name, cfg in config.get("cameras", {}).items():
+            shape = cfg.get("shape", [480, 640])
+            cam_config = OpenCVCameraConfig(
+                index_or_path=cfg.get("device_id", 0),
+                fps=self.fps,
+                width=shape[1],
+                height=shape[0],
+                color_mode=ColorMode.RGB,
+            )
+            camera = OpenCVCamera(cam_config)
+            camera.connect()
+            self.cameras[name] = camera
+            logger.info(f"Camera {name}: {shape[1]}x{shape[0]}")
+
+            # Create capture thread for this camera
+            capture_thread = CameraCaptureThread(camera, name)
+            self.capture_threads[name] = capture_thread
+
+        # ZMQ PUB socket
+        self.context = zmq.Context()
+        self.socket = self.context.socket(zmq.PUB)
+        self.socket.setsockopt(zmq.SNDHWM, 20)
+        self.socket.setsockopt(zmq.LINGER, 0)
+        self.socket.bind(f"tcp://*:{port}")
+
+        logger.info(f"ImageServer running on port {port}")
+
+    def run(self):
+        frame_count = 0
+        frame_times = deque(maxlen=60)
+        last_published_ts: dict[str, float] = {}
+
+        # Start all capture threads
+        for capture_thread in self.capture_threads.values():
+            capture_thread.start()
+
+        # Wait for first frames to be captured and encoded
+        logger.info("Waiting for cameras to start capturing...")
+        for name, capture_thread in self.capture_threads.items():
+            while capture_thread.get_latest()[0] is None:
+                time.sleep(0.01)
+            logger.info(f"Camera {name} ready (capture + encode in background)")
+
+        try:
+            while True:
+                t0 = time.time()
+
+                # Build message
+                message = {"timestamps": {}, "images": {}}
+                for name, capture_thread in self.capture_threads.items():
+                    encoded, timestamp = capture_thread.get_latest()
+                    if encoded is not None and timestamp > last_published_ts.get(name, 0.0):
+                        message["timestamps"][name] = timestamp
+                        message["images"][name] = encoded
+                        last_published_ts[name] = timestamp
+
+                # Send as JSON string (suppress if buffer full)
+                with contextlib.suppress(zmq.Again):
+                    self.socket.send_string(json.dumps(message), zmq.NOBLOCK)
+
+                frame_count += 1
+                frame_times.append(time.time() - t0)
+
+                if frame_count % 60 == 0:
+                    logger.debug(f"FPS: {len(frame_times) / sum(frame_times):.1f}")
+
+                sleep = (1.0 / self.fps) - (time.time() - t0)
+                if sleep > 0:
+                    time.sleep(sleep)
+
+        except KeyboardInterrupt:
+            pass
+        finally:
+            for capture_thread in self.capture_threads.values():
+                capture_thread.stop()
+            for cam in self.cameras.values():
+                cam.disconnect()
+            self.socket.close()
+            self.context.term()
+
+
+if __name__ == "__main__":
+    logging.basicConfig(level=logging.INFO)
+    config = {"fps": 30, "cameras": {"head_camera": {"device_id": 4, "shape": [480, 640]}}}
+    ImageServer(config, port=5555).run()
diff --git a/lerobot/src/lerobot/configs/default.py b/lerobot/src/lerobot/configs/default.py
new file mode 100644
index 0000000000000000000000000000000000000000..7f481b9ca7cf0fbac97b290e76ced1d5897e35c6
--- /dev/null
+++ b/lerobot/src/lerobot/configs/default.py
@@ -0,0 +1,108 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.datasets.transforms import ImageTransformsConfig
+from lerobot.datasets.video_utils import get_safe_default_codec
+
+
+@dataclass
+class DatasetConfig:
+    # You may provide a list of datasets here. `train.py` creates them all and concatenates them. Note: only data
+    # keys common between the datasets are kept. Each dataset gets and additional transform that inserts the
+    # "dataset_index" into the returned item. The index mapping is made according to the order in which the
+    # datasets are provided.
+    repo_id: str
+    # Root directory where the dataset will be stored (e.g. 'dataset/path'). If None, defaults to $HF_LEROBOT_HOME/repo_id.
+    root: str | None = None
+    episodes: list[int] | None = None
+    image_transforms: ImageTransformsConfig = field(default_factory=ImageTransformsConfig)
+    revision: str | None = None
+    use_imagenet_stats: bool = True
+    video_backend: str = field(default_factory=get_safe_default_codec)
+    streaming: bool = False
+
+    def __post_init__(self) -> None:
+        if self.episodes is not None:
+            if any(ep < 0 for ep in self.episodes):
+                raise ValueError(
+                    f"Episode indices must be non-negative, got: {[ep for ep in self.episodes if ep < 0]}"
+                )
+            if len(self.episodes) != len(set(self.episodes)):
+                duplicates = sorted({ep for ep in self.episodes if self.episodes.count(ep) > 1})
+                raise ValueError(f"Episode indices contain duplicates: {duplicates}")
+
+
+@dataclass
+class WandBConfig:
+    enable: bool = False
+    # Set to true to disable saving an artifact despite training.save_checkpoint=True
+    disable_artifact: bool = False
+    project: str = "lerobot"
+    entity: str | None = None
+    notes: str | None = None
+    run_id: str | None = None
+    mode: str | None = None  # Allowed values: 'online', 'offline' 'disabled'. Defaults to 'online'
+    add_tags: bool = True  # If True, save configuration as tags in the WandB run.
+
+
+@dataclass
+class EvalConfig:
+    n_episodes: int = 50
+    # `batch_size` specifies the number of environments to use in a gym.vector.VectorEnv.
+    batch_size: int = 50
+    # `use_async_envs` specifies whether to use asynchronous environments (multiprocessing).
+    use_async_envs: bool = False
+
+    def __post_init__(self) -> None:
+        if self.batch_size > self.n_episodes:
+            raise ValueError(
+                "The eval batch size is greater than the number of eval episodes "
+                f"({self.batch_size} > {self.n_episodes}). As a result, {self.batch_size} "
+                f"eval environments will be instantiated, but only {self.n_episodes} will be used. "
+                "This might significantly slow down evaluation. To fix this, you should update your command "
+                f"to increase the number of episodes to match the batch size (e.g. `eval.n_episodes={self.batch_size}`), "
+                f"or lower the batch size (e.g. `eval.batch_size={self.n_episodes}`)."
+            )
+
+
+@dataclass
+class PeftConfig:
+    # PEFT offers many fine-tuning methods, layer adapters being the most common and currently also the most
+    # effective methods so we'll focus on those in this high-level config interface.
+
+    # Either a string (module name suffix or 'all-linear'), a list of module name suffixes or a regular expression
+    # describing module names to target with the configured PEFT method. Some policies have a default value for this
+    # so that you don't *have* to choose which layers to adapt but it might still be worthwhile depending on your case.
+    target_modules: list[str] | str | None = None
+
+    # Names/suffixes of modules to fully fine-tune and store alongside adapter weights. Useful for layers that are
+    # not part of a pre-trained model (e.g., action state projections). Depending on the policy this defaults to layers
+    # that are newly created in pre-trained policies. If you're fine-tuning an already trained policy you might want
+    # to set this to `[]`. Corresponds to PEFT's `modules_to_save`.
+    full_training_modules: list[str] | None = None
+
+    # The PEFT (adapter) method to apply to the policy. Needs to be a valid PEFT type.
+    method_type: str = "LORA"
+
+    # Adapter initialization method. Look at the specific PEFT adapter documentation for defaults.
+    init_type: str | None = None
+
+    # We expect that all PEFT adapters are in some way doing rank-decomposition therefore this parameter specifies
+    # the rank used for the adapter. In general a higher rank means more trainable parameters and closer to full
+    # fine-tuning.
+    r: int = 16
diff --git a/lerobot/src/lerobot/configs/eval.py b/lerobot/src/lerobot/configs/eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..da8bee6b25b11a57af9d59dde8e1c6d8f2d8c03a
--- /dev/null
+++ b/lerobot/src/lerobot/configs/eval.py
@@ -0,0 +1,75 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import datetime as dt
+from dataclasses import dataclass, field
+from logging import getLogger
+from pathlib import Path
+
+from lerobot import envs, policies  # noqa: F401
+from lerobot.configs import parser
+from lerobot.configs.default import EvalConfig
+from lerobot.configs.policies import PreTrainedConfig
+
+logger = getLogger(__name__)
+
+
+@dataclass
+class EvalPipelineConfig:
+    # Either the repo ID of a model hosted on the Hub or a path to a directory containing weights
+    # saved using `Policy.save_pretrained`. If not provided, the policy is initialized from scratch
+    # (useful for debugging). This argument is mutually exclusive with `--config`.
+    env: envs.EnvConfig
+    eval: EvalConfig = field(default_factory=EvalConfig)
+    policy: PreTrainedConfig | None = None
+    output_dir: Path | None = None
+    job_name: str | None = None
+    seed: int | None = 1000
+    # Rename map for the observation to override the image and state keys
+    rename_map: dict[str, str] = field(default_factory=dict)
+    # Explicit consent to execute remote code from the Hub (required for hub environments).
+    trust_remote_code: bool = False
+
+    def __post_init__(self) -> None:
+        # HACK: We parse again the cli args here to get the pretrained path if there was one.
+        policy_path = parser.get_path_arg("policy")
+        if policy_path:
+            cli_overrides = parser.get_cli_overrides("policy")
+            self.policy = PreTrainedConfig.from_pretrained(policy_path, cli_overrides=cli_overrides)
+            self.policy.pretrained_path = Path(policy_path)
+
+        else:
+            logger.warning(
+                "No pretrained path was provided, evaluated policy will be built from scratch (random weights)."
+            )
+
+        if not self.job_name:
+            if self.env is None:
+                self.job_name = f"{self.policy.type if self.policy is not None else 'scratch'}"
+            else:
+                self.job_name = (
+                    f"{self.env.type}_{self.policy.type if self.policy is not None else 'scratch'}"
+                )
+
+            logger.warning(f"No job name provided, using '{self.job_name}' as job name.")
+
+        if not self.output_dir:
+            now = dt.datetime.now()
+            eval_dir = f"{now:%Y-%m-%d}/{now:%H-%M-%S}_{self.job_name}"
+            self.output_dir = Path("outputs/eval") / eval_dir
+
+    @classmethod
+    def __get_path_fields__(cls) -> list[str]:
+        """This enables the parser to load config from the policy using `--policy.path=local/dir`"""
+        return ["policy"]
diff --git a/lerobot/src/lerobot/configs/parser.py b/lerobot/src/lerobot/configs/parser.py
new file mode 100644
index 0000000000000000000000000000000000000000..57ebaf8fa41482d8820da3e5cf642cf5884a00c0
--- /dev/null
+++ b/lerobot/src/lerobot/configs/parser.py
@@ -0,0 +1,238 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import importlib
+import inspect
+import pkgutil
+import sys
+from argparse import ArgumentError
+from collections.abc import Callable, Iterable, Sequence
+from functools import wraps
+from pathlib import Path
+from pkgutil import ModuleInfo
+from types import ModuleType
+from typing import Any, TypeVar, cast
+
+import draccus
+
+from lerobot.utils.utils import has_method
+
+F = TypeVar("F", bound=Callable[..., object])
+
+PATH_KEY = "path"
+PLUGIN_DISCOVERY_SUFFIX = "discover_packages_path"
+
+
+def get_cli_overrides(field_name: str, args: Sequence[str] | None = None) -> list[str] | None:
+    """Parses arguments from cli at a given nested attribute level.
+
+    For example, supposing the main script was called with:
+    python myscript.py --arg1=1 --arg2.subarg1=abc --arg2.subarg2=some/path
+
+    If called during execution of myscript.py, get_cli_overrides("arg2") will return:
+    ["--subarg1=abc" "--subarg2=some/path"]
+    """
+    if args is None:
+        args = sys.argv[1:]
+    attr_level_args = []
+    detect_string = f"--{field_name}."
+    exclude_strings = (f"--{field_name}.{draccus.CHOICE_TYPE_KEY}=", f"--{field_name}.{PATH_KEY}=")
+    for arg in args:
+        if arg.startswith(detect_string) and not arg.startswith(exclude_strings):
+            denested_arg = f"--{arg.removeprefix(detect_string)}"
+            attr_level_args.append(denested_arg)
+
+    return attr_level_args
+
+
+def parse_arg(arg_name: str, args: Sequence[str] | None = None) -> str | None:
+    if args is None:
+        args = sys.argv[1:]
+    prefix = f"--{arg_name}="
+    for arg in args:
+        if arg.startswith(prefix):
+            return arg[len(prefix) :]
+    return None
+
+
+def parse_plugin_args(plugin_arg_suffix: str, args: Sequence[str]) -> dict[str, str]:
+    """Parse plugin-related arguments from command-line arguments.
+
+    This function extracts arguments from command-line arguments that match a specified suffix pattern.
+    It processes arguments in the format '--key=value' and returns them as a dictionary.
+
+    Args:
+        plugin_arg_suffix (str): The suffix to identify plugin-related arguments.
+        cli_args (Sequence[str]): A sequence of command-line arguments to parse.
+
+    Returns:
+        dict: A dictionary containing the parsed plugin arguments where:
+            - Keys are the argument names (with '--' prefix removed if present)
+            - Values are the corresponding argument values
+
+    Example:
+        >>> args = ["--env.discover_packages_path=my_package", "--other_arg=value"]
+        >>> parse_plugin_args("discover_packages_path", args)
+        {'env.discover_packages_path': 'my_package'}
+    """
+    plugin_args = {}
+    for arg in args:
+        if "=" in arg and plugin_arg_suffix in arg:
+            key, value = arg.split("=", 1)
+            # Remove leading '--' if present
+            if key.startswith("--"):
+                key = key[2:]
+            plugin_args[key] = value
+    return plugin_args
+
+
+class PluginLoadError(Exception):
+    """Raised when a plugin fails to load."""
+
+
+def load_plugin(plugin_path: str) -> None:
+    """Load and initialize a plugin from a given Python package path.
+
+    This function attempts to load a plugin by importing its package and any submodules.
+    Plugin registration is expected to happen during package initialization, i.e. when
+    the package is imported the gym environment should be registered and the config classes
+    registered with their parents using the `register_subclass` decorator.
+
+    Args:
+        plugin_path (str): The Python package path to the plugin (e.g. "mypackage.plugins.myplugin")
+
+    Raises:
+        PluginLoadError: If the plugin cannot be loaded due to import errors or if the package path is invalid.
+
+    Examples:
+        >>> load_plugin("external_plugin.core")  # Loads plugin from external package
+
+    Notes:
+        - The plugin package should handle its own registration during import
+        - All submodules in the plugin package will be imported
+        - Implementation follows the plugin discovery pattern from Python packaging guidelines
+
+    See Also:
+        https://packaging.python.org/en/latest/guides/creating-and-discovering-plugins/
+    """
+    try:
+        package_module = importlib.import_module(plugin_path, __package__)
+    except (ImportError, ModuleNotFoundError) as e:
+        raise PluginLoadError(
+            f"Failed to load plugin '{plugin_path}'. Verify the path and installation: {str(e)}"
+        ) from e
+
+    def iter_namespace(ns_pkg: ModuleType) -> Iterable[ModuleInfo]:
+        return pkgutil.iter_modules(ns_pkg.__path__, ns_pkg.__name__ + ".")
+
+    try:
+        for _finder, pkg_name, _ispkg in iter_namespace(package_module):
+            importlib.import_module(pkg_name)
+    except ImportError as e:
+        raise PluginLoadError(
+            f"Failed to load plugin '{plugin_path}'. Verify the path and installation: {str(e)}"
+        ) from e
+
+
+def get_path_arg(field_name: str, args: Sequence[str] | None = None) -> str | None:
+    return parse_arg(f"{field_name}.{PATH_KEY}", args)
+
+
+def get_type_arg(field_name: str, args: Sequence[str] | None = None) -> str | None:
+    return parse_arg(f"{field_name}.{draccus.CHOICE_TYPE_KEY}", args)
+
+
+def filter_arg(field_to_filter: str, args: Sequence[str] | None = None) -> list[str]:
+    if args is None:
+        return []
+    return [arg for arg in args if not arg.startswith(f"--{field_to_filter}=")]
+
+
+def filter_path_args(fields_to_filter: str | list[str], args: Sequence[str] | None = None) -> list[str]:
+    """
+    Filters command-line arguments related to fields with specific path arguments.
+
+    Args:
+        fields_to_filter (str | list[str]): A single str or a list of str whose arguments need to be filtered.
+        args (Sequence[str] | None): The sequence of command-line arguments to be filtered.
+            Defaults to None.
+
+    Returns:
+        list[str]: A filtered list of arguments, with arguments related to the specified
+        fields removed.
+
+    Raises:
+        ArgumentError: If both a path argument (e.g., `--field_name.path`) and a type
+            argument (e.g., `--field_name.type`) are specified for the same field.
+    """
+    if isinstance(fields_to_filter, str):
+        fields_to_filter = [fields_to_filter]
+
+    filtered_args = [] if args is None else list(args)
+
+    for field in fields_to_filter:
+        if get_path_arg(field, args):
+            if get_type_arg(field, args):
+                raise ArgumentError(
+                    argument=None,
+                    message=f"Cannot specify both --{field}.{PATH_KEY} and --{field}.{draccus.CHOICE_TYPE_KEY}",
+                )
+            filtered_args = [arg for arg in filtered_args if not arg.startswith(f"--{field}.")]
+
+    return filtered_args
+
+
+def wrap(config_path: Path | None = None) -> Callable[[F], F]:
+    """
+    HACK: Similar to draccus.wrap but does three additional things:
+        - Will remove '.path' arguments from CLI in order to process them later on.
+        - If a 'config_path' is passed and the main config class has a 'from_pretrained' method, will
+          initialize it from there to allow to fetch configs from the hub directly
+        - Will load plugins specified in the CLI arguments. These plugins will typically register
+            their own subclasses of config classes, so that draccus can find the right class to instantiate
+            from the CLI '.type' arguments
+    """
+
+    def wrapper_outer(fn: F) -> F:
+        @wraps(fn)
+        def wrapper_inner(*args: Any, **kwargs: Any) -> Any:
+            argspec = inspect.getfullargspec(fn)
+            argtype = argspec.annotations[argspec.args[0]]
+            if len(args) > 0 and type(args[0]) is argtype:
+                cfg = args[0]
+                args = args[1:]
+            else:
+                cli_args = sys.argv[1:]
+                plugin_args = parse_plugin_args(PLUGIN_DISCOVERY_SUFFIX, cli_args)
+                for plugin_cli_arg, plugin_path in plugin_args.items():
+                    try:
+                        load_plugin(plugin_path)
+                    except PluginLoadError as e:
+                        # add the relevant CLI arg to the error message
+                        raise PluginLoadError(f"{e}\nFailed plugin CLI Arg: {plugin_cli_arg}") from e
+                    cli_args = filter_arg(plugin_cli_arg, cli_args)
+                config_path_cli = parse_arg("config_path", cli_args)
+                if has_method(argtype, "__get_path_fields__"):
+                    path_fields = argtype.__get_path_fields__()
+                    cli_args = filter_path_args(path_fields, cli_args)
+                if has_method(argtype, "from_pretrained") and config_path_cli:
+                    cli_args = filter_arg("config_path", cli_args)
+                    cfg = argtype.from_pretrained(config_path_cli, cli_args=cli_args)
+                else:
+                    cfg = draccus.parse(config_class=argtype, config_path=config_path, args=cli_args)
+            response = fn(cfg, *args, **kwargs)
+            return response
+
+        return cast(F, wrapper_inner)
+
+    return cast(Callable[[F], F], wrapper_outer)
diff --git a/lerobot/src/lerobot/configs/policies.py b/lerobot/src/lerobot/configs/policies.py
new file mode 100644
index 0000000000000000000000000000000000000000..ce567b8f56372e83a2faf3836a70873d181965fd
--- /dev/null
+++ b/lerobot/src/lerobot/configs/policies.py
@@ -0,0 +1,226 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import abc
+import builtins
+import json
+import os
+import tempfile
+from dataclasses import dataclass, field
+from logging import getLogger
+from pathlib import Path
+from typing import Any, TypeVar
+
+import draccus
+from huggingface_hub import hf_hub_download
+from huggingface_hub.constants import CONFIG_NAME
+from huggingface_hub.errors import HfHubHTTPError
+
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.optim.optimizers import OptimizerConfig
+from lerobot.optim.schedulers import LRSchedulerConfig
+from lerobot.utils.constants import ACTION, OBS_STATE
+from lerobot.utils.device_utils import auto_select_torch_device, is_amp_available, is_torch_device_available
+from lerobot.utils.hub import HubMixin
+
+T = TypeVar("T", bound="PreTrainedConfig")
+logger = getLogger(__name__)
+
+
+@dataclass
+class PreTrainedConfig(draccus.ChoiceRegistry, HubMixin, abc.ABC):  # type: ignore[misc,name-defined] #TODO: draccus issue
+    """
+    Base configuration class for policy models.
+
+    Args:
+        n_obs_steps: Number of environment steps worth of observations to pass to the policy (takes the
+            current step and additional steps going back).
+        input_features: A dictionary defining the PolicyFeature of the input data for the policy. The key represents
+            the input data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        output_features: A dictionary defining the PolicyFeature of the output data for the policy. The key represents
+            the output data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        normalization_mapping: A dictionary that maps from a str value of FeatureType (e.g., "STATE", "VISUAL") to
+            a corresponding NormalizationMode (e.g., NormalizationMode.MIN_MAX)
+    """
+
+    n_obs_steps: int = 1
+
+    # `input_features` can be set to None/null in order to infer those values from the dataset.
+    input_features: dict[str, PolicyFeature] | None = field(default_factory=dict)
+    output_features: dict[str, PolicyFeature] | None = field(default_factory=dict)
+
+    device: str | None = None  # e.g. "cuda", "cuda:0", "cpu", or "mps"
+    # `use_amp` determines whether to use Automatic Mixed Precision (AMP) for training and evaluation. With AMP,
+    # automatic gradient scaling is used.
+    use_amp: bool = False
+
+    # Whether the policy employed PEFT for training.
+    use_peft: bool = False
+
+    push_to_hub: bool = True  # type: ignore[assignment] # TODO: use a different name to avoid override
+    repo_id: str | None = None
+
+    # Upload on private repository on the Hugging Face hub.
+    private: bool | None = None
+    # Add tags to your policy on the hub.
+    tags: list[str] | None = None
+    # Add tags to your policy on the hub.
+    license: str | None = None
+    # Either the repo ID of a model hosted on the Hub or a path to a directory containing weights
+    # saved using `Policy.save_pretrained`. If not provided, the policy is initialized from scratch.
+    pretrained_path: Path | None = None
+
+    def __post_init__(self) -> None:
+        if not self.device or not is_torch_device_available(self.device):
+            auto_device = auto_select_torch_device()
+            logger.warning(f"Device '{self.device}' is not available. Switching to '{auto_device}'.")
+            self.device = auto_device.type
+
+        # Automatically deactivate AMP if necessary
+        if self.use_amp and not is_amp_available(self.device):
+            logger.warning(
+                f"Automatic Mixed Precision (amp) is not available on device '{self.device}'. Deactivating AMP."
+            )
+            self.use_amp = False
+
+    @property
+    def type(self) -> str:
+        choice_name = self.get_choice_name(self.__class__)
+        if not isinstance(choice_name, str):
+            raise TypeError(f"Expected string from get_choice_name, got {type(choice_name)}")
+        return choice_name
+
+    @property
+    @abc.abstractmethod
+    def observation_delta_indices(self) -> list | None:  # type: ignore[type-arg] #TODO: No implementation
+        raise NotImplementedError
+
+    @property
+    @abc.abstractmethod
+    def action_delta_indices(self) -> list | None:  # type: ignore[type-arg]    #TODO: No implementation
+        raise NotImplementedError
+
+    @property
+    @abc.abstractmethod
+    def reward_delta_indices(self) -> list | None:  # type: ignore[type-arg]    #TODO: No implementation
+        raise NotImplementedError
+
+    @abc.abstractmethod
+    def get_optimizer_preset(self) -> OptimizerConfig:
+        raise NotImplementedError
+
+    @abc.abstractmethod
+    def get_scheduler_preset(self) -> LRSchedulerConfig | None:
+        raise NotImplementedError
+
+    @abc.abstractmethod
+    def validate_features(self) -> None:
+        raise NotImplementedError
+
+    @property
+    def robot_state_feature(self) -> PolicyFeature | None:
+        if not self.input_features:
+            return None
+        for ft_name, ft in self.input_features.items():
+            if ft.type is FeatureType.STATE and ft_name == OBS_STATE:
+                return ft
+        return None
+
+    @property
+    def env_state_feature(self) -> PolicyFeature | None:
+        if not self.input_features:
+            return None
+        for _, ft in self.input_features.items():
+            if ft.type is FeatureType.ENV:
+                return ft
+        return None
+
+    @property
+    def image_features(self) -> dict[str, PolicyFeature]:
+        if not self.input_features:
+            return {}
+        return {key: ft for key, ft in self.input_features.items() if ft.type is FeatureType.VISUAL}
+
+    @property
+    def action_feature(self) -> PolicyFeature | None:
+        if not self.output_features:
+            return None
+        for ft_name, ft in self.output_features.items():
+            if ft.type is FeatureType.ACTION and ft_name == ACTION:
+                return ft
+        return None
+
+    def _save_pretrained(self, save_directory: Path) -> None:
+        with open(save_directory / CONFIG_NAME, "w") as f, draccus.config_type("json"):
+            draccus.dump(self, f, indent=4)
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type[T],
+        pretrained_name_or_path: str | Path,
+        *,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict[Any, Any] | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        **policy_kwargs: Any,
+    ) -> T:
+        model_id = str(pretrained_name_or_path)
+        config_file: str | None = None
+        if Path(model_id).is_dir():
+            if CONFIG_NAME in os.listdir(model_id):
+                config_file = os.path.join(model_id, CONFIG_NAME)
+            else:
+                logger.error(f"{CONFIG_NAME} not found in {Path(model_id).resolve()}")
+        else:
+            try:
+                config_file = hf_hub_download(
+                    repo_id=model_id,
+                    filename=CONFIG_NAME,
+                    revision=revision,
+                    cache_dir=cache_dir,
+                    force_download=force_download,
+                    proxies=proxies,
+                    resume_download=resume_download,
+                    token=token,
+                    local_files_only=local_files_only,
+                )
+            except HfHubHTTPError as e:
+                raise FileNotFoundError(
+                    f"{CONFIG_NAME} not found on the HuggingFace Hub in {model_id}"
+                ) from e
+
+        # HACK: Parse the original config to get the config subclass, so that we can
+        # apply cli overrides.
+        # This is very ugly, ideally we'd like to be able to do that natively with draccus
+        # something like --policy.path (in addition to --policy.type)
+        with draccus.config_type("json"):
+            orig_config = draccus.parse(cls, config_file, args=[])
+
+        if config_file is None:
+            raise FileNotFoundError(f"{CONFIG_NAME} not found in {model_id}")
+
+        with open(config_file) as f:
+            config = json.load(f)
+
+        config.pop("type")
+        with tempfile.NamedTemporaryFile("w+", delete=False, suffix=".json") as f:
+            json.dump(config, f)
+            config_file = f.name
+
+        cli_overrides = policy_kwargs.pop("cli_overrides", [])
+        with draccus.config_type("json"):
+            return draccus.parse(orig_config.__class__, config_file, args=cli_overrides)
diff --git a/lerobot/src/lerobot/configs/train.py b/lerobot/src/lerobot/configs/train.py
new file mode 100644
index 0000000000000000000000000000000000000000..8b8aedb26beaf969d7ce3b4b061af1058bb1cacb
--- /dev/null
+++ b/lerobot/src/lerobot/configs/train.py
@@ -0,0 +1,216 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import builtins
+import datetime as dt
+import os
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import Any
+
+import draccus
+from huggingface_hub import hf_hub_download
+from huggingface_hub.errors import HfHubHTTPError
+
+from lerobot import envs
+from lerobot.configs import parser
+from lerobot.configs.default import DatasetConfig, EvalConfig, PeftConfig, WandBConfig
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.optim import OptimizerConfig
+from lerobot.optim.schedulers import LRSchedulerConfig
+from lerobot.utils.hub import HubMixin
+
+TRAIN_CONFIG_NAME = "train_config.json"
+
+
+@dataclass
+class TrainPipelineConfig(HubMixin):
+    dataset: DatasetConfig
+    env: envs.EnvConfig | None = None
+    policy: PreTrainedConfig | None = None
+    # Set `dir` to where you would like to save all of the run outputs. If you run another training session
+    # with the same value for `dir` its contents will be overwritten unless you set `resume` to true.
+    output_dir: Path | None = None
+    job_name: str | None = None
+    # Set `resume` to true to resume a previous run. In order for this to work, you will need to make sure
+    # `dir` is the directory of an existing run with at least one checkpoint in it.
+    # Note that when resuming a run, the default behavior is to use the configuration from the checkpoint,
+    # regardless of what's provided with the training command at the time of resumption.
+    resume: bool = False
+    # `seed` is used for training (eg: model initialization, dataset shuffling)
+    # AND for the evaluation environments.
+    seed: int | None = 1000
+    # Set to True to use deterministic cuDNN algorithms for reproducibility.
+    # This disables cudnn.benchmark and may reduce training speed by ~10-20 percent.
+    cudnn_deterministic: bool = False
+    # Number of workers for the dataloader.
+    num_workers: int = 4
+    batch_size: int = 8
+    steps: int = 100_000
+    eval_freq: int = 20_000
+    log_freq: int = 200
+    tolerance_s: float = 1e-4
+    save_checkpoint: bool = True
+    # Checkpoint is saved every `save_freq` training iterations and after the last training step.
+    save_freq: int = 20_000
+    use_policy_training_preset: bool = True
+    optimizer: OptimizerConfig | None = None
+    scheduler: LRSchedulerConfig | None = None
+    eval: EvalConfig = field(default_factory=EvalConfig)
+    wandb: WandBConfig = field(default_factory=WandBConfig)
+    peft: PeftConfig | None = None
+
+    # RA-BC (Reward-Aligned Behavior Cloning) parameters
+    use_rabc: bool = False  # Enable reward-weighted training
+    rabc_progress_path: str | None = None  # Path to precomputed SARM progress parquet file
+    rabc_kappa: float = 0.01  # Hard threshold for high-quality samples
+    rabc_epsilon: float = 1e-6  # Small constant for numerical stability
+    rabc_head_mode: str | None = "sparse"  # For dual-head models: "sparse" or "dense"
+
+    # Rename map for the observation to override the image and state keys
+    rename_map: dict[str, str] = field(default_factory=dict)
+    checkpoint_path: Path | None = field(init=False, default=None)
+
+    def validate(self) -> None:
+        # HACK: We parse again the cli args here to get the pretrained paths if there was some.
+        policy_path = parser.get_path_arg("policy")
+        if policy_path:
+            # Only load the policy config
+            cli_overrides = parser.get_cli_overrides("policy")
+            self.policy = PreTrainedConfig.from_pretrained(policy_path, cli_overrides=cli_overrides)
+            self.policy.pretrained_path = Path(policy_path)
+        elif self.resume:
+            # The entire train config is already loaded, we just need to get the checkpoint dir
+            config_path = parser.parse_arg("config_path")
+            if not config_path:
+                raise ValueError(
+                    f"A config_path is expected when resuming a run. Please specify path to {TRAIN_CONFIG_NAME}"
+                )
+
+            if not Path(config_path).resolve().exists():
+                raise NotADirectoryError(
+                    f"{config_path=} is expected to be a local path. "
+                    "Resuming from the hub is not supported for now."
+                )
+
+            policy_dir = Path(config_path).parent
+            if self.policy is not None:
+                self.policy.pretrained_path = policy_dir
+            self.checkpoint_path = policy_dir.parent
+
+        if self.policy is None:
+            raise ValueError(
+                "Policy is not configured. Please specify a pretrained policy with `--policy.path`."
+            )
+
+        if not self.job_name:
+            if self.env is None:
+                self.job_name = f"{self.policy.type}"
+            else:
+                self.job_name = f"{self.env.type}_{self.policy.type}"
+
+        if not self.resume and isinstance(self.output_dir, Path) and self.output_dir.is_dir():
+            raise FileExistsError(
+                f"Output directory {self.output_dir} already exists and resume is {self.resume}. "
+                f"Please change your output directory so that {self.output_dir} is not overwritten."
+            )
+        elif not self.output_dir:
+            now = dt.datetime.now()
+            train_dir = f"{now:%Y-%m-%d}/{now:%H-%M-%S}_{self.job_name}"
+            self.output_dir = Path("outputs/train") / train_dir
+
+        if isinstance(self.dataset.repo_id, list):
+            raise NotImplementedError("LeRobotMultiDataset is not currently implemented.")
+
+        if not self.use_policy_training_preset and (self.optimizer is None or self.scheduler is None):
+            raise ValueError("Optimizer and Scheduler must be set when the policy presets are not used.")
+        elif self.use_policy_training_preset and not self.resume:
+            self.optimizer = self.policy.get_optimizer_preset()
+            self.scheduler = self.policy.get_scheduler_preset()
+
+        if self.policy.push_to_hub and not self.policy.repo_id:
+            raise ValueError(
+                "'policy.repo_id' argument missing. Please specify it to push the model to the hub."
+            )
+
+        if self.use_rabc and not self.rabc_progress_path:
+            # Auto-detect from dataset path
+            repo_id = self.dataset.repo_id
+            if self.dataset.root:
+                self.rabc_progress_path = str(Path(self.dataset.root) / "sarm_progress.parquet")
+            else:
+                self.rabc_progress_path = f"hf://datasets/{repo_id}/sarm_progress.parquet"
+
+    @classmethod
+    def __get_path_fields__(cls) -> list[str]:
+        """This enables the parser to load config from the policy using `--policy.path=local/dir`"""
+        return ["policy"]
+
+    def to_dict(self) -> dict[str, Any]:
+        return draccus.encode(self)  # type: ignore[no-any-return]  # because of the third-party library draccus uses Any as the return type
+
+    def _save_pretrained(self, save_directory: Path) -> None:
+        with open(save_directory / TRAIN_CONFIG_NAME, "w") as f, draccus.config_type("json"):
+            draccus.dump(self, f, indent=4)
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type["TrainPipelineConfig"],
+        pretrained_name_or_path: str | Path,
+        *,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict[Any, Any] | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        **kwargs: Any,
+    ) -> "TrainPipelineConfig":
+        model_id = str(pretrained_name_or_path)
+        config_file: str | None = None
+        if Path(model_id).is_dir():
+            if TRAIN_CONFIG_NAME in os.listdir(model_id):
+                config_file = os.path.join(model_id, TRAIN_CONFIG_NAME)
+            else:
+                print(f"{TRAIN_CONFIG_NAME} not found in {Path(model_id).resolve()}")
+        elif Path(model_id).is_file():
+            config_file = model_id
+        else:
+            try:
+                config_file = hf_hub_download(
+                    repo_id=model_id,
+                    filename=TRAIN_CONFIG_NAME,
+                    revision=revision,
+                    cache_dir=cache_dir,
+                    force_download=force_download,
+                    proxies=proxies,
+                    resume_download=resume_download,
+                    token=token,
+                    local_files_only=local_files_only,
+                )
+            except HfHubHTTPError as e:
+                raise FileNotFoundError(
+                    f"{TRAIN_CONFIG_NAME} not found on the HuggingFace Hub in {model_id}"
+                ) from e
+
+        cli_args = kwargs.pop("cli_args", [])
+        with draccus.config_type("json"):
+            return draccus.parse(cls, config_file, args=cli_args)
+
+
+@dataclass(kw_only=True)
+class TrainRLServerPipelineConfig(TrainPipelineConfig):
+    # NOTE: In RL, we don't need an offline dataset
+    # TODO: Make `TrainPipelineConfig.dataset` optional
+    dataset: DatasetConfig | None = None  # type: ignore[assignment] # because the parent class has made it's type non-optional
diff --git a/lerobot/src/lerobot/configs/types.py b/lerobot/src/lerobot/configs/types.py
new file mode 100644
index 0000000000000000000000000000000000000000..18359ef0575a4f560a65f21e2d6ddf534edc770f
--- /dev/null
+++ b/lerobot/src/lerobot/configs/types.py
@@ -0,0 +1,52 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# Note: We subclass str so that serialization is straightforward
+# https://stackoverflow.com/questions/24481852/serialising-an-enum-member-to-json
+from dataclasses import dataclass
+from enum import Enum
+
+
+class FeatureType(str, Enum):
+    STATE = "STATE"
+    VISUAL = "VISUAL"
+    ENV = "ENV"
+    ACTION = "ACTION"
+    REWARD = "REWARD"
+    LANGUAGE = "LANGUAGE"
+
+
+class PipelineFeatureType(str, Enum):
+    ACTION = "ACTION"
+    OBSERVATION = "OBSERVATION"
+
+
+class NormalizationMode(str, Enum):
+    MIN_MAX = "MIN_MAX"
+    MEAN_STD = "MEAN_STD"
+    IDENTITY = "IDENTITY"
+    QUANTILES = "QUANTILES"
+    QUANTILE10 = "QUANTILE10"
+
+
+@dataclass
+class PolicyFeature:
+    type: FeatureType
+    shape: tuple[int, ...]
+
+
+class RTCAttentionSchedule(str, Enum):
+    ZEROS = "ZEROS"
+    ONES = "ONES"
+    LINEAR = "LINEAR"
+    EXP = "EXP"
diff --git a/lerobot/src/lerobot/data_processing/__init__.py b/lerobot/src/lerobot/data_processing/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..2f76d5676d10d8c064b78855cbfdc6b3c4ea45a7
--- /dev/null
+++ b/lerobot/src/lerobot/data_processing/__init__.py
@@ -0,0 +1,13 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
diff --git a/lerobot/src/lerobot/data_processing/sarm_annotations/__init__.py b/lerobot/src/lerobot/data_processing/sarm_annotations/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..2f76d5676d10d8c064b78855cbfdc6b3c4ea45a7
--- /dev/null
+++ b/lerobot/src/lerobot/data_processing/sarm_annotations/__init__.py
@@ -0,0 +1,13 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
diff --git a/lerobot/src/lerobot/data_processing/sarm_annotations/subtask_annotation.py b/lerobot/src/lerobot/data_processing/sarm_annotations/subtask_annotation.py
new file mode 100644
index 0000000000000000000000000000000000000000..8f3a65e3900897a9fa8e237bb1e46789e568bb3d
--- /dev/null
+++ b/lerobot/src/lerobot/data_processing/sarm_annotations/subtask_annotation.py
@@ -0,0 +1,1203 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+SARM Subtask Annotation using local GPU (Qwen3-VL).
+
+This script implements the annotation approach from the SARM paper using local GPU inference:
+"SARM: Stage-Aware Reward Modeling for Long Horizon Robot Manipulation"
+Paper: https://arxiv.org/pdf/2509.25358
+
+What it does:
+1. Takes videos from a LeRobot dataset
+2. Uses Qwen3-VL running locally on GPU to identify when subtasks occur
+3. Saves subtask timestamps to the dataset metadata
+4. Optionally pushes the annotated dataset to HuggingFace Hub
+
+SARM trains reward models that predict:
+  - Stage: Which subtask is currently being executed (discrete classification)
+  - Progress: How far along the subtask we are (continuous 0-1)
+
+Supports three annotation modes:
+  1. No annotations (no args): Auto-creates single sparse "task" stage covering full episode.
+     Use with SARM config annotation_mode="single_stage" for simple tasks.
+
+  2. Dense-only (--dense-only --dense-subtasks): Dense annotations from VLM, auto-generated
+     single sparse "task" stage. Use with annotation_mode="dense_only".
+
+  3. Dual mode (--sparse-subtasks + --dense-subtasks): Both sparse and dense annotations
+     from VLM. Use with annotation_mode="dual".
+
+Requirements:
+  - GPU with sufficient VRAM (16GB+ recommended for 30B model)
+  - `pip install transformers, torch, qwen-vl-utils`
+
+Run with:
+```bash
+python examples/dataset_annotation/subtask_annotation.py \
+  --repo-id your-username/your-dataset \
+  --sparse-subtasks "Do ..." \
+  --dense-subtasks "Do task 1, Do task 2, Do task 3" \
+  --video-key observation.images.base \
+  --push-to-hub
+```
+"""
+
+import argparse
+import json
+import multiprocessing as mp
+import random
+import re
+import subprocess
+import tempfile
+import textwrap
+import time
+from concurrent.futures import ProcessPoolExecutor, as_completed
+from pathlib import Path
+from typing import Any
+
+import cv2
+import numpy as np
+import pandas as pd
+import torch
+from pydantic import BaseModel, Field
+from transformers import AutoProcessor, Qwen3VLMoeForConditionalGeneration
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+
+# Pydantic Models for SARM Subtask Annotation
+class Timestamp(BaseModel):
+    """Timestamp in MM:SS or SS format"""
+
+    start: str = Field(description="Start timestamp (MM:SS or just seconds)")
+    end: str = Field(description="End timestamp (MM:SS or just seconds)")
+
+
+class Subtask(BaseModel):
+    """Individual subtask/stage - must use EXACT names from provided list"""
+
+    name: str = Field(description="Subtask name - MUST match one from the predefined list exactly")
+    timestamps: Timestamp
+
+
+class SubtaskAnnotation(BaseModel):
+    """Complete annotation for a robot manipulation episode"""
+
+    subtasks: list[Subtask] = Field(description="List of all subtasks in temporal order")
+
+
+def compute_temporal_proportions(
+    annotations: dict[int, Any], fps: int = 30, subtask_order: list[str] | None = None
+) -> dict[str, float]:
+    """
+    Compute dataset-level temporal proportions (priors) for each subtask.
+
+    Implements SARM Paper Formula (1): ᾱ_k = (1/M) × Σ_i (L_{i,k} / T_i)
+
+    Args:
+        annotations: Dict mapping episode index to SubtaskAnnotation object.
+        fps: Frames per second (unused, kept for API compatibility)
+        subtask_order: Optional list defining the output order of subtasks.
+
+    Returns:
+        Dict mapping subtask name to its temporal proportion (ᾱ_k), ordered by subtask_order if provided.
+    """
+    subtask_proportions: dict[str, list[float]] = {}
+
+    for annotation in annotations.values():
+        total_duration = 0
+        durations: dict[str, int] = {}
+
+        for subtask in annotation.subtasks:
+            start_parts = subtask.timestamps.start.split(":")
+            end_parts = subtask.timestamps.end.split(":")
+
+            start_seconds = (
+                int(start_parts[0]) * 60 + int(start_parts[1])
+                if len(start_parts) == 2
+                else int(start_parts[0])
+            )
+            end_seconds = (
+                int(end_parts[0]) * 60 + int(end_parts[1]) if len(end_parts) == 2 else int(end_parts[0])
+            )
+
+            duration = end_seconds - start_seconds
+            durations[subtask.name] = duration
+            total_duration += duration
+
+        if total_duration > 0:
+            for name, duration in durations.items():
+                if name not in subtask_proportions:
+                    subtask_proportions[name] = []
+                subtask_proportions[name].append(duration / total_duration)
+
+    if not subtask_proportions:
+        return {}
+
+    avg_proportions = {name: sum(props) / len(props) for name, props in subtask_proportions.items()}
+
+    total = sum(avg_proportions.values())
+    if total > 0:
+        avg_proportions = {name: prop / total for name, prop in avg_proportions.items()}
+
+    # Reorder according to subtask_order if provided
+    if subtask_order:
+        avg_proportions = {
+            name: avg_proportions.get(name, 0.0) for name in subtask_order if name in avg_proportions
+        }
+
+    return avg_proportions
+
+
+def create_sarm_prompt(subtask_list: list[str]) -> str:
+    subtask_str = "\n".join([f"  - {name}" for name in subtask_list])
+
+    return textwrap.dedent(f"""\
+        # Role
+        You are a Robotics Vision System specializing in temporal action localization for robot manipulation. Your job is to segment a single demonstration video into distinct, non-overlapping atomic actions from a fixed subtask list.
+
+        # Subtask Label Set (Closed Vocabulary)
+        You must strictly identify the video segments using ONLY the following labels. Do not create new labels or modify existing ones:
+
+        [
+        {subtask_str}
+        ]
+
+        The video shows one successful execution of all subtasks in a logical order.
+
+        # Ground-Truth Semantics (Very Important)
+        Use **visual state changes** to define when a subtask starts and ends. Do NOT assume equal durations for the subtasks.
+
+        - A subtask **starts** at the first frame where the robot's motion clearly initiates that subtask.
+        - A subtask **ends** at the first frame where that specific action is visually completed and the manipulated object reaches a temporary, stable configuration.
+
+        If there are short pauses or micro-motions that don't clearly correspond to a new subtask, they belong to the **current** subtask.
+
+        # Hard Constraints & Logic
+        1. **Continuous Coverage (No Gaps):**
+           - The entire video duration from "00:00" to the final timestamp must be covered by subtasks.
+           - There can be no gaps between subtasks.
+           - If there is any idle or ambiguous time between clear actions, extend the *preceding* subtask to cover it.
+
+        2. **Boundary Consistency:**
+           - The `"end"` timestamp of one subtask must be exactly equal to the `"start"` timestamp of the next subtask.
+           - Boundaries must coincide with a real visual state transition, not just a convenient time split.
+
+        3. **Chronological Order, One Occurrence Each:**
+           - This is a single successful demonstration.
+           - Each subtask from the vocabulary appears **exactly once**, in the correct logical order.
+           - **Durations may be very different** between subtasks. Never assume they are similar lengths. Base all boundaries only on the video.
+
+        4. **Reject Uniform Segmentation (Important):**
+           - Do NOT simply divide the video into equal or nearly equal time chunks.
+           - If your boundaries would result in subtasks with similar durations (e.g. all around 5 seconds), treat this as evidence that your segmentation is wrong and refine the boundaries.
+           - Only use nearly equal durations if the video truly shows each subtask taking the same amount of time (this is very rare).
+
+        5. **Timestamps:**
+           - Timestamps must be in `"MM:SS"` format.
+           - The first subtask always starts at `"00:00"`.
+           - The last subtask ends at the final visible frame of the video.
+
+        # Step 1 — Textual Timeline (must do this first)
+        First, write a extensive and detailed textual timeline describing what happens in the video with approximate timestamps.
+        For each subtask, include:
+        - its name
+        - an approximate start and end time,
+        - an description of the visual event at the boundary (e.g. "shirt fully folded to the left", "robot rotates folded shirt 90 degrees").
+
+        Format this as a bullet list.
+
+        # Step 2 — JSON Output (final answer)
+        After the textual timeline, output **only** valid JSON with this structure.
+        The JSON **must** be consistent with the textual timeline above:
+
+        {{
+          "subtasks": [
+            {{
+              "name": "EXACT_NAME_FROM_LIST",
+              "timestamps": {{
+                "start": "MM:SS",
+                "end":   "MM:SS"
+              }}
+            }},
+            {{
+              "name": "EXACT_NAME_FROM_LIST",
+              "timestamps": {{
+                "start": "MM:SS",
+                "end":   "MM:SS"
+              }}
+            }}
+          ]
+        }}
+
+        Do not add any extra keys to the JSON.
+        """)
+
+
+class VideoAnnotator:
+    """Annotates robot manipulation videos using local Qwen3-VL model on GPU"""
+
+    def __init__(
+        self,
+        subtask_list: list[str],
+        model_name: str = "Qwen/Qwen3-VL-30B-A3B-Instruct",
+        device: str = "cuda",
+        torch_dtype: torch.dtype = torch.bfloat16,
+        model: Qwen3VLMoeForConditionalGeneration | None = None,  # noqa: F821
+        processor: AutoProcessor | None = None,  # noqa: F821
+    ):
+        """
+        Initialize the video annotator with local model.
+
+        Args:
+            subtask_list: List of allowed subtask names (for consistency)
+            model_name: Hugging Face model name (default: Qwen/Qwen3-VL-30B-A3B-Instruct)
+            device: Device to use (cuda, cpu)
+            torch_dtype: Data type for model (bfloat16, float16, float32)
+            model: Pre-loaded model instance (optional, to share between annotators)
+            processor: Pre-loaded processor instance (optional, to share between annotators)
+        """
+        self.subtask_list = subtask_list
+        self.prompt = create_sarm_prompt(subtask_list)
+        self.device = device
+
+        # Use provided model/processor or load new ones
+        if model is not None and processor is not None:
+            self.model = model
+            self.processor = processor
+            print(f"Using shared model on {device}")
+        else:
+            from transformers import AutoProcessor, Qwen3VLMoeForConditionalGeneration
+
+            print(f"Loading model: {model_name}...")
+
+            self.model = Qwen3VLMoeForConditionalGeneration.from_pretrained(
+                model_name, torch_dtype=torch_dtype, device_map=device, trust_remote_code=True
+            )
+
+            self.processor = AutoProcessor.from_pretrained(model_name, trust_remote_code=True)
+
+            print(f"Model loaded successfully on {device}")
+
+    def extract_episode_segment(
+        self, file_path: Path, start_timestamp: float, end_timestamp: float, target_fps: int = 1
+    ) -> Path:
+        """
+        Extract a specific episode segment from concatenated video.
+        Uses minimal compression to preserve quality for local inference.
+
+        Args:
+            file_path: Path to the concatenated video file
+            start_timestamp: Starting timestamp in seconds (within this video file)
+            end_timestamp: Ending timestamp in seconds (within this video file)
+            target_fps: Target FPS (default: 1 for faster processing)
+
+        Returns:
+            Path to extracted video file
+        """
+        # Create temporary file for extracted video
+        with tempfile.NamedTemporaryFile(suffix=".mp4", delete=False) as tmp_file:
+            tmp_path = Path(tmp_file.name)
+
+        try:
+            # Check if ffmpeg is available
+            subprocess.run(  # nosec B607
+                ["ffmpeg", "-version"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True
+            )
+        except (subprocess.CalledProcessError, FileNotFoundError) as err:
+            raise RuntimeError("ffmpeg not found, cannot extract episode segment") from err
+
+        try:
+            # Calculate duration
+            duration = end_timestamp - start_timestamp
+
+            print(f"Extracting episode: {start_timestamp:.1f}s-{end_timestamp:.1f}s ({duration:.1f}s)")
+
+            # Use ffmpeg to extract segment with minimal quality loss
+            cmd = [
+                "ffmpeg",
+                "-i",
+                str(file_path),
+                "-ss",
+                str(start_timestamp),
+                "-t",
+                str(duration),
+                "-r",
+                str(target_fps),
+                "-c:v",
+                "libx264",
+                "-preset",
+                "ultrafast",
+                "-crf",
+                "23",
+                "-an",
+                "-y",
+                str(tmp_path),
+            ]
+
+            subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
+
+            # Verify the output file was created and is not empty
+            if not tmp_path.exists() or tmp_path.stat().st_size == 0:
+                print("Video extraction failed (0 bytes) - skipping episode")
+                if tmp_path.exists():
+                    tmp_path.unlink()
+                raise RuntimeError("FFmpeg produced empty video file")
+
+            # Show extraction results
+            file_size_mb = tmp_path.stat().st_size / (1024 * 1024)
+
+            # Fail if file is too small (< 100KB likely means extraction failed)
+            if file_size_mb < 0.1:
+                print(f"Extracted video too small ({file_size_mb:.2f}MB) - skipping episode")
+                tmp_path.unlink()
+                raise RuntimeError(f"Video extraction produced invalid file ({file_size_mb:.2f}MB)")
+
+            print(f"Extracted: {file_size_mb:.1f}MB ({target_fps} FPS)")
+
+            return tmp_path
+
+        except subprocess.CalledProcessError as e:
+            raise RuntimeError(f"ffmpeg failed ({e})") from e
+
+    def annotate(
+        self,
+        file_path: str | Path,
+        fps: int,
+        start_timestamp: float = 0.0,
+        end_timestamp: float | None = None,
+        max_retries: int = 3,
+    ) -> SubtaskAnnotation:
+        """Annotate a video segment using local GPU."""
+        from qwen_vl_utils import process_vision_info
+
+        file_path = Path(file_path)
+
+        if end_timestamp is None:
+            cap = cv2.VideoCapture(str(file_path))
+            end_timestamp = int(cap.get(cv2.CAP_PROP_FRAME_COUNT)) / (cap.get(cv2.CAP_PROP_FPS) or 1)
+            cap.release()
+
+        duration = end_timestamp - start_timestamp
+        duration_str = f"{int(duration // 60):02d}:{int(duration % 60):02d}"
+
+        extracted_path = self.extract_episode_segment(file_path, start_timestamp, end_timestamp, 1)
+        is_extracted = extracted_path != file_path
+
+        try:
+            messages = [
+                {"role": "system", "content": [{"type": "text", "text": self.prompt}]},
+                {
+                    "role": "user",
+                    "content": [
+                        {"type": "video", "video": str(extracted_path), "fps": 1.0},
+                        {
+                            "type": "text",
+                            "text": f"Video is {duration_str} (~{duration:.1f}s). Follow instructions.",
+                        },
+                    ],
+                },
+            ]
+
+            for attempt in range(max_retries):
+                try:
+                    text = self.processor.apply_chat_template(
+                        messages, tokenize=False, add_generation_prompt=True
+                    )
+                    image_inputs, video_inputs = process_vision_info(messages)
+                    inputs = self.processor(
+                        text=[text],
+                        images=image_inputs,
+                        videos=video_inputs,
+                        padding=True,
+                        return_tensors="pt",
+                    ).to(self.device)
+
+                    with torch.no_grad():
+                        generated_ids = self.model.generate(
+                            **inputs, max_new_tokens=1024, do_sample=True, temperature=0.7
+                        )
+
+                    response = self.processor.batch_decode(
+                        [out[len(inp) :] for inp, out in zip(inputs.input_ids, generated_ids, strict=True)],
+                        skip_special_tokens=True,
+                    )[0].strip()
+
+                    # Extract JSON
+                    if "```json" in response:
+                        response = response.split("```json")[1].split("```")[0]
+                    elif "```" in response:
+                        response = response.split("```")[1].split("```")[0]
+
+                    try:
+                        return SubtaskAnnotation.model_validate(json.loads(response))
+                    except json.JSONDecodeError:
+                        match = re.search(r"\{.*\}", response, re.DOTALL)
+                        if match:
+                            return SubtaskAnnotation.model_validate(json.loads(match.group()))
+                        raise ValueError("No JSON found") from None
+                except Exception as e:
+                    if attempt == max_retries - 1:
+                        raise RuntimeError(f"Failed after {max_retries} attempts") from e
+                    time.sleep(1)
+        finally:
+            if is_extracted and extracted_path.exists():
+                extracted_path.unlink()
+
+
+def display_annotation(annotation: SubtaskAnnotation, episode_idx: int, fps: int, prefix: str = ""):
+    """Display annotation summary."""
+    subtask_summary = ", ".join(
+        f"{s.name}({s.timestamps.start}-{s.timestamps.end})" for s in annotation.subtasks
+    )
+    print(f"Episode {episode_idx} {prefix}: {len(annotation.subtasks)} subtasks - {subtask_summary}")
+
+
+def timestamp_to_seconds(timestamp: str) -> float:
+    """Convert MM:SS or SS timestamp to seconds"""
+    parts = timestamp.split(":")
+    if len(parts) == 2:
+        return int(parts[0]) * 60 + int(parts[1])
+    else:
+        return int(parts[0])
+
+
+def extract_frame(video_path: Path, timestamp: float) -> np.ndarray | None:
+    """Extract a single frame from video at given timestamp."""
+    cap = cv2.VideoCapture(str(video_path))
+    if not cap.isOpened():
+        return None
+    cap.set(cv2.CAP_PROP_POS_MSEC, timestamp * 1000)
+    ret, frame = cap.read()
+    cap.release()
+    return cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) if ret else None
+
+
+def draw_timeline(ax, subtasks, total_duration, colors):
+    """Draw a timeline with color-coded subtask segments."""
+    import matplotlib.patches as mpatches
+
+    bar_height, bar_y = 0.6, 0.5
+
+    for i, subtask in enumerate(subtasks):
+        start = timestamp_to_seconds(subtask.timestamps.start)
+        end = timestamp_to_seconds(subtask.timestamps.end)
+        color = colors[i % len(colors)]
+
+        rect = mpatches.FancyBboxPatch(
+            (start, bar_y - bar_height / 2),
+            end - start,
+            bar_height,
+            boxstyle="round,pad=0.02,rounding_size=0.1",
+            facecolor=color,
+            edgecolor="white",
+            linewidth=1.5,
+            alpha=0.85,
+        )
+        ax.add_patch(rect)
+
+        # Add label if segment is wide enough
+        duration = end - start
+        if duration > total_duration * 0.06:
+            ax.text(
+                (start + end) / 2,
+                bar_y,
+                subtask.name,
+                ha="center",
+                va="center",
+                fontsize=8,
+                fontweight="bold",
+                color="white",
+                rotation=0 if duration > total_duration * 0.12 else 45,
+            )
+
+        if i > 0:
+            ax.axvline(x=start, ymin=0.1, ymax=0.9, color="white", linestyle="--", linewidth=1.5, alpha=0.7)
+
+    ax.axvline(x=0, ymin=0.1, ymax=0.9, color="#00ff00", linestyle="-", linewidth=2, alpha=0.9)
+    if subtasks:
+        ax.axvline(
+            x=timestamp_to_seconds(subtasks[-1].timestamps.end),
+            ymin=0.1,
+            ymax=0.9,
+            color="white",
+            linestyle="--",
+            linewidth=1.5,
+            alpha=0.7,
+        )
+
+    ax.set_xlim(-total_duration * 0.02, total_duration * 1.02)
+    ax.set_ylim(-0.1, 1.1)
+    ax.set_xlabel("Time (seconds)", fontsize=10, color="white", labelpad=5)
+    for spine in ["top", "right", "left"]:
+        ax.spines[spine].set_visible(False)
+    ax.spines["bottom"].set_color("#444444")
+    ax.tick_params(axis="x", colors="#888888", labelsize=8)
+    ax.tick_params(axis="y", left=False, labelleft=False)
+
+
+def visualize_episode(
+    ep_idx: int,
+    annotation: SubtaskAnnotation,
+    video_path: Path,
+    video_start: float,
+    video_end: float,
+    output_path: Path,
+    video_key: str,
+    ann_type: str,
+):
+    """Create visualization for a single episode with frames and timeline."""
+    import matplotlib.pyplot as plt
+
+    if annotation is None:
+        print(f"No {ann_type} annotation for episode {ep_idx}")
+        return
+
+    subtasks = annotation.subtasks
+    if not subtasks:
+        print(f"No subtasks for episode {ep_idx}")
+        return
+
+    colors = plt.cm.tab10(np.linspace(0, 1, max(len(subtasks), 10)))
+    total_duration = timestamp_to_seconds(subtasks[-1].timestamps.end)
+
+    # Extract middle frame from each subtask
+    sample_frames, frame_times = [], []
+    for subtask in subtasks:
+        start = timestamp_to_seconds(subtask.timestamps.start)
+        end = timestamp_to_seconds(subtask.timestamps.end)
+        mid = (start + end) / 2
+        frame_times.append(mid)
+        sample_frames.append(extract_frame(video_path, video_start + mid))
+
+    # Create figure
+    fig_width = max(16, len(subtasks) * 2.5)
+    fig = plt.figure(figsize=(fig_width, 10))
+    fig.patch.set_facecolor("#1a1a2e")
+
+    gs = fig.add_gridspec(
+        2,
+        max(len(subtasks), 1),
+        height_ratios=[2, 1],
+        hspace=0.3,
+        wspace=0.1,
+        left=0.05,
+        right=0.95,
+        top=0.88,
+        bottom=0.1,
+    )
+
+    fig.suptitle(
+        f"Episode {ep_idx} - {ann_type.capitalize()} Annotations",
+        fontsize=18,
+        fontweight="bold",
+        color="white",
+        y=0.96,
+    )
+    fig.text(
+        0.5,
+        0.91,
+        f"Camera: {video_key} | Duration: {video_end - video_start:.1f}s | {len(subtasks)} subtasks",
+        ha="center",
+        fontsize=11,
+        color="#888888",
+    )
+
+    # Plot frames
+    for i, (frame, subtask) in enumerate(zip(sample_frames, subtasks, strict=True)):
+        ax = fig.add_subplot(gs[0, i])
+        ax.set_facecolor("#16213e")
+        if frame is not None:
+            ax.imshow(frame)
+        else:
+            ax.text(
+                0.5, 0.5, "N/A", ha="center", va="center", fontsize=12, color="white", transform=ax.transAxes
+            )
+        ax.set_title(subtask.name, fontsize=10, fontweight="bold", color=colors[i % len(colors)], pad=8)
+        ax.axis("off")
+        ax.text(
+            0.5,
+            -0.08,
+            f"t={frame_times[i]:.1f}s",
+            ha="center",
+            fontsize=9,
+            color="#888888",
+            transform=ax.transAxes,
+        )
+
+    # Plot timeline
+    ax_timeline = fig.add_subplot(gs[1, :])
+    ax_timeline.set_facecolor("#16213e")
+    draw_timeline(ax_timeline, subtasks, total_duration, colors)
+
+    output_path.parent.mkdir(parents=True, exist_ok=True)
+    plt.savefig(output_path, dpi=150, facecolor=fig.get_facecolor(), edgecolor="none", bbox_inches="tight")
+    plt.close()
+    print(f"Saved: {output_path}")
+
+
+def visualize_annotations(
+    dataset: LeRobotDataset,
+    sparse_annotations: dict[int, SubtaskAnnotation],
+    dense_annotations: dict[int, SubtaskAnnotation] | None,
+    video_key: str,
+    output_dir: Path,
+    num_episodes: int = 5,
+    annotation_type: str = "sparse",
+    episode_indices: list[int] | None = None,
+):
+    """
+    Visualize subtask annotations for a set of episodes.
+
+    Args:
+        dataset: LeRobotDataset instance
+        sparse_annotations: Dict mapping episode index to sparse annotations
+        dense_annotations: Dict mapping episode index to dense annotations (or None)
+        video_key: Camera/video key to use
+        output_dir: Directory to save visualization images
+        num_episodes: Number of episodes to visualize (ignored if episode_indices provided)
+        annotation_type: "sparse", "dense", or "both"
+        episode_indices: Specific episode indices to visualize (optional)
+    """
+    # Determine available episodes based on annotation type
+    if annotation_type == "sparse":
+        available = set(sparse_annotations.keys())
+    elif annotation_type == "dense":
+        available = set(dense_annotations.keys()) if dense_annotations else set()
+    else:  # both
+        sparse_set = set(sparse_annotations.keys())
+        dense_set = set(dense_annotations.keys()) if dense_annotations else set()
+        available = sparse_set | dense_set
+
+    if not available:
+        print("Error: No annotations found to visualize.")
+        return
+
+    # Select episodes to visualize
+    if episode_indices:
+        episodes = sorted([e for e in episode_indices if e in available])
+        missing = set(episode_indices) - available
+        if missing:
+            print(f"Episodes not found in annotations: {sorted(missing)}")
+    else:
+        episodes = sorted(random.sample(list(available), min(num_episodes, len(available))))
+    print(f"Visualizing {len(episodes)} episodes: {episodes}")
+    output_dir.mkdir(parents=True, exist_ok=True)
+
+    # Generate visualizations
+    for i, ep_idx in enumerate(episodes, 1):
+        print(f"Processing episode {ep_idx} ({i}/{len(episodes)})")
+        video_path = dataset.root / dataset.meta.get_video_file_path(ep_idx, video_key)
+        if not video_path.exists():
+            print(f"Video not found: {video_path}")
+            continue
+
+        video_start = float(dataset.meta.episodes[f"videos/{video_key}/from_timestamp"][ep_idx])
+        video_end = float(dataset.meta.episodes[f"videos/{video_key}/to_timestamp"][ep_idx])
+
+        if annotation_type == "both":
+            # Visualize both sparse and dense
+            for ann_type, annotations in [("sparse", sparse_annotations), ("dense", dense_annotations)]:
+                if annotations and ep_idx in annotations:
+                    output_path = output_dir / f"episode_{ep_idx:04d}_{ann_type}.png"
+                    visualize_episode(
+                        ep_idx,
+                        annotations.get(ep_idx),
+                        video_path,
+                        video_start,
+                        video_end,
+                        output_path,
+                        video_key,
+                        ann_type,
+                    )
+        else:
+            annotations = sparse_annotations if annotation_type == "sparse" else dense_annotations
+            if annotations and ep_idx in annotations:
+                output_path = output_dir / f"episode_{ep_idx:04d}_{annotation_type}.png"
+                visualize_episode(
+                    ep_idx,
+                    annotations.get(ep_idx),
+                    video_path,
+                    video_start,
+                    video_end,
+                    output_path,
+                    video_key,
+                    annotation_type,
+                )
+
+    print(f"Visualizations saved to: {output_dir.absolute()}")
+
+
+def save_annotations_to_dataset(
+    dataset_path: Path, annotations: dict[int, SubtaskAnnotation], fps: int, prefix: str = "sparse"
+):
+    """Save annotations to LeRobot dataset parquet format."""
+    from lerobot.datasets.io_utils import load_episodes
+    from lerobot.datasets.utils import DEFAULT_EPISODES_PATH
+
+    episodes_dataset = load_episodes(dataset_path)
+    if not episodes_dataset or len(episodes_dataset) == 0:
+        return
+
+    episodes_df = episodes_dataset.to_pandas()
+    cols = [
+        f"{prefix}_{c}"
+        for c in [
+            "subtask_names",
+            "subtask_start_times",
+            "subtask_end_times",
+            "subtask_start_frames",
+            "subtask_end_frames",
+        ]
+    ]
+    for col in cols:
+        episodes_df[col] = None
+
+    for ep_idx, ann in annotations.items():
+        if ep_idx >= len(episodes_df):
+            continue
+        names, starts, ends, start_frames, end_frames = [], [], [], [], []
+        for s in ann.subtasks:
+            names.append(s.name)
+            st, et = timestamp_to_seconds(s.timestamps.start), timestamp_to_seconds(s.timestamps.end)
+            starts.append(st)
+            ends.append(et)
+            start_frames.append(int(st * fps))
+            end_frames.append(int(et * fps))
+        episodes_df.at[ep_idx, cols[0]] = names
+        episodes_df.at[ep_idx, cols[1]] = starts
+        episodes_df.at[ep_idx, cols[2]] = ends
+        episodes_df.at[ep_idx, cols[3]] = start_frames
+        episodes_df.at[ep_idx, cols[4]] = end_frames
+
+    # Group by file and write
+    for ep_idx in episodes_df.index:
+        key = (
+            episodes_df.loc[ep_idx, "meta/episodes/chunk_index"],
+            episodes_df.loc[ep_idx, "meta/episodes/file_index"],
+        )
+        path = dataset_path / DEFAULT_EPISODES_PATH.format(chunk_index=key[0], file_index=key[1])
+        if path.exists():
+            file_df = pd.read_parquet(path)
+            for col in cols + (
+                [
+                    "subtask_names",
+                    "subtask_start_times",
+                    "subtask_end_times",
+                    "subtask_start_frames",
+                    "subtask_end_frames",
+                ]
+                if prefix == "sparse"
+                else []
+            ):
+                if col not in file_df.columns:
+                    file_df[col] = None
+            if ep_idx in annotations:
+                for col in cols:
+                    file_df.at[ep_idx, col] = episodes_df.loc[ep_idx, col]
+                if prefix == "sparse":  # Legacy columns
+                    for i, legacy in enumerate(
+                        [
+                            "subtask_names",
+                            "subtask_start_times",
+                            "subtask_end_times",
+                            "subtask_start_frames",
+                            "subtask_end_frames",
+                        ]
+                    ):
+                        file_df.at[ep_idx, legacy] = episodes_df.loc[ep_idx, cols[i]]
+            file_df.to_parquet(path, engine="pyarrow", compression="snappy")
+
+
+def generate_auto_sparse_annotations(
+    dataset: LeRobotDataset, episode_indices: list[int], video_key: str
+) -> dict[int, SubtaskAnnotation]:
+    """Auto-generate single 'task' stage annotations for all episodes."""
+    annotations = {}
+    for ep_idx in episode_indices:
+        start = float(dataset.meta.episodes[f"videos/{video_key}/from_timestamp"][ep_idx])
+        end = float(dataset.meta.episodes[f"videos/{video_key}/to_timestamp"][ep_idx])
+        duration = end - start
+        end_str = f"{int(duration // 60):02d}:{int(duration % 60):02d}"
+        annotations[ep_idx] = SubtaskAnnotation(
+            subtasks=[Subtask(name="task", timestamps=Timestamp(start="00:00", end=end_str))]
+        )
+    return annotations
+
+
+def load_annotations_from_dataset(dataset_path: Path, prefix: str = "sparse") -> dict[int, SubtaskAnnotation]:
+    """Load annotations from LeRobot dataset parquet files."""
+    from lerobot.datasets.io_utils import load_episodes
+
+    episodes_dataset = load_episodes(dataset_path)
+    if not episodes_dataset or len(episodes_dataset) == 0:
+        return {}
+
+    col_names = f"{prefix}_subtask_names"
+    col_start = f"{prefix}_subtask_start_times"
+    col_end = f"{prefix}_subtask_end_times"
+
+    # Fall back to legacy columns for sparse
+    if col_names not in episodes_dataset.column_names:
+        if prefix == "sparse" and "subtask_names" in episodes_dataset.column_names:
+            col_names, col_start, col_end = "subtask_names", "subtask_start_times", "subtask_end_times"
+        else:
+            return {}
+
+    df = episodes_dataset.to_pandas()
+    annotations = {}
+    for ep_idx in df.index:
+        names = df.loc[ep_idx, col_names]
+        if names is None or (isinstance(names, float) and pd.isna(names)):
+            continue
+        starts, ends = df.loc[ep_idx, col_start], df.loc[ep_idx, col_end]
+        annotations[int(ep_idx)] = SubtaskAnnotation(
+            subtasks=[
+                Subtask(
+                    name=n,
+                    timestamps=Timestamp(
+                        start=f"{int(s) // 60:02d}:{int(s) % 60:02d}",
+                        end=f"{int(e) // 60:02d}:{int(e) % 60:02d}",
+                    ),
+                )
+                for n, s, e in zip(names, starts, ends, strict=True)
+            ]
+        )
+    return annotations
+
+
+def process_single_episode(
+    ep_idx: int,
+    dataset_root: Path,
+    dataset_meta,
+    video_key: str,
+    fps: int,
+    annotator: VideoAnnotator,
+) -> tuple[int, SubtaskAnnotation | None, str | None]:
+    """Process a single episode annotation."""
+    try:
+        video_path = dataset_root / dataset_meta.get_video_file_path(ep_idx, video_key)
+        if not video_path.exists():
+            return ep_idx, None, f"Video not found: {video_path}"
+
+        start = float(dataset_meta.episodes[f"videos/{video_key}/from_timestamp"][ep_idx])
+        end = float(dataset_meta.episodes[f"videos/{video_key}/to_timestamp"][ep_idx])
+        return ep_idx, annotator.annotate(video_path, fps, start, end), None
+    except Exception as e:
+        return ep_idx, None, str(e)
+
+
+def worker_process_episodes(
+    worker_id: int,
+    gpu_id: int,
+    episode_indices: list[int],
+    repo_id: str,
+    video_key: str,
+    sparse_subtask_list: list[str],
+    dense_subtask_list: list[str] | None,
+    model_name: str,
+    torch_dtype: torch.dtype,
+) -> tuple[dict, dict | None]:
+    """Worker for parallel processing across GPUs."""
+    device = f"cuda:{gpu_id}"
+    dataset = LeRobotDataset(repo_id, download_videos=False)
+
+    sparse_annotator = VideoAnnotator(sparse_subtask_list, model_name, device, torch_dtype)
+    dense_annotator = (
+        VideoAnnotator(
+            dense_subtask_list,
+            model_name,
+            device,
+            torch_dtype,
+            sparse_annotator.model,
+            sparse_annotator.processor,
+        )
+        if dense_subtask_list
+        else None
+    )
+
+    sparse_annotations, dense_annotations = {}, {} if dense_subtask_list else None
+
+    for ep_idx in episode_indices:
+        _, sparse_ann, err = process_single_episode(
+            ep_idx, dataset.root, dataset.meta, video_key, dataset.fps, sparse_annotator
+        )
+        if sparse_ann:
+            sparse_annotations[ep_idx] = sparse_ann
+
+        if dense_annotator:
+            _, dense_ann, _ = process_single_episode(
+                ep_idx, dataset.root, dataset.meta, video_key, dataset.fps, dense_annotator
+            )
+            if dense_ann:
+                dense_annotations[ep_idx] = dense_ann
+
+    return sparse_annotations, dense_annotations
+
+
+def main():
+    parser = argparse.ArgumentParser(description="SARM-style subtask annotation using local GPU (Qwen3-VL)")
+    parser.add_argument("--repo-id", type=str, required=True, help="HuggingFace dataset repository ID")
+    parser.add_argument(
+        "--sparse-subtasks", type=str, default=None, help="Comma-separated sparse subtask names"
+    )
+    parser.add_argument(
+        "--dense-subtasks", type=str, default=None, help="Comma-separated dense subtask names"
+    )
+    parser.add_argument(
+        "--dense-only", action="store_true", help="Dense-only mode with auto-generated sparse 'task' stage"
+    )
+    parser.add_argument("--episodes", type=int, nargs="+", default=None, help="Episode indices to annotate")
+    parser.add_argument("--model", type=str, default="Qwen/Qwen3-VL-30B-A3B-Instruct", help="VLM model")
+    parser.add_argument("--skip-existing", action="store_true", help="Skip already annotated episodes")
+    parser.add_argument("--video-key", type=str, default=None, help="Video key (default: first available)")
+    parser.add_argument("--push-to-hub", action="store_true", help="Push to HuggingFace Hub")
+    parser.add_argument("--output-repo-id", type=str, default=None, help="Output repo ID for push")
+    parser.add_argument("--device", type=str, default="cuda", help="Device (cuda/cpu)")
+    parser.add_argument("--dtype", type=str, default="bfloat16", choices=["bfloat16", "float16", "float32"])
+    parser.add_argument("--num-workers", type=int, default=1, help="Parallel workers for multi-GPU")
+    parser.add_argument("--gpu-ids", type=int, nargs="+", default=None, help="GPU IDs to use")
+    # Visualization options
+    parser.add_argument(
+        "--visualize-only",
+        action="store_true",
+        help="Only visualize existing annotations (no generation)",
+    )
+    parser.add_argument(
+        "--num-visualizations",
+        type=int,
+        default=5,
+        help="Number of episodes to visualize (default: 5)",
+    )
+    parser.add_argument(
+        "--visualize-type",
+        type=str,
+        default="sparse",
+        choices=["sparse", "dense", "both"],
+        help="Type of annotations to visualize (default: sparse)",
+    )
+    parser.add_argument(
+        "--output-dir",
+        type=str,
+        default="./subtask_viz",
+        help="Output directory for visualizations (default: ./subtask_viz)",
+    )
+
+    args = parser.parse_args()
+
+    # Load dataset first (needed for both annotation and visualization)
+    print(f"Loading dataset: {args.repo_id}")
+    dataset = LeRobotDataset(args.repo_id, download_videos=True)
+    fps = dataset.fps
+
+    if not dataset.meta.video_keys:
+        raise ValueError("No video keys found")
+
+    video_key = (
+        args.video_key if args.video_key in (dataset.meta.video_keys or []) else dataset.meta.video_keys[0]
+    )
+    print(f"Using camera: {video_key}, FPS: {fps}")
+
+    # Handle visualization-only mode
+    if args.visualize_only:
+        print("Visualization-only mode")
+        sparse_annotations = load_annotations_from_dataset(dataset.root, prefix="sparse")
+        dense_annotations = load_annotations_from_dataset(dataset.root, prefix="dense")
+
+        if not sparse_annotations and not dense_annotations:
+            return print("Error: No annotations found. Run annotation first.")
+
+        print(f"Found {len(sparse_annotations)} sparse, {len(dense_annotations)} dense annotations")
+
+        visualize_annotations(
+            dataset=dataset,
+            sparse_annotations=sparse_annotations,
+            dense_annotations=dense_annotations if dense_annotations else None,
+            video_key=video_key,
+            output_dir=Path(args.output_dir),
+            num_episodes=args.num_visualizations,
+            annotation_type=args.visualize_type,
+            episode_indices=args.episodes,
+        )
+        return
+
+    # Validate arguments for annotation mode
+    if args.dense_only and not args.dense_subtasks:
+        return print("Error: --dense-only requires --dense-subtasks")
+    if args.dense_subtasks and not args.sparse_subtasks and not args.dense_only:
+        return print("Error: --dense-subtasks requires --sparse-subtasks or --dense-only")
+
+    sparse_subtask_list = (
+        [s.strip() for s in args.sparse_subtasks.split(",")] if args.sparse_subtasks else None
+    )
+    dense_subtask_list = [s.strip() for s in args.dense_subtasks.split(",")] if args.dense_subtasks else None
+    auto_sparse = sparse_subtask_list is None
+    dense_mode = dense_subtask_list is not None
+    torch_dtype = {"bfloat16": torch.bfloat16, "float16": torch.float16, "float32": torch.float32}[args.dtype]
+
+    # Determine episodes
+    episode_indices = args.episodes or list(range(dataset.meta.total_episodes))
+
+    existing_annotations = load_annotations_from_dataset(dataset.root, prefix="sparse")
+    if args.skip_existing:
+        episode_indices = [ep for ep in episode_indices if ep not in existing_annotations]
+
+    if not episode_indices:
+        return print("All episodes already annotated!")
+    print(f"Annotating {len(episode_indices)} episodes")
+
+    # GPU setup
+    gpu_ids = args.gpu_ids or list(
+        range(min(args.num_workers, torch.cuda.device_count() if torch.cuda.is_available() else 1))
+    )
+    args.num_workers = len(gpu_ids)
+
+    sparse_annotations = existing_annotations.copy()
+    dense_annotations = {} if dense_mode else None
+
+    # Auto-sparse mode
+    if auto_sparse:
+        sparse_annotations.update(generate_auto_sparse_annotations(dataset, episode_indices, video_key))
+        save_annotations_to_dataset(dataset.root, sparse_annotations, fps, prefix="sparse")
+        print(f"Auto-generated {len(episode_indices)} sparse 'task' annotations")
+
+    # VLM annotation (for sparse if not auto, and for dense)
+    need_vlm = (not auto_sparse) or dense_mode
+
+    if need_vlm:
+        if args.num_workers > 1 and not auto_sparse:
+            # Parallel processing
+            print(f"Parallel processing with {args.num_workers} workers")
+            episodes_per_worker = [[] for _ in range(args.num_workers)]
+            for i, ep_idx in enumerate(episode_indices):
+                episodes_per_worker[i % args.num_workers].append(ep_idx)
+
+            with ProcessPoolExecutor(
+                max_workers=args.num_workers, mp_context=mp.get_context("spawn")
+            ) as executor:
+                futures = [
+                    executor.submit(
+                        worker_process_episodes,
+                        w,
+                        gpu_ids[w],
+                        episodes_per_worker[w],
+                        args.repo_id,
+                        video_key,
+                        sparse_subtask_list,
+                        dense_subtask_list,
+                        args.model,
+                        torch_dtype,
+                    )
+                    for w in range(args.num_workers)
+                    if episodes_per_worker[w]
+                ]
+
+                for future in as_completed(futures):
+                    try:
+                        worker_sparse, worker_dense = future.result()
+                        sparse_annotations.update(worker_sparse)
+                        if dense_mode and worker_dense:
+                            dense_annotations.update(worker_dense)
+                        save_annotations_to_dataset(dataset.root, sparse_annotations, fps, prefix="sparse")
+                        if dense_mode:
+                            save_annotations_to_dataset(dataset.root, dense_annotations, fps, prefix="dense")
+                    except Exception as e:
+                        raise RuntimeError(f"Worker failed: {e}") from e
+        else:
+            # Sequential processing
+            sparse_annotator = (
+                VideoAnnotator(sparse_subtask_list, args.model, args.device, torch_dtype)
+                if not auto_sparse and sparse_subtask_list
+                else None
+            )
+            dense_annotator = (
+                VideoAnnotator(
+                    dense_subtask_list,
+                    args.model,
+                    args.device,
+                    torch_dtype,
+                    sparse_annotator.model if sparse_annotator else None,
+                    sparse_annotator.processor if sparse_annotator else None,
+                )
+                if dense_mode
+                else None
+            )
+
+            for i, ep_idx in enumerate(episode_indices):
+                print(f"Episode {ep_idx} ({i + 1}/{len(episode_indices)})")
+
+                if sparse_annotator:
+                    _, sparse_ann, err = process_single_episode(
+                        ep_idx, dataset.root, dataset.meta, video_key, fps, sparse_annotator
+                    )
+                    if sparse_ann:
+                        sparse_annotations[ep_idx] = sparse_ann
+                        save_annotations_to_dataset(dataset.root, sparse_annotations, fps, prefix="sparse")
+                    elif err:
+                        print(f"Sparse failed: {err}")
+
+                if dense_annotator:
+                    _, dense_ann, err = process_single_episode(
+                        ep_idx, dataset.root, dataset.meta, video_key, fps, dense_annotator
+                    )
+                    if dense_ann:
+                        dense_annotations[ep_idx] = dense_ann
+                        save_annotations_to_dataset(dataset.root, dense_annotations, fps, prefix="dense")
+                    elif err:
+                        print(f"Dense failed: {err}")
+
+    # Save temporal proportions
+    def save_proportions(annotations, prefix, subtask_list=None, is_auto=False):
+        props: dict[str, float] = (
+            {"task": 1.0} if is_auto else compute_temporal_proportions(annotations, fps, subtask_list)
+        )
+        path = dataset.root / "meta" / f"temporal_proportions_{prefix}.json"
+        path.parent.mkdir(parents=True, exist_ok=True)
+        with open(path, "w") as f:
+            json.dump(props, f, indent=2)
+        print(f"Saved {prefix} temporal proportions")
+
+    save_proportions(sparse_annotations, "sparse", sparse_subtask_list, auto_sparse)
+    if dense_mode and dense_annotations:
+        save_proportions(dense_annotations, "dense", dense_subtask_list)
+
+    print(f"\nComplete! {len(sparse_annotations)} sparse, {len(dense_annotations or {})} dense annotations")
+
+    # Visualize annotations after generation
+    if args.num_visualizations > 0:
+        print(f"\nGenerating {args.num_visualizations} visualizations...")
+        visualize_type = "both" if dense_mode else "sparse"
+        visualize_annotations(
+            dataset=dataset,
+            sparse_annotations=sparse_annotations,
+            dense_annotations=dense_annotations,
+            video_key=video_key,
+            output_dir=Path(args.output_dir),
+            num_episodes=args.num_visualizations,
+            annotation_type=visualize_type,
+        )
+
+    if args.push_to_hub:
+        try:
+            dataset.push_to_hub(push_videos=True)
+            print(f"Pushed to {args.output_repo_id or args.repo_id}")
+        except Exception as e:
+            print(f"Push failed: {e}")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/datasets/aggregate.py b/lerobot/src/lerobot/datasets/aggregate.py
new file mode 100644
index 0000000000000000000000000000000000000000..66f055f047751400f4967ce446afbdda61523031
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/aggregate.py
@@ -0,0 +1,655 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team.
+# All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import shutil
+from pathlib import Path
+
+import datasets
+import pandas as pd
+import tqdm
+
+from lerobot.datasets.compute_stats import aggregate_stats
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.feature_utils import get_hf_features_from_features
+from lerobot.datasets.io_utils import (
+    get_file_size_in_mb,
+    get_parquet_file_size_in_mb,
+    to_parquet_with_hf_images,
+    write_info,
+    write_stats,
+    write_tasks,
+)
+from lerobot.datasets.utils import (
+    DEFAULT_CHUNK_SIZE,
+    DEFAULT_DATA_FILE_SIZE_IN_MB,
+    DEFAULT_DATA_PATH,
+    DEFAULT_EPISODES_PATH,
+    DEFAULT_VIDEO_FILE_SIZE_IN_MB,
+    DEFAULT_VIDEO_PATH,
+    update_chunk_file_indices,
+)
+from lerobot.datasets.video_utils import concatenate_video_files, get_video_duration_in_s
+
+
+def validate_all_metadata(all_metadata: list[LeRobotDatasetMetadata]):
+    """Validates that all dataset metadata have consistent properties.
+
+    Ensures all datasets have the same fps, robot_type, and features to guarantee
+    compatibility when aggregating them into a single dataset.
+
+    Args:
+        all_metadata: List of LeRobotDatasetMetadata objects to validate.
+
+    Returns:
+        tuple: A tuple containing (fps, robot_type, features) from the first metadata.
+
+    Raises:
+        ValueError: If any metadata has different fps, robot_type, or features
+                   than the first metadata in the list.
+    """
+
+    fps = all_metadata[0].fps
+    robot_type = all_metadata[0].robot_type
+    features = all_metadata[0].features
+
+    for meta in tqdm.tqdm(all_metadata, desc="Validate all meta data"):
+        if fps != meta.fps:
+            raise ValueError(f"Same fps is expected, but got fps={meta.fps} instead of {fps}.")
+        if robot_type != meta.robot_type:
+            raise ValueError(
+                f"Same robot_type is expected, but got robot_type={meta.robot_type} instead of {robot_type}."
+            )
+        if features != meta.features:
+            raise ValueError(
+                f"Same features is expected, but got features={meta.features} instead of {features}."
+            )
+
+    return fps, robot_type, features
+
+
+def update_data_df(df, src_meta, dst_meta):
+    """Updates a data DataFrame with new indices and task mappings for aggregation.
+
+    Adjusts episode indices, frame indices, and task indices to account for
+    previously aggregated data in the destination dataset.
+
+    Args:
+        df: DataFrame containing the data to be updated.
+        src_meta: Source dataset metadata.
+        dst_meta: Destination dataset metadata.
+
+    Returns:
+        pd.DataFrame: Updated DataFrame with adjusted indices.
+    """
+
+    df["episode_index"] = df["episode_index"] + dst_meta.info["total_episodes"]
+    df["index"] = df["index"] + dst_meta.info["total_frames"]
+
+    src_task_names = src_meta.tasks.index.take(df["task_index"].to_numpy())
+    df["task_index"] = dst_meta.tasks.loc[src_task_names, "task_index"].to_numpy()
+
+    return df
+
+
+def update_meta_data(
+    df,
+    dst_meta,
+    meta_idx,
+    data_idx,
+    videos_idx,
+):
+    """Updates metadata DataFrame with new chunk, file, and timestamp indices.
+
+    Adjusts all indices and timestamps to account for previously aggregated
+    data and videos in the destination dataset.
+
+    For data file indices, uses the 'src_to_dst' mapping from aggregate_data()
+    to correctly map source file indices to their destination locations.
+
+    Args:
+        df: DataFrame containing the metadata to be updated.
+        dst_meta: Destination dataset metadata.
+        meta_idx: Dictionary containing current metadata chunk and file indices.
+        data_idx: Dictionary containing current data chunk and file indices.
+        videos_idx: Dictionary containing current video indices and timestamps.
+
+    Returns:
+        pd.DataFrame: Updated DataFrame with adjusted indices and timestamps.
+    """
+
+    df["meta/episodes/chunk_index"] = df["meta/episodes/chunk_index"] + meta_idx["chunk"]
+    df["meta/episodes/file_index"] = df["meta/episodes/file_index"] + meta_idx["file"]
+
+    # Update data file indices using source-to-destination mapping
+    # This is critical for handling datasets that are already results of a merge
+    data_src_to_dst = data_idx.get("src_to_dst", {})
+    if data_src_to_dst:
+        # Store original indices for lookup
+        df["_orig_data_chunk"] = df["data/chunk_index"].copy()
+        df["_orig_data_file"] = df["data/file_index"].copy()
+
+        # Vectorized mapping from (src_chunk, src_file) to (dst_chunk, dst_file)
+        # This is much faster than per-row iteration for large metadata tables
+        mapping_index = pd.MultiIndex.from_tuples(
+            list(data_src_to_dst.keys()),
+            names=["chunk_index", "file_index"],
+        )
+        mapping_values = list(data_src_to_dst.values())
+        mapping_df = pd.DataFrame(
+            mapping_values,
+            index=mapping_index,
+            columns=["dst_chunk", "dst_file"],
+        )
+
+        # Construct a MultiIndex for each row based on original data indices
+        row_index = pd.MultiIndex.from_arrays(
+            [df["_orig_data_chunk"], df["_orig_data_file"]],
+            names=["chunk_index", "file_index"],
+        )
+
+        # Align mapping to rows; missing keys fall back to the default destination
+        reindexed = mapping_df.reindex(row_index)
+        reindexed[["dst_chunk", "dst_file"]] = reindexed[["dst_chunk", "dst_file"]].fillna(
+            {"dst_chunk": data_idx["chunk"], "dst_file": data_idx["file"]}
+        )
+
+        # Assign mapped destination indices back to the DataFrame
+        df["data/chunk_index"] = reindexed["dst_chunk"].to_numpy()
+        df["data/file_index"] = reindexed["dst_file"].to_numpy()
+
+        # Clean up temporary columns
+        df = df.drop(columns=["_orig_data_chunk", "_orig_data_file"])
+    else:
+        # Fallback to simple offset (backward compatibility for single-file sources)
+        df["data/chunk_index"] = df["data/chunk_index"] + data_idx["chunk"]
+        df["data/file_index"] = df["data/file_index"] + data_idx["file"]
+    for key, video_idx in videos_idx.items():
+        # Store original video file indices before updating
+        orig_chunk_col = f"videos/{key}/chunk_index"
+        orig_file_col = f"videos/{key}/file_index"
+        df["_orig_chunk"] = df[orig_chunk_col].copy()
+        df["_orig_file"] = df[orig_file_col].copy()
+
+        # Get mappings for this video key
+        src_to_offset = video_idx.get("src_to_offset", {})
+        src_to_dst = video_idx.get("src_to_dst", {})
+
+        # Apply per-source-file mappings
+        if src_to_dst:
+            # Map each episode to its correct destination file and apply offset
+            for idx in df.index:
+                src_key = (df.at[idx, "_orig_chunk"], df.at[idx, "_orig_file"])
+
+                # Get destination chunk/file for this source file
+                dst_chunk, dst_file = src_to_dst.get(src_key, (video_idx["chunk"], video_idx["file"]))
+                df.at[idx, orig_chunk_col] = dst_chunk
+                df.at[idx, orig_file_col] = dst_file
+
+                # Apply timestamp offset
+                offset = src_to_offset.get(src_key, 0)
+                df.at[idx, f"videos/{key}/from_timestamp"] += offset
+                df.at[idx, f"videos/{key}/to_timestamp"] += offset
+        elif src_to_offset:
+            # Fallback: use same destination for all, but apply per-file offsets
+            df[orig_chunk_col] = video_idx["chunk"]
+            df[orig_file_col] = video_idx["file"]
+            for idx in df.index:
+                src_key = (df.at[idx, "_orig_chunk"], df.at[idx, "_orig_file"])
+                offset = src_to_offset.get(src_key, 0)
+                df.at[idx, f"videos/{key}/from_timestamp"] += offset
+                df.at[idx, f"videos/{key}/to_timestamp"] += offset
+        else:
+            # Fallback to simple offset (for backward compatibility)
+            df[orig_chunk_col] = video_idx["chunk"]
+            df[orig_file_col] = video_idx["file"]
+            df[f"videos/{key}/from_timestamp"] = (
+                df[f"videos/{key}/from_timestamp"] + video_idx["latest_duration"]
+            )
+            df[f"videos/{key}/to_timestamp"] = df[f"videos/{key}/to_timestamp"] + video_idx["latest_duration"]
+
+        # Clean up temporary columns
+        df = df.drop(columns=["_orig_chunk", "_orig_file"])
+
+    df["dataset_from_index"] = df["dataset_from_index"] + dst_meta.info["total_frames"]
+    df["dataset_to_index"] = df["dataset_to_index"] + dst_meta.info["total_frames"]
+    df["episode_index"] = df["episode_index"] + dst_meta.info["total_episodes"]
+
+    return df
+
+
+def aggregate_datasets(
+    repo_ids: list[str],
+    aggr_repo_id: str,
+    roots: list[Path] | None = None,
+    aggr_root: Path | None = None,
+    data_files_size_in_mb: float | None = None,
+    video_files_size_in_mb: float | None = None,
+    chunk_size: int | None = None,
+):
+    """Aggregates multiple LeRobot datasets into a single unified dataset.
+
+    This is the main function that orchestrates the aggregation process by:
+    1. Loading and validating all source dataset metadata
+    2. Creating a new destination dataset with unified tasks
+    3. Aggregating videos, data, and metadata from all source datasets
+    4. Finalizing the aggregated dataset with proper statistics
+
+    Args:
+        repo_ids: List of repository IDs for the datasets to aggregate.
+        aggr_repo_id: Repository ID for the aggregated output dataset.
+        roots: Optional list of root paths for the source datasets.
+        aggr_root: Optional root path for the aggregated dataset.
+        data_files_size_in_mb: Maximum size for data files in MB (defaults to DEFAULT_DATA_FILE_SIZE_IN_MB)
+        video_files_size_in_mb: Maximum size for video files in MB (defaults to DEFAULT_VIDEO_FILE_SIZE_IN_MB)
+        chunk_size: Maximum number of files per chunk (defaults to DEFAULT_CHUNK_SIZE)
+    """
+    logging.info("Start aggregate_datasets")
+
+    if data_files_size_in_mb is None:
+        data_files_size_in_mb = DEFAULT_DATA_FILE_SIZE_IN_MB
+    if video_files_size_in_mb is None:
+        video_files_size_in_mb = DEFAULT_VIDEO_FILE_SIZE_IN_MB
+    if chunk_size is None:
+        chunk_size = DEFAULT_CHUNK_SIZE
+
+    all_metadata = (
+        [LeRobotDatasetMetadata(repo_id) for repo_id in repo_ids]
+        if roots is None
+        else [
+            LeRobotDatasetMetadata(repo_id, root=root) for repo_id, root in zip(repo_ids, roots, strict=False)
+        ]
+    )
+    fps, robot_type, features = validate_all_metadata(all_metadata)
+    video_keys = [key for key in features if features[key]["dtype"] == "video"]
+
+    dst_meta = LeRobotDatasetMetadata.create(
+        repo_id=aggr_repo_id,
+        fps=fps,
+        robot_type=robot_type,
+        features=features,
+        root=aggr_root,
+        use_videos=len(video_keys) > 0,
+        chunks_size=chunk_size,
+        data_files_size_in_mb=data_files_size_in_mb,
+        video_files_size_in_mb=video_files_size_in_mb,
+    )
+
+    logging.info("Find all tasks")
+    unique_tasks = pd.concat([m.tasks for m in all_metadata]).index.unique()
+    dst_meta.tasks = pd.DataFrame(
+        {"task_index": range(len(unique_tasks))}, index=pd.Index(unique_tasks, name="task")
+    )
+
+    meta_idx = {"chunk": 0, "file": 0}
+    data_idx = {"chunk": 0, "file": 0}
+    videos_idx = {
+        key: {"chunk": 0, "file": 0, "latest_duration": 0, "episode_duration": 0} for key in video_keys
+    }
+
+    dst_meta.episodes = {}
+
+    for src_meta in tqdm.tqdm(all_metadata, desc="Copy data and videos"):
+        videos_idx = aggregate_videos(src_meta, dst_meta, videos_idx, video_files_size_in_mb, chunk_size)
+        data_idx = aggregate_data(src_meta, dst_meta, data_idx, data_files_size_in_mb, chunk_size)
+
+        meta_idx = aggregate_metadata(src_meta, dst_meta, meta_idx, data_idx, videos_idx)
+
+        # Clear the src_to_dst mapping after processing each source dataset
+        # to avoid interference between different source datasets
+        data_idx.pop("src_to_dst", None)
+
+        dst_meta.info["total_episodes"] += src_meta.total_episodes
+        dst_meta.info["total_frames"] += src_meta.total_frames
+
+    finalize_aggregation(dst_meta, all_metadata)
+    logging.info("Aggregation complete.")
+
+
+def aggregate_videos(src_meta, dst_meta, videos_idx, video_files_size_in_mb, chunk_size):
+    """Aggregates video chunks from a source dataset into the destination dataset.
+
+    Handles video file concatenation and rotation based on file size limits.
+    Creates new video files when size limits are exceeded.
+
+    Args:
+        src_meta: Source dataset metadata.
+        dst_meta: Destination dataset metadata.
+        videos_idx: Dictionary tracking video chunk and file indices.
+        video_files_size_in_mb: Maximum size for video files in MB (defaults to DEFAULT_VIDEO_FILE_SIZE_IN_MB)
+        chunk_size: Maximum number of files per chunk (defaults to DEFAULT_CHUNK_SIZE)
+
+    Returns:
+        dict: Updated videos_idx with current chunk and file indices.
+    """
+    for key in videos_idx:
+        videos_idx[key]["episode_duration"] = 0
+        # Track offset for each source (chunk, file) pair
+        videos_idx[key]["src_to_offset"] = {}
+        # Track destination (chunk, file) for each source (chunk, file) pair
+        videos_idx[key]["src_to_dst"] = {}
+        # Initialize dst_file_durations if not present
+        # dst_file_durations tracks duration of each destination file
+        if "dst_file_durations" not in videos_idx[key]:
+            videos_idx[key]["dst_file_durations"] = {}
+
+    for key, video_idx in videos_idx.items():
+        unique_chunk_file_pairs = {
+            (chunk, file)
+            for chunk, file in zip(
+                src_meta.episodes[f"videos/{key}/chunk_index"],
+                src_meta.episodes[f"videos/{key}/file_index"],
+                strict=False,
+            )
+        }
+        unique_chunk_file_pairs = sorted(unique_chunk_file_pairs)
+
+        chunk_idx = video_idx["chunk"]
+        file_idx = video_idx["file"]
+        dst_file_durations = video_idx["dst_file_durations"]
+
+        for src_chunk_idx, src_file_idx in unique_chunk_file_pairs:
+            src_path = src_meta.root / DEFAULT_VIDEO_PATH.format(
+                video_key=key,
+                chunk_index=src_chunk_idx,
+                file_index=src_file_idx,
+            )
+
+            dst_path = dst_meta.root / DEFAULT_VIDEO_PATH.format(
+                video_key=key,
+                chunk_index=chunk_idx,
+                file_index=file_idx,
+            )
+
+            src_duration = get_video_duration_in_s(src_path)
+            dst_key = (chunk_idx, file_idx)
+
+            if not dst_path.exists():
+                # New destination file: offset is 0
+                videos_idx[key]["src_to_offset"][(src_chunk_idx, src_file_idx)] = 0
+                videos_idx[key]["src_to_dst"][(src_chunk_idx, src_file_idx)] = dst_key
+                dst_path.parent.mkdir(parents=True, exist_ok=True)
+                shutil.copy(str(src_path), str(dst_path))
+                # Track duration of this destination file
+                dst_file_durations[dst_key] = src_duration
+                videos_idx[key]["episode_duration"] += src_duration
+                continue
+
+            # Check file sizes before appending
+            src_size = get_file_size_in_mb(src_path)
+            dst_size = get_file_size_in_mb(dst_path)
+
+            if dst_size + src_size >= video_files_size_in_mb:
+                # Rotate to a new file - offset is 0
+                chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, chunk_size)
+                dst_key = (chunk_idx, file_idx)
+                videos_idx[key]["src_to_offset"][(src_chunk_idx, src_file_idx)] = 0
+                videos_idx[key]["src_to_dst"][(src_chunk_idx, src_file_idx)] = dst_key
+                dst_path = dst_meta.root / DEFAULT_VIDEO_PATH.format(
+                    video_key=key,
+                    chunk_index=chunk_idx,
+                    file_index=file_idx,
+                )
+                dst_path.parent.mkdir(parents=True, exist_ok=True)
+                shutil.copy(str(src_path), str(dst_path))
+                # Track duration of this new destination file
+                dst_file_durations[dst_key] = src_duration
+            else:
+                # Append to existing destination file
+                # Offset is the current duration of this destination file
+                current_dst_duration = dst_file_durations.get(dst_key, 0)
+                videos_idx[key]["src_to_offset"][(src_chunk_idx, src_file_idx)] = current_dst_duration
+                videos_idx[key]["src_to_dst"][(src_chunk_idx, src_file_idx)] = dst_key
+                concatenate_video_files(
+                    [dst_path, src_path],
+                    dst_path,
+                )
+                # Update duration of this destination file
+                dst_file_durations[dst_key] = current_dst_duration + src_duration
+
+            videos_idx[key]["episode_duration"] += src_duration
+
+        videos_idx[key]["chunk"] = chunk_idx
+        videos_idx[key]["file"] = file_idx
+
+    return videos_idx
+
+
+def aggregate_data(src_meta, dst_meta, data_idx, data_files_size_in_mb, chunk_size):
+    """Aggregates data chunks from a source dataset into the destination dataset.
+
+    Reads source data files, updates indices to match the aggregated dataset,
+    and writes them to the destination with proper file rotation.
+
+    Tracks a `src_to_dst` mapping from source (chunk, file) to destination (chunk, file)
+    which is critical for correctly updating episode metadata when source datasets
+    have multiple data files (e.g., from a previous merge operation).
+
+    Args:
+        src_meta: Source dataset metadata.
+        dst_meta: Destination dataset metadata.
+        data_idx: Dictionary tracking data chunk and file indices.
+        data_files_size_in_mb: Maximum size for data files in MB.
+        chunk_size: Maximum number of files per chunk.
+
+    Returns:
+        dict: Updated data_idx with current chunk and file indices.
+    """
+    unique_chunk_file_ids = {
+        (c, f)
+        for c, f in zip(
+            src_meta.episodes["data/chunk_index"], src_meta.episodes["data/file_index"], strict=False
+        )
+    }
+
+    unique_chunk_file_ids = sorted(unique_chunk_file_ids)
+    contains_images = len(dst_meta.image_keys) > 0
+
+    # retrieve features schema for proper image typing in parquet
+    hf_features = get_hf_features_from_features(dst_meta.features) if contains_images else None
+
+    # Track source to destination file mapping for metadata update
+    # This is critical for handling datasets that are already results of a merge
+    src_to_dst: dict[tuple[int, int], tuple[int, int]] = {}
+
+    for src_chunk_idx, src_file_idx in unique_chunk_file_ids:
+        src_path = src_meta.root / DEFAULT_DATA_PATH.format(
+            chunk_index=src_chunk_idx, file_index=src_file_idx
+        )
+        if contains_images:
+            # Use HuggingFace datasets to read source data to preserve image format
+            src_ds = datasets.Dataset.from_parquet(str(src_path))
+            df = src_ds.to_pandas()
+        else:
+            df = pd.read_parquet(src_path)
+        df = update_data_df(df, src_meta, dst_meta)
+
+        # Write data and get the actual destination file it was written to
+        # This avoids duplicating the rotation logic here
+        data_idx, (dst_chunk, dst_file) = append_or_create_parquet_file(
+            df,
+            src_path,
+            data_idx,
+            data_files_size_in_mb,
+            chunk_size,
+            DEFAULT_DATA_PATH,
+            contains_images=contains_images,
+            aggr_root=dst_meta.root,
+            hf_features=hf_features,
+        )
+
+        # Record the mapping from source to actual destination
+        src_to_dst[(src_chunk_idx, src_file_idx)] = (dst_chunk, dst_file)
+
+    # Add the mapping to data_idx for use in metadata update
+    data_idx["src_to_dst"] = src_to_dst
+
+    return data_idx
+
+
+def aggregate_metadata(src_meta, dst_meta, meta_idx, data_idx, videos_idx):
+    """Aggregates metadata from a source dataset into the destination dataset.
+
+    Reads source metadata files, updates all indices and timestamps,
+    and writes them to the destination with proper file rotation.
+
+    Args:
+        src_meta: Source dataset metadata.
+        dst_meta: Destination dataset metadata.
+        meta_idx: Dictionary tracking metadata chunk and file indices.
+        data_idx: Dictionary tracking data chunk and file indices.
+        videos_idx: Dictionary tracking video indices and timestamps.
+
+    Returns:
+        dict: Updated meta_idx with current chunk and file indices.
+    """
+    chunk_file_ids = {
+        (c, f)
+        for c, f in zip(
+            src_meta.episodes["meta/episodes/chunk_index"],
+            src_meta.episodes["meta/episodes/file_index"],
+            strict=False,
+        )
+    }
+
+    chunk_file_ids = sorted(chunk_file_ids)
+    for chunk_idx, file_idx in chunk_file_ids:
+        src_path = src_meta.root / DEFAULT_EPISODES_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
+        df = pd.read_parquet(src_path)
+        df = update_meta_data(
+            df,
+            dst_meta,
+            meta_idx,
+            data_idx,
+            videos_idx,
+        )
+
+        meta_idx, _ = append_or_create_parquet_file(
+            df,
+            src_path,
+            meta_idx,
+            DEFAULT_DATA_FILE_SIZE_IN_MB,
+            DEFAULT_CHUNK_SIZE,
+            DEFAULT_EPISODES_PATH,
+            contains_images=False,
+            aggr_root=dst_meta.root,
+        )
+
+    # Increment latest_duration by the total duration added from this source dataset
+    for k in videos_idx:
+        videos_idx[k]["latest_duration"] += videos_idx[k]["episode_duration"]
+
+    return meta_idx
+
+
+def append_or_create_parquet_file(
+    df: pd.DataFrame,
+    src_path: Path,
+    idx: dict[str, int],
+    max_mb: float,
+    chunk_size: int,
+    default_path: str,
+    contains_images: bool = False,
+    aggr_root: Path = None,
+    hf_features: datasets.Features | None = None,
+) -> tuple[dict[str, int], tuple[int, int]]:
+    """Appends data to an existing parquet file or creates a new one based on size constraints.
+
+    Manages file rotation when size limits are exceeded to prevent individual files
+    from becoming too large. Handles both regular parquet files and those containing images.
+
+    Args:
+        df: DataFrame to write to the parquet file.
+        src_path: Path to the source file (used for size estimation).
+        idx: Dictionary containing current 'chunk' and 'file' indices.
+        max_mb: Maximum allowed file size in MB before rotation.
+        chunk_size: Maximum number of files per chunk before incrementing chunk index.
+        default_path: Format string for generating file paths.
+        contains_images: Whether the data contains images requiring special handling.
+        aggr_root: Root path for the aggregated dataset.
+        hf_features: Optional HuggingFace Features schema for proper image typing.
+
+    Returns:
+        tuple: (updated_idx, (dst_chunk, dst_file)) where updated_idx is the index dict
+               and (dst_chunk, dst_file) is the actual destination file the data was written to.
+    """
+    dst_chunk, dst_file = idx["chunk"], idx["file"]
+    dst_path = aggr_root / default_path.format(chunk_index=dst_chunk, file_index=dst_file)
+
+    if not dst_path.exists():
+        dst_path.parent.mkdir(parents=True, exist_ok=True)
+        if contains_images:
+            to_parquet_with_hf_images(df, dst_path, features=hf_features)
+        else:
+            df.to_parquet(dst_path)
+        return idx, (dst_chunk, dst_file)
+
+    src_size = get_parquet_file_size_in_mb(src_path)
+    dst_size = get_parquet_file_size_in_mb(dst_path)
+
+    if dst_size + src_size >= max_mb:
+        idx["chunk"], idx["file"] = update_chunk_file_indices(idx["chunk"], idx["file"], chunk_size)
+        dst_chunk, dst_file = idx["chunk"], idx["file"]
+        new_path = aggr_root / default_path.format(chunk_index=dst_chunk, file_index=dst_file)
+        new_path.parent.mkdir(parents=True, exist_ok=True)
+        final_df = df
+        target_path = new_path
+    else:
+        if contains_images:
+            # Use HuggingFace datasets to read existing data to preserve image format
+            existing_ds = datasets.Dataset.from_parquet(str(dst_path))
+            existing_df = existing_ds.to_pandas()
+        else:
+            existing_df = pd.read_parquet(dst_path)
+        final_df = pd.concat([existing_df, df], ignore_index=True)
+        target_path = dst_path
+
+    if contains_images:
+        to_parquet_with_hf_images(final_df, target_path, features=hf_features)
+    else:
+        final_df.to_parquet(target_path)
+
+    return idx, (dst_chunk, dst_file)
+
+
+def finalize_aggregation(aggr_meta, all_metadata):
+    """Finalizes the dataset aggregation by writing summary files and statistics.
+
+    Writes the tasks file, info file with total counts and splits, and
+    aggregated statistics from all source datasets.
+
+    Args:
+        aggr_meta: Aggregated dataset metadata.
+        all_metadata: List of all source dataset metadata objects.
+    """
+    logging.info("write tasks")
+    write_tasks(aggr_meta.tasks, aggr_meta.root)
+
+    logging.info("write info")
+    aggr_meta.info.update(
+        {
+            "total_tasks": len(aggr_meta.tasks),
+            "total_episodes": sum(m.total_episodes for m in all_metadata),
+            "total_frames": sum(m.total_frames for m in all_metadata),
+            "splits": {"train": f"0:{sum(m.total_episodes for m in all_metadata)}"},
+        }
+    )
+    write_info(aggr_meta.info, aggr_meta.root)
+
+    logging.info("write stats")
+    aggr_meta.stats = aggregate_stats([m.stats for m in all_metadata])
+    write_stats(aggr_meta.stats, aggr_meta.root)
diff --git a/lerobot/src/lerobot/datasets/card_template.md b/lerobot/src/lerobot/datasets/card_template.md
new file mode 100644
index 0000000000000000000000000000000000000000..1eced9f4c3dc0b8c39f3cacf1c5c804c183c20ac
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/card_template.md
@@ -0,0 +1,35 @@
+---
+# For reference on dataset card metadata, see the spec: https://github.com/huggingface/hub-docs/blob/main/datasetcard.md?plain=1
+# Doc / guide: https://huggingface.co/docs/hub/datasets-cards
+# prettier-ignore
+{{card_data}}
+---
+
+This dataset was created using [LeRobot](https://github.com/huggingface/lerobot).
+
+{% if repo_id is defined and repo_id %}
+<a class="flex" href="https://huggingface.co/spaces/lerobot/visualize_dataset?path={{ repo_id }}">
+<img class="block dark:hidden" src="https://huggingface.co/datasets/huggingface/badges/resolve/main/visualize-this-dataset-xl.svg"/>
+<img class="hidden dark:block" src="https://huggingface.co/datasets/huggingface/badges/resolve/main/visualize-this-dataset-xl-dark.svg"/>
+</a>
+{% endif %}
+
+## Dataset Description
+
+{{ dataset_description | default("", true) }}
+
+- **Homepage:** {{ url | default("[More Information Needed]", true)}}
+- **Paper:** {{ paper | default("[More Information Needed]", true)}}
+- **License:** {{ license | default("[More Information Needed]", true)}}
+
+## Dataset Structure
+
+{{ dataset_structure | default("[More Information Needed]", true)}}
+
+## Citation
+
+**BibTeX:**
+
+```bibtex
+{{ citation_bibtex | default("[More Information Needed]", true)}}
+```
diff --git a/lerobot/src/lerobot/datasets/compute_stats.py b/lerobot/src/lerobot/datasets/compute_stats.py
new file mode 100644
index 0000000000000000000000000000000000000000..5bd95810b295095752d8b672a95c80ffe67d7fa3
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/compute_stats.py
@@ -0,0 +1,626 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import numpy as np
+
+from lerobot.datasets.io_utils import load_image_as_numpy
+
+DEFAULT_QUANTILES = [0.01, 0.10, 0.50, 0.90, 0.99]
+
+
+class RunningQuantileStats:
+    """
+    Maintains running statistics for batches of vectors, including mean,
+    standard deviation, min, max, and approximate quantiles.
+
+    Statistics are computed per feature dimension and updated incrementally
+    as new batches are observed. Quantiles are estimated using histograms,
+    which adapt dynamically if the observed data range expands.
+    """
+
+    def __init__(self, quantile_list: list[float] | None = None, num_quantile_bins: int = 5000):
+        self._count = 0
+        self._mean = None
+        self._mean_of_squares = None
+        self._min = None
+        self._max = None
+        self._histograms = None
+        self._bin_edges = None
+        self._num_quantile_bins = num_quantile_bins
+
+        self._quantile_list = quantile_list
+        if self._quantile_list is None:
+            self._quantile_list = DEFAULT_QUANTILES
+        self._quantile_keys = [f"q{int(q * 100):02d}" for q in self._quantile_list]
+
+    def update(self, batch: np.ndarray) -> None:
+        """Update the running statistics with a batch of vectors.
+
+        Args:
+            batch: An array where all dimensions except the last are batch dimensions.
+        """
+        batch = batch.reshape(-1, batch.shape[-1])
+        num_elements, vector_length = batch.shape
+
+        if self._count == 0:
+            self._mean = np.mean(batch, axis=0)
+            self._mean_of_squares = np.mean(batch**2, axis=0)
+            self._min = np.min(batch, axis=0)
+            self._max = np.max(batch, axis=0)
+            self._histograms = [np.zeros(self._num_quantile_bins) for _ in range(vector_length)]
+            self._bin_edges = [
+                np.linspace(self._min[i] - 1e-10, self._max[i] + 1e-10, self._num_quantile_bins + 1)
+                for i in range(vector_length)
+            ]
+        else:
+            if vector_length != self._mean.size:
+                raise ValueError("The length of new vectors does not match the initialized vector length.")
+
+            new_max = np.max(batch, axis=0)
+            new_min = np.min(batch, axis=0)
+            max_changed = np.any(new_max > self._max)
+            min_changed = np.any(new_min < self._min)
+            self._max = np.maximum(self._max, new_max)
+            self._min = np.minimum(self._min, new_min)
+
+            if max_changed or min_changed:
+                self._adjust_histograms()
+
+        self._count += num_elements
+
+        batch_mean = np.mean(batch, axis=0)
+        batch_mean_of_squares = np.mean(batch**2, axis=0)
+
+        # Update running mean and mean of squares
+        self._mean += (batch_mean - self._mean) * (num_elements / self._count)
+        self._mean_of_squares += (batch_mean_of_squares - self._mean_of_squares) * (
+            num_elements / self._count
+        )
+
+        self._update_histograms(batch)
+
+    def get_statistics(self) -> dict[str, np.ndarray]:
+        """Compute and return the statistics of the vectors processed so far.
+
+        Args:
+            quantiles: List of quantiles to compute (e.g., [0.01, 0.10, 0.50, 0.90, 0.99]). If None, no quantiles computed.
+
+        Returns:
+            Dictionary containing the computed statistics.
+        """
+        if self._count < 2:
+            raise ValueError("Cannot compute statistics for less than 2 vectors.")
+
+        variance = self._mean_of_squares - self._mean**2
+
+        stddev = np.sqrt(np.maximum(0, variance))
+
+        stats = {
+            "min": self._min.copy(),
+            "max": self._max.copy(),
+            "mean": self._mean.copy(),
+            "std": stddev,
+            "count": np.array([self._count]),
+        }
+
+        quantile_results = self._compute_quantiles()
+        for i, q in enumerate(self._quantile_keys):
+            stats[q] = quantile_results[i]
+
+        return stats
+
+    def _adjust_histograms(self):
+        """Adjust histograms when min or max changes."""
+        for i in range(len(self._histograms)):
+            old_edges = self._bin_edges[i]
+            old_hist = self._histograms[i]
+
+            # Create new edges with small padding to ensure range coverage
+            padding = (self._max[i] - self._min[i]) * 1e-10
+            new_edges = np.linspace(
+                self._min[i] - padding, self._max[i] + padding, self._num_quantile_bins + 1
+            )
+
+            # Redistribute existing histogram counts to new bins
+            # We need to map each old bin center to the new bins
+            old_centers = (old_edges[:-1] + old_edges[1:]) / 2
+            new_hist = np.zeros(self._num_quantile_bins)
+
+            for old_center, count in zip(old_centers, old_hist, strict=False):
+                if count > 0:
+                    # Find which new bin this old center belongs to
+                    bin_idx = np.searchsorted(new_edges, old_center) - 1
+                    bin_idx = max(0, min(bin_idx, self._num_quantile_bins - 1))
+                    new_hist[bin_idx] += count
+
+            self._histograms[i] = new_hist
+            self._bin_edges[i] = new_edges
+
+    def _update_histograms(self, batch: np.ndarray) -> None:
+        """Update histograms with new vectors."""
+        for i in range(batch.shape[1]):
+            hist, _ = np.histogram(batch[:, i], bins=self._bin_edges[i])
+            self._histograms[i] += hist
+
+    def _compute_quantiles(self) -> list[np.ndarray]:
+        """Compute quantiles based on histograms."""
+        results = []
+        for q in self._quantile_list:
+            target_count = q * self._count
+            q_values = []
+
+            for hist, edges in zip(self._histograms, self._bin_edges, strict=True):
+                q_value = self._compute_single_quantile(hist, edges, target_count)
+                q_values.append(q_value)
+
+            results.append(np.array(q_values))
+        return results
+
+    def _compute_single_quantile(self, hist: np.ndarray, edges: np.ndarray, target_count: float) -> float:
+        """Compute a single quantile value from histogram and bin edges."""
+        cumsum = np.cumsum(hist)
+        idx = np.searchsorted(cumsum, target_count)
+
+        if idx == 0:
+            return edges[0]
+        if idx >= len(cumsum):
+            return edges[-1]
+
+        # If not edge case, interpolate within the bin
+        count_before = cumsum[idx - 1]
+        count_in_bin = cumsum[idx] - count_before
+
+        # If no samples in this bin, use the bin edge
+        if count_in_bin == 0:
+            return edges[idx]
+
+        # Linear interpolation within the bin
+        fraction = (target_count - count_before) / count_in_bin
+        return edges[idx] + fraction * (edges[idx + 1] - edges[idx])
+
+
+def estimate_num_samples(
+    dataset_len: int, min_num_samples: int = 100, max_num_samples: int = 10_000, power: float = 0.75
+) -> int:
+    """Heuristic to estimate the number of samples based on dataset size.
+    The power controls the sample growth relative to dataset size.
+    Lower the power for less number of samples.
+
+    For default arguments, we have:
+    - from 1 to ~500, num_samples=100
+    - at 1000, num_samples=177
+    - at 2000, num_samples=299
+    - at 5000, num_samples=594
+    - at 10000, num_samples=1000
+    - at 20000, num_samples=1681
+    """
+    if dataset_len < min_num_samples:
+        min_num_samples = dataset_len
+    return max(min_num_samples, min(int(dataset_len**power), max_num_samples))
+
+
+def sample_indices(data_len: int) -> list[int]:
+    num_samples = estimate_num_samples(data_len)
+    return np.round(np.linspace(0, data_len - 1, num_samples)).astype(int).tolist()
+
+
+def auto_downsample_height_width(img: np.ndarray, target_size: int = 150, max_size_threshold: int = 300):
+    _, height, width = img.shape
+
+    if max(width, height) < max_size_threshold:
+        # no downsampling needed
+        return img
+
+    downsample_factor = int(width / target_size) if width > height else int(height / target_size)
+    return img[:, ::downsample_factor, ::downsample_factor]
+
+
+def sample_images(image_paths: list[str]) -> np.ndarray:
+    sampled_indices = sample_indices(len(image_paths))
+
+    images = None
+    for i, idx in enumerate(sampled_indices):
+        path = image_paths[idx]
+        # we load as uint8 to reduce memory usage
+        img = load_image_as_numpy(path, dtype=np.uint8, channel_first=True)
+        img = auto_downsample_height_width(img)
+
+        if images is None:
+            images = np.empty((len(sampled_indices), *img.shape), dtype=np.uint8)
+
+        images[i] = img
+
+    return images
+
+
+def _reshape_stats_by_axis(
+    stats: dict[str, np.ndarray],
+    axis: int | tuple[int, ...] | None,
+    keepdims: bool,
+    original_shape: tuple[int, ...],
+) -> dict[str, np.ndarray]:
+    """Reshape all statistics to match NumPy's output conventions.
+
+    Applies consistent reshaping to all statistics (except 'count') based on the
+    axis and keepdims parameters. This ensures statistics have the correct shape
+    for broadcasting with the original data.
+
+    Args:
+        stats: Dictionary of computed statistics
+        axis: Axis or axes along which statistics were computed
+        keepdims: Whether to keep reduced dimensions as size-1 dimensions
+        original_shape: Shape of the original array
+
+    Returns:
+        Dictionary with reshaped statistics
+
+    Note:
+        The 'count' statistic is never reshaped as it represents metadata
+        rather than per-feature statistics.
+    """
+    if axis == (1,) and not keepdims:
+        return stats
+
+    result = {}
+    for key, value in stats.items():
+        if key == "count":
+            result[key] = value
+        else:
+            result[key] = _reshape_single_stat(value, axis, keepdims, original_shape)
+
+    return result
+
+
+def _reshape_for_image_stats(value: np.ndarray, keepdims: bool) -> np.ndarray:
+    """Reshape statistics for image data (axis=(0,2,3))."""
+    if keepdims and value.ndim == 1:
+        return value.reshape(1, -1, 1, 1)
+    return value
+
+
+def _reshape_for_vector_stats(
+    value: np.ndarray, keepdims: bool, original_shape: tuple[int, ...]
+) -> np.ndarray:
+    """Reshape statistics for vector data (axis=0 or axis=(0,))."""
+    if not keepdims:
+        return value
+
+    if len(original_shape) == 1 and value.ndim > 0:
+        return value.reshape(1)
+    elif len(original_shape) >= 2 and value.ndim == 1:
+        return value.reshape(1, -1)
+    return value
+
+
+def _reshape_for_feature_stats(value: np.ndarray, keepdims: bool) -> np.ndarray:
+    """Reshape statistics for feature-wise computation (axis=(1,))."""
+    if not keepdims:
+        return value
+
+    if value.ndim == 0:
+        return value.reshape(1, 1)
+    elif value.ndim == 1:
+        return value.reshape(-1, 1)
+    return value
+
+
+def _reshape_for_global_stats(
+    value: np.ndarray, keepdims: bool, original_shape: tuple[int, ...]
+) -> np.ndarray | float:
+    """Reshape statistics for global reduction (axis=None)."""
+    if keepdims:
+        target_shape = tuple(1 for _ in original_shape)
+        return value.reshape(target_shape)
+    # Keep at least 1-D arrays to satisfy validator
+    return np.atleast_1d(value)
+
+
+def _reshape_single_stat(
+    value: np.ndarray, axis: int | tuple[int, ...] | None, keepdims: bool, original_shape: tuple[int, ...]
+) -> np.ndarray | float:
+    """Apply appropriate reshaping to a single statistic array.
+
+    This function transforms statistic arrays to match expected output shapes
+    based on the axis configuration and keepdims parameter.
+
+    Args:
+        value: The statistic array to reshape
+        axis: Axis or axes that were reduced during computation
+        keepdims: Whether to maintain reduced dimensions as size-1 dimensions
+        original_shape: Shape of the original data before reduction
+
+    Returns:
+        Reshaped array following NumPy broadcasting conventions
+
+    """
+    if axis == (0, 2, 3):
+        return _reshape_for_image_stats(value, keepdims)
+
+    if axis in [0, (0,)]:
+        return _reshape_for_vector_stats(value, keepdims, original_shape)
+
+    if axis == (1,):
+        return _reshape_for_feature_stats(value, keepdims)
+
+    if axis is None:
+        return _reshape_for_global_stats(value, keepdims, original_shape)
+
+    return value
+
+
+def _prepare_array_for_stats(array: np.ndarray, axis: int | tuple[int, ...] | None) -> tuple[np.ndarray, int]:
+    """Prepare array for statistics computation by reshaping according to axis.
+
+    Args:
+        array: Input data array
+        axis: Axis or axes along which to compute statistics
+
+    Returns:
+        Tuple of (reshaped_array, sample_count)
+    """
+    if axis == (0, 2, 3):  # Image data
+        batch_size, channels, height, width = array.shape
+        reshaped = array.transpose(0, 2, 3, 1).reshape(-1, channels)
+        return reshaped, batch_size
+
+    if axis == 0 or axis == (0,):  # Vector data
+        reshaped = array
+        if array.ndim == 1:
+            reshaped = array.reshape(-1, 1)
+        return reshaped, array.shape[0]
+
+    if axis == (1,):  # Feature-wise statistics
+        return array.T, array.shape[1]
+
+    if axis is None:  # Global statistics
+        reshaped = array.reshape(-1, 1)
+        # For backward compatibility, count represents the first dimension size
+        return reshaped, array.shape[0] if array.ndim > 0 else 1
+
+    raise ValueError(f"Unsupported axis configuration: {axis}")
+
+
+def _compute_basic_stats(
+    array: np.ndarray, sample_count: int, quantile_list: list[float] | None = None
+) -> dict[str, np.ndarray]:
+    """Compute basic statistics for arrays with insufficient samples for quantiles.
+
+    Args:
+        array: Reshaped array ready for statistics computation
+        sample_count: Number of samples represented in the data
+
+    Returns:
+        Dictionary with basic statistics and quantiles set to mean values
+    """
+    if quantile_list is None:
+        quantile_list = DEFAULT_QUANTILES
+    quantile_list_keys = [f"q{int(q * 100):02d}" for q in quantile_list]
+
+    stats = {
+        "min": np.min(array, axis=0),
+        "max": np.max(array, axis=0),
+        "mean": np.mean(array, axis=0),
+        "std": np.std(array, axis=0),
+        "count": np.array([sample_count]),
+    }
+
+    for q in quantile_list_keys:
+        stats[q] = stats["mean"].copy()
+
+    return stats
+
+
+def get_feature_stats(
+    array: np.ndarray,
+    axis: int | tuple[int, ...] | None,
+    keepdims: bool,
+    quantile_list: list[float] | None = None,
+) -> dict[str, np.ndarray]:
+    """Compute comprehensive statistics for array features along specified axes.
+
+    This function calculates min, max, mean, std, and quantiles (1%, 10%, 50%, 90%, 99%)
+    for the input array along the specified axes. It handles different data layouts:
+    - Image data: axis=(0,2,3) computes per-channel statistics
+    - Vector data: axis=0 computes per-feature statistics
+    - Feature-wise: axis=1 computes statistics across features
+    - Global: axis=None computes statistics over entire array
+
+    Args:
+        array: Input data array with shape appropriate for the specified axis
+        axis: Axis or axes along which to compute statistics
+            - (0, 2, 3): For image data (batch, channels, height, width)
+            - 0 or (0,): For vector/tabular data (samples, features)
+            - (1,): For computing across features
+            - None: For global statistics over entire array
+        keepdims: If True, reduced axes are kept as dimensions with size 1
+
+    Returns:
+        Dictionary containing:
+            - 'min': Minimum values
+            - 'max': Maximum values
+            - 'mean': Mean values
+            - 'std': Standard deviation
+            - 'count': Number of samples (always shape (1,))
+            - 'q01', 'q10', 'q50', 'q90', 'q99': Quantile values
+
+    """
+    if quantile_list is None:
+        quantile_list = DEFAULT_QUANTILES
+
+    original_shape = array.shape
+    reshaped, sample_count = _prepare_array_for_stats(array, axis)
+
+    if reshaped.shape[0] < 2:
+        stats = _compute_basic_stats(reshaped, sample_count, quantile_list)
+    else:
+        running_stats = RunningQuantileStats()
+        running_stats.update(reshaped)
+        stats = running_stats.get_statistics()
+        stats["count"] = np.array([sample_count])
+
+    stats = _reshape_stats_by_axis(stats, axis, keepdims, original_shape)
+    return stats
+
+
+def compute_episode_stats(
+    episode_data: dict[str, list[str] | np.ndarray],
+    features: dict,
+    quantile_list: list[float] | None = None,
+) -> dict:
+    """Compute comprehensive statistics for all features in an episode.
+
+    Processes different data types appropriately:
+    - Images/videos: Samples from paths, computes per-channel stats, normalizes to [0,1]
+    - Numerical arrays: Computes per-feature statistics
+    - Strings: Skipped (no statistics computed)
+
+    Args:
+        episode_data: Dictionary mapping feature names to data
+            - For images/videos: list of file paths
+            - For numerical data: numpy arrays
+        features: Dictionary describing each feature's dtype and shape
+
+    Returns:
+        Dictionary mapping feature names to their statistics dictionaries.
+        Each statistics dictionary contains min, max, mean, std, count, and quantiles.
+
+    Note:
+        Image statistics are normalized to [0,1] range and have shape (3,1,1) for
+        per-channel values when dtype is 'image' or 'video'.
+    """
+    if quantile_list is None:
+        quantile_list = DEFAULT_QUANTILES
+
+    ep_stats = {}
+    for key, data in episode_data.items():
+        if features[key]["dtype"] == "string":
+            continue
+
+        if features[key]["dtype"] in ["image", "video"]:
+            ep_ft_array = sample_images(data)
+            axes_to_reduce = (0, 2, 3)
+            keepdims = True
+        else:
+            ep_ft_array = data
+            axes_to_reduce = 0
+            keepdims = data.ndim == 1
+
+        ep_stats[key] = get_feature_stats(
+            ep_ft_array, axis=axes_to_reduce, keepdims=keepdims, quantile_list=quantile_list
+        )
+
+        if features[key]["dtype"] in ["image", "video"]:
+            ep_stats[key] = {
+                k: v if k == "count" else np.squeeze(v / 255.0, axis=0) for k, v in ep_stats[key].items()
+            }
+
+    return ep_stats
+
+
+def _validate_stat_value(value: np.ndarray, key: str, feature_key: str) -> None:
+    """Validate a single statistic value."""
+    if not isinstance(value, np.ndarray):
+        raise ValueError(
+            f"Stats must be composed of numpy array, but key '{key}' of feature '{feature_key}' "
+            f"is of type '{type(value)}' instead."
+        )
+
+    if value.ndim == 0:
+        raise ValueError("Number of dimensions must be at least 1, and is 0 instead.")
+
+    if key == "count" and value.shape != (1,):
+        raise ValueError(f"Shape of 'count' must be (1), but is {value.shape} instead.")
+
+    if "image" in feature_key and key != "count" and value.shape != (3, 1, 1):
+        raise ValueError(f"Shape of quantile '{key}' must be (3,1,1), but is {value.shape} instead.")
+
+
+def _assert_type_and_shape(stats_list: list[dict[str, dict]]):
+    """Validate that all statistics have correct types and shapes.
+
+    Args:
+        stats_list: List of statistics dictionaries to validate
+
+    Raises:
+        ValueError: If any statistic has incorrect type or shape
+    """
+    for stats in stats_list:
+        for feature_key, feature_stats in stats.items():
+            for stat_key, stat_value in feature_stats.items():
+                _validate_stat_value(stat_value, stat_key, feature_key)
+
+
+def aggregate_feature_stats(stats_ft_list: list[dict[str, dict]]) -> dict[str, dict[str, np.ndarray]]:
+    """Aggregates stats for a single feature."""
+    means = np.stack([s["mean"] for s in stats_ft_list])
+    variances = np.stack([s["std"] ** 2 for s in stats_ft_list])
+    counts = np.stack([s["count"] for s in stats_ft_list])
+    total_count = counts.sum(axis=0)
+
+    # Prepare weighted mean by matching number of dimensions
+    while counts.ndim < means.ndim:
+        counts = np.expand_dims(counts, axis=-1)
+
+    # Compute the weighted mean
+    weighted_means = means * counts
+    total_mean = weighted_means.sum(axis=0) / total_count
+
+    # Compute the variance using the parallel algorithm
+    delta_means = means - total_mean
+    weighted_variances = (variances + delta_means**2) * counts
+    total_variance = weighted_variances.sum(axis=0) / total_count
+
+    aggregated = {
+        "min": np.min(np.stack([s["min"] for s in stats_ft_list]), axis=0),
+        "max": np.max(np.stack([s["max"] for s in stats_ft_list]), axis=0),
+        "mean": total_mean,
+        "std": np.sqrt(total_variance),
+        "count": total_count,
+    }
+
+    if stats_ft_list:
+        quantile_keys = [k for k in stats_ft_list[0] if k.startswith("q") and k[1:].isdigit()]
+
+        for q_key in quantile_keys:
+            if all(q_key in s for s in stats_ft_list):
+                quantile_values = np.stack([s[q_key] for s in stats_ft_list])
+                weighted_quantiles = quantile_values * counts
+                aggregated[q_key] = weighted_quantiles.sum(axis=0) / total_count
+
+    return aggregated
+
+
+def aggregate_stats(stats_list: list[dict[str, dict]]) -> dict[str, dict[str, np.ndarray]]:
+    """Aggregate stats from multiple compute_stats outputs into a single set of stats.
+
+    The final stats will have the union of all data keys from each of the stats dicts.
+
+    For instance:
+    - new_min = min(min_dataset_0, min_dataset_1, ...)
+    - new_max = max(max_dataset_0, max_dataset_1, ...)
+    - new_mean = (mean of all data, weighted by counts)
+    - new_std = (std of all data)
+    """
+
+    _assert_type_and_shape(stats_list)
+
+    data_keys = {key for stats in stats_list for key in stats}
+    aggregated_stats = {key: {} for key in data_keys}
+
+    for key in data_keys:
+        stats_with_key = [stats[key] for stats in stats_list if key in stats]
+        aggregated_stats[key] = aggregate_feature_stats(stats_with_key)
+
+    return aggregated_stats
diff --git a/lerobot/src/lerobot/datasets/dataset_metadata.py b/lerobot/src/lerobot/datasets/dataset_metadata.py
new file mode 100644
index 0000000000000000000000000000000000000000..560a90a6ee926649d9e3a23cb8b1534bbef2a075
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/dataset_metadata.py
@@ -0,0 +1,517 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from pathlib import Path
+
+import numpy as np
+import packaging.version
+import pandas as pd
+import pyarrow as pa
+import pyarrow.parquet as pq
+from huggingface_hub import snapshot_download
+
+from lerobot.datasets.compute_stats import aggregate_stats
+from lerobot.datasets.feature_utils import _validate_feature_names, create_empty_dataset_info
+from lerobot.datasets.io_utils import (
+    get_file_size_in_mb,
+    load_episodes,
+    load_info,
+    load_stats,
+    load_subtasks,
+    load_tasks,
+    write_info,
+    write_json,
+    write_stats,
+    write_tasks,
+)
+from lerobot.datasets.utils import (
+    DEFAULT_EPISODES_PATH,
+    DEFAULT_FEATURES,
+    INFO_PATH,
+    check_version_compatibility,
+    flatten_dict,
+    get_safe_version,
+    is_valid_version,
+    update_chunk_file_indices,
+)
+from lerobot.datasets.video_utils import get_video_info
+from lerobot.utils.constants import HF_LEROBOT_HOME
+
+CODEBASE_VERSION = "v3.0"
+
+
+class LeRobotDatasetMetadata:
+    def __init__(
+        self,
+        repo_id: str,
+        root: str | Path | None = None,
+        revision: str | None = None,
+        force_cache_sync: bool = False,
+        metadata_buffer_size: int = 10,
+    ):
+        self.repo_id = repo_id
+        self.revision = revision if revision else CODEBASE_VERSION
+        self.root = Path(root) if root is not None else HF_LEROBOT_HOME / repo_id
+        self.writer = None
+        self.latest_episode = None
+        self.metadata_buffer: list[dict] = []
+        self.metadata_buffer_size = metadata_buffer_size
+
+        try:
+            if force_cache_sync:
+                raise FileNotFoundError
+            self.load_metadata()
+        except (FileNotFoundError, NotADirectoryError):
+            if is_valid_version(self.revision):
+                self.revision = get_safe_version(self.repo_id, self.revision)
+
+            (self.root / "meta").mkdir(exist_ok=True, parents=True)
+            self.pull_from_repo(allow_patterns="meta/")
+            self.load_metadata()
+
+    def _flush_metadata_buffer(self) -> None:
+        """Write all buffered episode metadata to parquet file."""
+        if not hasattr(self, "metadata_buffer") or len(self.metadata_buffer) == 0:
+            return
+
+        combined_dict = {}
+        for episode_dict in self.metadata_buffer:
+            for key, value in episode_dict.items():
+                if key not in combined_dict:
+                    combined_dict[key] = []
+                # Extract value and serialize numpy arrays
+                # because PyArrow's from_pydict function doesn't support numpy arrays
+                val = value[0] if isinstance(value, list) else value
+                combined_dict[key].append(val.tolist() if isinstance(val, np.ndarray) else val)
+
+        first_ep = self.metadata_buffer[0]
+        chunk_idx = first_ep["meta/episodes/chunk_index"][0]
+        file_idx = first_ep["meta/episodes/file_index"][0]
+
+        table = pa.Table.from_pydict(combined_dict)
+
+        if not self.writer:
+            path = Path(self.root / DEFAULT_EPISODES_PATH.format(chunk_index=chunk_idx, file_index=file_idx))
+            path.parent.mkdir(parents=True, exist_ok=True)
+            self.writer = pq.ParquetWriter(
+                path, schema=table.schema, compression="snappy", use_dictionary=True
+            )
+
+        self.writer.write_table(table)
+
+        self.latest_episode = self.metadata_buffer[-1]
+        self.metadata_buffer.clear()
+
+    def _close_writer(self) -> None:
+        """Close and cleanup the parquet writer if it exists."""
+        self._flush_metadata_buffer()
+
+        writer = getattr(self, "writer", None)
+        if writer is not None:
+            writer.close()
+            self.writer = None
+
+    def __del__(self):
+        """
+        Trust the user to call .finalize() but as an added safety check call the parquet writer to stop when calling the destructor
+        """
+        self._close_writer()
+
+    def load_metadata(self):
+        self.info = load_info(self.root)
+        check_version_compatibility(self.repo_id, self._version, CODEBASE_VERSION)
+        self.tasks = load_tasks(self.root)
+        self.subtasks = load_subtasks(self.root)
+        self.episodes = load_episodes(self.root)
+        self.stats = load_stats(self.root)
+
+    def pull_from_repo(
+        self,
+        allow_patterns: list[str] | str | None = None,
+        ignore_patterns: list[str] | str | None = None,
+    ) -> None:
+        snapshot_download(
+            self.repo_id,
+            repo_type="dataset",
+            revision=self.revision,
+            local_dir=self.root,
+            allow_patterns=allow_patterns,
+            ignore_patterns=ignore_patterns,
+        )
+
+    @property
+    def url_root(self) -> str:
+        return f"hf://datasets/{self.repo_id}"
+
+    @property
+    def _version(self) -> packaging.version.Version:
+        """Codebase version used to create this dataset."""
+        return packaging.version.parse(self.info["codebase_version"])
+
+    def get_data_file_path(self, ep_index: int) -> Path:
+        if self.episodes is None:
+            self.episodes = load_episodes(self.root)
+        if ep_index >= len(self.episodes):
+            raise IndexError(
+                f"Episode index {ep_index} out of range. Episodes: {len(self.episodes) if self.episodes else 0}"
+            )
+        ep = self.episodes[ep_index]
+        chunk_idx = ep["data/chunk_index"]
+        file_idx = ep["data/file_index"]
+        fpath = self.data_path.format(chunk_index=chunk_idx, file_index=file_idx)
+        return Path(fpath)
+
+    def get_video_file_path(self, ep_index: int, vid_key: str) -> Path:
+        if self.episodes is None:
+            self.episodes = load_episodes(self.root)
+        if ep_index >= len(self.episodes):
+            raise IndexError(
+                f"Episode index {ep_index} out of range. Episodes: {len(self.episodes) if self.episodes else 0}"
+            )
+        ep = self.episodes[ep_index]
+        chunk_idx = ep[f"videos/{vid_key}/chunk_index"]
+        file_idx = ep[f"videos/{vid_key}/file_index"]
+        fpath = self.video_path.format(video_key=vid_key, chunk_index=chunk_idx, file_index=file_idx)
+        return Path(fpath)
+
+    @property
+    def data_path(self) -> str:
+        """Formattable string for the parquet files."""
+        return self.info["data_path"]
+
+    @property
+    def video_path(self) -> str | None:
+        """Formattable string for the video files."""
+        return self.info["video_path"]
+
+    @property
+    def robot_type(self) -> str | None:
+        """Robot type used in recording this dataset."""
+        return self.info["robot_type"]
+
+    @property
+    def fps(self) -> int:
+        """Frames per second used during data collection."""
+        return self.info["fps"]
+
+    @property
+    def features(self) -> dict[str, dict]:
+        """All features contained in the dataset."""
+        return self.info["features"]
+
+    @property
+    def image_keys(self) -> list[str]:
+        """Keys to access visual modalities stored as images."""
+        return [key for key, ft in self.features.items() if ft["dtype"] == "image"]
+
+    @property
+    def video_keys(self) -> list[str]:
+        """Keys to access visual modalities stored as videos."""
+        return [key for key, ft in self.features.items() if ft["dtype"] == "video"]
+
+    @property
+    def camera_keys(self) -> list[str]:
+        """Keys to access visual modalities (regardless of their storage method)."""
+        return [key for key, ft in self.features.items() if ft["dtype"] in ["video", "image"]]
+
+    @property
+    def names(self) -> dict[str, list | dict]:
+        """Names of the various dimensions of vector modalities."""
+        return {key: ft["names"] for key, ft in self.features.items()}
+
+    @property
+    def shapes(self) -> dict:
+        """Shapes for the different features."""
+        return {key: tuple(ft["shape"]) for key, ft in self.features.items()}
+
+    @property
+    def total_episodes(self) -> int:
+        """Total number of episodes available."""
+        return self.info["total_episodes"]
+
+    @property
+    def total_frames(self) -> int:
+        """Total number of frames saved in this dataset."""
+        return self.info["total_frames"]
+
+    @property
+    def total_tasks(self) -> int:
+        """Total number of different tasks performed in this dataset."""
+        return self.info["total_tasks"]
+
+    @property
+    def chunks_size(self) -> int:
+        """Max number of files per chunk."""
+        return self.info["chunks_size"]
+
+    @property
+    def data_files_size_in_mb(self) -> int:
+        """Max size of data file in mega bytes."""
+        return self.info["data_files_size_in_mb"]
+
+    @property
+    def video_files_size_in_mb(self) -> int:
+        """Max size of video file in mega bytes."""
+        return self.info["video_files_size_in_mb"]
+
+    def get_task_index(self, task: str) -> int | None:
+        """
+        Given a task in natural language, returns its task_index if the task already exists in the dataset,
+        otherwise return None.
+        """
+        if task in self.tasks.index:
+            return int(self.tasks.loc[task].task_index)
+        else:
+            return None
+
+    def save_episode_tasks(self, tasks: list[str]):
+        if len(set(tasks)) != len(tasks):
+            raise ValueError(f"Tasks are not unique: {tasks}")
+
+        if self.tasks is None:
+            new_tasks = tasks
+            task_indices = range(len(tasks))
+            self.tasks = pd.DataFrame({"task_index": task_indices}, index=pd.Index(tasks, name="task"))
+        else:
+            new_tasks = [task for task in tasks if task not in self.tasks.index]
+            new_task_indices = range(len(self.tasks), len(self.tasks) + len(new_tasks))
+            for task_idx, task in zip(new_task_indices, new_tasks, strict=False):
+                self.tasks.loc[task] = task_idx
+
+        if len(new_tasks) > 0:
+            # Update on disk
+            write_tasks(self.tasks, self.root)
+
+    def _save_episode_metadata(self, episode_dict: dict) -> None:
+        """Buffer episode metadata and write to parquet in batches for efficiency.
+
+        This function accumulates episode metadata in a buffer and flushes it when the buffer
+        reaches the configured size. This reduces I/O overhead by writing multiple episodes
+        at once instead of one row at a time.
+
+        Notes: We both need to update parquet files and HF dataset:
+        - `pandas` loads parquet file in RAM
+        - `datasets` relies on a memory mapping from pyarrow (no RAM). It either converts parquet files to a pyarrow cache on disk,
+          or loads directly from pyarrow cache.
+        """
+        # Convert to list format for each value
+        episode_dict = {key: [value] for key, value in episode_dict.items()}
+        num_frames = episode_dict["length"][0]
+
+        if self.latest_episode is None:
+            # Initialize indices and frame count for a new dataset made of the first episode data
+            chunk_idx, file_idx = 0, 0
+            if self.episodes is not None and len(self.episodes) > 0:
+                # It means we are resuming recording, so we need to load the latest episode
+                # Update the indices to avoid overwriting the latest episode
+                chunk_idx = self.episodes[-1]["meta/episodes/chunk_index"]
+                file_idx = self.episodes[-1]["meta/episodes/file_index"]
+                latest_num_frames = self.episodes[-1]["dataset_to_index"]
+                episode_dict["dataset_from_index"] = [latest_num_frames]
+                episode_dict["dataset_to_index"] = [latest_num_frames + num_frames]
+
+                # When resuming, move to the next file
+                chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, self.chunks_size)
+            else:
+                episode_dict["dataset_from_index"] = [0]
+                episode_dict["dataset_to_index"] = [num_frames]
+
+            episode_dict["meta/episodes/chunk_index"] = [chunk_idx]
+            episode_dict["meta/episodes/file_index"] = [file_idx]
+        else:
+            chunk_idx = self.latest_episode["meta/episodes/chunk_index"][0]
+            file_idx = self.latest_episode["meta/episodes/file_index"][0]
+
+            latest_path = (
+                self.root / DEFAULT_EPISODES_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
+                if self.writer is None
+                else self.writer.where
+            )
+
+            if Path(latest_path).exists():
+                latest_size_in_mb = get_file_size_in_mb(Path(latest_path))
+                latest_num_frames = self.latest_episode["episode_index"][0]
+
+                av_size_per_frame = latest_size_in_mb / latest_num_frames if latest_num_frames > 0 else 0.0
+
+                if latest_size_in_mb + av_size_per_frame * num_frames >= self.data_files_size_in_mb:
+                    # Size limit is reached, flush buffer and prepare new parquet file
+                    self._flush_metadata_buffer()
+                    chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, self.chunks_size)
+                    self._close_writer()
+
+            # Update the existing pandas dataframe with new row
+            episode_dict["meta/episodes/chunk_index"] = [chunk_idx]
+            episode_dict["meta/episodes/file_index"] = [file_idx]
+            episode_dict["dataset_from_index"] = [self.latest_episode["dataset_to_index"][0]]
+            episode_dict["dataset_to_index"] = [self.latest_episode["dataset_to_index"][0] + num_frames]
+
+        # Add to buffer
+        self.metadata_buffer.append(episode_dict)
+        self.latest_episode = episode_dict
+
+        if len(self.metadata_buffer) >= self.metadata_buffer_size:
+            self._flush_metadata_buffer()
+
+    def save_episode(
+        self,
+        episode_index: int,
+        episode_length: int,
+        episode_tasks: list[str],
+        episode_stats: dict[str, dict],
+        episode_metadata: dict,
+    ) -> None:
+        episode_dict = {
+            "episode_index": episode_index,
+            "tasks": episode_tasks,
+            "length": episode_length,
+        }
+        episode_dict.update(episode_metadata)
+        episode_dict.update(flatten_dict({"stats": episode_stats}))
+        self._save_episode_metadata(episode_dict)
+
+        # Update info
+        self.info["total_episodes"] += 1
+        self.info["total_frames"] += episode_length
+        self.info["total_tasks"] = len(self.tasks)
+        self.info["splits"] = {"train": f"0:{self.info['total_episodes']}"}
+
+        write_info(self.info, self.root)
+
+        self.stats = aggregate_stats([self.stats, episode_stats]) if self.stats is not None else episode_stats
+        write_stats(self.stats, self.root)
+
+    def update_video_info(self, video_key: str | None = None) -> None:
+        """
+        Warning: this function writes info from first episode videos, implicitly assuming that all videos have
+        been encoded the same way. Also, this means it assumes the first episode exists.
+        """
+        if video_key is not None and video_key not in self.video_keys:
+            raise ValueError(f"Video key {video_key} not found in dataset")
+
+        video_keys = [video_key] if video_key is not None else self.video_keys
+        for key in video_keys:
+            if not self.features[key].get("info", None):
+                video_path = self.root / self.video_path.format(video_key=key, chunk_index=0, file_index=0)
+                self.info["features"][key]["info"] = get_video_info(video_path)
+
+    def update_chunk_settings(
+        self,
+        chunks_size: int | None = None,
+        data_files_size_in_mb: int | None = None,
+        video_files_size_in_mb: int | None = None,
+    ) -> None:
+        """Update chunk and file size settings after dataset creation.
+
+        This allows users to customize storage organization without modifying the constructor.
+        These settings control how episodes are chunked and how large files can grow before
+        creating new ones.
+
+        Args:
+            chunks_size: Maximum number of files per chunk directory. If None, keeps current value.
+            data_files_size_in_mb: Maximum size for data parquet files in MB. If None, keeps current value.
+            video_files_size_in_mb: Maximum size for video files in MB. If None, keeps current value.
+        """
+        if chunks_size is not None:
+            if chunks_size <= 0:
+                raise ValueError(f"chunks_size must be positive, got {chunks_size}")
+            self.info["chunks_size"] = chunks_size
+
+        if data_files_size_in_mb is not None:
+            if data_files_size_in_mb <= 0:
+                raise ValueError(f"data_files_size_in_mb must be positive, got {data_files_size_in_mb}")
+            self.info["data_files_size_in_mb"] = data_files_size_in_mb
+
+        if video_files_size_in_mb is not None:
+            if video_files_size_in_mb <= 0:
+                raise ValueError(f"video_files_size_in_mb must be positive, got {video_files_size_in_mb}")
+            self.info["video_files_size_in_mb"] = video_files_size_in_mb
+
+        # Update the info file on disk
+        write_info(self.info, self.root)
+
+    def get_chunk_settings(self) -> dict[str, int]:
+        """Get current chunk and file size settings.
+
+        Returns:
+            Dict containing chunks_size, data_files_size_in_mb, and video_files_size_in_mb.
+        """
+        return {
+            "chunks_size": self.chunks_size,
+            "data_files_size_in_mb": self.data_files_size_in_mb,
+            "video_files_size_in_mb": self.video_files_size_in_mb,
+        }
+
+    def __repr__(self):
+        feature_keys = list(self.features)
+        return (
+            f"{self.__class__.__name__}({{\n"
+            f"    Repository ID: '{self.repo_id}',\n"
+            f"    Total episodes: '{self.total_episodes}',\n"
+            f"    Total frames: '{self.total_frames}',\n"
+            f"    Features: '{feature_keys}',\n"
+            "})',\n"
+        )
+
+    @classmethod
+    def create(
+        cls,
+        repo_id: str,
+        fps: int,
+        features: dict,
+        robot_type: str | None = None,
+        root: str | Path | None = None,
+        use_videos: bool = True,
+        metadata_buffer_size: int = 10,
+        chunks_size: int | None = None,
+        data_files_size_in_mb: int | None = None,
+        video_files_size_in_mb: int | None = None,
+    ) -> "LeRobotDatasetMetadata":
+        """Creates metadata for a LeRobotDataset."""
+        obj = cls.__new__(cls)
+        obj.repo_id = repo_id
+        obj.root = Path(root) if root is not None else HF_LEROBOT_HOME / repo_id
+
+        obj.root.mkdir(parents=True, exist_ok=False)
+
+        features = {**features, **DEFAULT_FEATURES}
+        _validate_feature_names(features)
+
+        obj.tasks = None
+        obj.subtasks = None
+        obj.episodes = None
+        obj.stats = None
+        obj.info = create_empty_dataset_info(
+            CODEBASE_VERSION,
+            fps,
+            features,
+            use_videos,
+            robot_type,
+            chunks_size,
+            data_files_size_in_mb,
+            video_files_size_in_mb,
+        )
+        if len(obj.video_keys) > 0 and not use_videos:
+            raise ValueError(
+                f"Features contain video keys {obj.video_keys}, but 'use_videos' is set to False. "
+                "Either remove video features from the features dict, or set 'use_videos=True'."
+            )
+        write_json(obj.info, obj.root / INFO_PATH)
+        obj.revision = None
+        obj.writer = None
+        obj.latest_episode = None
+        obj.metadata_buffer = []
+        obj.metadata_buffer_size = metadata_buffer_size
+        return obj
diff --git a/lerobot/src/lerobot/datasets/dataset_tools.py b/lerobot/src/lerobot/datasets/dataset_tools.py
new file mode 100644
index 0000000000000000000000000000000000000000..87cdc18e5e39f3b26c91c04159a16ce34778b323
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/dataset_tools.py
@@ -0,0 +1,1783 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Dataset tools utilities for LeRobotDataset.
+
+This module provides utilities for:
+- Deleting episodes from datasets
+- Splitting datasets into multiple smaller datasets
+- Adding/removing features from datasets
+- Merging datasets (wrapper around aggregate functionality)
+"""
+
+import logging
+import shutil
+from collections.abc import Callable
+from concurrent.futures import ThreadPoolExecutor, as_completed
+from pathlib import Path
+
+import datasets
+import numpy as np
+import pandas as pd
+import pyarrow.parquet as pq
+import torch
+from tqdm import tqdm
+
+from lerobot.datasets.aggregate import aggregate_datasets
+from lerobot.datasets.compute_stats import aggregate_stats
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.io_utils import (
+    get_parquet_file_size_in_mb,
+    load_episodes,
+    write_info,
+    write_stats,
+    write_tasks,
+)
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.utils import (
+    DATA_DIR,
+    DEFAULT_CHUNK_SIZE,
+    DEFAULT_DATA_FILE_SIZE_IN_MB,
+    DEFAULT_DATA_PATH,
+    DEFAULT_EPISODES_PATH,
+    update_chunk_file_indices,
+)
+from lerobot.datasets.video_utils import encode_video_frames, get_video_info
+from lerobot.utils.constants import HF_LEROBOT_HOME, OBS_IMAGE
+
+
+def _load_episode_with_stats(src_dataset: LeRobotDataset, episode_idx: int) -> dict:
+    """Load a single episode's metadata including stats from parquet file.
+
+    Args:
+        src_dataset: Source dataset
+        episode_idx: Episode index to load
+
+    Returns:
+        dict containing episode metadata and stats
+    """
+    ep_meta = src_dataset.meta.episodes[episode_idx]
+    chunk_idx = ep_meta["meta/episodes/chunk_index"]
+    file_idx = ep_meta["meta/episodes/file_index"]
+
+    parquet_path = src_dataset.root / DEFAULT_EPISODES_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
+    df = pd.read_parquet(parquet_path)
+
+    episode_row = df[df["episode_index"] == episode_idx].iloc[0]
+
+    return episode_row.to_dict()
+
+
+def delete_episodes(
+    dataset: LeRobotDataset,
+    episode_indices: list[int],
+    output_dir: str | Path | None = None,
+    repo_id: str | None = None,
+) -> LeRobotDataset:
+    """Delete episodes from a LeRobotDataset and create a new dataset.
+
+    Args:
+        dataset: The source LeRobotDataset.
+        episode_indices: List of episode indices to delete.
+        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
+        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+    """
+    if not episode_indices:
+        raise ValueError("No episodes to delete")
+
+    valid_indices = set(range(dataset.meta.total_episodes))
+    invalid = set(episode_indices) - valid_indices
+    if invalid:
+        raise ValueError(f"Invalid episode indices: {invalid}")
+
+    logging.info(f"Deleting {len(episode_indices)} episodes from dataset")
+
+    if repo_id is None:
+        repo_id = f"{dataset.repo_id}_modified"
+    output_dir = Path(output_dir) if output_dir is not None else HF_LEROBOT_HOME / repo_id
+
+    episodes_to_keep = [i for i in range(dataset.meta.total_episodes) if i not in episode_indices]
+    if not episodes_to_keep:
+        raise ValueError("Cannot delete all episodes from dataset")
+
+    new_meta = LeRobotDatasetMetadata.create(
+        repo_id=repo_id,
+        fps=dataset.meta.fps,
+        features=dataset.meta.features,
+        robot_type=dataset.meta.robot_type,
+        root=output_dir,
+        use_videos=len(dataset.meta.video_keys) > 0,
+    )
+
+    episode_mapping = {old_idx: new_idx for new_idx, old_idx in enumerate(episodes_to_keep)}
+
+    video_metadata = None
+    if dataset.meta.video_keys:
+        video_metadata = _copy_and_reindex_videos(dataset, new_meta, episode_mapping)
+
+    data_metadata = _copy_and_reindex_data(dataset, new_meta, episode_mapping)
+
+    _copy_and_reindex_episodes_metadata(dataset, new_meta, episode_mapping, data_metadata, video_metadata)
+
+    new_dataset = LeRobotDataset(
+        repo_id=repo_id,
+        root=output_dir,
+        image_transforms=dataset.image_transforms,
+        delta_timestamps=dataset.delta_timestamps,
+        tolerance_s=dataset.tolerance_s,
+    )
+
+    logging.info(f"Created new dataset with {len(episodes_to_keep)} episodes")
+    return new_dataset
+
+
+def split_dataset(
+    dataset: LeRobotDataset,
+    splits: dict[str, float | list[int]],
+    output_dir: str | Path | None = None,
+) -> dict[str, LeRobotDataset]:
+    """Split a LeRobotDataset into multiple smaller datasets.
+
+    Args:
+        dataset: The source LeRobotDataset to split.
+        splits: Either a dict mapping split names to episode indices, or a dict mapping
+                split names to fractions (must sum to <= 1.0).
+        output_dir: Root directory where the split datasets will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id.
+
+    Examples:
+      Split by specific episodes
+        splits = {"train": [0, 1, 2], "val": [3, 4]}
+        datasets = split_dataset(dataset, splits)
+
+      Split by fractions
+        splits = {"train": 0.8, "val": 0.2}
+        datasets = split_dataset(dataset, splits)
+    """
+    if not splits:
+        raise ValueError("No splits provided")
+
+    if all(isinstance(v, float) for v in splits.values()):
+        splits = _fractions_to_episode_indices(dataset.meta.total_episodes, splits)
+
+    all_episodes = set()
+    for split_name, episodes in splits.items():
+        if not episodes:
+            raise ValueError(f"Split '{split_name}' has no episodes")
+        episode_set = set(episodes)
+        if episode_set & all_episodes:
+            raise ValueError("Episodes cannot appear in multiple splits")
+        all_episodes.update(episode_set)
+
+    valid_indices = set(range(dataset.meta.total_episodes))
+    invalid = all_episodes - valid_indices
+    if invalid:
+        raise ValueError(f"Invalid episode indices: {invalid}")
+
+    if output_dir is not None:
+        output_dir = Path(output_dir)
+
+    result_datasets = {}
+
+    for split_name, episodes in splits.items():
+        logging.info(f"Creating split '{split_name}' with {len(episodes)} episodes")
+
+        split_repo_id = f"{dataset.repo_id}_{split_name}"
+
+        split_output_dir = (
+            output_dir / split_name if output_dir is not None else HF_LEROBOT_HOME / split_repo_id
+        )
+
+        episode_mapping = {old_idx: new_idx for new_idx, old_idx in enumerate(sorted(episodes))}
+
+        new_meta = LeRobotDatasetMetadata.create(
+            repo_id=split_repo_id,
+            fps=dataset.meta.fps,
+            features=dataset.meta.features,
+            robot_type=dataset.meta.robot_type,
+            root=split_output_dir,
+            use_videos=len(dataset.meta.video_keys) > 0,
+            chunks_size=dataset.meta.chunks_size,
+            data_files_size_in_mb=dataset.meta.data_files_size_in_mb,
+            video_files_size_in_mb=dataset.meta.video_files_size_in_mb,
+        )
+
+        video_metadata = None
+        if dataset.meta.video_keys:
+            video_metadata = _copy_and_reindex_videos(dataset, new_meta, episode_mapping)
+
+        data_metadata = _copy_and_reindex_data(dataset, new_meta, episode_mapping)
+
+        _copy_and_reindex_episodes_metadata(dataset, new_meta, episode_mapping, data_metadata, video_metadata)
+
+        new_dataset = LeRobotDataset(
+            repo_id=split_repo_id,
+            root=split_output_dir,
+            image_transforms=dataset.image_transforms,
+            delta_timestamps=dataset.delta_timestamps,
+            tolerance_s=dataset.tolerance_s,
+        )
+
+        result_datasets[split_name] = new_dataset
+
+    return result_datasets
+
+
+def merge_datasets(
+    datasets: list[LeRobotDataset],
+    output_repo_id: str,
+    output_dir: str | Path | None = None,
+) -> LeRobotDataset:
+    """Merge multiple LeRobotDatasets into a single dataset.
+
+    This is a wrapper around the aggregate_datasets functionality with a cleaner API.
+
+    Args:
+        datasets: List of LeRobotDatasets to merge.
+        output_repo_id: Merged dataset identifier.
+        output_dir: Root directory where the merged dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/output_repo_id.
+    """
+    if not datasets:
+        raise ValueError("No datasets to merge")
+
+    output_dir = Path(output_dir) if output_dir is not None else HF_LEROBOT_HOME / output_repo_id
+
+    repo_ids = [ds.repo_id for ds in datasets]
+    roots = [ds.root for ds in datasets]
+
+    aggregate_datasets(
+        repo_ids=repo_ids,
+        aggr_repo_id=output_repo_id,
+        roots=roots,
+        aggr_root=output_dir,
+    )
+
+    merged_dataset = LeRobotDataset(
+        repo_id=output_repo_id,
+        root=output_dir,
+        image_transforms=datasets[0].image_transforms,
+        delta_timestamps=datasets[0].delta_timestamps,
+        tolerance_s=datasets[0].tolerance_s,
+    )
+
+    return merged_dataset
+
+
+def modify_features(
+    dataset: LeRobotDataset,
+    add_features: dict[str, tuple[np.ndarray | torch.Tensor | Callable, dict]] | None = None,
+    remove_features: str | list[str] | None = None,
+    output_dir: str | Path | None = None,
+    repo_id: str | None = None,
+) -> LeRobotDataset:
+    """Modify a LeRobotDataset by adding and/or removing features in a single pass.
+
+    This is the most efficient way to modify features, as it only copies the dataset once
+    regardless of how many features are being added or removed.
+
+    Args:
+        dataset: The source LeRobotDataset.
+        add_features: Optional dict mapping feature names to (feature_values, feature_info) tuples.
+        remove_features: Optional feature name(s) to remove. Can be a single string or list.
+        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
+        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+
+    Returns:
+        New dataset with features modified.
+
+    Example:
+        new_dataset = modify_features(
+            dataset,
+            add_features={
+                "reward": (reward_array, {"dtype": "float32", "shape": [1], "names": None}),
+            },
+            remove_features=["old_feature"],
+            output_dir="./output",
+        )
+    """
+    if add_features is None and remove_features is None:
+        raise ValueError("Must specify at least one of add_features or remove_features")
+
+    remove_features_list: list[str] = []
+    if remove_features is not None:
+        remove_features_list = [remove_features] if isinstance(remove_features, str) else remove_features
+
+    if add_features:
+        required_keys = {"dtype", "shape"}
+        for feature_name, (_, feature_info) in add_features.items():
+            if feature_name in dataset.meta.features:
+                raise ValueError(f"Feature '{feature_name}' already exists in dataset")
+
+            if not required_keys.issubset(feature_info.keys()):
+                raise ValueError(f"feature_info for '{feature_name}' must contain keys: {required_keys}")
+
+    if remove_features_list:
+        for name in remove_features_list:
+            if name not in dataset.meta.features:
+                raise ValueError(f"Feature '{name}' not found in dataset")
+
+        required_features = {"timestamp", "frame_index", "episode_index", "index", "task_index"}
+        if any(name in required_features for name in remove_features_list):
+            raise ValueError(f"Cannot remove required features: {required_features}")
+
+    if repo_id is None:
+        repo_id = f"{dataset.repo_id}_modified"
+    output_dir = Path(output_dir) if output_dir is not None else HF_LEROBOT_HOME / repo_id
+
+    new_features = dataset.meta.features.copy()
+
+    if remove_features_list:
+        for name in remove_features_list:
+            new_features.pop(name, None)
+
+    if add_features:
+        for feature_name, (_, feature_info) in add_features.items():
+            new_features[feature_name] = feature_info
+
+    video_keys_to_remove = [name for name in remove_features_list if name in dataset.meta.video_keys]
+    remaining_video_keys = [k for k in dataset.meta.video_keys if k not in video_keys_to_remove]
+
+    new_meta = LeRobotDatasetMetadata.create(
+        repo_id=repo_id,
+        fps=dataset.meta.fps,
+        features=new_features,
+        robot_type=dataset.meta.robot_type,
+        root=output_dir,
+        use_videos=len(remaining_video_keys) > 0,
+    )
+
+    _copy_data_with_feature_changes(
+        dataset=dataset,
+        new_meta=new_meta,
+        add_features=add_features,
+        remove_features=remove_features_list if remove_features_list else None,
+    )
+
+    if new_meta.video_keys:
+        _copy_videos(dataset, new_meta, exclude_keys=video_keys_to_remove if video_keys_to_remove else None)
+
+    new_dataset = LeRobotDataset(
+        repo_id=repo_id,
+        root=output_dir,
+        image_transforms=dataset.image_transforms,
+        delta_timestamps=dataset.delta_timestamps,
+        tolerance_s=dataset.tolerance_s,
+    )
+
+    return new_dataset
+
+
+def add_features(
+    dataset: LeRobotDataset,
+    features: dict[str, tuple[np.ndarray | torch.Tensor | Callable, dict]],
+    output_dir: str | Path | None = None,
+    repo_id: str | None = None,
+) -> LeRobotDataset:
+    """Add multiple features to a LeRobotDataset in a single pass.
+
+    This is more efficient than calling add_feature() multiple times, as it only
+    copies the dataset once regardless of how many features are being added.
+
+    Args:
+        dataset: The source LeRobotDataset.
+        features: Dictionary mapping feature names to (feature_values, feature_info) tuples.
+        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
+        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+
+    Returns:
+        New dataset with all features added.
+
+    Example:
+        features = {
+            "task_embedding": (task_emb_array, {"dtype": "float32", "shape": [384], "names": None}),
+            "cam1_embedding": (cam1_emb_array, {"dtype": "float32", "shape": [768], "names": None}),
+            "cam2_embedding": (cam2_emb_array, {"dtype": "float32", "shape": [768], "names": None}),
+        }
+        new_dataset = add_features(dataset, features, output_dir="./output", repo_id="my_dataset")
+    """
+    if not features:
+        raise ValueError("No features provided")
+
+    return modify_features(
+        dataset=dataset,
+        add_features=features,
+        remove_features=None,
+        output_dir=output_dir,
+        repo_id=repo_id,
+    )
+
+
+def remove_feature(
+    dataset: LeRobotDataset,
+    feature_names: str | list[str],
+    output_dir: str | Path | None = None,
+    repo_id: str | None = None,
+) -> LeRobotDataset:
+    """Remove features from a LeRobotDataset.
+
+    Args:
+        dataset: The source LeRobotDataset.
+        feature_names: Name(s) of features to remove. Can be a single string or list.
+        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
+        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+
+    Returns:
+        New dataset with features removed.
+    """
+    return modify_features(
+        dataset=dataset,
+        add_features=None,
+        remove_features=feature_names,
+        output_dir=output_dir,
+        repo_id=repo_id,
+    )
+
+
+def _fractions_to_episode_indices(
+    total_episodes: int,
+    splits: dict[str, float],
+) -> dict[str, list[int]]:
+    """Convert split fractions to episode indices."""
+    if sum(splits.values()) > 1.0:
+        raise ValueError("Split fractions must sum to <= 1.0")
+
+    indices = list(range(total_episodes))
+    result = {}
+    start_idx = 0
+
+    for split_name, fraction in splits.items():
+        num_episodes = int(total_episodes * fraction)
+        if num_episodes == 0:
+            logging.warning(f"Split '{split_name}' has no episodes, skipping...")
+            continue
+        end_idx = start_idx + num_episodes
+        if split_name == list(splits.keys())[-1]:
+            end_idx = total_episodes
+        result[split_name] = indices[start_idx:end_idx]
+        start_idx = end_idx
+
+    return result
+
+
+def _copy_and_reindex_data(
+    src_dataset: LeRobotDataset,
+    dst_meta: LeRobotDatasetMetadata,
+    episode_mapping: dict[int, int],
+) -> dict[int, dict]:
+    """Copy and filter data files, only modifying files with deleted episodes.
+
+    Args:
+        src_dataset: Source dataset to copy from
+        dst_meta: Destination metadata object
+        episode_mapping: Mapping from old episode indices to new indices
+
+    Returns:
+        dict mapping episode index to its data file metadata (chunk_index, file_index, etc.)
+    """
+    if src_dataset.meta.episodes is None:
+        src_dataset.meta.episodes = load_episodes(src_dataset.meta.root)
+
+    file_to_episodes: dict[Path, set[int]] = {}
+    for old_idx in episode_mapping:
+        file_path = src_dataset.meta.get_data_file_path(old_idx)
+        if file_path not in file_to_episodes:
+            file_to_episodes[file_path] = set()
+        file_to_episodes[file_path].add(old_idx)
+
+    global_index = 0
+    episode_data_metadata: dict[int, dict] = {}
+
+    if dst_meta.tasks is None:
+        all_task_indices = set()
+        for src_path in file_to_episodes:
+            df = pd.read_parquet(src_dataset.root / src_path)
+            mask = df["episode_index"].isin(list(episode_mapping.keys()))
+            task_series: pd.Series = df[mask]["task_index"]
+            all_task_indices.update(task_series.unique().tolist())
+        tasks = [src_dataset.meta.tasks.iloc[idx].name for idx in all_task_indices]
+        dst_meta.save_episode_tasks(list(set(tasks)))
+
+    task_mapping = {}
+    for old_task_idx in range(len(src_dataset.meta.tasks)):
+        task_name = src_dataset.meta.tasks.iloc[old_task_idx].name
+        new_task_idx = dst_meta.get_task_index(task_name)
+        if new_task_idx is not None:
+            task_mapping[old_task_idx] = new_task_idx
+
+    for src_path in tqdm(sorted(file_to_episodes.keys()), desc="Processing data files"):
+        df = pd.read_parquet(src_dataset.root / src_path)
+
+        all_episodes_in_file = set(df["episode_index"].unique())
+        episodes_to_keep = file_to_episodes[src_path]
+
+        if all_episodes_in_file == episodes_to_keep:
+            df["episode_index"] = df["episode_index"].replace(episode_mapping)
+            df["index"] = range(global_index, global_index + len(df))
+            df["task_index"] = df["task_index"].replace(task_mapping)
+
+            first_ep_old_idx = min(episodes_to_keep)
+            src_ep = src_dataset.meta.episodes[first_ep_old_idx]
+            chunk_idx = src_ep["data/chunk_index"]
+            file_idx = src_ep["data/file_index"]
+        else:
+            mask = df["episode_index"].isin(list(episode_mapping.keys()))
+            df = df[mask].copy().reset_index(drop=True)
+
+            if len(df) == 0:
+                continue
+
+            df["episode_index"] = df["episode_index"].replace(episode_mapping)
+            df["index"] = range(global_index, global_index + len(df))
+            df["task_index"] = df["task_index"].replace(task_mapping)
+
+            first_ep_old_idx = min(episodes_to_keep)
+            src_ep = src_dataset.meta.episodes[first_ep_old_idx]
+            chunk_idx = src_ep["data/chunk_index"]
+            file_idx = src_ep["data/file_index"]
+
+        dst_path = dst_meta.root / DEFAULT_DATA_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
+        dst_path.parent.mkdir(parents=True, exist_ok=True)
+
+        _write_parquet(df, dst_path, dst_meta)
+
+        for ep_old_idx in episodes_to_keep:
+            ep_new_idx = episode_mapping[ep_old_idx]
+            ep_df = df[df["episode_index"] == ep_new_idx]
+            episode_data_metadata[ep_new_idx] = {
+                "data/chunk_index": chunk_idx,
+                "data/file_index": file_idx,
+                "dataset_from_index": int(ep_df["index"].min()),
+                "dataset_to_index": int(ep_df["index"].max() + 1),
+            }
+
+        global_index += len(df)
+
+    return episode_data_metadata
+
+
+def _keep_episodes_from_video_with_av(
+    input_path: Path,
+    output_path: Path,
+    episodes_to_keep: list[tuple[int, int]],
+    fps: float,
+    vcodec: str = "libsvtav1",
+    pix_fmt: str = "yuv420p",
+) -> None:
+    """Keep only specified episodes from a video file using PyAV.
+
+    This function decodes frames from specified frame ranges and re-encodes them with
+    properly reset timestamps to ensure monotonic progression.
+
+    Args:
+        input_path: Source video file path.
+        output_path: Destination video file path.
+        episodes_to_keep: List of (start_frame, end_frame) tuples for episodes to keep.
+            Ranges are half-open intervals: [start_frame, end_frame), where start_frame
+            is inclusive and end_frame is exclusive.
+        fps: Frame rate of the video.
+        vcodec: Video codec to use for encoding.
+        pix_fmt: Pixel format for output video.
+    """
+    from fractions import Fraction
+
+    import av
+
+    if not episodes_to_keep:
+        raise ValueError("No episodes to keep")
+
+    in_container = av.open(str(input_path))
+
+    # Check if video stream exists.
+    if not in_container.streams.video:
+        raise ValueError(
+            f"No video streams found in {input_path}. "
+            "The video file may be corrupted or empty. "
+            "Try re-downloading the dataset or checking the video file."
+        )
+
+    v_in = in_container.streams.video[0]
+
+    out = av.open(str(output_path), mode="w")
+
+    # Convert fps to Fraction for PyAV compatibility.
+    fps_fraction = Fraction(fps).limit_denominator(1000)
+    v_out = out.add_stream(vcodec, rate=fps_fraction)
+
+    # PyAV type stubs don't distinguish video streams from audio/subtitle streams.
+    v_out.width = v_in.codec_context.width
+    v_out.height = v_in.codec_context.height
+    v_out.pix_fmt = pix_fmt
+
+    # Set time_base to match the frame rate for proper timestamp handling.
+    v_out.time_base = Fraction(1, int(fps))
+
+    out.start_encoding()
+
+    # Create set of (start, end) ranges for fast lookup.
+    # Convert to a sorted list for efficient checking.
+    frame_ranges = sorted(episodes_to_keep)
+
+    # Track frame index for setting PTS and current range being processed.
+    src_frame_count = 0
+    frame_count = 0
+    range_idx = 0
+
+    # Read through entire video once and filter frames.
+    for packet in in_container.demux(v_in):
+        for frame in packet.decode():
+            if frame is None:
+                continue
+
+            # Check if frame is in any of our desired frame ranges.
+            # Skip ranges that have already passed.
+            while range_idx < len(frame_ranges) and src_frame_count >= frame_ranges[range_idx][1]:
+                range_idx += 1
+
+            # If we've passed all ranges, stop processing.
+            if range_idx >= len(frame_ranges):
+                break
+
+            # Check if frame is in current range.
+            start_frame = frame_ranges[range_idx][0]
+
+            if src_frame_count < start_frame:
+                src_frame_count += 1
+                continue
+
+            # Frame is in range - create a new frame with reset timestamps.
+            # We need to create a copy to avoid modifying the original.
+            new_frame = frame.reformat(width=v_out.width, height=v_out.height, format=v_out.pix_fmt)
+            new_frame.pts = frame_count
+            new_frame.time_base = Fraction(1, int(fps))
+
+            # Encode and mux the frame.
+            for pkt in v_out.encode(new_frame):
+                out.mux(pkt)
+
+            src_frame_count += 1
+            frame_count += 1
+
+    # Flush encoder.
+    for pkt in v_out.encode():
+        out.mux(pkt)
+
+    out.close()
+    in_container.close()
+
+
+def _copy_and_reindex_videos(
+    src_dataset: LeRobotDataset,
+    dst_meta: LeRobotDatasetMetadata,
+    episode_mapping: dict[int, int],
+    vcodec: str = "libsvtav1",
+    pix_fmt: str = "yuv420p",
+) -> dict[int, dict]:
+    """Copy and filter video files, only re-encoding files with deleted episodes.
+
+    For video files that only contain kept episodes, we copy them directly.
+    For files with mixed kept/deleted episodes, we use PyAV filters to efficiently
+    re-encode only the desired segments.
+
+    Args:
+        src_dataset: Source dataset to copy from
+        dst_meta: Destination metadata object
+        episode_mapping: Mapping from old episode indices to new indices
+
+    Returns:
+        dict mapping episode index to its video metadata (chunk_index, file_index, timestamps)
+    """
+    if src_dataset.meta.episodes is None:
+        src_dataset.meta.episodes = load_episodes(src_dataset.meta.root)
+
+    episodes_video_metadata: dict[int, dict] = {new_idx: {} for new_idx in episode_mapping.values()}
+
+    for video_key in src_dataset.meta.video_keys:
+        logging.info(f"Processing videos for {video_key}")
+
+        if dst_meta.video_path is None:
+            raise ValueError("Destination metadata has no video_path defined")
+
+        file_to_episodes: dict[tuple[int, int], list[int]] = {}
+        for old_idx in episode_mapping:
+            src_ep = src_dataset.meta.episodes[old_idx]
+            chunk_idx = src_ep[f"videos/{video_key}/chunk_index"]
+            file_idx = src_ep[f"videos/{video_key}/file_index"]
+            file_key = (chunk_idx, file_idx)
+            if file_key not in file_to_episodes:
+                file_to_episodes[file_key] = []
+            file_to_episodes[file_key].append(old_idx)
+
+        for (src_chunk_idx, src_file_idx), episodes_in_file in tqdm(
+            sorted(file_to_episodes.items()), desc=f"Processing {video_key} video files"
+        ):
+            all_episodes_in_file = [
+                ep_idx
+                for ep_idx in range(src_dataset.meta.total_episodes)
+                if src_dataset.meta.episodes[ep_idx].get(f"videos/{video_key}/chunk_index") == src_chunk_idx
+                and src_dataset.meta.episodes[ep_idx].get(f"videos/{video_key}/file_index") == src_file_idx
+            ]
+
+            episodes_to_keep_set = set(episodes_in_file)
+            all_in_file_set = set(all_episodes_in_file)
+
+            if all_in_file_set == episodes_to_keep_set:
+                assert src_dataset.meta.video_path is not None
+                src_video_path = src_dataset.root / src_dataset.meta.video_path.format(
+                    video_key=video_key, chunk_index=src_chunk_idx, file_index=src_file_idx
+                )
+                dst_video_path = dst_meta.root / dst_meta.video_path.format(
+                    video_key=video_key, chunk_index=src_chunk_idx, file_index=src_file_idx
+                )
+                dst_video_path.parent.mkdir(parents=True, exist_ok=True)
+                shutil.copy(src_video_path, dst_video_path)
+
+                for old_idx in episodes_in_file:
+                    new_idx = episode_mapping[old_idx]
+                    src_ep = src_dataset.meta.episodes[old_idx]
+                    episodes_video_metadata[new_idx][f"videos/{video_key}/chunk_index"] = src_chunk_idx
+                    episodes_video_metadata[new_idx][f"videos/{video_key}/file_index"] = src_file_idx
+                    episodes_video_metadata[new_idx][f"videos/{video_key}/from_timestamp"] = src_ep[
+                        f"videos/{video_key}/from_timestamp"
+                    ]
+                    episodes_video_metadata[new_idx][f"videos/{video_key}/to_timestamp"] = src_ep[
+                        f"videos/{video_key}/to_timestamp"
+                    ]
+            else:
+                # Build list of frame ranges to keep, in sorted order.
+                sorted_keep_episodes = sorted(episodes_in_file, key=lambda x: episode_mapping[x])
+                episodes_to_keep_ranges: list[tuple[int, int]] = []
+                for old_idx in sorted_keep_episodes:
+                    src_ep = src_dataset.meta.episodes[old_idx]
+                    from_frame = round(src_ep[f"videos/{video_key}/from_timestamp"] * src_dataset.meta.fps)
+                    to_frame = round(src_ep[f"videos/{video_key}/to_timestamp"] * src_dataset.meta.fps)
+                    assert src_ep["length"] == to_frame - from_frame, (
+                        f"Episode length mismatch: {src_ep['length']} vs {to_frame - from_frame}"
+                    )
+                    episodes_to_keep_ranges.append((from_frame, to_frame))
+
+                # Use PyAV filters to efficiently re-encode only the desired segments.
+                assert src_dataset.meta.video_path is not None
+                src_video_path = src_dataset.root / src_dataset.meta.video_path.format(
+                    video_key=video_key, chunk_index=src_chunk_idx, file_index=src_file_idx
+                )
+                dst_video_path = dst_meta.root / dst_meta.video_path.format(
+                    video_key=video_key, chunk_index=src_chunk_idx, file_index=src_file_idx
+                )
+                dst_video_path.parent.mkdir(parents=True, exist_ok=True)
+
+                logging.info(
+                    f"Re-encoding {video_key} (chunk {src_chunk_idx}, file {src_file_idx}) "
+                    f"with {len(episodes_to_keep_ranges)} episodes"
+                )
+                _keep_episodes_from_video_with_av(
+                    src_video_path,
+                    dst_video_path,
+                    episodes_to_keep_ranges,
+                    src_dataset.meta.fps,
+                    vcodec,
+                    pix_fmt,
+                )
+
+                cumulative_ts = 0.0
+                for old_idx in sorted_keep_episodes:
+                    new_idx = episode_mapping[old_idx]
+                    src_ep = src_dataset.meta.episodes[old_idx]
+                    ep_length = src_ep["length"]
+                    ep_duration = ep_length / src_dataset.meta.fps
+
+                    episodes_video_metadata[new_idx][f"videos/{video_key}/chunk_index"] = src_chunk_idx
+                    episodes_video_metadata[new_idx][f"videos/{video_key}/file_index"] = src_file_idx
+                    episodes_video_metadata[new_idx][f"videos/{video_key}/from_timestamp"] = cumulative_ts
+                    episodes_video_metadata[new_idx][f"videos/{video_key}/to_timestamp"] = (
+                        cumulative_ts + ep_duration
+                    )
+
+                    cumulative_ts += ep_duration
+
+    return episodes_video_metadata
+
+
+def _copy_and_reindex_episodes_metadata(
+    src_dataset: LeRobotDataset,
+    dst_meta: LeRobotDatasetMetadata,
+    episode_mapping: dict[int, int],
+    data_metadata: dict[int, dict],
+    video_metadata: dict[int, dict] | None = None,
+) -> None:
+    """Copy and reindex episodes metadata using provided data and video metadata.
+
+    Args:
+        src_dataset: Source dataset to copy from
+        dst_meta: Destination metadata object
+        episode_mapping: Mapping from old episode indices to new indices
+        data_metadata: Dict mapping new episode index to its data file metadata
+        video_metadata: Optional dict mapping new episode index to its video metadata
+    """
+    from lerobot.datasets.utils import flatten_dict
+
+    if src_dataset.meta.episodes is None:
+        src_dataset.meta.episodes = load_episodes(src_dataset.meta.root)
+
+    all_stats = []
+    total_frames = 0
+
+    for old_idx, new_idx in tqdm(
+        sorted(episode_mapping.items(), key=lambda x: x[1]), desc="Processing episodes metadata"
+    ):
+        src_episode_full = _load_episode_with_stats(src_dataset, old_idx)
+
+        src_episode = src_dataset.meta.episodes[old_idx]
+
+        episode_meta = data_metadata[new_idx].copy()
+
+        if video_metadata and new_idx in video_metadata:
+            episode_meta.update(video_metadata[new_idx])
+
+        # Extract episode statistics from parquet metadata.
+        # Note (maractingi): When pandas/pyarrow serializes numpy arrays with shape (3, 1, 1) to parquet,
+        # they are being deserialized as nested object arrays like:
+        #   array([array([array([0.])]), array([array([0.])]), array([array([0.])])])
+        # This happens particularly with image/video statistics. We need to detect and flatten
+        # these nested structures back to proper (3, 1, 1) arrays so aggregate_stats can process them.
+        episode_stats = {}
+        for key in src_episode_full:
+            if key.startswith("stats/"):
+                stat_key = key.replace("stats/", "")
+                parts = stat_key.split("/")
+                if len(parts) == 2:
+                    feature_name, stat_name = parts
+                    if feature_name not in episode_stats:
+                        episode_stats[feature_name] = {}
+
+                    value = src_episode_full[key]
+
+                    if feature_name in src_dataset.meta.features:
+                        feature_dtype = src_dataset.meta.features[feature_name]["dtype"]
+                        if feature_dtype in ["image", "video"] and stat_name != "count":
+                            if isinstance(value, np.ndarray) and value.dtype == object:
+                                flat_values = []
+                                for item in value:
+                                    while isinstance(item, np.ndarray):
+                                        item = item.flatten()[0]
+                                    flat_values.append(item)
+                                value = np.array(flat_values, dtype=np.float64).reshape(3, 1, 1)
+                            elif isinstance(value, np.ndarray) and value.shape == (3,):
+                                value = value.reshape(3, 1, 1)
+
+                    episode_stats[feature_name][stat_name] = value
+
+        all_stats.append(episode_stats)
+
+        episode_dict = {
+            "episode_index": new_idx,
+            "tasks": src_episode["tasks"],
+            "length": src_episode["length"],
+        }
+        episode_dict.update(episode_meta)
+        episode_dict.update(flatten_dict({"stats": episode_stats}))
+        dst_meta._save_episode_metadata(episode_dict)
+
+        total_frames += src_episode["length"]
+
+    dst_meta._close_writer()
+
+    dst_meta.info.update(
+        {
+            "total_episodes": len(episode_mapping),
+            "total_frames": total_frames,
+            "total_tasks": len(dst_meta.tasks) if dst_meta.tasks is not None else 0,
+            "splits": {"train": f"0:{len(episode_mapping)}"},
+        }
+    )
+    write_info(dst_meta.info, dst_meta.root)
+
+    if not all_stats:
+        logging.warning("No statistics found to aggregate")
+        return
+
+    logging.info(f"Aggregating statistics for {len(all_stats)} episodes")
+    aggregated_stats = aggregate_stats(all_stats)
+    filtered_stats = {k: v for k, v in aggregated_stats.items() if k in dst_meta.features}
+    write_stats(filtered_stats, dst_meta.root)
+
+
+def _write_parquet(df: pd.DataFrame, path: Path, meta: LeRobotDatasetMetadata) -> None:
+    """Write DataFrame to parquet
+
+    This ensures images are properly embedded and the file can be loaded correctly by HF datasets.
+    """
+    from lerobot.datasets.feature_utils import get_hf_features_from_features
+    from lerobot.datasets.io_utils import embed_images
+
+    hf_features = get_hf_features_from_features(meta.features)
+    ep_dataset = datasets.Dataset.from_dict(df.to_dict(orient="list"), features=hf_features, split="train")
+
+    if len(meta.image_keys) > 0:
+        ep_dataset = embed_images(ep_dataset)
+
+    table = ep_dataset.with_format("arrow")[:]
+    writer = pq.ParquetWriter(path, schema=table.schema, compression="snappy", use_dictionary=True)
+    writer.write_table(table)
+    writer.close()
+
+
+def _save_data_chunk(
+    df: pd.DataFrame,
+    meta: LeRobotDatasetMetadata,
+    chunk_idx: int = 0,
+    file_idx: int = 0,
+) -> tuple[int, int, dict[int, dict]]:
+    """Save a data chunk and return updated indices and episode metadata.
+
+    Returns:
+        tuple: (next_chunk_idx, next_file_idx, episode_metadata_dict)
+            where episode_metadata_dict maps episode_index to its data file metadata
+    """
+    path = meta.root / DEFAULT_DATA_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
+    path.parent.mkdir(parents=True, exist_ok=True)
+
+    _write_parquet(df, path, meta)
+
+    episode_metadata = {}
+    for ep_idx in df["episode_index"].unique():
+        ep_df = df[df["episode_index"] == ep_idx]
+        episode_metadata[ep_idx] = {
+            "data/chunk_index": chunk_idx,
+            "data/file_index": file_idx,
+            "dataset_from_index": int(ep_df["index"].min()),
+            "dataset_to_index": int(ep_df["index"].max() + 1),
+        }
+
+    file_size = get_parquet_file_size_in_mb(path)
+    if file_size >= DEFAULT_DATA_FILE_SIZE_IN_MB * 0.9:
+        chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, DEFAULT_CHUNK_SIZE)
+
+    return chunk_idx, file_idx, episode_metadata
+
+
+def _copy_data_with_feature_changes(
+    dataset: LeRobotDataset,
+    new_meta: LeRobotDatasetMetadata,
+    add_features: dict[str, tuple] | None = None,
+    remove_features: list[str] | None = None,
+) -> None:
+    """Copy data while adding or removing features."""
+    data_dir = dataset.root / DATA_DIR
+    parquet_files = sorted(data_dir.glob("*/*.parquet"))
+
+    if not parquet_files:
+        raise ValueError(f"No parquet files found in {data_dir}")
+
+    frame_idx = 0
+
+    for src_path in tqdm(parquet_files, desc="Processing data files"):
+        df = pd.read_parquet(src_path).reset_index(drop=True)
+
+        relative_path = src_path.relative_to(dataset.root)
+        chunk_dir = relative_path.parts[1]
+        file_name = relative_path.parts[2]
+
+        chunk_idx = int(chunk_dir.split("-")[1])
+        file_idx = int(file_name.split("-")[1].split(".")[0])
+
+        if remove_features:
+            df = df.drop(columns=remove_features, errors="ignore")
+
+        if add_features:
+            end_idx = frame_idx + len(df)
+            for feature_name, (values, _) in add_features.items():
+                if callable(values):
+                    feature_values = []
+                    for _, row in df.iterrows():
+                        ep_idx = row["episode_index"]
+                        frame_in_ep = row["frame_index"]
+                        value = values(row.to_dict(), ep_idx, frame_in_ep)
+                        if isinstance(value, np.ndarray) and value.size == 1:
+                            value = value.item()
+                        feature_values.append(value)
+                    df[feature_name] = feature_values
+                else:
+                    feature_slice = values[frame_idx:end_idx]
+                    if len(feature_slice.shape) > 1 and feature_slice.shape[1] == 1:
+                        df[feature_name] = feature_slice.flatten()
+                    else:
+                        df[feature_name] = feature_slice
+            frame_idx = end_idx
+
+        # Write using the same chunk/file structure as source
+        dst_path = new_meta.root / DEFAULT_DATA_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
+        dst_path.parent.mkdir(parents=True, exist_ok=True)
+
+        _write_parquet(df, dst_path, new_meta)
+
+    _copy_episodes_metadata_and_stats(dataset, new_meta)
+
+
+def _copy_videos(
+    src_dataset: LeRobotDataset,
+    dst_meta: LeRobotDatasetMetadata,
+    exclude_keys: list[str] | None = None,
+) -> None:
+    """Copy video files, optionally excluding certain keys."""
+    if exclude_keys is None:
+        exclude_keys = []
+
+    for video_key in src_dataset.meta.video_keys:
+        if video_key in exclude_keys:
+            continue
+
+        video_files = set()
+        for ep_idx in range(len(src_dataset.meta.episodes)):
+            try:
+                video_files.add(src_dataset.meta.get_video_file_path(ep_idx, video_key))
+            except KeyError:
+                continue
+
+        for src_path in tqdm(sorted(video_files), desc=f"Copying {video_key} videos"):
+            dst_path = dst_meta.root / src_path
+            dst_path.parent.mkdir(parents=True, exist_ok=True)
+            shutil.copy(src_dataset.root / src_path, dst_path)
+
+
+def _copy_episodes_metadata_and_stats(
+    src_dataset: LeRobotDataset,
+    dst_meta: LeRobotDatasetMetadata,
+) -> None:
+    """Copy episodes metadata and recalculate stats."""
+    if src_dataset.meta.tasks is not None:
+        write_tasks(src_dataset.meta.tasks, dst_meta.root)
+        dst_meta.tasks = src_dataset.meta.tasks.copy()
+
+    episodes_dir = src_dataset.root / "meta/episodes"
+    dst_episodes_dir = dst_meta.root / "meta/episodes"
+    if episodes_dir.exists():
+        shutil.copytree(episodes_dir, dst_episodes_dir, dirs_exist_ok=True)
+
+    dst_meta.info.update(
+        {
+            "total_episodes": src_dataset.meta.total_episodes,
+            "total_frames": src_dataset.meta.total_frames,
+            "total_tasks": src_dataset.meta.total_tasks,
+            "splits": src_dataset.meta.info.get("splits", {"train": f"0:{src_dataset.meta.total_episodes}"}),
+        }
+    )
+
+    if dst_meta.video_keys and src_dataset.meta.video_keys:
+        for key in dst_meta.video_keys:
+            if key in src_dataset.meta.features:
+                dst_meta.info["features"][key]["info"] = src_dataset.meta.info["features"][key].get(
+                    "info", {}
+                )
+
+    write_info(dst_meta.info, dst_meta.root)
+
+    if set(dst_meta.features.keys()) != set(src_dataset.meta.features.keys()):
+        logging.info("Recalculating dataset statistics...")
+        if src_dataset.meta.stats:
+            new_stats = {}
+            for key in dst_meta.features:
+                if key in src_dataset.meta.stats:
+                    new_stats[key] = src_dataset.meta.stats[key]
+            write_stats(new_stats, dst_meta.root)
+    else:
+        if src_dataset.meta.stats:
+            write_stats(src_dataset.meta.stats, dst_meta.root)
+
+
+def _save_episode_images_for_video(
+    dataset: LeRobotDataset,
+    imgs_dir: Path,
+    img_key: str,
+    episode_index: int,
+    num_workers: int = 4,
+) -> None:
+    """Save images from a specific episode and camera to disk for video encoding.
+
+    Args:
+        dataset: The LeRobot dataset to extract images from
+        imgs_dir: Directory to save images to
+        img_key: The image key (camera) to extract
+        episode_index: Index of the episode to save
+        num_workers: Number of threads for parallel image saving
+    """
+    # Create directory
+    imgs_dir.mkdir(parents=True, exist_ok=True)
+
+    # Get dataset without torch format for PIL image access
+    hf_dataset = dataset.hf_dataset.with_format(None)
+
+    # Select only this camera's images
+    imgs_dataset = hf_dataset.select_columns(img_key)
+
+    # Get episode start and end indices
+    from_idx = dataset.meta.episodes["dataset_from_index"][episode_index]
+    to_idx = dataset.meta.episodes["dataset_to_index"][episode_index]
+
+    # Get all items for this episode
+    episode_dataset = imgs_dataset.select(range(from_idx, to_idx))
+
+    # Define function to save a single image
+    def save_single_image(i_item_tuple):
+        i, item = i_item_tuple
+        img = item[img_key]
+        # Use frame-XXXXXX.png format to match encode_video_frames expectations
+        img.save(str(imgs_dir / f"frame-{i:06d}.png"), quality=100)
+        return i
+
+    # Save images with proper naming convention for encode_video_frames (frame-XXXXXX.png)
+    items = list(enumerate(episode_dataset))
+
+    with ThreadPoolExecutor(max_workers=num_workers) as executor:
+        futures = [executor.submit(save_single_image, item) for item in items]
+        for future in as_completed(futures):
+            future.result()  # This will raise any exceptions that occurred
+
+
+def _save_batch_episodes_images(
+    dataset: LeRobotDataset,
+    imgs_dir: Path,
+    img_key: str,
+    episode_indices: list[int],
+    num_workers: int = 4,
+) -> list[float]:
+    """Save images from multiple episodes to disk for batch video encoding.
+
+    Args:
+        dataset: The LeRobot dataset to extract images from
+        imgs_dir: Directory to save images to
+        img_key: The image key (camera) to extract
+        episode_indices: List of episode indices to save
+        num_workers: Number of threads for parallel image saving
+
+    Returns:
+        List of episode durations in seconds
+    """
+    imgs_dir.mkdir(parents=True, exist_ok=True)
+    hf_dataset = dataset.hf_dataset.with_format(None)
+    imgs_dataset = hf_dataset.select_columns(img_key)
+
+    # Define function to save a single image with global frame index
+    # Defined once outside the loop to avoid repeated closure creation
+    def save_single_image(i_item_tuple, base_frame_idx, img_key_param):
+        i, item = i_item_tuple
+        img = item[img_key_param]
+        # Use global frame index for naming
+        img.save(str(imgs_dir / f"frame-{base_frame_idx + i:06d}.png"), quality=100)
+        return i
+
+    episode_durations = []
+    frame_idx = 0
+
+    for ep_idx in episode_indices:
+        # Get episode range
+        from_idx = dataset.meta.episodes["dataset_from_index"][ep_idx]
+        to_idx = dataset.meta.episodes["dataset_to_index"][ep_idx]
+        episode_length = to_idx - from_idx
+        episode_durations.append(episode_length / dataset.fps)
+
+        # Get episode images
+        episode_dataset = imgs_dataset.select(range(from_idx, to_idx))
+
+        # Save images
+        items = list(enumerate(episode_dataset))
+        with ThreadPoolExecutor(max_workers=num_workers) as executor:
+            futures = [executor.submit(save_single_image, item, frame_idx, img_key) for item in items]
+            for future in as_completed(futures):
+                future.result()
+
+        frame_idx += episode_length
+
+    return episode_durations
+
+
+def _iter_episode_batches(
+    episode_indices: list[int],
+    episode_lengths: dict[int, int],
+    size_per_frame_mb: float,
+    video_file_size_limit: float,
+    max_episodes: int | None,
+    max_frames: int | None,
+):
+    """Generator that yields batches of episode indices for video encoding.
+
+    Groups episodes into batches that respect size and memory constraints:
+    - Stays under video file size limit
+    - Respects maximum episodes per batch (if specified)
+    - Respects maximum frames per batch (if specified)
+
+    Args:
+        episode_indices: List of episode indices to batch
+        episode_lengths: Dictionary mapping episode index to episode length
+        size_per_frame_mb: Estimated size per frame in MB
+        video_file_size_limit: Maximum video file size in MB
+        max_episodes: Maximum number of episodes per batch (None = no limit)
+        max_frames: Maximum number of frames per batch (None = no limit)
+
+    Yields:
+        List of episode indices for each batch
+    """
+    batch_episodes = []
+    estimated_size = 0.0
+    total_frames = 0
+
+    for ep_idx in episode_indices:
+        ep_length = episode_lengths[ep_idx]
+        ep_estimated_size = ep_length * size_per_frame_mb
+
+        # we check if adding this episode would exceed any constraint
+        would_exceed_size = estimated_size > 0 and estimated_size + ep_estimated_size >= video_file_size_limit
+        would_exceed_episodes = max_episodes is not None and len(batch_episodes) >= max_episodes
+        would_exceed_frames = max_frames is not None and total_frames + ep_length > max_frames
+
+        if batch_episodes and (would_exceed_size or would_exceed_episodes or would_exceed_frames):
+            # yield current batch before adding this episode
+            yield batch_episodes
+            # start a new batch with current episode
+            batch_episodes = [ep_idx]
+            estimated_size = ep_estimated_size
+            total_frames = ep_length
+        else:
+            # add to current batch
+            batch_episodes.append(ep_idx)
+            estimated_size += ep_estimated_size
+            total_frames += ep_length
+
+    # yield final batch if not empty
+    if batch_episodes:
+        yield batch_episodes
+
+
+def _estimate_frame_size_via_calibration(
+    dataset: LeRobotDataset,
+    img_key: str,
+    episode_indices: list[int],
+    temp_dir: Path,
+    fps: int,
+    vcodec: str,
+    pix_fmt: str,
+    g: int,
+    crf: int,
+    fast_decode: int,
+    num_calibration_frames: int = 30,
+) -> float:
+    """Estimate MB per frame by encoding a small calibration sample.
+
+    Encodes a representative sample of frames using the exact codec parameters
+    to measure actual compression ratio, which is more accurate than heuristics.
+
+    Args:
+        dataset: Source dataset with images.
+        img_key: Image key to calibrate (e.g., "observation.images.top").
+        episode_indices: List of episode indices being processed.
+        temp_dir: Temporary directory for calibration files.
+        fps: Frames per second for video encoding.
+        vcodec: Video codec (libsvtav1, h264, hevc).
+        pix_fmt: Pixel format (yuv420p, etc.).
+        g: GOP size (group of pictures).
+        crf: Constant Rate Factor (quality).
+        fast_decode: Fast decode tuning parameter.
+        num_calibration_frames: Number of frames to use for calibration (default: 30).
+
+    Returns:
+        Estimated size in MB per frame based on actual encoding.
+    """
+    calibration_dir = temp_dir / "calibration" / img_key
+    calibration_dir.mkdir(parents=True, exist_ok=True)
+
+    try:
+        # Select a representative episode (prefer middle episode if available)
+        calibration_ep_idx = episode_indices[len(episode_indices) // 2]
+
+        # Get episode range
+        from_idx = dataset.meta.episodes["dataset_from_index"][calibration_ep_idx]
+        to_idx = dataset.meta.episodes["dataset_to_index"][calibration_ep_idx]
+        episode_length = to_idx - from_idx
+
+        # Use up to num_calibration_frames from this episode
+        num_frames = min(num_calibration_frames, episode_length)
+
+        # Get frames from dataset
+        hf_dataset = dataset.hf_dataset.with_format(None)
+        sample_indices = range(from_idx, from_idx + num_frames)
+
+        # Save calibration frames
+        for i, idx in enumerate(sample_indices):
+            img = hf_dataset[idx][img_key]
+            img.save(str(calibration_dir / f"frame-{i:06d}.png"), quality=100)
+
+        # Encode calibration video
+        calibration_video_path = calibration_dir / "calibration.mp4"
+        encode_video_frames(
+            imgs_dir=calibration_dir,
+            video_path=calibration_video_path,
+            fps=fps,
+            vcodec=vcodec,
+            pix_fmt=pix_fmt,
+            g=g,
+            crf=crf,
+            fast_decode=fast_decode,
+            overwrite=True,
+        )
+
+        # Measure actual compressed size
+        video_size_bytes = calibration_video_path.stat().st_size
+        video_size_mb = video_size_bytes / BYTES_PER_MIB
+        size_per_frame_mb = video_size_mb / num_frames
+
+        logging.info(
+            f"  Calibration: {num_frames} frames -> {video_size_mb:.2f} MB "
+            f"= {size_per_frame_mb:.4f} MB/frame for {img_key}"
+        )
+
+        return size_per_frame_mb
+
+    finally:
+        # Clean up calibration files
+        if calibration_dir.exists():
+            shutil.rmtree(calibration_dir)
+
+
+def _copy_data_without_images(
+    src_dataset: LeRobotDataset,
+    dst_meta: LeRobotDatasetMetadata,
+    episode_indices: list[int],
+    img_keys: list[str],
+) -> None:
+    """Copy data files without image columns.
+
+    Args:
+        src_dataset: Source dataset
+        dst_meta: Destination metadata
+        episode_indices: Episodes to include
+        img_keys: Image keys to remove
+    """
+    from lerobot.datasets.utils import DATA_DIR
+
+    data_dir = src_dataset.root / DATA_DIR
+    parquet_files = sorted(data_dir.glob("*/*.parquet"))
+
+    if not parquet_files:
+        raise ValueError(f"No parquet files found in {data_dir}")
+
+    episode_set = set(episode_indices)
+
+    for src_path in tqdm(parquet_files, desc="Processing data files"):
+        df = pd.read_parquet(src_path).reset_index(drop=True)
+
+        # Filter to only include selected episodes
+        df = df[df["episode_index"].isin(episode_set)].copy()
+
+        if len(df) == 0:
+            continue
+
+        # Remove image columns
+        columns_to_drop = [col for col in img_keys if col in df.columns]
+        if columns_to_drop:
+            df = df.drop(columns=columns_to_drop)
+
+        # Get chunk and file indices from path
+        relative_path = src_path.relative_to(src_dataset.root)
+        chunk_dir = relative_path.parts[1]
+        file_name = relative_path.parts[2]
+        chunk_idx = int(chunk_dir.split("-")[1])
+        file_idx = int(file_name.split("-")[1].split(".")[0])
+
+        # Write to destination without pandas index
+        dst_path = dst_meta.root / f"data/chunk-{chunk_idx:03d}/file-{file_idx:03d}.parquet"
+        dst_path.parent.mkdir(parents=True, exist_ok=True)
+        df.to_parquet(dst_path, index=False)
+
+
+# Video conversion constants
+BYTES_PER_KIB = 1024
+BYTES_PER_MIB = BYTES_PER_KIB * BYTES_PER_KIB
+
+
+def modify_tasks(
+    dataset: LeRobotDataset,
+    new_task: str | None = None,
+    episode_tasks: dict[int, str] | None = None,
+) -> LeRobotDataset:
+    """Modify tasks in a LeRobotDataset.
+
+    This function allows you to either:
+    1. Set a single task for the entire dataset (using `new_task`)
+    2. Set specific tasks for specific episodes (using `episode_tasks`)
+
+    You can combine both: `new_task` sets the default, and `episode_tasks` overrides
+    specific episodes.
+
+    The dataset is modified in-place, updating only the task-related files:
+    - meta/tasks.parquet
+    - data/**/*.parquet (task_index column)
+    - meta/episodes/**/*.parquet (tasks column)
+    - meta/info.json (total_tasks)
+
+    Args:
+        dataset: The source LeRobotDataset to modify.
+        new_task: A single task string to apply to all episodes. If None and episode_tasks
+            is also None, raises an error.
+        episode_tasks: Optional dict mapping episode indices to their task strings.
+            Overrides `new_task` for specific episodes.
+
+
+    Examples:
+        Set a single task for all episodes:
+            dataset = modify_tasks(dataset, new_task="Pick up the cube")
+
+        Set different tasks for specific episodes:
+            dataset = modify_tasks(
+                dataset,
+                episode_tasks={0: "Task A", 1: "Task B", 2: "Task A"}
+            )
+
+        Set a default task with overrides:
+            dataset = modify_tasks(
+                dataset,
+                new_task="Default task",
+                episode_tasks={5: "Special task for episode 5"}
+            )
+    """
+    if new_task is None and episode_tasks is None:
+        raise ValueError("Must specify at least one of new_task or episode_tasks")
+
+    if episode_tasks is not None:
+        valid_indices = set(range(dataset.meta.total_episodes))
+        invalid = set(episode_tasks.keys()) - valid_indices
+        if invalid:
+            raise ValueError(f"Invalid episode indices: {invalid}")
+
+    # Ensure episodes metadata is loaded
+    if dataset.meta.episodes is None:
+        dataset.meta.episodes = load_episodes(dataset.root)
+
+    # Build the mapping from episode index to task string
+    episode_to_task: dict[int, str] = {}
+    for ep_idx in range(dataset.meta.total_episodes):
+        if episode_tasks and ep_idx in episode_tasks:
+            episode_to_task[ep_idx] = episode_tasks[ep_idx]
+        elif new_task is not None:
+            episode_to_task[ep_idx] = new_task
+        else:
+            # Keep original task if not overridden and no default provided
+            original_tasks = dataset.meta.episodes[ep_idx]["tasks"]
+            if not original_tasks:
+                raise ValueError(f"Episode {ep_idx} has no tasks and no default task was provided")
+            episode_to_task[ep_idx] = original_tasks[0]
+
+    # Collect all unique tasks and create new task mapping
+    unique_tasks = sorted(set(episode_to_task.values()))
+    new_task_df = pd.DataFrame(
+        {"task_index": list(range(len(unique_tasks)))}, index=pd.Index(unique_tasks, name="task")
+    )
+    task_to_index = {task: idx for idx, task in enumerate(unique_tasks)}
+
+    logging.info(f"Modifying tasks in {dataset.repo_id}")
+    logging.info(f"New tasks: {unique_tasks}")
+
+    root = dataset.root
+
+    # Update data files - modify task_index column
+    logging.info("Updating data files...")
+    data_dir = root / DATA_DIR
+
+    for parquet_path in tqdm(sorted(data_dir.rglob("*.parquet")), desc="Updating data"):
+        df = pd.read_parquet(parquet_path)
+
+        # Build a mapping from episode_index to new task_index for rows in this file
+        episode_indices_in_file = df["episode_index"].unique()
+        ep_to_new_task_idx = {
+            ep_idx: task_to_index[episode_to_task[ep_idx]] for ep_idx in episode_indices_in_file
+        }
+
+        # Update task_index column
+        df["task_index"] = df["episode_index"].map(ep_to_new_task_idx)
+        df.to_parquet(parquet_path, index=False)
+
+    # Update episodes metadata - modify tasks column
+    logging.info("Updating episodes metadata...")
+    episodes_dir = root / "meta" / "episodes"
+
+    for parquet_path in tqdm(sorted(episodes_dir.rglob("*.parquet")), desc="Updating episodes"):
+        df = pd.read_parquet(parquet_path)
+
+        # Update tasks column
+        df["tasks"] = df["episode_index"].apply(lambda ep_idx: [episode_to_task[ep_idx]])
+        df.to_parquet(parquet_path, index=False)
+
+    # Write new tasks.parquet
+    write_tasks(new_task_df, root)
+
+    # Update info.json
+    dataset.meta.info["total_tasks"] = len(unique_tasks)
+    write_info(dataset.meta.info, root)
+
+    # Reload metadata to reflect changes
+    dataset.meta.tasks = new_task_df
+    dataset.meta.episodes = load_episodes(root)
+
+    logging.info(f"Tasks: {unique_tasks}")
+
+    return dataset
+
+
+def convert_image_to_video_dataset(
+    dataset: LeRobotDataset,
+    output_dir: Path | None = None,
+    repo_id: str | None = None,
+    vcodec: str = "libsvtav1",
+    pix_fmt: str = "yuv420p",
+    g: int = 2,
+    crf: int = 30,
+    fast_decode: int = 0,
+    episode_indices: list[int] | None = None,
+    num_workers: int = 4,
+    max_episodes_per_batch: int | None = None,
+    max_frames_per_batch: int | None = None,
+) -> LeRobotDataset:
+    """Convert image-to-video dataset.
+
+    Creates a new LeRobotDataset with images encoded as videos, following the proper
+    LeRobot dataset structure with videos stored in chunked MP4 files.
+
+    Args:
+        dataset: The source LeRobot dataset with images
+        output_dir: Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id. Equivalent to new_root in EditDatasetConfig.
+        repo_id: Edited dataset identifier. Equivalent to new_repo_id in EditDatasetConfig.
+        vcodec: Video codec (default: libsvtav1)
+        pix_fmt: Pixel format (default: yuv420p)
+        g: Group of pictures size (default: 2)
+        crf: Constant rate factor (default: 30)
+        fast_decode: Fast decode tuning (default: 0)
+        episode_indices: List of episode indices to convert (None = all episodes)
+        num_workers: Number of threads for parallel processing (default: 4)
+        max_episodes_per_batch: Maximum episodes per video batch to avoid memory issues (None = no limit)
+        max_frames_per_batch: Maximum frames per video batch to avoid memory issues (None = no limit)
+
+    Returns:
+        New LeRobotDataset with images encoded as videos
+    """
+    # Check that it's an image dataset
+    if len(dataset.meta.video_keys) > 0:
+        raise ValueError(
+            f"This operation is for image datasets only. Video dataset provided: {dataset.repo_id}"
+        )
+
+    # Get all image keys
+    hf_dataset = dataset.hf_dataset.with_format(None)
+    img_keys = [key for key in hf_dataset.features if key.startswith(OBS_IMAGE)]
+
+    if len(img_keys) == 0:
+        raise ValueError(f"No image keys found in dataset {dataset.repo_id}")
+
+    # Determine which episodes to process
+    if episode_indices is None:
+        episode_indices = list(range(dataset.meta.total_episodes))
+
+    if repo_id is None:
+        repo_id = f"{dataset.repo_id}_video"
+
+    logging.info(
+        f"Converting {len(episode_indices)} episodes with {len(img_keys)} cameras from {dataset.repo_id}"
+    )
+    logging.info(f"Video codec: {vcodec}, pixel format: {pix_fmt}, GOP: {g}, CRF: {crf}")
+
+    # Create new features dict, converting image features to video features
+    new_features = {}
+    for key, value in dataset.meta.features.items():
+        if key not in img_keys:
+            new_features[key] = value
+        else:
+            # Convert image key to video format
+            new_features[key] = value.copy()
+            new_features[key]["dtype"] = "video"  # Change dtype from "image" to "video"
+            # Video info will be updated after episodes are encoded
+
+    # Create new metadata for video dataset
+    output_dir = Path(output_dir) if output_dir is not None else HF_LEROBOT_HOME / repo_id
+    new_meta = LeRobotDatasetMetadata.create(
+        repo_id=repo_id,
+        fps=dataset.meta.fps,
+        features=new_features,
+        robot_type=dataset.meta.robot_type,
+        root=output_dir,
+        use_videos=True,
+        chunks_size=dataset.meta.chunks_size,
+        data_files_size_in_mb=dataset.meta.data_files_size_in_mb,
+        video_files_size_in_mb=dataset.meta.video_files_size_in_mb,
+    )
+
+    # Create temporary directory for image extraction
+    temp_dir = output_dir / "temp_images"
+    temp_dir.mkdir(parents=True, exist_ok=True)
+
+    # Process all episodes and batch encode videos
+    # Use dictionary for O(1) episode metadata lookups instead of O(n) linear search
+    all_episode_metadata = {}
+    fps = int(dataset.fps)
+
+    try:
+        # Build episode metadata entries first
+        logging.info("Building episode metadata...")
+        cumulative_frame_idx = 0
+        for ep_idx in episode_indices:
+            src_episode = dataset.meta.episodes[ep_idx]
+            ep_length = src_episode["length"]
+            ep_meta = {
+                "episode_index": ep_idx,
+                "length": ep_length,
+                "dataset_from_index": cumulative_frame_idx,
+                "dataset_to_index": cumulative_frame_idx + ep_length,
+            }
+            if "data/chunk_index" in src_episode:
+                ep_meta["data/chunk_index"] = src_episode["data/chunk_index"]
+                ep_meta["data/file_index"] = src_episode["data/file_index"]
+            all_episode_metadata[ep_idx] = ep_meta
+            cumulative_frame_idx += ep_length
+
+        # Process each camera and batch encode multiple episodes together
+        video_file_size_limit = new_meta.video_files_size_in_mb
+
+        # Pre-compute episode lengths for batching
+        episode_lengths = {ep_idx: dataset.meta.episodes["length"][ep_idx] for ep_idx in episode_indices}
+
+        for img_key in tqdm(img_keys, desc="Processing cameras"):
+            # Estimate size per frame by encoding a small calibration sample
+            # This provides accurate compression ratio for the specific codec parameters
+            size_per_frame_mb = _estimate_frame_size_via_calibration(
+                dataset=dataset,
+                img_key=img_key,
+                episode_indices=episode_indices,
+                temp_dir=temp_dir,
+                fps=fps,
+                vcodec=vcodec,
+                pix_fmt=pix_fmt,
+                g=g,
+                crf=crf,
+                fast_decode=fast_decode,
+            )
+
+            logging.info(f"Processing camera: {img_key}")
+            chunk_idx, file_idx = 0, 0
+            cumulative_timestamp = 0.0
+
+            # Process episodes in batches to stay under size limit
+            for batch_episodes in _iter_episode_batches(
+                episode_indices=episode_indices,
+                episode_lengths=episode_lengths,
+                size_per_frame_mb=size_per_frame_mb,
+                video_file_size_limit=video_file_size_limit,
+                max_episodes=max_episodes_per_batch,
+                max_frames=max_frames_per_batch,
+            ):
+                total_frames_in_batch = sum(episode_lengths[idx] for idx in batch_episodes)
+                logging.info(
+                    f"  Encoding batch of {len(batch_episodes)} episodes "
+                    f"({batch_episodes[0]}-{batch_episodes[-1]}) = {total_frames_in_batch} frames"
+                )
+
+                # Save images for all episodes in this batch
+                imgs_dir = temp_dir / f"batch_{chunk_idx}_{file_idx}" / img_key
+                episode_durations = _save_batch_episodes_images(
+                    dataset=dataset,
+                    imgs_dir=imgs_dir,
+                    img_key=img_key,
+                    episode_indices=batch_episodes,
+                    num_workers=num_workers,
+                )
+
+                # Encode all batched episodes into single video
+                video_path = new_meta.root / new_meta.video_path.format(
+                    video_key=img_key, chunk_index=chunk_idx, file_index=file_idx
+                )
+                video_path.parent.mkdir(parents=True, exist_ok=True)
+
+                encode_video_frames(
+                    imgs_dir=imgs_dir,
+                    video_path=video_path,
+                    fps=fps,
+                    vcodec=vcodec,
+                    pix_fmt=pix_fmt,
+                    g=g,
+                    crf=crf,
+                    fast_decode=fast_decode,
+                    overwrite=True,
+                )
+
+                # Clean up temporary images
+                shutil.rmtree(imgs_dir)
+
+                # Update metadata for each episode in the batch
+                for ep_idx, duration in zip(batch_episodes, episode_durations, strict=True):
+                    from_timestamp = cumulative_timestamp
+                    to_timestamp = cumulative_timestamp + duration
+                    cumulative_timestamp = to_timestamp
+
+                    # Find episode metadata entry and add video metadata (O(1) dictionary lookup)
+                    ep_meta = all_episode_metadata[ep_idx]
+                    ep_meta[f"videos/{img_key}/chunk_index"] = chunk_idx
+                    ep_meta[f"videos/{img_key}/file_index"] = file_idx
+                    ep_meta[f"videos/{img_key}/from_timestamp"] = from_timestamp
+                    ep_meta[f"videos/{img_key}/to_timestamp"] = to_timestamp
+
+                # Move to next video file for next batch
+                chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, new_meta.chunks_size)
+                cumulative_timestamp = 0.0
+
+        # Copy and transform data files (removing image columns)
+        _copy_data_without_images(dataset, new_meta, episode_indices, img_keys)
+
+        # Save episode metadata
+        episodes_df = pd.DataFrame(list(all_episode_metadata.values()))
+        episodes_path = new_meta.root / "meta" / "episodes" / "chunk-000" / "file-000.parquet"
+        episodes_path.parent.mkdir(parents=True, exist_ok=True)
+        episodes_df.to_parquet(episodes_path, index=False)
+
+        # Update metadata info
+        new_meta.info["total_episodes"] = len(episode_indices)
+        new_meta.info["total_frames"] = sum(ep["length"] for ep in all_episode_metadata.values())
+        new_meta.info["total_tasks"] = dataset.meta.total_tasks
+        new_meta.info["splits"] = {"train": f"0:{len(episode_indices)}"}
+
+        # Update video info for all image keys (now videos)
+        # We need to manually set video info since update_video_info() checks video_keys first
+        for img_key in img_keys:
+            if not new_meta.features[img_key].get("info", None):
+                video_path = new_meta.root / new_meta.video_path.format(
+                    video_key=img_key, chunk_index=0, file_index=0
+                )
+                new_meta.info["features"][img_key]["info"] = get_video_info(video_path)
+
+        write_info(new_meta.info, new_meta.root)
+
+        # Copy stats and tasks
+        if dataset.meta.stats is not None:
+            # Remove image stats
+            new_stats = {k: v for k, v in dataset.meta.stats.items() if k not in img_keys}
+            write_stats(new_stats, new_meta.root)
+
+        if dataset.meta.tasks is not None:
+            write_tasks(dataset.meta.tasks, new_meta.root)
+
+    finally:
+        # Clean up temporary directory
+        if temp_dir.exists():
+            shutil.rmtree(temp_dir)
+
+    logging.info(f"Completed converting {dataset.repo_id} to video format")
+    logging.info(f"New dataset saved to: {output_dir}")
+
+    # Return new dataset
+    return LeRobotDataset(repo_id=repo_id, root=output_dir)
diff --git a/lerobot/src/lerobot/datasets/factory.py b/lerobot/src/lerobot/datasets/factory.py
new file mode 100644
index 0000000000000000000000000000000000000000..4fbaab0a75a2a57408935106a8e8f87d8561f80c
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/factory.py
@@ -0,0 +1,154 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+from pprint import pformat
+
+import torch
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.multi_dataset import MultiLeRobotDataset
+from lerobot.datasets.streaming_dataset import StreamingLeRobotDataset
+from lerobot.datasets.transforms import ImageTransforms
+from lerobot.utils.constants import ACTION, OBS_PREFIX, REWARD
+
+IMAGENET_STATS = {
+    "mean": [[[0.485]], [[0.456]], [[0.406]]],  # (c,1,1)
+    "std": [[[0.229]], [[0.224]], [[0.225]]],  # (c,1,1)
+}
+
+
+def resolve_delta_timestamps(
+    cfg: PreTrainedConfig, ds_meta: LeRobotDatasetMetadata
+) -> dict[str, list] | None:
+    """Resolves delta_timestamps by reading from the 'delta_indices' properties of the PreTrainedConfig.
+
+    Args:
+        cfg (PreTrainedConfig): The PreTrainedConfig to read delta_indices from.
+        ds_meta (LeRobotDatasetMetadata): The dataset from which features and fps are used to build
+            delta_timestamps against.
+
+    Returns:
+        dict[str, list] | None: A dictionary of delta_timestamps, e.g.:
+            {
+                "observation.state": [-0.04, -0.02, 0]
+                "observation.action": [-0.02, 0, 0.02]
+            }
+            returns `None` if the resulting dict is empty.
+    """
+    delta_timestamps = {}
+    for key in ds_meta.features:
+        if key == REWARD and cfg.reward_delta_indices is not None:
+            delta_timestamps[key] = [i / ds_meta.fps for i in cfg.reward_delta_indices]
+        if key == ACTION and cfg.action_delta_indices is not None:
+            delta_timestamps[key] = [i / ds_meta.fps for i in cfg.action_delta_indices]
+        if key.startswith(OBS_PREFIX) and cfg.observation_delta_indices is not None:
+            delta_timestamps[key] = [i / ds_meta.fps for i in cfg.observation_delta_indices]
+
+    if len(delta_timestamps) == 0:
+        delta_timestamps = None
+
+    return delta_timestamps
+
+
+def make_dataset(cfg: TrainPipelineConfig) -> LeRobotDataset | MultiLeRobotDataset:
+    """Handles the logic of setting up delta timestamps and image transforms before creating a dataset.
+
+    Args:
+        cfg (TrainPipelineConfig): A TrainPipelineConfig config which contains a DatasetConfig and a PreTrainedConfig.
+
+    Raises:
+        NotImplementedError: The MultiLeRobotDataset is currently deactivated.
+
+    Returns:
+        LeRobotDataset | MultiLeRobotDataset
+    """
+    image_transforms = (
+        ImageTransforms(cfg.dataset.image_transforms) if cfg.dataset.image_transforms.enable else None
+    )
+
+    # Support SO100Dataset via repo_id starting with "so100:"
+    # Format: "so100:/path/to/data_root:/path/to/index.json:/path/to/stats.json"
+    if isinstance(cfg.dataset.repo_id, str) and cfg.dataset.repo_id.startswith("so100:"):
+        from so100_dataset import SO100Dataset
+
+        parts = cfg.dataset.repo_id.split(":")
+        data_root = parts[1]
+        index_path = parts[2] if len(parts) > 2 else None
+        stats_path = parts[3] if len(parts) > 3 else None
+        # Set rename_map for SO100 cameras -> pi05 expected names
+        if not cfg.rename_map:
+            cfg.rename_map = {
+                "observation.images.image": "observation.images.base_0_rgb",
+                "observation.images.image2": "observation.images.left_wrist_0_rgb",
+            }
+        dataset = SO100Dataset(
+            data_root=data_root,
+            index_path=index_path,
+            stats_path=stats_path,
+            image_transforms=image_transforms,
+        )
+        return dataset
+
+    if isinstance(cfg.dataset.repo_id, str):
+        ds_meta = LeRobotDatasetMetadata(
+            cfg.dataset.repo_id, root=cfg.dataset.root, revision=cfg.dataset.revision
+        )
+        delta_timestamps = resolve_delta_timestamps(cfg.policy, ds_meta)
+        if not cfg.dataset.streaming:
+            dataset = LeRobotDataset(
+                cfg.dataset.repo_id,
+                root=cfg.dataset.root,
+                episodes=cfg.dataset.episodes,
+                delta_timestamps=delta_timestamps,
+                image_transforms=image_transforms,
+                revision=cfg.dataset.revision,
+                video_backend=cfg.dataset.video_backend,
+                tolerance_s=cfg.tolerance_s,
+            )
+        else:
+            dataset = StreamingLeRobotDataset(
+                cfg.dataset.repo_id,
+                root=cfg.dataset.root,
+                episodes=cfg.dataset.episodes,
+                delta_timestamps=delta_timestamps,
+                image_transforms=image_transforms,
+                revision=cfg.dataset.revision,
+                max_num_shards=cfg.num_workers,
+                tolerance_s=cfg.tolerance_s,
+            )
+    else:
+        raise NotImplementedError("The MultiLeRobotDataset isn't supported for now.")
+        dataset = MultiLeRobotDataset(
+            cfg.dataset.repo_id,
+            # TODO(aliberts): add proper support for multi dataset
+            # delta_timestamps=delta_timestamps,
+            image_transforms=image_transforms,
+            video_backend=cfg.dataset.video_backend,
+        )
+        logging.info(
+            "Multiple datasets were provided. Applied the following index mapping to the provided datasets: "
+            f"{pformat(dataset.repo_id_to_index, indent=2)}"
+        )
+
+    if cfg.dataset.use_imagenet_stats:
+        for key in dataset.meta.camera_keys:
+            for stats_type, stats in IMAGENET_STATS.items():
+                dataset.meta.stats[key][stats_type] = torch.tensor(stats, dtype=torch.float32)
+
+    return dataset
diff --git a/lerobot/src/lerobot/datasets/feature_utils.py b/lerobot/src/lerobot/datasets/feature_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..d9a3c63015dc5ca472139132a4484b44293044fb
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/feature_utils.py
@@ -0,0 +1,552 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from pprint import pformat
+from typing import Any
+
+import datasets
+import numpy as np
+from PIL import Image as PILImage
+
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.datasets.utils import (
+    DEFAULT_CHUNK_SIZE,
+    DEFAULT_DATA_FILE_SIZE_IN_MB,
+    DEFAULT_DATA_PATH,
+    DEFAULT_FEATURES,
+    DEFAULT_VIDEO_FILE_SIZE_IN_MB,
+    DEFAULT_VIDEO_PATH,
+)
+from lerobot.utils.constants import ACTION, OBS_ENV_STATE, OBS_STR
+from lerobot.utils.utils import is_valid_numpy_dtype_string
+
+
+def get_hf_features_from_features(features: dict) -> datasets.Features:
+    """Convert a LeRobot features dictionary to a `datasets.Features` object.
+
+    Args:
+        features (dict): A LeRobot-style feature dictionary.
+
+    Returns:
+        datasets.Features: The corresponding Hugging Face `datasets.Features` object.
+
+    Raises:
+        ValueError: If a feature has an unsupported shape.
+    """
+    hf_features = {}
+    for key, ft in features.items():
+        if ft["dtype"] == "video":
+            continue
+        elif ft["dtype"] == "image":
+            hf_features[key] = datasets.Image()
+        elif ft["shape"] == (1,):
+            hf_features[key] = datasets.Value(dtype=ft["dtype"])
+        elif len(ft["shape"]) == 1:
+            hf_features[key] = datasets.Sequence(
+                length=ft["shape"][0], feature=datasets.Value(dtype=ft["dtype"])
+            )
+        elif len(ft["shape"]) == 2:
+            hf_features[key] = datasets.Array2D(shape=ft["shape"], dtype=ft["dtype"])
+        elif len(ft["shape"]) == 3:
+            hf_features[key] = datasets.Array3D(shape=ft["shape"], dtype=ft["dtype"])
+        elif len(ft["shape"]) == 4:
+            hf_features[key] = datasets.Array4D(shape=ft["shape"], dtype=ft["dtype"])
+        elif len(ft["shape"]) == 5:
+            hf_features[key] = datasets.Array5D(shape=ft["shape"], dtype=ft["dtype"])
+        else:
+            raise ValueError(f"Corresponding feature is not valid: {ft}")
+
+    return datasets.Features(hf_features)
+
+
+def _validate_feature_names(features: dict[str, dict]) -> None:
+    """Validate that feature names do not contain invalid characters.
+
+    Args:
+        features (dict): The LeRobot features dictionary.
+
+    Raises:
+        ValueError: If any feature name contains '/'.
+    """
+    invalid_features = {name: ft for name, ft in features.items() if "/" in name}
+    if invalid_features:
+        raise ValueError(f"Feature names should not contain '/'. Found '/' in '{invalid_features}'.")
+
+
+def hw_to_dataset_features(
+    hw_features: dict[str, type | tuple], prefix: str, use_video: bool = True
+) -> dict[str, dict]:
+    """Convert hardware-specific features to a LeRobot dataset feature dictionary.
+
+    This function takes a dictionary describing hardware outputs (like joint states
+    or camera image shapes) and formats it into the standard LeRobot feature
+    specification.
+
+    Args:
+        hw_features (dict): Dictionary mapping feature names to their type (float for
+            joints) or shape (tuple for images).
+        prefix (str): The prefix to add to the feature keys (e.g., "observation"
+            or "action").
+        use_video (bool): If True, image features are marked as "video", otherwise "image".
+
+    Returns:
+        dict: A LeRobot features dictionary.
+    """
+    features = {}
+    joint_fts = {
+        key: ftype
+        for key, ftype in hw_features.items()
+        if ftype is float or (isinstance(ftype, PolicyFeature) and ftype.type != FeatureType.VISUAL)
+    }
+    cam_fts = {key: shape for key, shape in hw_features.items() if isinstance(shape, tuple)}
+
+    if joint_fts and prefix == ACTION:
+        features[prefix] = {
+            "dtype": "float32",
+            "shape": (len(joint_fts),),
+            "names": list(joint_fts),
+        }
+
+    if joint_fts and prefix == OBS_STR:
+        features[f"{prefix}.state"] = {
+            "dtype": "float32",
+            "shape": (len(joint_fts),),
+            "names": list(joint_fts),
+        }
+
+    for key, shape in cam_fts.items():
+        features[f"{prefix}.images.{key}"] = {
+            "dtype": "video" if use_video else "image",
+            "shape": shape,
+            "names": ["height", "width", "channels"],
+        }
+
+    _validate_feature_names(features)
+    return features
+
+
+def build_dataset_frame(
+    ds_features: dict[str, dict], values: dict[str, Any], prefix: str
+) -> dict[str, np.ndarray]:
+    """Construct a single data frame from raw values based on dataset features.
+
+    A "frame" is a dictionary containing all the data for a single timestep,
+    formatted as numpy arrays according to the feature specification.
+
+    Args:
+        ds_features (dict): The LeRobot dataset features dictionary.
+        values (dict): A dictionary of raw values from the hardware/environment.
+        prefix (str): The prefix to filter features by (e.g., "observation"
+            or "action").
+
+    Returns:
+        dict: A dictionary representing a single frame of data.
+    """
+    frame = {}
+    for key, ft in ds_features.items():
+        if key in DEFAULT_FEATURES or not key.startswith(prefix):
+            continue
+        elif ft["dtype"] == "float32" and len(ft["shape"]) == 1:
+            frame[key] = np.array([values[name] for name in ft["names"]], dtype=np.float32)
+        elif ft["dtype"] in ["image", "video"]:
+            frame[key] = values[key.removeprefix(f"{prefix}.images.")]
+
+    return frame
+
+
+def dataset_to_policy_features(features: dict[str, dict]) -> dict[str, PolicyFeature]:
+    """Convert dataset features to policy features.
+
+    This function transforms the dataset's feature specification into a format
+    that a policy can use, classifying features by type (e.g., visual, state,
+    action) and ensuring correct shapes (e.g., channel-first for images).
+
+    Args:
+        features (dict): The LeRobot dataset features dictionary.
+
+    Returns:
+        dict: A dictionary mapping feature keys to `PolicyFeature` objects.
+
+    Raises:
+        ValueError: If an image feature does not have a 3D shape.
+    """
+    # TODO(aliberts): Implement "type" in dataset features and simplify this
+    policy_features = {}
+    for key, ft in features.items():
+        shape = ft["shape"]
+        if ft["dtype"] in ["image", "video"]:
+            type = FeatureType.VISUAL
+            if len(shape) != 3:
+                raise ValueError(f"Number of dimensions of {key} != 3 (shape={shape})")
+
+            names = ft["names"]
+            # Backward compatibility for "channel" which is an error introduced in LeRobotDataset v2.0 for ported datasets.
+            if names[2] in ["channel", "channels"]:  # (h, w, c) -> (c, h, w)
+                shape = (shape[2], shape[0], shape[1])
+        elif key == OBS_ENV_STATE:
+            type = FeatureType.ENV
+        elif key.startswith(OBS_STR):
+            type = FeatureType.STATE
+        elif key.startswith(ACTION):
+            type = FeatureType.ACTION
+        else:
+            continue
+
+        policy_features[key] = PolicyFeature(
+            type=type,
+            shape=shape,
+        )
+
+    return policy_features
+
+
+def combine_feature_dicts(*dicts: dict) -> dict:
+    """Merge LeRobot grouped feature dicts.
+
+    - For 1D numeric specs (dtype not image/video/string) with "names": we merge the names and recompute the shape.
+    - For others (e.g. `observation.images.*`), the last one wins (if they are identical).
+
+    Args:
+        *dicts: A variable number of LeRobot feature dictionaries to merge.
+
+    Returns:
+        dict: A single merged feature dictionary.
+
+    Raises:
+        ValueError: If there's a dtype mismatch for a feature being merged.
+    """
+    out: dict = {}
+    for d in dicts:
+        for key, value in d.items():
+            if not isinstance(value, dict):
+                out[key] = value
+                continue
+
+            dtype = value.get("dtype")
+            shape = value.get("shape")
+            is_vector = (
+                dtype not in ("image", "video", "string")
+                and isinstance(shape, tuple)
+                and len(shape) == 1
+                and "names" in value
+            )
+
+            if is_vector:
+                # Initialize or retrieve the accumulating dict for this feature key
+                target = out.setdefault(key, {"dtype": dtype, "names": [], "shape": (0,)})
+                # Ensure consistent data types across merged entries
+                if "dtype" in target and dtype != target["dtype"]:
+                    raise ValueError(f"dtype mismatch for '{key}': {target['dtype']} vs {dtype}")
+
+                # Merge feature names: append only new ones to preserve order without duplicates
+                seen = set(target["names"])
+                for n in value["names"]:
+                    if n not in seen:
+                        target["names"].append(n)
+                        seen.add(n)
+                # Recompute the shape to reflect the updated number of features
+                target["shape"] = (len(target["names"]),)
+            else:
+                # For images/videos and non-1D entries: override with the latest definition
+                out[key] = value
+    return out
+
+
+def create_empty_dataset_info(
+    codebase_version: str,
+    fps: int,
+    features: dict,
+    use_videos: bool,
+    robot_type: str | None = None,
+    chunks_size: int | None = None,
+    data_files_size_in_mb: int | None = None,
+    video_files_size_in_mb: int | None = None,
+) -> dict:
+    """Create a template dictionary for a new dataset's `info.json`.
+
+    Args:
+        codebase_version (str): The version of the LeRobot codebase.
+        fps (int): The frames per second of the data.
+        features (dict): The LeRobot features dictionary for the dataset.
+        use_videos (bool): Whether the dataset will store videos.
+        robot_type (str | None): The type of robot used, if any.
+
+    Returns:
+        dict: A dictionary with the initial dataset metadata.
+    """
+    return {
+        "codebase_version": codebase_version,
+        "robot_type": robot_type,
+        "total_episodes": 0,
+        "total_frames": 0,
+        "total_tasks": 0,
+        "chunks_size": chunks_size or DEFAULT_CHUNK_SIZE,
+        "data_files_size_in_mb": data_files_size_in_mb or DEFAULT_DATA_FILE_SIZE_IN_MB,
+        "video_files_size_in_mb": video_files_size_in_mb or DEFAULT_VIDEO_FILE_SIZE_IN_MB,
+        "fps": fps,
+        "splits": {},
+        "data_path": DEFAULT_DATA_PATH,
+        "video_path": DEFAULT_VIDEO_PATH if use_videos else None,
+        "features": features,
+    }
+
+
+def check_delta_timestamps(
+    delta_timestamps: dict[str, list[float]], fps: int, tolerance_s: float, raise_value_error: bool = True
+) -> bool:
+    """Check if delta timestamps are multiples of 1/fps +/- tolerance.
+
+    This ensures that adding these delta timestamps to any existing timestamp in
+    the dataset will result in a value that aligns with the dataset's frame rate.
+
+    Args:
+        delta_timestamps (dict): A dictionary where values are lists of time
+            deltas in seconds.
+        fps (int): The frames per second of the dataset.
+        tolerance_s (float): The allowed tolerance in seconds.
+        raise_value_error (bool): If True, raises an error on failure.
+
+    Returns:
+        bool: True if all deltas are valid, False otherwise.
+
+    Raises:
+        ValueError: If any delta is outside the tolerance and `raise_value_error` is True.
+    """
+    outside_tolerance = {}
+    for key, delta_ts in delta_timestamps.items():
+        within_tolerance = [abs(ts * fps - round(ts * fps)) / fps <= tolerance_s for ts in delta_ts]
+        if not all(within_tolerance):
+            outside_tolerance[key] = [
+                ts for ts, is_within in zip(delta_ts, within_tolerance, strict=True) if not is_within
+            ]
+
+    if len(outside_tolerance) > 0:
+        if raise_value_error:
+            raise ValueError(
+                f"""
+                The following delta_timestamps are found outside of tolerance range.
+                Please make sure they are multiples of 1/{fps} +/- tolerance and adjust
+                their values accordingly.
+                \n{pformat(outside_tolerance)}
+                """
+            )
+        return False
+
+    return True
+
+
+def get_delta_indices(delta_timestamps: dict[str, list[float]], fps: int) -> dict[str, list[int]]:
+    """Convert delta timestamps in seconds to delta indices in frames.
+
+    Args:
+        delta_timestamps (dict): A dictionary of time deltas in seconds.
+        fps (int): The frames per second of the dataset.
+
+    Returns:
+        dict: A dictionary of frame delta indices.
+    """
+    delta_indices = {}
+    for key, delta_ts in delta_timestamps.items():
+        delta_indices[key] = [round(d * fps) for d in delta_ts]
+
+    return delta_indices
+
+
+def validate_frame(frame: dict, features: dict) -> None:
+    expected_features = set(features) - set(DEFAULT_FEATURES)
+    actual_features = set(frame)
+
+    # task is a special required field that's not part of regular features
+    if "task" not in actual_features:
+        raise ValueError("Feature mismatch in `frame` dictionary:\nMissing features: {'task'}\n")
+
+    # Remove task from actual_features for regular feature validation
+    actual_features_for_validation = actual_features - {"task"}
+
+    error_message = validate_features_presence(actual_features_for_validation, expected_features)
+
+    common_features = actual_features_for_validation & expected_features
+    for name in common_features:
+        error_message += validate_feature_dtype_and_shape(name, features[name], frame[name])
+
+    if error_message:
+        raise ValueError(error_message)
+
+
+def validate_features_presence(actual_features: set[str], expected_features: set[str]) -> str:
+    """Check for missing or extra features in a frame.
+
+    Args:
+        actual_features (set[str]): The set of feature names present in the frame.
+        expected_features (set[str]): The set of feature names expected in the frame.
+
+    Returns:
+        str: An error message string if there's a mismatch, otherwise an empty string.
+    """
+    error_message = ""
+    missing_features = expected_features - actual_features
+    extra_features = actual_features - expected_features
+
+    if missing_features or extra_features:
+        error_message += "Feature mismatch in `frame` dictionary:\n"
+        if missing_features:
+            error_message += f"Missing features: {missing_features}\n"
+        if extra_features:
+            error_message += f"Extra features: {extra_features}\n"
+
+    return error_message
+
+
+def validate_feature_dtype_and_shape(
+    name: str, feature: dict, value: np.ndarray | PILImage.Image | str
+) -> str:
+    """Validate the dtype and shape of a single feature's value.
+
+    Args:
+        name (str): The name of the feature.
+        feature (dict): The feature specification from the LeRobot features dictionary.
+        value: The value of the feature to validate.
+
+    Returns:
+        str: An error message if validation fails, otherwise an empty string.
+
+    Raises:
+        NotImplementedError: If the feature dtype is not supported for validation.
+    """
+    expected_dtype = feature["dtype"]
+    expected_shape = feature["shape"]
+    if is_valid_numpy_dtype_string(expected_dtype):
+        return validate_feature_numpy_array(name, expected_dtype, expected_shape, value)
+    elif expected_dtype in ["image", "video"]:
+        return validate_feature_image_or_video(name, expected_shape, value)
+    elif expected_dtype == "string":
+        return validate_feature_string(name, value)
+    else:
+        raise NotImplementedError(f"The feature dtype '{expected_dtype}' is not implemented yet.")
+
+
+def validate_feature_numpy_array(
+    name: str, expected_dtype: str, expected_shape: list[int], value: np.ndarray
+) -> str:
+    """Validate a feature that is expected to be a numpy array.
+
+    Args:
+        name (str): The name of the feature.
+        expected_dtype (str): The expected numpy dtype as a string.
+        expected_shape (list[int]): The expected shape.
+        value (np.ndarray): The numpy array to validate.
+
+    Returns:
+        str: An error message if validation fails, otherwise an empty string.
+    """
+    error_message = ""
+    if isinstance(value, np.ndarray):
+        actual_dtype = value.dtype
+        actual_shape = value.shape
+
+        if actual_dtype != np.dtype(expected_dtype):
+            error_message += f"The feature '{name}' of dtype '{actual_dtype}' is not of the expected dtype '{expected_dtype}'.\n"
+
+        if actual_shape != expected_shape:
+            error_message += f"The feature '{name}' of shape '{actual_shape}' does not have the expected shape '{expected_shape}'.\n"
+    else:
+        error_message += f"The feature '{name}' is not a 'np.ndarray'. Expected type is '{expected_dtype}', but type '{type(value)}' provided instead.\n"
+
+    return error_message
+
+
+def validate_feature_image_or_video(
+    name: str, expected_shape: list[str], value: np.ndarray | PILImage.Image
+) -> str:
+    """Validate a feature that is expected to be an image or video frame.
+
+    Accepts `np.ndarray` (channel-first or channel-last) or `PIL.Image.Image`.
+
+    Args:
+        name (str): The name of the feature.
+        expected_shape (list[str]): The expected shape (C, H, W).
+        value: The image data to validate.
+
+    Returns:
+        str: An error message if validation fails, otherwise an empty string.
+    """
+    # Note: The check of pixels range ([0,1] for float and [0,255] for uint8) is done by the image writer threads.
+    error_message = ""
+    if isinstance(value, np.ndarray):
+        actual_shape = value.shape
+        c, h, w = expected_shape
+        if len(actual_shape) != 3 or (actual_shape != (c, h, w) and actual_shape != (h, w, c)):
+            error_message += f"The feature '{name}' of shape '{actual_shape}' does not have the expected shape '{(c, h, w)}' or '{(h, w, c)}'.\n"
+    elif isinstance(value, PILImage.Image):
+        pass
+    else:
+        error_message += f"The feature '{name}' is expected to be of type 'PIL.Image' or 'np.ndarray' channel first or channel last, but type '{type(value)}' provided instead.\n"
+
+    return error_message
+
+
+def validate_feature_string(name: str, value: str) -> str:
+    """Validate a feature that is expected to be a string.
+
+    Args:
+        name (str): The name of the feature.
+        value (str): The value to validate.
+
+    Returns:
+        str: An error message if validation fails, otherwise an empty string.
+    """
+    if not isinstance(value, str):
+        return f"The feature '{name}' is expected to be of type 'str', but type '{type(value)}' provided instead.\n"
+    return ""
+
+
+def validate_episode_buffer(episode_buffer: dict, total_episodes: int, features: dict) -> None:
+    """Validate the episode buffer before it's written to disk.
+
+    Ensures the buffer has the required keys, contains at least one frame, and
+    has features consistent with the dataset's specification.
+
+    Args:
+        episode_buffer (dict): The buffer containing data for a single episode.
+        total_episodes (int): The current total number of episodes in the dataset.
+        features (dict): The LeRobot features dictionary for the dataset.
+
+    Raises:
+        ValueError: If the buffer is invalid.
+        NotImplementedError: If the episode index is manually set and doesn't match.
+    """
+    if "size" not in episode_buffer:
+        raise ValueError("size key not found in episode_buffer")
+
+    if "task" not in episode_buffer:
+        raise ValueError("task key not found in episode_buffer")
+
+    if episode_buffer["episode_index"] != total_episodes:
+        # TODO(aliberts): Add option to use existing episode_index
+        raise NotImplementedError(
+            "You might have manually provided the episode_buffer with an episode_index that doesn't "
+            "match the total number of episodes already in the dataset. This is not supported for now."
+        )
+
+    if episode_buffer["size"] == 0:
+        raise ValueError("You must add one or several frames with `add_frame` before calling `add_episode`.")
+
+    buffer_keys = set(episode_buffer.keys()) - {"task", "size"}
+    if not buffer_keys == set(features):
+        raise ValueError(
+            f"Features from `episode_buffer` don't match the ones in `features`."
+            f"In episode_buffer not in features: {buffer_keys - set(features)}"
+            f"In features not in episode_buffer: {set(features) - buffer_keys}"
+        )
diff --git a/lerobot/src/lerobot/datasets/image_writer.py b/lerobot/src/lerobot/datasets/image_writer.py
new file mode 100644
index 0000000000000000000000000000000000000000..9f40394de3e6f08d07c416ddd6e806b8eeade765
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/image_writer.py
@@ -0,0 +1,205 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+import multiprocessing
+import queue
+import threading
+from pathlib import Path
+
+import numpy as np
+import PIL.Image
+import torch
+
+logger = logging.getLogger(__name__)
+
+
+def safe_stop_image_writer(func):
+    def wrapper(*args, **kwargs):
+        try:
+            return func(*args, **kwargs)
+        except Exception as e:
+            dataset = kwargs.get("dataset")
+            image_writer = getattr(dataset, "image_writer", None) if dataset else None
+            if image_writer is not None:
+                logger.warning("Waiting for image writer to terminate...")
+                image_writer.stop()
+            raise e
+
+    return wrapper
+
+
+def image_array_to_pil_image(image_array: np.ndarray, range_check: bool = True) -> PIL.Image.Image:
+    # TODO(aliberts): handle 1 channel and 4 for depth images
+    if image_array.ndim != 3:
+        raise ValueError(f"The array has {image_array.ndim} dimensions, but 3 is expected for an image.")
+
+    if image_array.shape[0] == 3:
+        # Transpose from pytorch convention (C, H, W) to (H, W, C)
+        image_array = image_array.transpose(1, 2, 0)
+
+    elif image_array.shape[-1] != 3:
+        raise NotImplementedError(
+            f"The image has {image_array.shape[-1]} channels, but 3 is required for now."
+        )
+
+    if image_array.dtype != np.uint8:
+        if range_check:
+            max_ = image_array.max().item()
+            min_ = image_array.min().item()
+            if max_ > 1.0 or min_ < 0.0:
+                raise ValueError(
+                    "The image data type is float, which requires values in the range [0.0, 1.0]. "
+                    f"However, the provided range is [{min_}, {max_}]. Please adjust the range or "
+                    "provide a uint8 image with values in the range [0, 255]."
+                )
+
+        image_array = (image_array * 255).astype(np.uint8)
+
+    return PIL.Image.fromarray(image_array)
+
+
+def write_image(image: np.ndarray | PIL.Image.Image, fpath: Path, compress_level: int = 1):
+    """
+    Saves a NumPy array or PIL Image to a file.
+
+    This function handles both NumPy arrays and PIL Image objects, converting
+    the former to a PIL Image before saving. It includes error handling for
+    the save operation.
+
+    Args:
+        image (np.ndarray | PIL.Image.Image): The image data to save.
+        fpath (Path): The destination file path for the image.
+        compress_level (int, optional): The compression level for the saved
+            image, as used by PIL.Image.save(). Defaults to 1.
+            Refer to: https://github.com/huggingface/lerobot/pull/2135
+            for more details on the default value rationale.
+
+    Raises:
+        TypeError: If the input 'image' is not a NumPy array or a
+            PIL.Image.Image object.
+
+    Side Effects:
+        Logs an error message if the image writing process fails for any reason.
+    """
+    try:
+        if isinstance(image, np.ndarray):
+            img = image_array_to_pil_image(image)
+        elif isinstance(image, PIL.Image.Image):
+            img = image
+        else:
+            raise TypeError(f"Unsupported image type: {type(image)}")
+        img.save(fpath, compress_level=compress_level)
+    except Exception as e:
+        logger.error("Error writing image %s: %s", fpath, e)
+
+
+def worker_thread_loop(queue: queue.Queue):
+    while True:
+        item = queue.get()
+        if item is None:
+            queue.task_done()
+            break
+        image_array, fpath, compress_level = item
+        write_image(image_array, fpath, compress_level)
+        queue.task_done()
+
+
+def worker_process(queue: queue.Queue, num_threads: int):
+    threads = []
+    for _ in range(num_threads):
+        t = threading.Thread(target=worker_thread_loop, args=(queue,))
+        t.daemon = True
+        t.start()
+        threads.append(t)
+    for t in threads:
+        t.join()
+
+
+class AsyncImageWriter:
+    """
+    This class abstract away the initialisation of processes or/and threads to
+    save images on disk asynchronously, which is critical to control a robot and record data
+    at a high frame rate.
+
+    When `num_processes=0`, it creates a threads pool of size `num_threads`.
+    When `num_processes>0`, it creates processes pool of size `num_processes`, where each subprocess starts
+    their own threads pool of size `num_threads`.
+
+    The optimal number of processes and threads depends on your computer capabilities.
+    We advise to use 4 threads per camera with 0 processes. If the fps is not stable, try to increase or lower
+    the number of threads. If it is still not stable, try to use 1 subprocess, or more.
+    """
+
+    def __init__(self, num_processes: int = 0, num_threads: int = 1):
+        self.num_processes = num_processes
+        self.num_threads = num_threads
+        self.queue = None
+        self.threads = []
+        self.processes = []
+        self._stopped = False
+
+        if num_threads <= 0 and num_processes <= 0:
+            raise ValueError("Number of threads and processes must be greater than zero.")
+
+        if self.num_processes == 0:
+            # Use threading
+            self.queue = queue.Queue()
+            for _ in range(self.num_threads):
+                t = threading.Thread(target=worker_thread_loop, args=(self.queue,))
+                t.daemon = True
+                t.start()
+                self.threads.append(t)
+        else:
+            # Use multiprocessing
+            self.queue = multiprocessing.JoinableQueue()
+            for _ in range(self.num_processes):
+                p = multiprocessing.Process(target=worker_process, args=(self.queue, self.num_threads))
+                p.daemon = True
+                p.start()
+                self.processes.append(p)
+
+    def save_image(
+        self, image: torch.Tensor | np.ndarray | PIL.Image.Image, fpath: Path, compress_level: int = 1
+    ):
+        if isinstance(image, torch.Tensor):
+            # Convert tensor to numpy array to minimize main process time
+            image = image.cpu().numpy()
+        self.queue.put((image, fpath, compress_level))
+
+    def wait_until_done(self):
+        self.queue.join()
+
+    def stop(self):
+        if self._stopped:
+            return
+
+        if self.num_processes == 0:
+            for _ in self.threads:
+                self.queue.put(None)
+            for t in self.threads:
+                t.join()
+        else:
+            num_nones = self.num_processes * self.num_threads
+            for _ in range(num_nones):
+                self.queue.put(None)
+            for p in self.processes:
+                p.join()
+                if p.is_alive():
+                    p.terminate()
+            self.queue.close()
+            self.queue.join_thread()
+
+        self._stopped = True
diff --git a/lerobot/src/lerobot/datasets/io_utils.py b/lerobot/src/lerobot/datasets/io_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..cee6cfba8c3b3008f99fc8eda28b078e12877ff5
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/io_utils.py
@@ -0,0 +1,342 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import json
+from pathlib import Path
+from typing import Any
+
+import datasets
+import numpy as np
+import pandas
+import pandas as pd
+import pyarrow.dataset as pa_ds
+import pyarrow.parquet as pq
+import torch
+from datasets import Dataset
+from datasets.table import embed_table_storage
+from PIL import Image as PILImage
+from torchvision import transforms
+
+from lerobot.datasets.utils import (
+    DEFAULT_DATA_FILE_SIZE_IN_MB,
+    DEFAULT_EPISODES_PATH,
+    DEFAULT_SUBTASKS_PATH,
+    DEFAULT_TASKS_PATH,
+    EPISODES_DIR,
+    INFO_PATH,
+    STATS_PATH,
+    flatten_dict,
+    serialize_dict,
+    unflatten_dict,
+)
+from lerobot.utils.utils import SuppressProgressBars
+
+
+def get_parquet_file_size_in_mb(parquet_path: str | Path) -> float:
+    metadata = pq.read_metadata(parquet_path)
+    total_uncompressed_size = 0
+    for row_group in range(metadata.num_row_groups):
+        rg_metadata = metadata.row_group(row_group)
+        for column in range(rg_metadata.num_columns):
+            col_metadata = rg_metadata.column(column)
+            total_uncompressed_size += col_metadata.total_uncompressed_size
+    return total_uncompressed_size / (1024**2)
+
+
+def get_hf_dataset_size_in_mb(hf_ds: Dataset) -> int:
+    return hf_ds.data.nbytes // (1024**2)
+
+
+def load_nested_dataset(
+    pq_dir: Path, features: datasets.Features | None = None, episodes: list[int] | None = None
+) -> Dataset:
+    """Find parquet files in provided directory {pq_dir}/chunk-xxx/file-xxx.parquet
+    Convert parquet files to pyarrow memory mapped in a cache folder for efficient RAM usage
+    Concatenate all pyarrow references to return HF Dataset format
+
+    Args:
+        pq_dir: Directory containing parquet files
+        features: Optional features schema to ensure consistent loading of complex types like images
+        episodes: Optional list of episode indices to filter. Uses PyArrow predicate pushdown for efficiency.
+    """
+    paths = sorted(pq_dir.glob("*/*.parquet"))
+    if len(paths) == 0:
+        raise FileNotFoundError(f"Provided directory does not contain any parquet file: {pq_dir}")
+
+    with SuppressProgressBars():
+        # We use .from_parquet() memory-mapped loading for efficiency
+        filters = pa_ds.field("episode_index").isin(episodes) if episodes is not None else None
+        return Dataset.from_parquet([str(path) for path in paths], filters=filters, features=features)
+
+
+def get_parquet_num_frames(parquet_path: str | Path) -> int:
+    metadata = pq.read_metadata(parquet_path)
+    return metadata.num_rows
+
+
+def get_file_size_in_mb(file_path: Path) -> float:
+    """Get file size on disk in megabytes.
+
+    Args:
+        file_path (Path): Path to the file.
+    """
+    file_size_bytes = file_path.stat().st_size
+    return file_size_bytes / (1024**2)
+
+
+def embed_images(dataset: datasets.Dataset) -> datasets.Dataset:
+    """Embed image bytes into the dataset table before saving to Parquet.
+
+    This function prepares a Hugging Face dataset for serialization by converting
+    image objects into an embedded format that can be stored in Arrow/Parquet.
+
+    Args:
+        dataset (datasets.Dataset): The input dataset, possibly containing image features.
+
+    Returns:
+        datasets.Dataset: The dataset with images embedded in the table storage.
+    """
+    # Embed image bytes into the table before saving to parquet
+    format = dataset.format
+    dataset = dataset.with_format("arrow")
+    dataset = dataset.map(embed_table_storage, batched=False)
+    dataset = dataset.with_format(**format)
+    return dataset
+
+
+def load_json(fpath: Path) -> Any:
+    """Load data from a JSON file.
+
+    Args:
+        fpath (Path): Path to the JSON file.
+
+    Returns:
+        Any: The data loaded from the JSON file.
+    """
+    with open(fpath) as f:
+        return json.load(f)
+
+
+def write_json(data: dict, fpath: Path) -> None:
+    """Write data to a JSON file.
+
+    Creates parent directories if they don't exist.
+
+    Args:
+        data (dict): The dictionary to write.
+        fpath (Path): The path to the output JSON file.
+    """
+    fpath.parent.mkdir(exist_ok=True, parents=True)
+    with open(fpath, "w") as f:
+        json.dump(data, f, indent=4, ensure_ascii=False)
+
+
+def write_info(info: dict, local_dir: Path) -> None:
+    write_json(info, local_dir / INFO_PATH)
+
+
+def load_info(local_dir: Path) -> dict:
+    """Load dataset info metadata from its standard file path.
+
+    Also converts shape lists to tuples for consistency.
+
+    Args:
+        local_dir (Path): The root directory of the dataset.
+
+    Returns:
+        dict: The dataset information dictionary.
+    """
+    info = load_json(local_dir / INFO_PATH)
+    for ft in info["features"].values():
+        ft["shape"] = tuple(ft["shape"])
+    return info
+
+
+def write_stats(stats: dict, local_dir: Path) -> None:
+    """Serialize and write dataset statistics to their standard file path.
+
+    Args:
+        stats (dict): The statistics dictionary (can contain tensors/numpy arrays).
+        local_dir (Path): The root directory of the dataset.
+    """
+    serialized_stats = serialize_dict(stats)
+    write_json(serialized_stats, local_dir / STATS_PATH)
+
+
+def cast_stats_to_numpy(stats: dict) -> dict[str, dict[str, np.ndarray]]:
+    """Recursively cast numerical values in a stats dictionary to numpy arrays.
+
+    Args:
+        stats (dict): The statistics dictionary.
+
+    Returns:
+        dict: The statistics dictionary with values cast to numpy arrays.
+    """
+    stats = {key: np.array(value) for key, value in flatten_dict(stats).items()}
+    return unflatten_dict(stats)
+
+
+def load_stats(local_dir: Path) -> dict[str, dict[str, np.ndarray]] | None:
+    """Load dataset statistics and cast numerical values to numpy arrays.
+
+    Returns None if the stats file doesn't exist.
+
+    Args:
+        local_dir (Path): The root directory of the dataset.
+
+    Returns:
+        A dictionary of statistics or None if the file is not found.
+    """
+    if not (local_dir / STATS_PATH).exists():
+        return None
+    stats = load_json(local_dir / STATS_PATH)
+    return cast_stats_to_numpy(stats)
+
+
+def write_tasks(tasks: pandas.DataFrame, local_dir: Path) -> None:
+    path = local_dir / DEFAULT_TASKS_PATH
+    path.parent.mkdir(parents=True, exist_ok=True)
+    tasks.to_parquet(path)
+
+
+def load_tasks(local_dir: Path) -> pandas.DataFrame:
+    tasks = pd.read_parquet(local_dir / DEFAULT_TASKS_PATH)
+    tasks.index.name = "task"
+    return tasks
+
+
+def load_subtasks(local_dir: Path) -> pandas.DataFrame | None:
+    """Load subtasks from subtasks.parquet if it exists."""
+    subtasks_path = local_dir / DEFAULT_SUBTASKS_PATH
+    if subtasks_path.exists():
+        return pd.read_parquet(subtasks_path)
+    return None
+
+
+def write_episodes(episodes: Dataset, local_dir: Path) -> None:
+    """Write episode metadata to a parquet file in the LeRobot v3.0 format.
+    This function writes episode-level metadata to a single parquet file.
+    Used primarily during dataset conversion (v2.1 → v3.0) and in test fixtures.
+
+    Args:
+        episodes: HuggingFace Dataset containing episode metadata
+        local_dir: Root directory where the dataset will be stored
+    """
+    episode_size_mb = get_hf_dataset_size_in_mb(episodes)
+    if episode_size_mb > DEFAULT_DATA_FILE_SIZE_IN_MB:
+        raise NotImplementedError(
+            f"Episodes dataset is too large ({episode_size_mb} MB) to write to a single file. "
+            f"The current limit is {DEFAULT_DATA_FILE_SIZE_IN_MB} MB. "
+            "This function only supports single-file episode metadata. "
+        )
+
+    fpath = local_dir / DEFAULT_EPISODES_PATH.format(chunk_index=0, file_index=0)
+    fpath.parent.mkdir(parents=True, exist_ok=True)
+    episodes.to_parquet(fpath)
+
+
+def load_episodes(local_dir: Path) -> datasets.Dataset:
+    episodes = load_nested_dataset(local_dir / EPISODES_DIR)
+    # Select episode features/columns containing references to episode data and videos
+    # (e.g. tasks, dataset_from_index, dataset_to_index, data/chunk_index, data/file_index, etc.)
+    # This is to speedup access to these data, instead of having to load episode stats.
+    episodes = episodes.select_columns([key for key in episodes.features if not key.startswith("stats/")])
+    return episodes
+
+
+def load_image_as_numpy(
+    fpath: str | Path, dtype: np.dtype = np.float32, channel_first: bool = True
+) -> np.ndarray:
+    """Load an image from a file into a numpy array.
+
+    Args:
+        fpath (str | Path): Path to the image file.
+        dtype (np.dtype): The desired data type of the output array. If floating,
+            pixels are scaled to [0, 1].
+        channel_first (bool): If True, converts the image to (C, H, W) format.
+            Otherwise, it remains in (H, W, C) format.
+
+    Returns:
+        np.ndarray: The image as a numpy array.
+    """
+    img = PILImage.open(fpath).convert("RGB")
+    img_array = np.array(img, dtype=dtype)
+    if channel_first:  # (H, W, C) -> (C, H, W)
+        img_array = np.transpose(img_array, (2, 0, 1))
+    if np.issubdtype(dtype, np.floating):
+        img_array /= 255.0
+    return img_array
+
+
+def hf_transform_to_torch(items_dict: dict[str, list[Any]]) -> dict[str, list[torch.Tensor | str]]:
+    """Convert a batch from a Hugging Face dataset to torch tensors.
+
+    This transform function converts items from Hugging Face dataset format (pyarrow)
+    to torch tensors. Importantly, images are converted from PIL objects (H, W, C, uint8)
+    to a torch image representation (C, H, W, float32) in the range [0, 1]. Other
+    types are converted to torch.tensor.
+
+    Args:
+        items_dict (dict): A dictionary representing a batch of data from a
+            Hugging Face dataset.
+
+    Returns:
+        dict: The batch with items converted to torch tensors.
+    """
+    for key in items_dict:
+        first_item = items_dict[key][0]
+        if isinstance(first_item, PILImage.Image):
+            to_tensor = transforms.ToTensor()
+            items_dict[key] = [to_tensor(img) for img in items_dict[key]]
+        elif first_item is None:
+            pass
+        else:
+            items_dict[key] = [x if isinstance(x, str) else torch.tensor(x) for x in items_dict[key]]
+    return items_dict
+
+
+def to_parquet_with_hf_images(
+    df: pandas.DataFrame, path: Path, features: datasets.Features | None = None
+) -> None:
+    """This function correctly writes to parquet a panda DataFrame that contains images encoded by HF dataset.
+    This way, it can be loaded by HF dataset and correctly formatted images are returned.
+
+    Args:
+        df: DataFrame to write to parquet.
+        path: Path to write the parquet file.
+        features: Optional HuggingFace Features schema. If provided, ensures image columns
+                  are properly typed as Image() in the parquet schema.
+    """
+    # TODO(qlhoest): replace this weird synthax by `df.to_parquet(path)` only
+    ds = datasets.Dataset.from_dict(df.to_dict(orient="list"), features=features)
+    ds.to_parquet(path)
+
+
+def item_to_torch(item: dict) -> dict:
+    """Convert all items in a dictionary to PyTorch tensors where appropriate.
+
+    This function is used to convert an item from a streaming dataset to PyTorch tensors.
+
+    Args:
+        item (dict): Dictionary of items from a dataset.
+
+    Returns:
+        dict: Dictionary with all tensor-like items converted to torch.Tensor.
+    """
+    for key, val in item.items():
+        if isinstance(val, (np.ndarray | list)) and key not in ["task"]:
+            # Convert numpy arrays and lists to torch tensors
+            item[key] = torch.tensor(val)
+    return item
diff --git a/lerobot/src/lerobot/datasets/lerobot_dataset.py b/lerobot/src/lerobot/datasets/lerobot_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..8f0600ba8635e6db12ee81d19d944d518c53ea80
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/lerobot_dataset.py
@@ -0,0 +1,1244 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import concurrent.futures
+import contextlib
+import logging
+import shutil
+import tempfile
+from collections.abc import Callable
+from pathlib import Path
+
+import datasets
+import numpy as np
+import pandas as pd
+import PIL.Image
+import pyarrow.parquet as pq
+import torch
+import torch.utils
+from huggingface_hub import HfApi, snapshot_download
+from huggingface_hub.errors import RevisionNotFoundError
+
+from lerobot.datasets.compute_stats import compute_episode_stats
+from lerobot.datasets.dataset_metadata import CODEBASE_VERSION, LeRobotDatasetMetadata
+from lerobot.datasets.feature_utils import (
+    check_delta_timestamps,
+    get_delta_indices,
+    get_hf_features_from_features,
+    validate_episode_buffer,
+    validate_frame,
+)
+from lerobot.datasets.image_writer import AsyncImageWriter, write_image
+from lerobot.datasets.io_utils import (
+    embed_images,
+    get_file_size_in_mb,
+    hf_transform_to_torch,
+    load_episodes,
+    load_nested_dataset,
+    write_info,
+)
+from lerobot.datasets.utils import (
+    DEFAULT_EPISODES_PATH,
+    DEFAULT_IMAGE_PATH,
+    create_lerobot_dataset_card,
+    get_safe_version,
+    is_valid_version,
+    update_chunk_file_indices,
+)
+from lerobot.datasets.video_utils import (
+    StreamingVideoEncoder,
+    concatenate_video_files,
+    decode_video_frames,
+    encode_video_frames,
+    get_safe_default_codec,
+    get_video_duration_in_s,
+    resolve_vcodec,
+)
+from lerobot.utils.constants import HF_LEROBOT_HOME
+
+logger = logging.getLogger(__name__)
+
+
+def _encode_video_worker(
+    video_key: str,
+    episode_index: int,
+    root: Path,
+    fps: int,
+    vcodec: str = "libsvtav1",
+    encoder_threads: int | None = None,
+) -> Path:
+    temp_path = Path(tempfile.mkdtemp(dir=root)) / f"{video_key}_{episode_index:03d}.mp4"
+    fpath = DEFAULT_IMAGE_PATH.format(image_key=video_key, episode_index=episode_index, frame_index=0)
+    img_dir = (root / fpath).parent
+    encode_video_frames(
+        img_dir, temp_path, fps, vcodec=vcodec, overwrite=True, encoder_threads=encoder_threads
+    )
+    shutil.rmtree(img_dir)
+    return temp_path
+
+
+class LeRobotDataset(torch.utils.data.Dataset):
+    def __init__(
+        self,
+        repo_id: str,
+        root: str | Path | None = None,
+        episodes: list[int] | None = None,
+        image_transforms: Callable | None = None,
+        delta_timestamps: dict[str, list[float]] | None = None,
+        tolerance_s: float = 1e-4,
+        revision: str | None = None,
+        force_cache_sync: bool = False,
+        download_videos: bool = True,
+        video_backend: str | None = None,
+        batch_encoding_size: int = 1,
+        vcodec: str = "libsvtav1",
+        streaming_encoding: bool = False,
+        encoder_queue_maxsize: int = 30,
+        encoder_threads: int | None = None,
+    ):
+        """
+        2 modes are available for instantiating this class, depending on 2 different use cases:
+
+        1. Your dataset already exists:
+            - On your local disk in the 'root' folder. This is typically the case when you recorded your
+              dataset locally and you may or may not have pushed it to the hub yet. Instantiating this class
+              with 'root' will load your dataset directly from disk. This can happen while you're offline (no
+              internet connection).
+
+            - On the Hugging Face Hub at the address https://huggingface.co/datasets/{repo_id} and not on
+              your local disk in the 'root' folder. Instantiating this class with this 'repo_id' will download
+              the dataset from that address and load it, pending your dataset is compliant with
+              codebase_version v3.0. If your dataset has been created before this new format, you will be
+              prompted to convert it using our conversion script from v2.1 to v3.0, which you can find at
+              lerobot/scripts/convert_dataset_v21_to_v30.py.
+
+
+        2. Your dataset doesn't already exists (either on local disk or on the Hub): you can create an empty
+           LeRobotDataset with the 'create' classmethod. This can be used for recording a dataset or port an
+           existing dataset to the LeRobotDataset format.
+
+
+        In terms of files, LeRobotDataset encapsulates 3 main things:
+            - metadata:
+                - info contains various information about the dataset like shapes, keys, fps etc.
+                - stats stores the dataset statistics of the different modalities for normalization
+                - tasks contains the prompts for each task of the dataset, which can be used for
+                  task-conditioned training.
+            - hf_dataset (from datasets.Dataset), which will read any values from parquet files.
+            - videos (optional) from which frames are loaded to be synchronous with data from parquet files.
+
+        A typical LeRobotDataset looks like this from its root path:
+        .
+        ├── data
+        │   ├── chunk-000
+        │   │   ├── file-000.parquet
+        │   │   ├── file-001.parquet
+        │   │   └── ...
+        │   ├── chunk-001
+        │   │   ├── file-000.parquet
+        │   │   ├── file-001.parquet
+        │   │   └── ...
+        │   └── ...
+        ├── meta
+        │   ├── episodes
+        │   │   ├── chunk-000
+        │   │   │   ├── file-000.parquet
+        │   │   │   ├── file-001.parquet
+        │   │   │   └── ...
+        │   │   ├── chunk-001
+        │   │   │   └── ...
+        │   │   └── ...
+        │   ├── info.json
+        │   ├── stats.json
+        │   └── tasks.parquet
+        └── videos
+            ├── observation.images.laptop
+            │   ├── chunk-000
+            │   │   ├── file-000.mp4
+            │   │   ├── file-001.mp4
+            │   │   └── ...
+            │   ├── chunk-001
+            │   │   └── ...
+            │   └── ...
+            ├── observation.images.phone
+            │   ├── chunk-000
+            │   │   ├── file-000.mp4
+            │   │   ├── file-001.mp4
+            │   │   └── ...
+            │   ├── chunk-001
+            │   │   └── ...
+            │   └── ...
+            └── ...
+
+        Note that this file-based structure is designed to be as versatile as possible. Multiple episodes are
+        consolidated into chunked files which improves storage efficiency and loading performance. The
+        structure of the dataset is entirely described in the info.json file, which can be easily downloaded
+        or viewed directly on the hub before downloading any actual data. The type of files used are very
+        simple and do not need complex tools to be read, it only uses .parquet, .json and .mp4 files (and .md
+        for the README).
+
+        Args:
+            repo_id (str): This is the repo id that will be used to fetch the dataset.
+            root (Path | None, optional): Local directory where the dataset will be downloaded and
+                stored. If set, all dataset files will be stored directly under this path. If not set, the
+                dataset files will be stored under $HF_LEROBOT_HOME/repo_id (configurable via the
+                HF_LEROBOT_HOME environment variable).
+            episodes (list[int] | None, optional): If specified, this will only load episodes specified by
+                their episode_index in this list. Defaults to None.
+            image_transforms (Callable | None, optional): You can pass standard v2 image transforms from
+                torchvision.transforms.v2 here which will be applied to visual modalities (whether they come
+                from videos or images). Defaults to None.
+            delta_timestamps (dict[list[float]] | None, optional): _description_. Defaults to None.
+            tolerance_s (float, optional): Tolerance in seconds used to ensure data timestamps are actually in
+                sync with the fps value. It is used at the init of the dataset to make sure that each
+                timestamps is separated to the next by 1/fps +/- tolerance_s. This also applies to frames
+                decoded from video files. It is also used to check that `delta_timestamps` (when provided) are
+                multiples of 1/fps. Defaults to 1e-4.
+            revision (str, optional): An optional Git revision id which can be a branch name, a tag, or a
+                commit hash. Defaults to current codebase version tag.
+            force_cache_sync (bool, optional): Flag to sync and refresh local files first. If True and files
+                are already present in the local cache, this will be faster. However, files loaded might not
+                be in sync with the version on the hub, especially if you specified 'revision'. Defaults to
+                False.
+            download_videos (bool, optional): Flag to download the videos. Note that when set to True but the
+                video files are already present on local disk, they won't be downloaded again. Defaults to
+                True.
+            video_backend (str | None, optional): Video backend to use for decoding videos. Defaults to torchcodec when available int the platform; otherwise, defaults to 'pyav'.
+                You can also use the 'pyav' decoder used by Torchvision, which used to be the default option, or 'video_reader' which is another decoder of Torchvision.
+            batch_encoding_size (int, optional): Number of episodes to accumulate before batch encoding videos.
+                Set to 1 for immediate encoding (default), or higher for batched encoding. Defaults to 1.
+            vcodec (str, optional): Video codec for encoding videos during recording. Options: 'h264', 'hevc',
+                'libsvtav1', 'auto', or hardware-specific codecs like 'h264_videotoolbox', 'h264_nvenc'.
+                Defaults to 'libsvtav1'. Use 'auto' to auto-detect the best available hardware encoder.
+            streaming_encoding (bool, optional): If True, encode video frames in real-time during capture
+                instead of writing PNG images first. This makes save_episode() near-instant. Defaults to False.
+            encoder_queue_maxsize (int, optional): Maximum number of frames to buffer per camera when using
+                streaming encoding. Defaults to 30 (~1s at 30fps).
+            encoder_threads (int | None, optional): Number of threads per encoder instance. None lets the
+                codec auto-detect (default). Lower values reduce CPU usage per encoder. Maps to 'lp' (via svtav1-params) for
+                libsvtav1 and 'threads' for h264/hevc.
+        """
+        super().__init__()
+        self.repo_id = repo_id
+        self.root = Path(root) if root else HF_LEROBOT_HOME / repo_id
+        self.image_transforms = image_transforms
+        self.delta_timestamps = delta_timestamps
+        self.episodes = episodes
+        self.tolerance_s = tolerance_s
+        self.revision = revision if revision else CODEBASE_VERSION
+        self.video_backend = video_backend if video_backend else get_safe_default_codec()
+        self.delta_indices = None
+        self.batch_encoding_size = batch_encoding_size
+        self.episodes_since_last_encoding = 0
+        self.vcodec = resolve_vcodec(vcodec)
+        self._encoder_threads = encoder_threads
+
+        # Unused attributes
+        self.image_writer = None
+        self.episode_buffer = None
+        self.writer = None
+        self.latest_episode = None
+        self._current_file_start_frame = None  # Track the starting frame index of the current parquet file
+        self._streaming_encoder = None
+
+        self.root.mkdir(exist_ok=True, parents=True)
+
+        # Load metadata
+        self.meta = LeRobotDatasetMetadata(
+            self.repo_id, self.root, self.revision, force_cache_sync=force_cache_sync
+        )
+
+        # Track dataset state for efficient incremental writing
+        self._lazy_loading = False
+        self._recorded_frames = self.meta.total_frames
+        self._writer_closed_for_reading = False
+
+        # Load actual data
+        try:
+            if force_cache_sync:
+                raise FileNotFoundError
+            self.hf_dataset = self.load_hf_dataset()
+            # Check if cached dataset contains all requested episodes
+            if not self._check_cached_episodes_sufficient():
+                raise FileNotFoundError("Cached dataset doesn't contain all requested episodes")
+        except (FileNotFoundError, NotADirectoryError):
+            if is_valid_version(self.revision):
+                self.revision = get_safe_version(self.repo_id, self.revision)
+            self.download(download_videos)
+            self.hf_dataset = self.load_hf_dataset()
+
+        # Create mapping from absolute indices to relative indices when only a subset of the episodes are loaded
+        # Build a mapping: absolute_index -> relative_index_in_filtered_dataset
+        self._absolute_to_relative_idx = None
+        if self.episodes is not None:
+            self._absolute_to_relative_idx = {
+                abs_idx.item() if isinstance(abs_idx, torch.Tensor) else abs_idx: rel_idx
+                for rel_idx, abs_idx in enumerate(self.hf_dataset["index"])
+            }
+
+        # Setup delta_indices
+        if self.delta_timestamps is not None:
+            check_delta_timestamps(self.delta_timestamps, self.fps, self.tolerance_s)
+            self.delta_indices = get_delta_indices(self.delta_timestamps, self.fps)
+
+        # Initialize streaming encoder for resumed recording
+        if streaming_encoding and len(self.meta.video_keys) > 0:
+            self._streaming_encoder = StreamingVideoEncoder(
+                fps=self.meta.fps,
+                vcodec=self.vcodec,
+                pix_fmt="yuv420p",
+                g=2,
+                crf=30,
+                preset=None,
+                queue_maxsize=encoder_queue_maxsize,
+                encoder_threads=encoder_threads,
+            )
+
+    def _close_writer(self) -> None:
+        """Close and cleanup the parquet writer if it exists."""
+        writer = getattr(self, "writer", None)
+        if writer is not None:
+            writer.close()
+            self.writer = None
+
+    def __del__(self):
+        """
+        Trust the user to call .finalize() but as an added safety check call the parquet writer to stop when calling the destructor
+        """
+        self._close_writer()
+
+    def push_to_hub(
+        self,
+        branch: str | None = None,
+        tags: list | None = None,
+        license: str | None = "apache-2.0",
+        tag_version: bool = True,
+        push_videos: bool = True,
+        private: bool = False,
+        allow_patterns: list[str] | str | None = None,
+        upload_large_folder: bool = False,
+        **card_kwargs,
+    ) -> None:
+        ignore_patterns = ["images/"]
+        if not push_videos:
+            ignore_patterns.append("videos/")
+
+        hub_api = HfApi()
+        hub_api.create_repo(
+            repo_id=self.repo_id,
+            private=private,
+            repo_type="dataset",
+            exist_ok=True,
+        )
+        if branch:
+            hub_api.create_branch(
+                repo_id=self.repo_id,
+                branch=branch,
+                revision=self.revision,
+                repo_type="dataset",
+                exist_ok=True,
+            )
+
+        upload_kwargs = {
+            "repo_id": self.repo_id,
+            "folder_path": self.root,
+            "repo_type": "dataset",
+            "revision": branch,
+            "allow_patterns": allow_patterns,
+            "ignore_patterns": ignore_patterns,
+        }
+        if upload_large_folder:
+            hub_api.upload_large_folder(**upload_kwargs)
+        else:
+            hub_api.upload_folder(**upload_kwargs)
+
+        card = create_lerobot_dataset_card(
+            tags=tags, dataset_info=self.meta.info, license=license, repo_id=self.repo_id, **card_kwargs
+        )
+        card.push_to_hub(repo_id=self.repo_id, repo_type="dataset", revision=branch)
+
+        if tag_version:
+            with contextlib.suppress(RevisionNotFoundError):
+                hub_api.delete_tag(self.repo_id, tag=CODEBASE_VERSION, repo_type="dataset")
+            hub_api.create_tag(self.repo_id, tag=CODEBASE_VERSION, revision=branch, repo_type="dataset")
+
+    def pull_from_repo(
+        self,
+        allow_patterns: list[str] | str | None = None,
+        ignore_patterns: list[str] | str | None = None,
+    ) -> None:
+        snapshot_download(
+            self.repo_id,
+            repo_type="dataset",
+            revision=self.revision,
+            local_dir=self.root,
+            allow_patterns=allow_patterns,
+            ignore_patterns=ignore_patterns,
+        )
+
+    def download(self, download_videos: bool = True) -> None:
+        """Downloads the dataset from the given 'repo_id' at the provided version. If 'episodes' is given, this
+        will only download those episodes (selected by their episode_index). If 'episodes' is None, the whole
+        dataset will be downloaded. Thanks to the behavior of snapshot_download, if the files are already present
+        in 'local_dir', they won't be downloaded again.
+        """
+        # TODO(rcadene, aliberts): implement faster transfer
+        # https://huggingface.co/docs/huggingface_hub/en/guides/download#faster-downloads
+        ignore_patterns = None if download_videos else "videos/"
+        files = None
+        if self.episodes is not None:
+            files = self.get_episodes_file_paths()
+        self.pull_from_repo(allow_patterns=files, ignore_patterns=ignore_patterns)
+
+    def get_episodes_file_paths(self) -> list[Path]:
+        episodes = self.episodes if self.episodes is not None else list(range(self.meta.total_episodes))
+        fpaths = [str(self.meta.get_data_file_path(ep_idx)) for ep_idx in episodes]
+        if len(self.meta.video_keys) > 0:
+            video_files = [
+                str(self.meta.get_video_file_path(ep_idx, vid_key))
+                for vid_key in self.meta.video_keys
+                for ep_idx in episodes
+            ]
+            fpaths += video_files
+        # episodes are stored in the same files, so we return unique paths only
+        fpaths = list(set(fpaths))
+        return fpaths
+
+    def load_hf_dataset(self) -> datasets.Dataset:
+        """hf_dataset contains all the observations, states, actions, rewards, etc."""
+        features = get_hf_features_from_features(self.features)
+        hf_dataset = load_nested_dataset(self.root / "data", features=features, episodes=self.episodes)
+        hf_dataset.set_transform(hf_transform_to_torch)
+        return hf_dataset
+
+    def _check_cached_episodes_sufficient(self) -> bool:
+        """Check if the cached dataset contains all requested episodes and their video files."""
+        if self.hf_dataset is None or len(self.hf_dataset) == 0:
+            return False
+
+        # Get available episode indices from cached dataset
+        available_episodes = {
+            ep_idx.item() if isinstance(ep_idx, torch.Tensor) else ep_idx
+            for ep_idx in self.hf_dataset.unique("episode_index")
+        }
+
+        # Determine requested episodes
+        if self.episodes is None:
+            requested_episodes = set(range(self.meta.total_episodes))
+        else:
+            requested_episodes = set(self.episodes)
+
+        # Check if all requested episodes are available in cached data
+        if not requested_episodes.issubset(available_episodes):
+            return False
+
+        # Check if all required video files exist
+        if len(self.meta.video_keys) > 0:
+            for ep_idx in requested_episodes:
+                for vid_key in self.meta.video_keys:
+                    video_path = self.root / self.meta.get_video_file_path(ep_idx, vid_key)
+                    if not video_path.exists():
+                        return False
+
+        return True
+
+    def create_hf_dataset(self) -> datasets.Dataset:
+        features = get_hf_features_from_features(self.features)
+        ft_dict = {col: [] for col in features}
+        hf_dataset = datasets.Dataset.from_dict(ft_dict, features=features, split="train")
+        hf_dataset.set_transform(hf_transform_to_torch)
+        return hf_dataset
+
+    @property
+    def fps(self) -> int:
+        """Frames per second used during data collection."""
+        return self.meta.fps
+
+    @property
+    def num_frames(self) -> int:
+        """Number of frames in selected episodes.
+
+        Note: When episodes a subset of the full dataset is requested, we must return the
+        actual loaded data length (len(self.hf_dataset)) rather than metadata total_frames.
+        self.meta.total_frames is the total number of frames in the full dataset.
+        """
+        if self.episodes is not None and self.hf_dataset is not None:
+            return len(self.hf_dataset)
+        return self.meta.total_frames
+
+    @property
+    def num_episodes(self) -> int:
+        """Number of episodes selected."""
+        return len(self.episodes) if self.episodes is not None else self.meta.total_episodes
+
+    @property
+    def features(self) -> dict[str, dict]:
+        return self.meta.features
+
+    @property
+    def hf_features(self) -> datasets.Features:
+        """Features of the hf_dataset."""
+        if self.hf_dataset is not None:
+            return self.hf_dataset.features
+        else:
+            return get_hf_features_from_features(self.features)
+
+    def _get_query_indices(
+        self, abs_idx: int, ep_idx: int
+    ) -> tuple[dict[str, list[int]], dict[str, torch.Tensor]]:
+        """Compute query indices for delta timestamps.
+
+        Args:
+            abs_idx: The absolute index in the full dataset (not the relative index in filtered episodes).
+            ep_idx: The episode index.
+
+        Returns:
+            A tuple of (query_indices, padding) where:
+            - query_indices: Dict mapping keys to lists of absolute indices to query
+            - padding: Dict mapping "{key}_is_pad" to boolean tensors indicating padded positions
+        """
+        ep = self.meta.episodes[ep_idx]
+        ep_start = ep["dataset_from_index"]
+        ep_end = ep["dataset_to_index"]
+        query_indices = {
+            key: [max(ep_start, min(ep_end - 1, abs_idx + delta)) for delta in delta_idx]
+            for key, delta_idx in self.delta_indices.items()
+        }
+        padding = {  # Pad values outside of current episode range
+            f"{key}_is_pad": torch.BoolTensor(
+                [(abs_idx + delta < ep_start) | (abs_idx + delta >= ep_end) for delta in delta_idx]
+            )
+            for key, delta_idx in self.delta_indices.items()
+        }
+        return query_indices, padding
+
+    def _get_query_timestamps(
+        self,
+        current_ts: float,
+        query_indices: dict[str, list[int]] | None = None,
+    ) -> dict[str, list[float]]:
+        query_timestamps = {}
+        for key in self.meta.video_keys:
+            if query_indices is not None and key in query_indices:
+                if self._absolute_to_relative_idx is not None:
+                    relative_indices = [self._absolute_to_relative_idx[idx] for idx in query_indices[key]]
+                    timestamps = self.hf_dataset[relative_indices]["timestamp"]
+                else:
+                    timestamps = self.hf_dataset[query_indices[key]]["timestamp"]
+                query_timestamps[key] = torch.stack(timestamps).tolist()
+            else:
+                query_timestamps[key] = [current_ts]
+
+        return query_timestamps
+
+    def _query_hf_dataset(self, query_indices: dict[str, list[int]]) -> dict:
+        """
+        Query dataset for indices across keys, skipping video keys.
+
+        Tries column-first [key][indices] for speed, falls back to row-first.
+
+        Args:
+            query_indices: Dict mapping keys to index lists to retrieve
+
+        Returns:
+            Dict with stacked tensors of queried data (video keys excluded)
+        """
+        result: dict = {}
+        for key, q_idx in query_indices.items():
+            if key in self.meta.video_keys:
+                continue
+            # Map absolute indices to relative indices if needed
+            relative_indices = (
+                q_idx
+                if self._absolute_to_relative_idx is None
+                else [self._absolute_to_relative_idx[idx] for idx in q_idx]
+            )
+            try:
+                result[key] = torch.stack(self.hf_dataset[key][relative_indices])
+            except (KeyError, TypeError, IndexError):
+                result[key] = torch.stack(self.hf_dataset[relative_indices][key])
+        return result
+
+    def _query_videos(self, query_timestamps: dict[str, list[float]], ep_idx: int) -> dict[str, torch.Tensor]:
+        """Note: When using data workers (e.g. DataLoader with num_workers>0), do not call this function
+        in the main process (e.g. by using a second Dataloader with num_workers=0). It will result in a
+        Segmentation Fault. This probably happens because a memory reference to the video loader is created in
+        the main process and a subprocess fails to access it.
+        """
+        ep = self.meta.episodes[ep_idx]
+        item = {}
+        for vid_key, query_ts in query_timestamps.items():
+            # Episodes are stored sequentially on a single mp4 to reduce the number of files.
+            # Thus we load the start timestamp of the episode on this mp4 and,
+            # shift the query timestamp accordingly.
+            from_timestamp = ep[f"videos/{vid_key}/from_timestamp"]
+            shifted_query_ts = [from_timestamp + ts for ts in query_ts]
+
+            video_path = self.root / self.meta.get_video_file_path(ep_idx, vid_key)
+            frames = decode_video_frames(video_path, shifted_query_ts, self.tolerance_s, self.video_backend)
+            item[vid_key] = frames.squeeze(0)
+
+        return item
+
+    def _ensure_hf_dataset_loaded(self):
+        """Lazy load the HF dataset only when needed for reading."""
+        if self._lazy_loading or self.hf_dataset is None:
+            # Close the writer before loading to ensure parquet file is properly finalized
+            if self.writer is not None:
+                self._close_writer()
+                self._writer_closed_for_reading = True
+            self.hf_dataset = self.load_hf_dataset()
+            self._lazy_loading = False
+
+    def __len__(self):
+        return self.num_frames
+
+    def __getitem__(self, idx) -> dict:
+        # Ensure dataset is loaded when we actually need to read from it
+        self._ensure_hf_dataset_loaded()
+        item = self.hf_dataset[idx]
+        ep_idx = item["episode_index"].item()
+        # Use the absolute index from the dataset for delta timestamp calculations
+        abs_idx = item["index"].item()
+
+        query_indices = None
+        if self.delta_indices is not None:
+            query_indices, padding = self._get_query_indices(abs_idx, ep_idx)
+            query_result = self._query_hf_dataset(query_indices)
+            item = {**item, **padding}
+            for key, val in query_result.items():
+                item[key] = val
+
+        if len(self.meta.video_keys) > 0:
+            current_ts = item["timestamp"].item()
+            query_timestamps = self._get_query_timestamps(current_ts, query_indices)
+            video_frames = self._query_videos(query_timestamps, ep_idx)
+            item = {**video_frames, **item}
+
+        if self.image_transforms is not None:
+            image_keys = self.meta.camera_keys
+            for cam in image_keys:
+                item[cam] = self.image_transforms(item[cam])
+
+        # Add task as a string
+        task_idx = item["task_index"].item()
+        item["task"] = self.meta.tasks.iloc[task_idx].name
+
+        # add subtask information if available
+        if "subtask_index" in self.features and self.meta.subtasks is not None:
+            subtask_idx = item["subtask_index"].item()
+            item["subtask"] = self.meta.subtasks.iloc[subtask_idx].name
+
+        return item
+
+    def __repr__(self):
+        feature_keys = list(self.features)
+        return (
+            f"{self.__class__.__name__}({{\n"
+            f"    Repository ID: '{self.repo_id}',\n"
+            f"    Number of selected episodes: '{self.num_episodes}',\n"
+            f"    Number of selected samples: '{self.num_frames}',\n"
+            f"    Features: '{feature_keys}',\n"
+            "})',\n"
+        )
+
+    def finalize(self):
+        """
+        Close the parquet writers. This function needs to be called after data collection/conversion, else footer metadata won't be written to the parquet files.
+        The dataset won't be valid and can't be loaded as ds = LeRobotDataset(repo_id=repo, root=HF_LEROBOT_HOME.joinpath(repo))
+        """
+        self._close_writer()
+        self.meta._close_writer()
+        if self._streaming_encoder is not None:
+            self._streaming_encoder.close()
+
+    def create_episode_buffer(self, episode_index: int | None = None) -> dict:
+        current_ep_idx = self.meta.total_episodes if episode_index is None else episode_index
+        ep_buffer = {}
+        # size and task are special cases that are not in self.features
+        ep_buffer["size"] = 0
+        ep_buffer["task"] = []
+        for key in self.features:
+            ep_buffer[key] = current_ep_idx if key == "episode_index" else []
+        return ep_buffer
+
+    # TODO(Steven): consider move this to utils
+    def _get_image_file_path(self, episode_index: int, image_key: str, frame_index: int) -> Path:
+        fpath = DEFAULT_IMAGE_PATH.format(
+            image_key=image_key, episode_index=episode_index, frame_index=frame_index
+        )
+        return self.root / fpath
+
+    def _get_image_file_dir(self, episode_index: int, image_key: str) -> Path:
+        return self._get_image_file_path(episode_index, image_key, frame_index=0).parent
+
+    def _save_image(
+        self, image: torch.Tensor | np.ndarray | PIL.Image.Image, fpath: Path, compress_level: int = 1
+    ) -> None:
+        if self.image_writer is None:
+            if isinstance(image, torch.Tensor):
+                image = image.cpu().numpy()
+            write_image(image, fpath, compress_level=compress_level)
+        else:
+            self.image_writer.save_image(image=image, fpath=fpath, compress_level=compress_level)
+
+    def add_frame(self, frame: dict) -> None:
+        """
+        This function only adds the frame to the episode_buffer. Apart from images — which are written in a
+        temporary directory — nothing is written to disk. To save those frames, the 'save_episode()' method
+        then needs to be called.
+        """
+        # Convert torch to numpy if needed
+        for name in frame:
+            if isinstance(frame[name], torch.Tensor):
+                frame[name] = frame[name].numpy()
+
+        validate_frame(frame, self.features)
+
+        if self.episode_buffer is None:
+            self.episode_buffer = self.create_episode_buffer()
+
+        # Automatically add frame_index and timestamp to episode buffer
+        frame_index = self.episode_buffer["size"]
+        timestamp = frame.pop("timestamp") if "timestamp" in frame else frame_index / self.fps
+        self.episode_buffer["frame_index"].append(frame_index)
+        self.episode_buffer["timestamp"].append(timestamp)
+        self.episode_buffer["task"].append(frame.pop("task"))  # Remove task from frame after processing
+
+        # Start streaming encoder on first frame of episode (once, before iterating keys)
+        if frame_index == 0 and self._streaming_encoder is not None:
+            self._streaming_encoder.start_episode(
+                video_keys=list(self.meta.video_keys),
+                temp_dir=self.root,
+            )
+
+        # Add frame features to episode_buffer
+        for key in frame:
+            if key not in self.features:
+                raise ValueError(
+                    f"An element of the frame is not in the features. '{key}' not in '{self.features.keys()}'."
+                )
+
+            if self.features[key]["dtype"] == "video" and self._streaming_encoder is not None:
+                self._streaming_encoder.feed_frame(key, frame[key])
+                self.episode_buffer[key].append(None)  # Placeholder (video keys are skipped in parquet)
+            elif self.features[key]["dtype"] in ["image", "video"]:
+                img_path = self._get_image_file_path(
+                    episode_index=self.episode_buffer["episode_index"], image_key=key, frame_index=frame_index
+                )
+                if frame_index == 0:
+                    img_path.parent.mkdir(parents=True, exist_ok=True)
+                compress_level = 1 if self.features[key]["dtype"] == "video" else 6
+                self._save_image(frame[key], img_path, compress_level)
+                self.episode_buffer[key].append(str(img_path))
+            else:
+                self.episode_buffer[key].append(frame[key])
+
+        self.episode_buffer["size"] += 1
+
+    def save_episode(
+        self,
+        episode_data: dict | None = None,
+        parallel_encoding: bool = True,
+    ) -> None:
+        """
+        This will save to disk the current episode in self.episode_buffer.
+
+        Video encoding is handled automatically based on batch_encoding_size:
+        - If batch_encoding_size == 1: Videos are encoded immediately after each episode
+        - If batch_encoding_size > 1: Videos are encoded in batches.
+
+        Args:
+            episode_data (dict | None, optional): Dict containing the episode data to save. If None, this will
+                save the current episode in self.episode_buffer, which is filled with 'add_frame'. Defaults to
+                None.
+            parallel_encoding (bool, optional): If True, encode videos in parallel using ProcessPoolExecutor.
+                Defaults to True on Linux, False on macOS as it tends to use all the CPU available already.
+        """
+        episode_buffer = episode_data if episode_data is not None else self.episode_buffer
+
+        validate_episode_buffer(episode_buffer, self.meta.total_episodes, self.features)
+
+        # size and task are special cases that won't be added to hf_dataset
+        episode_length = episode_buffer.pop("size")
+        tasks = episode_buffer.pop("task")
+        episode_tasks = list(set(tasks))
+        episode_index = episode_buffer["episode_index"]
+
+        episode_buffer["index"] = np.arange(self.meta.total_frames, self.meta.total_frames + episode_length)
+        episode_buffer["episode_index"] = np.full((episode_length,), episode_index)
+
+        # Update tasks and task indices with new tasks if any
+        self.meta.save_episode_tasks(episode_tasks)
+
+        # Given tasks in natural language, find their corresponding task indices
+        episode_buffer["task_index"] = np.array([self.meta.get_task_index(task) for task in tasks])
+
+        for key, ft in self.features.items():
+            # index, episode_index, task_index are already processed above, and image and video
+            # are processed separately by storing image path and frame info as meta data
+            if key in ["index", "episode_index", "task_index"] or ft["dtype"] in ["image", "video"]:
+                continue
+            episode_buffer[key] = np.stack(episode_buffer[key])
+
+        # Wait for image writer to end, so that episode stats over images can be computed
+        self._wait_image_writer()
+
+        has_video_keys = len(self.meta.video_keys) > 0
+        use_streaming = self._streaming_encoder is not None and has_video_keys
+        use_batched_encoding = self.batch_encoding_size > 1
+
+        if use_streaming:
+            # Compute stats for non-video features only (video stats come from encoder)
+            non_video_buffer = {
+                k: v
+                for k, v in episode_buffer.items()
+                if self.features.get(k, {}).get("dtype") not in ("video",)
+            }
+            non_video_features = {k: v for k, v in self.features.items() if v["dtype"] != "video"}
+            ep_stats = compute_episode_stats(non_video_buffer, non_video_features)
+        else:
+            ep_stats = compute_episode_stats(episode_buffer, self.features)
+
+        ep_metadata = self._save_episode_data(episode_buffer)
+
+        if use_streaming:
+            # Finish streaming encoding and collect results
+            streaming_results = self._streaming_encoder.finish_episode()
+            for video_key in self.meta.video_keys:
+                temp_path, video_stats = streaming_results[video_key]
+                if video_stats is not None:
+                    # Format stats same as compute_episode_stats: normalize to [0,1], reshape to (C,1,1)
+                    ep_stats[video_key] = {
+                        k: v if k == "count" else np.squeeze(v.reshape(1, -1, 1, 1) / 255.0, axis=0)
+                        for k, v in video_stats.items()
+                    }
+                ep_metadata.update(self._save_episode_video(video_key, episode_index, temp_path=temp_path))
+        elif has_video_keys and not use_batched_encoding:
+            num_cameras = len(self.meta.video_keys)
+            if parallel_encoding and num_cameras > 1:
+                # TODO(Steven): Ideally we would like to control the number of threads per encoding such that:
+                # num_cameras * num_threads = (total_cpu -1)
+                with concurrent.futures.ProcessPoolExecutor(max_workers=num_cameras) as executor:
+                    future_to_key = {
+                        executor.submit(
+                            _encode_video_worker,
+                            video_key,
+                            episode_index,
+                            self.root,
+                            self.fps,
+                            self.vcodec,
+                            self._encoder_threads,
+                        ): video_key
+                        for video_key in self.meta.video_keys
+                    }
+
+                    results = {}
+                    for future in concurrent.futures.as_completed(future_to_key):
+                        video_key = future_to_key[future]
+                        try:
+                            temp_path = future.result()
+                            results[video_key] = temp_path
+                        except Exception as exc:
+                            logger.error(f"Video encoding failed for {video_key}: {exc}")
+                            raise exc
+
+                for video_key in self.meta.video_keys:
+                    temp_path = results[video_key]
+                    ep_metadata.update(
+                        self._save_episode_video(video_key, episode_index, temp_path=temp_path)
+                    )
+            else:
+                for video_key in self.meta.video_keys:
+                    ep_metadata.update(self._save_episode_video(video_key, episode_index))
+
+        # `meta.save_episode` need to be executed after encoding the videos
+        self.meta.save_episode(episode_index, episode_length, episode_tasks, ep_stats, ep_metadata)
+
+        if has_video_keys and use_batched_encoding:
+            # Check if we should trigger batch encoding
+            self.episodes_since_last_encoding += 1
+            if self.episodes_since_last_encoding == self.batch_encoding_size:
+                start_ep = self.num_episodes - self.batch_encoding_size
+                end_ep = self.num_episodes
+                self._batch_save_episode_video(start_ep, end_ep)
+                self.episodes_since_last_encoding = 0
+
+        if not episode_data:
+            # Reset episode buffer and clean up temporary images (if not already deleted during video encoding)
+            self.clear_episode_buffer(delete_images=len(self.meta.image_keys) > 0)
+
+    def _batch_save_episode_video(self, start_episode: int, end_episode: int | None = None) -> None:
+        """
+        Batch save videos for multiple episodes.
+
+        Args:
+            start_episode: Starting episode index (inclusive)
+            end_episode: Ending episode index (exclusive). If None, encodes all episodes from start_episode to the current episode.
+        """
+        if end_episode is None:
+            end_episode = self.num_episodes
+
+        logger.info(
+            f"Batch encoding {self.batch_encoding_size} videos for episodes {start_episode} to {end_episode - 1}"
+        )
+
+        chunk_idx = self.meta.episodes[start_episode]["data/chunk_index"]
+        file_idx = self.meta.episodes[start_episode]["data/file_index"]
+        episode_df_path = self.root / DEFAULT_EPISODES_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
+        episode_df = pd.read_parquet(episode_df_path)
+
+        for ep_idx in range(start_episode, end_episode):
+            logger.info(f"Encoding videos for episode {ep_idx}")
+
+            if (
+                self.meta.episodes[ep_idx]["data/chunk_index"] != chunk_idx
+                or self.meta.episodes[ep_idx]["data/file_index"] != file_idx
+            ):
+                # The current episode is in a new chunk or file.
+                # Save previous episode dataframe and update the Hugging Face dataset by reloading it.
+                episode_df.to_parquet(episode_df_path)
+                self.meta.episodes = load_episodes(self.root)
+
+                # Load new episode dataframe
+                chunk_idx = self.meta.episodes[ep_idx]["data/chunk_index"]
+                file_idx = self.meta.episodes[ep_idx]["data/file_index"]
+                episode_df_path = self.root / DEFAULT_EPISODES_PATH.format(
+                    chunk_index=chunk_idx, file_index=file_idx
+                )
+                episode_df = pd.read_parquet(episode_df_path)
+
+            # Save the current episode's video metadata to the dataframe
+            video_ep_metadata = {}
+            for video_key in self.meta.video_keys:
+                video_ep_metadata.update(self._save_episode_video(video_key, ep_idx))
+            video_ep_metadata.pop("episode_index")
+            video_ep_df = pd.DataFrame(video_ep_metadata, index=[ep_idx]).convert_dtypes(
+                dtype_backend="pyarrow"
+            )  # allows NaN values along with integers
+
+            episode_df = episode_df.combine_first(video_ep_df)
+            episode_df.to_parquet(episode_df_path)
+            self.meta.episodes = load_episodes(self.root)
+
+    def _save_episode_data(self, episode_buffer: dict) -> dict:
+        """Save episode data to a parquet file and update the Hugging Face dataset of frames data.
+
+        This function processes episodes data from a buffer, converts it into a Hugging Face dataset,
+        and saves it as a parquet file. It handles both the creation of new parquet files and the
+        updating of existing ones based on size constraints. After saving the data, it reloads
+        the Hugging Face dataset to ensure it is up-to-date.
+
+        Notes: We both need to update parquet files and HF dataset:
+        - `pandas` loads parquet file in RAM
+        - `datasets` relies on a memory mapping from pyarrow (no RAM). It either converts parquet files to a pyarrow cache on disk,
+          or loads directly from pyarrow cache.
+        """
+        # Convert buffer into HF Dataset
+        ep_dict = {key: episode_buffer[key] for key in self.hf_features}
+        ep_dataset = datasets.Dataset.from_dict(ep_dict, features=self.hf_features, split="train")
+        ep_dataset = embed_images(ep_dataset)
+        ep_num_frames = len(ep_dataset)
+
+        if self.latest_episode is None:
+            # Initialize indices and frame count for a new dataset made of the first episode data
+            chunk_idx, file_idx = 0, 0
+            global_frame_index = 0
+            self._current_file_start_frame = 0
+            # However, if the episodes already exists
+            # It means we are resuming recording, so we need to load the latest episode
+            # Update the indices to avoid overwriting the latest episode
+            if self.meta.episodes is not None and len(self.meta.episodes) > 0:
+                latest_ep = self.meta.episodes[-1]
+                global_frame_index = latest_ep["dataset_to_index"]
+                chunk_idx = latest_ep["data/chunk_index"]
+                file_idx = latest_ep["data/file_index"]
+
+                # When resuming, move to the next file
+                chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, self.meta.chunks_size)
+                self._current_file_start_frame = global_frame_index
+        else:
+            # Retrieve information from the latest parquet file
+            latest_ep = self.latest_episode
+            chunk_idx = latest_ep["data/chunk_index"]
+            file_idx = latest_ep["data/file_index"]
+            global_frame_index = latest_ep["index"][-1] + 1
+
+            latest_path = self.root / self.meta.data_path.format(chunk_index=chunk_idx, file_index=file_idx)
+            latest_size_in_mb = get_file_size_in_mb(latest_path)
+
+            frames_in_current_file = global_frame_index - self._current_file_start_frame
+            av_size_per_frame = (
+                latest_size_in_mb / frames_in_current_file if frames_in_current_file > 0 else 0
+            )
+
+            # Determine if a new parquet file is needed
+            if (
+                latest_size_in_mb + av_size_per_frame * ep_num_frames >= self.meta.data_files_size_in_mb
+                or self._writer_closed_for_reading
+            ):
+                # Size limit is reached or writer was closed for reading, prepare new parquet file
+                chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, self.meta.chunks_size)
+                self._close_writer()
+                self._writer_closed_for_reading = False
+                self._current_file_start_frame = global_frame_index
+
+        ep_dict["data/chunk_index"] = chunk_idx
+        ep_dict["data/file_index"] = file_idx
+
+        # Write the resulting dataframe from RAM to disk
+        path = self.root / self.meta.data_path.format(chunk_index=chunk_idx, file_index=file_idx)
+        path.parent.mkdir(parents=True, exist_ok=True)
+
+        table = ep_dataset.with_format("arrow")[:]
+        if not self.writer:
+            self.writer = pq.ParquetWriter(
+                path, schema=table.schema, compression="snappy", use_dictionary=True
+            )
+        self.writer.write_table(table)
+
+        metadata = {
+            "data/chunk_index": chunk_idx,
+            "data/file_index": file_idx,
+            "dataset_from_index": global_frame_index,
+            "dataset_to_index": global_frame_index + ep_num_frames,
+        }
+
+        # Store metadata with episode data for next episode
+        self.latest_episode = {**ep_dict, **metadata}
+
+        # Mark that the HF dataset needs reloading (lazy loading approach)
+        # This avoids expensive reloading during sequential recording
+        self._lazy_loading = True
+        # Update recorded frames count for efficient length tracking
+        self._recorded_frames += ep_num_frames
+
+        return metadata
+
+    def _save_episode_video(
+        self,
+        video_key: str,
+        episode_index: int,
+        temp_path: Path | None = None,
+    ) -> dict:
+        # Encode episode frames into a temporary video
+        if temp_path is None:
+            ep_path = self._encode_temporary_episode_video(video_key, episode_index)
+        else:
+            ep_path = temp_path
+
+        ep_size_in_mb = get_file_size_in_mb(ep_path)
+        ep_duration_in_s = get_video_duration_in_s(ep_path)
+
+        if (
+            episode_index == 0
+            or self.meta.latest_episode is None
+            or f"videos/{video_key}/chunk_index" not in self.meta.latest_episode
+        ):
+            # Initialize indices for a new dataset made of the first episode data
+            chunk_idx, file_idx = 0, 0
+            if self.meta.episodes is not None and len(self.meta.episodes) > 0:
+                # It means we are resuming recording, so we need to load the latest episode
+                # Update the indices to avoid overwriting the latest episode
+                old_chunk_idx = self.meta.episodes[-1][f"videos/{video_key}/chunk_index"]
+                old_file_idx = self.meta.episodes[-1][f"videos/{video_key}/file_index"]
+                chunk_idx, file_idx = update_chunk_file_indices(
+                    old_chunk_idx, old_file_idx, self.meta.chunks_size
+                )
+            latest_duration_in_s = 0.0
+            new_path = self.root / self.meta.video_path.format(
+                video_key=video_key, chunk_index=chunk_idx, file_index=file_idx
+            )
+            new_path.parent.mkdir(parents=True, exist_ok=True)
+            shutil.move(str(ep_path), str(new_path))
+        else:
+            # Retrieve information from the latest updated video file using latest_episode
+            latest_ep = self.meta.latest_episode
+            chunk_idx = latest_ep[f"videos/{video_key}/chunk_index"][0]
+            file_idx = latest_ep[f"videos/{video_key}/file_index"][0]
+
+            latest_path = self.root / self.meta.video_path.format(
+                video_key=video_key, chunk_index=chunk_idx, file_index=file_idx
+            )
+            latest_size_in_mb = get_file_size_in_mb(latest_path)
+            latest_duration_in_s = latest_ep[f"videos/{video_key}/to_timestamp"][0]
+
+            if latest_size_in_mb + ep_size_in_mb >= self.meta.video_files_size_in_mb:
+                # Move temporary episode video to a new video file in the dataset
+                chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, self.meta.chunks_size)
+                new_path = self.root / self.meta.video_path.format(
+                    video_key=video_key, chunk_index=chunk_idx, file_index=file_idx
+                )
+                new_path.parent.mkdir(parents=True, exist_ok=True)
+                shutil.move(str(ep_path), str(new_path))
+                latest_duration_in_s = 0.0
+            else:
+                # Update latest video file
+                concatenate_video_files(
+                    [latest_path, ep_path],
+                    latest_path,
+                )
+
+        # Remove temporary directory
+        shutil.rmtree(str(ep_path.parent))
+
+        # Update video info (only needed when first episode is encoded since it reads from episode 0)
+        if episode_index == 0:
+            self.meta.update_video_info(video_key)
+            write_info(self.meta.info, self.meta.root)  # ensure video info always written properly
+
+        metadata = {
+            "episode_index": episode_index,
+            f"videos/{video_key}/chunk_index": chunk_idx,
+            f"videos/{video_key}/file_index": file_idx,
+            f"videos/{video_key}/from_timestamp": latest_duration_in_s,
+            f"videos/{video_key}/to_timestamp": latest_duration_in_s + ep_duration_in_s,
+        }
+        return metadata
+
+    def clear_episode_buffer(self, delete_images: bool = True) -> None:
+        # Cancel streaming encoder if active
+        if self._streaming_encoder is not None:
+            self._streaming_encoder.cancel_episode()
+
+        # Clean up image files for the current episode buffer
+        if delete_images:
+            # Wait for the async image writer to finish
+            if self.image_writer is not None:
+                self._wait_image_writer()
+            episode_index = self.episode_buffer["episode_index"]
+            if isinstance(episode_index, np.ndarray):
+                episode_index = episode_index.item() if episode_index.size == 1 else episode_index[0]
+            for cam_key in self.meta.image_keys:
+                img_dir = self._get_image_file_dir(episode_index, cam_key)
+                if img_dir.is_dir():
+                    shutil.rmtree(img_dir)
+
+        # Reset the buffer
+        self.episode_buffer = self.create_episode_buffer()
+
+    def start_image_writer(self, num_processes: int = 0, num_threads: int = 4) -> None:
+        if isinstance(self.image_writer, AsyncImageWriter):
+            logger.warning(
+                "You are starting a new AsyncImageWriter that is replacing an already existing one in the dataset."
+            )
+
+        self.image_writer = AsyncImageWriter(
+            num_processes=num_processes,
+            num_threads=num_threads,
+        )
+
+    def stop_image_writer(self) -> None:
+        """
+        Whenever wrapping this dataset inside a parallelized DataLoader, this needs to be called first to
+        remove the image_writer in order for the LeRobotDataset object to be pickleable and parallelized.
+        """
+        if self.image_writer is not None:
+            self.image_writer.stop()
+            self.image_writer = None
+
+    def _wait_image_writer(self) -> None:
+        """Wait for asynchronous image writer to finish."""
+        if self.image_writer is not None:
+            self.image_writer.wait_until_done()
+
+    def _encode_temporary_episode_video(self, video_key: str, episode_index: int) -> Path:
+        """
+        Use ffmpeg to convert frames stored as png into mp4 videos.
+        Note: `encode_video_frames` is a blocking call. Making it asynchronous shouldn't speedup encoding,
+        since video encoding with ffmpeg is already using multithreading.
+        """
+        return _encode_video_worker(
+            video_key, episode_index, self.root, self.fps, self.vcodec, self._encoder_threads
+        )
+
+    @classmethod
+    def create(
+        cls,
+        repo_id: str,
+        fps: int,
+        features: dict,
+        root: str | Path | None = None,
+        robot_type: str | None = None,
+        use_videos: bool = True,
+        tolerance_s: float = 1e-4,
+        image_writer_processes: int = 0,
+        image_writer_threads: int = 0,
+        video_backend: str | None = None,
+        batch_encoding_size: int = 1,
+        vcodec: str = "libsvtav1",
+        metadata_buffer_size: int = 10,
+        streaming_encoding: bool = False,
+        encoder_queue_maxsize: int = 30,
+        encoder_threads: int | None = None,
+    ) -> "LeRobotDataset":
+        """Create a LeRobot Dataset from scratch in order to record data."""
+        vcodec = resolve_vcodec(vcodec)
+        obj = cls.__new__(cls)
+        obj.meta = LeRobotDatasetMetadata.create(
+            repo_id=repo_id,
+            fps=fps,
+            robot_type=robot_type,
+            features=features,
+            root=root,
+            use_videos=use_videos,
+            metadata_buffer_size=metadata_buffer_size,
+        )
+        obj.repo_id = obj.meta.repo_id
+        obj.root = obj.meta.root
+        obj.revision = None
+        obj.tolerance_s = tolerance_s
+        obj.image_writer = None
+        obj.batch_encoding_size = batch_encoding_size
+        obj.episodes_since_last_encoding = 0
+        obj.vcodec = vcodec
+        obj._encoder_threads = encoder_threads
+
+        if image_writer_processes or image_writer_threads:
+            obj.start_image_writer(image_writer_processes, image_writer_threads)
+
+        obj.episode_buffer = obj.create_episode_buffer()
+
+        obj.episodes = None
+        obj.hf_dataset = obj.create_hf_dataset()
+        obj.image_transforms = None
+        obj.delta_timestamps = None
+        obj.delta_indices = None
+        obj._absolute_to_relative_idx = None
+        obj.video_backend = video_backend if video_backend is not None else get_safe_default_codec()
+        obj.writer = None
+        obj.latest_episode = None
+        obj._current_file_start_frame = None
+        # Initialize tracking for incremental recording
+        obj._lazy_loading = False
+        obj._recorded_frames = 0
+        obj._writer_closed_for_reading = False
+
+        # Initialize streaming encoder
+        if streaming_encoding and len(obj.meta.video_keys) > 0:
+            obj._streaming_encoder = StreamingVideoEncoder(
+                fps=fps,
+                vcodec=vcodec,
+                pix_fmt="yuv420p",
+                g=2,
+                crf=30,
+                preset=None,
+                queue_maxsize=encoder_queue_maxsize,
+                encoder_threads=encoder_threads,
+            )
+        else:
+            obj._streaming_encoder = None
+
+        return obj
diff --git a/lerobot/src/lerobot/datasets/multi_dataset.py b/lerobot/src/lerobot/datasets/multi_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..917d5c5ebc8c499863ae56f1151bf38891a27df0
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/multi_dataset.py
@@ -0,0 +1,210 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+from collections.abc import Callable
+from pathlib import Path
+
+import datasets
+import torch
+import torch.utils
+
+from lerobot.datasets.compute_stats import aggregate_stats
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.video_utils import VideoFrame
+from lerobot.utils.constants import HF_LEROBOT_HOME
+
+logger = logging.getLogger(__name__)
+
+
+class MultiLeRobotDataset(torch.utils.data.Dataset):
+    """A dataset consisting of multiple underlying `LeRobotDataset`s.
+
+    The underlying `LeRobotDataset`s are effectively concatenated, and this class adopts much of the API
+    structure of `LeRobotDataset`.
+    """
+
+    def __init__(
+        self,
+        repo_ids: list[str],
+        root: str | Path | None = None,
+        episodes: dict | None = None,
+        image_transforms: Callable | None = None,
+        delta_timestamps: dict[str, list[float]] | None = None,
+        tolerances_s: dict | None = None,
+        download_videos: bool = True,
+        video_backend: str | None = None,
+    ):
+        super().__init__()
+        self.repo_ids = repo_ids
+        self.root = Path(root) if root else HF_LEROBOT_HOME
+        self.tolerances_s = tolerances_s if tolerances_s else dict.fromkeys(repo_ids, 0.0001)
+        # Construct the underlying datasets passing everything but `transform` and `delta_timestamps` which
+        # are handled by this class.
+        self._datasets = [
+            LeRobotDataset(
+                repo_id,
+                root=self.root / repo_id,
+                episodes=episodes[repo_id] if episodes else None,
+                image_transforms=image_transforms,
+                delta_timestamps=delta_timestamps,
+                tolerance_s=self.tolerances_s[repo_id],
+                download_videos=download_videos,
+                video_backend=video_backend,
+            )
+            for repo_id in repo_ids
+        ]
+
+        # Disable any data keys that are not common across all of the datasets. Note: we may relax this
+        # restriction in future iterations of this class. For now, this is necessary at least for being able
+        # to use PyTorch's default DataLoader collate function.
+        self.disabled_features = set()
+        intersection_features = set(self._datasets[0].features)
+        for ds in self._datasets:
+            intersection_features.intersection_update(ds.features)
+        if len(intersection_features) == 0:
+            raise RuntimeError(
+                "Multiple datasets were provided but they had no keys common to all of them. "
+                "The multi-dataset functionality currently only keeps common keys."
+            )
+        for repo_id, ds in zip(self.repo_ids, self._datasets, strict=True):
+            extra_keys = set(ds.features).difference(intersection_features)
+            if extra_keys:
+                logger.warning(
+                    f"keys {extra_keys} of {repo_id} were disabled as they are not contained in all the "
+                    "other datasets."
+                )
+                self.disabled_features.update(extra_keys)
+
+        self.image_transforms = image_transforms
+        self.delta_timestamps = delta_timestamps
+        # TODO(rcadene, aliberts): We should not perform this aggregation for datasets
+        # with multiple robots of different ranges. Instead we should have one normalization
+        # per robot.
+        self.stats = aggregate_stats([dataset.meta.stats for dataset in self._datasets])
+
+    @property
+    def repo_id_to_index(self):
+        """Return a mapping from dataset repo_id to a dataset index automatically created by this class.
+
+        This index is incorporated as a data key in the dictionary returned by `__getitem__`.
+        """
+        return {repo_id: i for i, repo_id in enumerate(self.repo_ids)}
+
+    @property
+    def fps(self) -> int:
+        """Frames per second used during data collection.
+
+        NOTE: Fow now, this relies on a check in __init__ to make sure all sub-datasets have the same info.
+        """
+        return self._datasets[0].meta.info["fps"]
+
+    @property
+    def video(self) -> bool:
+        """Returns True if this dataset loads video frames from mp4 files.
+
+        Returns False if it only loads images from png files.
+
+        NOTE: Fow now, this relies on a check in __init__ to make sure all sub-datasets have the same info.
+        """
+        return self._datasets[0].meta.info.get("video", False)
+
+    @property
+    def features(self) -> datasets.Features:
+        features = {}
+        for dataset in self._datasets:
+            features.update({k: v for k, v in dataset.hf_features.items() if k not in self.disabled_features})
+        return features
+
+    @property
+    def camera_keys(self) -> list[str]:
+        """Keys to access image and video stream from cameras."""
+        keys = []
+        for key, feats in self.features.items():
+            if isinstance(feats, (datasets.Image | VideoFrame)):
+                keys.append(key)
+        return keys
+
+    @property
+    def video_frame_keys(self) -> list[str]:
+        """Keys to access video frames that requires to be decoded into images.
+
+        Note: It is empty if the dataset contains images only,
+        or equal to `self.cameras` if the dataset contains videos only,
+        or can even be a subset of `self.cameras` in a case of a mixed image/video dataset.
+        """
+        video_frame_keys = []
+        for key, feats in self.features.items():
+            if isinstance(feats, VideoFrame):
+                video_frame_keys.append(key)
+        return video_frame_keys
+
+    @property
+    def num_frames(self) -> int:
+        """Number of samples/frames."""
+        return sum(d.num_frames for d in self._datasets)
+
+    @property
+    def num_episodes(self) -> int:
+        """Number of episodes."""
+        return sum(d.num_episodes for d in self._datasets)
+
+    @property
+    def tolerance_s(self) -> float:
+        """Tolerance in seconds used to discard loaded frames when their timestamps
+        are not close enough from the requested frames. It is only used when `delta_timestamps`
+        is provided or when loading video frames from mp4 files.
+        """
+        # 1e-4 to account for possible numerical error
+        return 1 / self.fps - 1e-4
+
+    def __len__(self):
+        return self.num_frames
+
+    def __getitem__(self, idx: int) -> dict[str, torch.Tensor]:
+        if idx >= len(self):
+            raise IndexError(f"Index {idx} out of bounds.")
+        # Determine which dataset to get an item from based on the index.
+        start_idx = 0
+        dataset_idx = 0
+        for dataset in self._datasets:
+            if idx >= start_idx + dataset.num_frames:
+                start_idx += dataset.num_frames
+                dataset_idx += 1
+                continue
+            break
+        else:
+            raise AssertionError("We expect the loop to break out as long as the index is within bounds.")
+        item = self._datasets[dataset_idx][idx - start_idx]
+        item["dataset_index"] = torch.tensor(dataset_idx)
+        for data_key in self.disabled_features:
+            if data_key in item:
+                del item[data_key]
+
+        return item
+
+    def __repr__(self):
+        return (
+            f"{self.__class__.__name__}(\n"
+            f"  Repository IDs: '{self.repo_ids}',\n"
+            f"  Number of Samples: {self.num_frames},\n"
+            f"  Number of Episodes: {self.num_episodes},\n"
+            f"  Type: {'video (.mp4)' if self.video else 'image (.png)'},\n"
+            f"  Recorded Frames per Second: {self.fps},\n"
+            f"  Camera Keys: {self.camera_keys},\n"
+            f"  Video Frame Keys: {self.video_frame_keys if self.video else 'N/A'},\n"
+            f"  Transformations: {self.image_transforms},\n"
+            f")"
+        )
diff --git a/lerobot/src/lerobot/datasets/pipeline_features.py b/lerobot/src/lerobot/datasets/pipeline_features.py
new file mode 100644
index 0000000000000000000000000000000000000000..96779fdc64b5038d9543a7196a1b2c67c872b222
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/pipeline_features.py
@@ -0,0 +1,142 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import re
+from collections.abc import Sequence
+from typing import Any
+
+from lerobot.configs.types import PipelineFeatureType
+from lerobot.datasets.feature_utils import hw_to_dataset_features
+from lerobot.processor import DataProcessorPipeline
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.constants import ACTION, OBS_IMAGES, OBS_STATE, OBS_STR
+
+
+def create_initial_features(
+    action: RobotAction | None = None, observation: RobotObservation | None = None
+) -> dict[PipelineFeatureType, dict[str, Any]]:
+    """
+    Creates the initial features dict for the dataset from action and observation specs.
+
+    Args:
+        action: A dictionary of action feature names to their types/shapes.
+        observation: A dictionary of observation feature names to their types/shapes.
+
+    Returns:
+        The initial features dictionary structured by PipelineFeatureType.
+    """
+    features = {PipelineFeatureType.ACTION: {}, PipelineFeatureType.OBSERVATION: {}}
+    if action:
+        features[PipelineFeatureType.ACTION] = action
+    if observation:
+        features[PipelineFeatureType.OBSERVATION] = observation
+    return features
+
+
+# Helper to filter state/action keys based on compiled regex patterns.
+def should_keep(key: str, patterns: tuple[re.Pattern] | None) -> bool:
+    if patterns is None:
+        return True
+    return any(pat.search(key) for pat in patterns)
+
+
+def strip_prefix(key: str, prefixes_to_strip: tuple[str]) -> str:
+    for prefix in prefixes_to_strip:
+        if key.startswith(prefix):
+            return key[len(prefix) :]
+    return key
+
+
+# Define prefixes to strip from feature keys for clean names.
+# Handles both fully qualified (e.g., "action.state") and short (e.g., "state") forms.
+PREFIXES_TO_STRIP = tuple(
+    f"{token}." for const in (ACTION, OBS_STATE, OBS_IMAGES) for token in (const, const.split(".")[-1])
+)
+
+
+def aggregate_pipeline_dataset_features(
+    pipeline: DataProcessorPipeline,
+    initial_features: dict[PipelineFeatureType, dict[str, Any]],
+    *,
+    use_videos: bool = True,
+    patterns: Sequence[str] | None = None,
+) -> dict[str, dict]:
+    """
+    Aggregates and filters pipeline features to create a dataset-ready features dictionary.
+
+    This function transforms initial features using the pipeline, categorizes them as action or observations
+    (image or state), filters them based on `use_videos` and `patterns`, and finally
+    formats them for use with a Hugging Face LeRobot Dataset.
+
+    Args:
+        pipeline: The DataProcessorPipeline to apply.
+        initial_features: A dictionary of raw feature specs for actions and observations.
+        use_videos: If False, image features are excluded.
+        patterns: A sequence of regex patterns to filter action and state features.
+                  Image features are not affected by this filter.
+
+    Returns:
+        A dictionary of features formatted for a Hugging Face LeRobot Dataset.
+    """
+    compiled_patterns = tuple(re.compile(p) for p in patterns) if patterns is not None else None
+
+    all_features = pipeline.transform_features(initial_features)
+
+    # Intermediate storage for categorized and filtered features.
+    processed_features: dict[str, dict[str, Any]] = {
+        ACTION: {},
+        OBS_STR: {},
+    }
+    images_token = OBS_IMAGES.split(".")[-1]
+
+    # Iterate through all features transformed by the pipeline.
+    for ptype, feats in all_features.items():
+        if ptype not in [PipelineFeatureType.ACTION, PipelineFeatureType.OBSERVATION]:
+            continue
+
+        for key, value in feats.items():
+            # 1. Categorize the feature.
+            is_action = ptype == PipelineFeatureType.ACTION
+            # Observations are classified as images if their key matches image-related tokens or if the shape of the feature is 3.
+            # All other observations are treated as state.
+            is_image = not is_action and (
+                (isinstance(value, tuple) and len(value) == 3)
+                or (
+                    key.startswith(f"{OBS_IMAGES}.")
+                    or key.startswith(f"{images_token}.")
+                    or f".{images_token}." in key
+                )
+            )
+
+            # 2. Apply filtering rules.
+            if is_image and not use_videos:
+                continue
+            if not is_image and not should_keep(key, compiled_patterns):
+                continue
+
+            # 3. Add the feature to the appropriate group with a clean name.
+            name = strip_prefix(key, PREFIXES_TO_STRIP)
+            if is_action:
+                processed_features[ACTION][name] = value
+            else:
+                processed_features[OBS_STR][name] = value
+
+    # Convert the processed features into the final dataset format.
+    dataset_features = {}
+    if processed_features[ACTION]:
+        dataset_features.update(hw_to_dataset_features(processed_features[ACTION], ACTION, use_videos))
+    if processed_features[OBS_STR]:
+        dataset_features.update(hw_to_dataset_features(processed_features[OBS_STR], OBS_STR, use_videos))
+
+    return dataset_features
diff --git a/lerobot/src/lerobot/datasets/sampler.py b/lerobot/src/lerobot/datasets/sampler.py
new file mode 100644
index 0000000000000000000000000000000000000000..2bf7ab922da62be88a5758d395665e255ead5cfc
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/sampler.py
@@ -0,0 +1,86 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+from collections.abc import Iterator
+
+import torch
+
+logger = logging.getLogger(__name__)
+
+
+class EpisodeAwareSampler:
+    def __init__(
+        self,
+        dataset_from_indices: list[int],
+        dataset_to_indices: list[int],
+        episode_indices_to_use: list | None = None,
+        drop_n_first_frames: int = 0,
+        drop_n_last_frames: int = 0,
+        shuffle: bool = False,
+    ):
+        """Sampler that optionally incorporates episode boundary information.
+
+        Args:
+            dataset_from_indices: List of indices containing the start of each episode in the dataset.
+            dataset_to_indices: List of indices containing the end of each episode in the dataset.
+            episode_indices_to_use: List of episode indices to use. If None, all episodes are used.
+                                    Assumes that episodes are indexed from 0 to N-1.
+            drop_n_first_frames: Number of frames to drop from the start of each episode.
+            drop_n_last_frames: Number of frames to drop from the end of each episode.
+            shuffle: Whether to shuffle the indices.
+        """
+        if drop_n_first_frames < 0:
+            raise ValueError(f"drop_n_first_frames must be >= 0, got {drop_n_first_frames}")
+        if drop_n_last_frames < 0:
+            raise ValueError(f"drop_n_last_frames must be >= 0, got {drop_n_last_frames}")
+
+        indices = []
+        for episode_idx, (start_index, end_index) in enumerate(
+            zip(dataset_from_indices, dataset_to_indices, strict=True)
+        ):
+            if episode_indices_to_use is None or episode_idx in episode_indices_to_use:
+                ep_length = end_index - start_index
+                if drop_n_first_frames + drop_n_last_frames >= ep_length:
+                    logger.warning(
+                        "Episode %d has %d frames but drop_n_first_frames=%d and "
+                        "drop_n_last_frames=%d removes all frames. Skipping.",
+                        episode_idx,
+                        ep_length,
+                        drop_n_first_frames,
+                        drop_n_last_frames,
+                    )
+                    continue
+                indices.extend(range(start_index + drop_n_first_frames, end_index - drop_n_last_frames))
+
+        if not indices:
+            raise ValueError(
+                "No valid frames remain after applying drop_n_first_frames and drop_n_last_frames. "
+                "All episodes were either filtered out or had too few frames."
+            )
+
+        self.indices = indices
+        self.shuffle = shuffle
+
+    def __iter__(self) -> Iterator[int]:
+        if self.shuffle:
+            for i in torch.randperm(len(self.indices)):
+                yield self.indices[i]
+        else:
+            for i in self.indices:
+                yield i
+
+    def __len__(self) -> int:
+        return len(self.indices)
diff --git a/lerobot/src/lerobot/datasets/streaming_dataset.py b/lerobot/src/lerobot/datasets/streaming_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..62e00558a6926a8e919196a882b681ddf5474aaf
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/streaming_dataset.py
@@ -0,0 +1,689 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from collections import deque
+from collections.abc import Callable, Generator, Iterable, Iterator
+from pathlib import Path
+
+import datasets
+import numpy as np
+import torch
+from datasets import load_dataset
+
+from lerobot.datasets.dataset_metadata import CODEBASE_VERSION, LeRobotDatasetMetadata
+from lerobot.datasets.feature_utils import get_delta_indices
+from lerobot.datasets.io_utils import item_to_torch
+from lerobot.datasets.utils import (
+    check_version_compatibility,
+    find_float_index,
+    is_float_in_list,
+    safe_shard,
+)
+from lerobot.datasets.video_utils import (
+    VideoDecoderCache,
+    decode_video_frames_torchcodec,
+)
+from lerobot.utils.constants import HF_LEROBOT_HOME, LOOKAHEAD_BACKTRACKTABLE, LOOKBACK_BACKTRACKTABLE
+
+
+class LookBackError(Exception):
+    """
+    Exception raised when trying to look back in the history of a Backtrackable object.
+    """
+
+    pass
+
+
+class LookAheadError(Exception):
+    """
+    Exception raised when trying to look ahead in the future of a Backtrackable object.
+    """
+
+    pass
+
+
+class Backtrackable[T]:
+    """
+    Wrap any iterator/iterable so you can step back up to `history` items
+    and look ahead up to `lookahead` items.
+
+    This is useful for streaming datasets where you need to access previous and future items
+    but can't load the entire dataset into memory.
+
+    Example:
+    -------
+    ```python
+    ds = load_dataset("c4", "en", streaming=True, split="train")
+    rev = Backtrackable(ds, history=3, lookahead=2)
+
+    x0 = next(rev)  # forward
+    x1 = next(rev)
+    x2 = next(rev)
+
+    # Look ahead
+    x3_peek = rev.peek_ahead(1)  # next item without moving cursor
+    x4_peek = rev.peek_ahead(2)  # two items ahead
+
+    # Look back
+    x1_again = rev.peek_back(1)  # previous item without moving cursor
+    x0_again = rev.peek_back(2)  # two items back
+
+    # Move backward
+    x1_back = rev.prev()  # back one step
+    next(rev)  # returns x2, continues forward from where we were
+    ```
+    """
+
+    __slots__ = ("_source", "_back_buf", "_ahead_buf", "_cursor", "_history", "_lookahead")
+
+    def __init__(self, iterable: Iterable[T], *, history: int = 1, lookahead: int = 0):
+        if history < 1:
+            raise ValueError("history must be >= 1")
+        if lookahead <= 0:
+            raise ValueError("lookahead must be > 0")
+
+        self._source: Iterator[T] = iter(iterable)
+        self._back_buf: deque[T] = deque(maxlen=history)
+        self._ahead_buf: deque[T] = deque(maxlen=lookahead) if lookahead > 0 else deque()
+        self._cursor: int = 0
+        self._history = history
+        self._lookahead = lookahead
+
+    def __iter__(self) -> "Backtrackable[T]":
+        return self
+
+    def __next__(self) -> T:
+        # If we've stepped back, consume from back buffer first
+        if self._cursor < 0:  # -1 means "last item", etc.
+            self._cursor += 1
+            return self._back_buf[self._cursor]
+
+        # If we have items in the ahead buffer, use them first
+        item = self._ahead_buf.popleft() if self._ahead_buf else next(self._source)
+
+        # Add current item to back buffer and reset cursor
+        self._back_buf.append(item)
+        self._cursor = 0
+        return item
+
+    def prev(self) -> T:
+        """
+        Step one item back in history and return it.
+        Raises IndexError if already at the oldest buffered item.
+        """
+        if len(self._back_buf) + self._cursor <= 1:
+            raise LookBackError("At start of history")
+
+        self._cursor -= 1
+        return self._back_buf[self._cursor]
+
+    def peek_back(self, n: int = 1) -> T:
+        """
+        Look `n` items back (n=1 == previous item) without moving the cursor.
+        """
+        if n < 0 or n + 1 > len(self._back_buf) + self._cursor:
+            raise LookBackError("peek_back distance out of range")
+
+        return self._back_buf[self._cursor - (n + 1)]
+
+    def peek_ahead(self, n: int = 1) -> T:
+        """
+        Look `n` items ahead (n=1 == next item) without moving the cursor.
+        Fills the ahead buffer if necessary.
+        """
+        if n < 1:
+            raise LookAheadError("peek_ahead distance must be 1 or more")
+        elif n > self._lookahead:
+            raise LookAheadError("peek_ahead distance exceeds lookahead limit")
+
+        # Fill ahead buffer if we don't have enough items
+        while len(self._ahead_buf) < n:
+            try:
+                item = next(self._source)
+                self._ahead_buf.append(item)
+
+            except StopIteration as err:
+                raise LookAheadError("peek_ahead: not enough items in source") from err
+
+        return self._ahead_buf[n - 1]
+
+    def history(self) -> list[T]:
+        """
+        Return a copy of the buffered history (most recent last).
+        The list length ≤ `history` argument passed at construction.
+        """
+        if self._cursor == 0:
+            return list(self._back_buf)
+
+        # When cursor<0, slice so the order remains chronological
+        return list(self._back_buf)[: self._cursor or None]
+
+    def can_peek_back(self, steps: int = 1) -> bool:
+        """
+        Check if we can go back `steps` items without raising an IndexError.
+        """
+        return steps <= len(self._back_buf) + self._cursor
+
+    def can_peek_ahead(self, steps: int = 1) -> bool:
+        """
+        Check if we can peek ahead `steps` items.
+        This may involve trying to fill the ahead buffer.
+        """
+        if self._lookahead > 0 and steps > self._lookahead:
+            return False
+
+        # Try to fill ahead buffer to check if we can peek that far
+        try:
+            while len(self._ahead_buf) < steps:
+                if self._lookahead > 0 and len(self._ahead_buf) >= self._lookahead:
+                    return False
+                item = next(self._source)
+                self._ahead_buf.append(item)
+            return True
+        except StopIteration:
+            return False
+
+
+class StreamingLeRobotDataset(torch.utils.data.IterableDataset):
+    """LeRobotDataset with streaming capabilities.
+
+    This class extends LeRobotDataset to add streaming functionality, allowing data to be streamed
+    rather than loaded entirely into memory. This is especially useful for large datasets that may
+    not fit in memory or when you want to quickly explore a dataset without downloading it completely.
+
+    The key innovation is using a Backtrackable iterator that maintains a bounded buffer of recent
+    items, allowing us to access previous frames for delta timestamps without loading the entire
+    dataset into memory.
+
+    Example:
+        Basic usage:
+        ```python
+        from lerobot.common.datasets.streaming_dataset import StreamingLeRobotDataset
+
+        # Create a streaming dataset with delta timestamps
+        delta_timestamps = {
+            "observation.image": [-1.0, -0.5, 0.0],  # 1 sec ago, 0.5 sec ago, current
+            "action": [0.0, 0.1, 0.2],  # current, 0.1 sec future, 0.2 sec future
+        }
+
+        dataset = StreamingLeRobotDataset(
+            repo_id="your-dataset-repo-id",
+            delta_timestamps=delta_timestamps,
+            streaming=True,
+            buffer_size=1000,
+        )
+
+        # Iterate over the dataset
+        for i, item in enumerate(dataset):
+            print(f"Sample {i}: Episode {item['episode_index']} Frame {item['frame_index']}")
+            # item will contain stacked frames according to delta_timestamps
+            if i >= 10:
+                break
+        ```
+    """
+
+    def __init__(
+        self,
+        repo_id: str,
+        root: str | Path | None = None,
+        episodes: list[int] | None = None,
+        image_transforms: Callable | None = None,
+        delta_timestamps: dict[list[float]] | None = None,
+        tolerance_s: float = 1e-4,
+        revision: str | None = None,
+        force_cache_sync: bool = False,
+        streaming: bool = True,
+        buffer_size: int = 1000,
+        max_num_shards: int = 16,
+        seed: int = 42,
+        rng: np.random.Generator | None = None,
+        shuffle: bool = True,
+    ):
+        """Initialize a StreamingLeRobotDataset.
+
+        Args:
+            repo_id (str): This is the repo id that will be used to fetch the dataset.
+            root (Path | None, optional): Local directory to use for downloading/writing files.
+            episodes (list[int] | None, optional): If specified, this will only load episodes specified by
+                their episode_index in this list.
+            image_transforms (Callable | None, optional): Transform to apply to image data.
+            tolerance_s (float, optional): Tolerance in seconds for timestamp matching.
+            revision (str, optional): Git revision id (branch name, tag, or commit hash).
+            force_cache_sync (bool, optional): Flag to sync and refresh local files first.
+            streaming (bool, optional): Whether to stream the dataset or load it all. Defaults to True.
+            buffer_size (int, optional): Buffer size for shuffling when streaming. Defaults to 1000.
+            max_num_shards (int, optional): Number of shards to re-shard the input dataset into. Defaults to 16.
+            seed (int, optional): Reproducibility random seed.
+            rng (np.random.Generator | None, optional): Random number generator.
+            shuffle (bool, optional): Whether to shuffle the dataset across exhaustions. Defaults to True.
+        """
+        super().__init__()
+        self.repo_id = repo_id
+        self.root = Path(root) if root else HF_LEROBOT_HOME / repo_id
+        self.streaming_from_local = root is not None
+
+        self.image_transforms = image_transforms
+        self.episodes = episodes
+        self.tolerance_s = tolerance_s
+        self.revision = revision if revision else CODEBASE_VERSION
+        self.seed = seed
+        self.rng = rng if rng is not None else np.random.default_rng(seed)
+        self.shuffle = shuffle
+
+        self.streaming = streaming
+        self.buffer_size = buffer_size
+
+        # We cache the video decoders to avoid re-initializing them at each frame (avoiding a ~10x slowdown)
+        self.video_decoder_cache = None
+
+        self.root.mkdir(exist_ok=True, parents=True)
+
+        # Load metadata
+        self.meta = LeRobotDatasetMetadata(
+            self.repo_id, self.root, self.revision, force_cache_sync=force_cache_sync
+        )
+        # Check version
+        check_version_compatibility(self.repo_id, self.meta._version, CODEBASE_VERSION)
+
+        self.delta_timestamps = None
+        self.delta_indices = None
+
+        if delta_timestamps is not None:
+            self._validate_delta_timestamp_keys(delta_timestamps)  # raises ValueError if invalid
+            self.delta_timestamps = delta_timestamps
+            self.delta_indices = get_delta_indices(self.delta_timestamps, self.fps)
+
+        self.hf_dataset: datasets.IterableDataset = load_dataset(
+            self.repo_id if not self.streaming_from_local else str(self.root),
+            split="train",
+            streaming=self.streaming,
+            data_files="data/*/*.parquet",
+            revision=self.revision,
+        )
+
+        self.num_shards = min(self.hf_dataset.num_shards, max_num_shards)
+
+    @property
+    def num_frames(self):
+        return self.meta.total_frames
+
+    @property
+    def num_episodes(self):
+        return self.meta.total_episodes
+
+    @property
+    def fps(self):
+        return self.meta.fps
+
+    @staticmethod
+    def _iter_random_indices(
+        rng: np.random.Generator, buffer_size: int, random_batch_size=100
+    ) -> Iterator[int]:
+        while True:
+            yield from (int(i) for i in rng.integers(0, buffer_size, size=random_batch_size))
+
+    @staticmethod
+    def _infinite_generator_over_elements(rng: np.random.Generator, elements: list[int]) -> Iterator[int]:
+        while True:
+            yield rng.choice(elements)
+
+    # TODO(fracapuano): Implement multi-threaded prefetching to accelerate data loading.
+    # The current sequential iteration is a bottleneck. A producer-consumer pattern
+    # could be used with a ThreadPoolExecutor to run `make_frame` (especially video decoding)
+    # in parallel, feeding a queue from which this iterator will yield processed items.
+    def __iter__(self) -> Iterator[dict[str, torch.Tensor]]:
+        if self.video_decoder_cache is None:
+            self.video_decoder_cache = VideoDecoderCache()
+
+        # keep the same seed across exhaustions if shuffle is False, otherwise shuffle data across exhaustions
+        rng = np.random.default_rng(self.seed) if not self.shuffle else self.rng
+
+        buffer_indices_generator = self._iter_random_indices(rng, self.buffer_size)
+
+        idx_to_backtrack_dataset = {
+            idx: self._make_backtrackable_dataset(safe_shard(self.hf_dataset, idx, self.num_shards))
+            for idx in range(self.num_shards)
+        }
+
+        # This buffer is populated while iterating on the dataset's shards
+        # the logic is to add 2 levels of randomness:
+        # (1) sample one shard at random from the ones available, and
+        # (2) sample one frame from the shard sampled at (1)
+        frames_buffer = []
+        while available_shards := list(idx_to_backtrack_dataset.keys()):
+            shard_key = next(self._infinite_generator_over_elements(rng, available_shards))
+            backtrack_dataset = idx_to_backtrack_dataset[shard_key]  # selects which shard to iterate on
+
+            try:
+                for frame in self.make_frame(backtrack_dataset):
+                    if len(frames_buffer) == self.buffer_size:
+                        i = next(buffer_indices_generator)  # samples a element from the buffer
+                        yield frames_buffer[i]
+                        frames_buffer[i] = frame
+                    else:
+                        frames_buffer.append(frame)
+                    break  # random shard sampled, switch shard
+            except (
+                RuntimeError,
+                StopIteration,
+            ):  # NOTE: StopIteration inside a generator throws a RuntimeError since python 3.7
+                del idx_to_backtrack_dataset[shard_key]  # Remove exhausted shard, onto another shard
+
+        # Once shards are all exhausted, shuffle the buffer and yield the remaining frames
+        rng.shuffle(frames_buffer)
+        yield from frames_buffer
+
+    def _get_window_steps(
+        self, delta_timestamps: dict[str, list[float]] | None = None, dynamic_bounds: bool = False
+    ) -> tuple[int, int]:
+        if delta_timestamps is None:
+            return 1, 1
+
+        if not dynamic_bounds:
+            # Fix the windows
+            lookback = LOOKBACK_BACKTRACKTABLE
+            lookahead = LOOKAHEAD_BACKTRACKTABLE
+        else:
+            # Dynamically adjust the windows based on the given delta_timesteps
+            all_timestamps = sum(delta_timestamps.values(), [])
+            lookback = min(all_timestamps) * self.fps
+            lookahead = max(all_timestamps) * self.fps
+
+            # When lookback is >=0 it means no negative timesteps have been provided
+            lookback = 0 if lookback >= 0 else (lookback * -1)
+
+        return lookback, lookahead
+
+    def _make_backtrackable_dataset(self, dataset: datasets.IterableDataset) -> Backtrackable:
+        lookback, lookahead = self._get_window_steps(self.delta_timestamps)
+        return Backtrackable(dataset, history=lookback, lookahead=lookahead)
+
+    def _make_timestamps_from_indices(
+        self, start_ts: float, indices: dict[str, list[int]] | None = None
+    ) -> dict[str, list[float]]:
+        if indices is not None:
+            return {
+                key: (
+                    start_ts + torch.tensor(indices[key]) / self.fps
+                ).tolist()  # NOTE: why not delta_timestamps directly?
+                for key in self.delta_timestamps
+            }
+        else:
+            return dict.fromkeys(self.meta.video_keys, [start_ts])
+
+    def _make_padding_camera_frame(self, camera_key: str):
+        """Variable-shape padding frame for given camera keys, given in (H, W, C)"""
+        return torch.zeros(self.meta.info["features"][camera_key]["shape"]).permute(-1, 0, 1)
+
+    def _get_video_frame_padding_mask(
+        self,
+        video_frames: dict[str, torch.Tensor],
+        query_timestamps: dict[str, list[float]],
+        original_timestamps: dict[str, list[float]],
+    ) -> dict[str, torch.BoolTensor]:
+        padding_mask = {}
+
+        for video_key, timestamps in original_timestamps.items():
+            if video_key not in video_frames:
+                continue  # only padding on video keys that are available
+            frames = []
+            mask = []
+            padding_frame = self._make_padding_camera_frame(video_key)
+            for ts in timestamps:
+                if is_float_in_list(ts, query_timestamps[video_key]):
+                    idx = find_float_index(ts, query_timestamps[video_key])
+                    frames.append(video_frames[video_key][idx, :])
+                    mask.append(False)
+                else:
+                    frames.append(padding_frame)
+                    mask.append(True)
+
+            padding_mask[f"{video_key}_is_pad"] = torch.BoolTensor(mask)
+
+        return padding_mask
+
+    def make_frame(self, dataset_iterator: Backtrackable) -> Generator:
+        """Makes a frame starting from a dataset iterator"""
+        item = next(dataset_iterator)
+        item = item_to_torch(item)
+
+        updates = []  # list of "updates" to apply to the item retrieved from hf_dataset (w/o camera features)
+
+        # Get episode index from the item
+        ep_idx = item["episode_index"]
+
+        # "timestamp" restarts from 0 for each episode, whereas we need a global timestep within the single .mp4 file (given by index/fps)
+        current_ts = item["index"] / self.fps
+
+        episode_boundaries_ts = {
+            key: (
+                self.meta.episodes[ep_idx][f"videos/{key}/from_timestamp"],
+                self.meta.episodes[ep_idx][f"videos/{key}/to_timestamp"],
+            )
+            for key in self.meta.video_keys
+        }
+
+        # Apply delta querying logic if necessary
+        if self.delta_indices is not None:
+            query_result, padding = self._get_delta_frames(dataset_iterator, item)
+            updates.append(query_result)
+            updates.append(padding)
+
+        # Load video frames, when needed
+        if len(self.meta.video_keys) > 0:
+            original_timestamps = self._make_timestamps_from_indices(current_ts, self.delta_indices)
+
+            # Some timestamps might not result available considering the episode's boundaries
+            query_timestamps = self._get_query_timestamps(
+                current_ts, self.delta_indices, episode_boundaries_ts
+            )
+            video_frames = self._query_videos(query_timestamps, ep_idx)
+
+            if self.image_transforms is not None:
+                image_keys = self.meta.camera_keys
+                for cam in image_keys:
+                    video_frames[cam] = self.image_transforms(video_frames[cam])
+
+            updates.append(video_frames)
+
+            if self.delta_indices is not None:
+                # We always return the same number of frames. Unavailable frames are padded.
+                padding_mask = self._get_video_frame_padding_mask(
+                    video_frames, query_timestamps, original_timestamps
+                )
+                updates.append(padding_mask)
+
+        result = item.copy()
+        for update in updates:
+            result.update(update)
+
+        result["task"] = self.meta.tasks.iloc[item["task_index"]].name
+
+        yield result
+
+    def _get_query_timestamps(
+        self,
+        current_ts: float,
+        query_indices: dict[str, list[int]] | None = None,
+        episode_boundaries_ts: dict[str, tuple[float, float]] | None = None,
+    ) -> dict[str, list[float]]:
+        query_timestamps = {}
+        keys_to_timestamps = self._make_timestamps_from_indices(current_ts, query_indices)
+        for key in self.meta.video_keys:
+            if query_indices is not None and key in query_indices:
+                timestamps = keys_to_timestamps[key]
+                # Clamp out timesteps outside of episode boundaries
+                query_timestamps[key] = torch.clamp(
+                    torch.tensor(timestamps), *episode_boundaries_ts[key]
+                ).tolist()
+
+            else:
+                query_timestamps[key] = [current_ts]
+
+        return query_timestamps
+
+    def _query_videos(self, query_timestamps: dict[str, list[float]], ep_idx: int) -> dict:
+        """Note: When using data workers (e.g. DataLoader with num_workers>0), do not call this function
+        in the main process (e.g. by using a second Dataloader with num_workers=0). It will result in a
+        Segmentation Fault. This probably happens because a memory reference to the video loader is created in
+        the main process and a subprocess fails to access it.
+        """
+
+        item = {}
+        for video_key, query_ts in query_timestamps.items():
+            root = self.meta.url_root if self.streaming and not self.streaming_from_local else self.root
+            video_path = f"{root}/{self.meta.get_video_file_path(ep_idx, video_key)}"
+            frames = decode_video_frames_torchcodec(
+                video_path, query_ts, self.tolerance_s, decoder_cache=self.video_decoder_cache
+            )
+
+            item[video_key] = frames.squeeze(0) if len(query_ts) == 1 else frames
+
+        return item
+
+    def _get_delta_frames(self, dataset_iterator: Backtrackable, current_item: dict):
+        # TODO(fracapuano): Modularize this function, refactor the code
+        """Get frames with delta offsets using the backtrackable iterator.
+
+        Args:
+            current_item (dict): Current item from the iterator.
+            ep_idx (int): Episode index.
+
+        Returns:
+            tuple: (query_result, padding) - frames at delta offsets and padding info.
+        """
+        current_episode_idx = current_item["episode_index"]
+
+        # Prepare results
+        query_result = {}
+        padding = {}
+
+        for key, delta_indices in self.delta_indices.items():
+            if key in self.meta.video_keys:
+                continue  # visual frames are decoded separately
+
+            target_frames = []
+            is_pad = []
+
+            # Create a results dictionary to store frames in processing order, then reconstruct original order for stacking
+            delta_results = {}
+
+            # Separate and sort deltas by difficulty (easier operations first)
+            negative_deltas = sorted([d for d in delta_indices if d < 0], reverse=True)  # [-1, -2, -3, ...]
+            positive_deltas = sorted([d for d in delta_indices if d > 0])  # [1, 2, 3, ...]
+            zero_deltas = [d for d in delta_indices if d == 0]
+
+            # Process zero deltas (current frame)
+            for delta in zero_deltas:
+                delta_results[delta] = (
+                    current_item[key],
+                    False,
+                )
+
+            # Process negative deltas in order of increasing difficulty
+            lookback_failed = False
+
+            last_successful_frame = current_item[key]
+
+            for delta in negative_deltas:
+                if lookback_failed:
+                    delta_results[delta] = (last_successful_frame, True)
+                    continue
+
+                try:
+                    steps_back = abs(delta)
+                    if dataset_iterator.can_peek_back(steps_back):
+                        past_item = dataset_iterator.peek_back(steps_back)
+                        past_item = item_to_torch(past_item)
+
+                        if past_item["episode_index"] == current_episode_idx:
+                            delta_results[delta] = (past_item[key], False)
+                            last_successful_frame = past_item[key]
+
+                        else:
+                            raise LookBackError("Retrieved frame is from different episode!")
+                    else:
+                        raise LookBackError("Cannot go back further than the history buffer!")
+
+                except LookBackError:
+                    delta_results[delta] = (last_successful_frame, True)
+                    lookback_failed = True  # All subsequent negative deltas will also fail
+
+            # Process positive deltas in order of increasing difficulty
+            lookahead_failed = False
+            last_successful_frame = current_item[key]
+
+            for delta in positive_deltas:
+                if lookahead_failed:
+                    delta_results[delta] = (last_successful_frame, True)
+                    continue
+
+                try:
+                    if dataset_iterator.can_peek_ahead(delta):
+                        future_item = dataset_iterator.peek_ahead(delta)
+                        future_item = item_to_torch(future_item)
+
+                        if future_item["episode_index"] == current_episode_idx:
+                            delta_results[delta] = (future_item[key], False)
+                            last_successful_frame = future_item[key]
+
+                        else:
+                            raise LookAheadError("Retrieved frame is from different episode!")
+                    else:
+                        raise LookAheadError("Cannot go ahead further than the lookahead buffer!")
+
+                except LookAheadError:
+                    delta_results[delta] = (last_successful_frame, True)
+                    lookahead_failed = True  # All subsequent positive deltas will also fail
+
+            # Reconstruct original order for stacking
+            for delta in delta_indices:
+                frame, is_padded = delta_results[delta]
+
+                # add batch dimension for stacking
+                target_frames.append(frame)  # frame.unsqueeze(0))
+                is_pad.append(is_padded)
+
+            # Stack frames and add to results
+            if target_frames:
+                query_result[key] = torch.stack(target_frames)
+                padding[f"{key}_is_pad"] = torch.BoolTensor(is_pad)
+
+        return query_result, padding
+
+    def _validate_delta_timestamp_keys(self, delta_timestamps: dict[list[float]]) -> None:
+        """
+        Validate that all keys in delta_timestamps correspond to actual features in the dataset.
+
+        Raises:
+            ValueError: If any delta timestamp key doesn't correspond to a dataset feature.
+        """
+        if delta_timestamps is None:
+            return
+
+        # Get all available feature keys from the dataset metadata
+        available_features = set(self.meta.features.keys())
+
+        # Get all keys from delta_timestamps
+        delta_keys = set(delta_timestamps.keys())
+
+        # Find any keys that don't correspond to features
+        invalid_keys = delta_keys - available_features
+
+        if invalid_keys:
+            raise ValueError(
+                f"The following delta_timestamp keys do not correspond to dataset features: {invalid_keys}. "
+                f"Available features are: {sorted(available_features)}"
+            )
diff --git a/lerobot/src/lerobot/datasets/transforms.py b/lerobot/src/lerobot/datasets/transforms.py
new file mode 100644
index 0000000000000000000000000000000000000000..5240619cb6caf732fa19306d6b32b3c59f0465ac
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/transforms.py
@@ -0,0 +1,260 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import collections
+from collections.abc import Callable, Sequence
+from dataclasses import dataclass, field
+from typing import Any
+
+import torch
+from torchvision.transforms import v2
+from torchvision.transforms.v2 import (
+    Transform,
+    functional as F,  # noqa: N812
+)
+
+
+class RandomSubsetApply(Transform):
+    """Apply a random subset of N transformations from a list of transformations.
+
+    Args:
+        transforms: list of transformations.
+        p: represents the multinomial probabilities (with no replacement) used for sampling the transform.
+            If the sum of the weights is not 1, they will be normalized. If ``None`` (default), all transforms
+            have the same probability.
+        n_subset: number of transformations to apply. If ``None``, all transforms are applied.
+            Must be in [1, len(transforms)].
+        random_order: apply transformations in a random order.
+    """
+
+    def __init__(
+        self,
+        transforms: Sequence[Callable],
+        p: list[float] | None = None,
+        n_subset: int | None = None,
+        random_order: bool = False,
+    ) -> None:
+        super().__init__()
+        if not isinstance(transforms, Sequence):
+            raise TypeError("Argument transforms should be a sequence of callables")
+        if p is None:
+            p = [1] * len(transforms)
+        elif len(p) != len(transforms):
+            raise ValueError(
+                f"Length of p doesn't match the number of transforms: {len(p)} != {len(transforms)}"
+            )
+
+        if n_subset is None:
+            n_subset = len(transforms)
+        elif not isinstance(n_subset, int):
+            raise TypeError("n_subset should be an int or None")
+        elif not (1 <= n_subset <= len(transforms)):
+            raise ValueError(f"n_subset should be in the interval [1, {len(transforms)}]")
+
+        self.transforms = transforms
+        total = sum(p)
+        self.p = [prob / total for prob in p]
+        self.n_subset = n_subset
+        self.random_order = random_order
+
+        self.selected_transforms = None
+
+    def forward(self, *inputs: Any) -> Any:
+        needs_unpacking = len(inputs) > 1
+
+        selected_indices = torch.multinomial(torch.tensor(self.p), self.n_subset)
+        if not self.random_order:
+            selected_indices = selected_indices.sort().values
+
+        self.selected_transforms = [self.transforms[i] for i in selected_indices]
+
+        for transform in self.selected_transforms:
+            outputs = transform(*inputs)
+            inputs = outputs if needs_unpacking else (outputs,)
+
+        return outputs
+
+    def extra_repr(self) -> str:
+        return (
+            f"transforms={self.transforms}, "
+            f"p={self.p}, "
+            f"n_subset={self.n_subset}, "
+            f"random_order={self.random_order}"
+        )
+
+
+class SharpnessJitter(Transform):
+    """Randomly change the sharpness of an image or video.
+
+    Similar to a v2.RandomAdjustSharpness with p=1 and a sharpness_factor sampled randomly.
+    While v2.RandomAdjustSharpness applies — with a given probability — a fixed sharpness_factor to an image,
+    SharpnessJitter applies a random sharpness_factor each time. This is to have a more diverse set of
+    augmentations as a result.
+
+    A sharpness_factor of 0 gives a blurred image, 1 gives the original image while 2 increases the sharpness
+    by a factor of 2.
+
+    If the input is a :class:`torch.Tensor`,
+    it is expected to have [..., 1 or 3, H, W] shape, where ... means an arbitrary number of leading dimensions.
+
+    Args:
+        sharpness: How much to jitter sharpness. sharpness_factor is chosen uniformly from
+            [max(0, 1 - sharpness), 1 + sharpness] or the given
+            [min, max]. Should be non negative numbers.
+    """
+
+    def __init__(self, sharpness: float | Sequence[float]) -> None:
+        super().__init__()
+        self.sharpness = self._check_input(sharpness)
+
+    def _check_input(self, sharpness):
+        if isinstance(sharpness, (int | float)):
+            if sharpness < 0:
+                raise ValueError("If sharpness is a single number, it must be non negative.")
+            sharpness = [1.0 - sharpness, 1.0 + sharpness]
+            sharpness[0] = max(sharpness[0], 0.0)
+        elif isinstance(sharpness, collections.abc.Sequence) and len(sharpness) == 2:
+            sharpness = [float(v) for v in sharpness]
+        else:
+            raise TypeError(f"{sharpness=} should be a single number or a sequence with length 2.")
+
+        if not 0.0 <= sharpness[0] <= sharpness[1]:
+            raise ValueError(f"sharpness values should be between (0., inf), but got {sharpness}.")
+
+        return float(sharpness[0]), float(sharpness[1])
+
+    def make_params(self, flat_inputs: list[Any]) -> dict[str, Any]:
+        sharpness_factor = torch.empty(1).uniform_(self.sharpness[0], self.sharpness[1]).item()
+        return {"sharpness_factor": sharpness_factor}
+
+    def transform(self, inpt: Any, params: dict[str, Any]) -> Any:
+        sharpness_factor = params["sharpness_factor"]
+        return self._call_kernel(F.adjust_sharpness, inpt, sharpness_factor=sharpness_factor)
+
+
+@dataclass
+class ImageTransformConfig:
+    """
+    For each transform, the following parameters are available:
+      weight: This represents the multinomial probability (with no replacement)
+            used for sampling the transform. If the sum of the weights is not 1,
+            they will be normalized.
+      type: The name of the class used. This is either a class available under torchvision.transforms.v2 or a
+            custom transform defined here.
+      kwargs: Lower & upper bound respectively used for sampling the transform's parameter
+            (following uniform distribution) when it's applied.
+    """
+
+    weight: float = 1.0
+    type: str = "Identity"
+    kwargs: dict[str, Any] = field(default_factory=dict)
+
+
+@dataclass
+class ImageTransformsConfig:
+    """
+    These transforms are all using standard torchvision.transforms.v2
+    You can find out how these transformations affect images here:
+    https://pytorch.org/vision/0.18/auto_examples/transforms/plot_transforms_illustrations.html
+    We use a custom RandomSubsetApply container to sample them.
+    """
+
+    # Set this flag to `true` to enable transforms during training
+    enable: bool = False
+    # This is the maximum number of transforms (sampled from these below) that will be applied to each frame.
+    # It's an integer in the interval [1, number_of_available_transforms].
+    max_num_transforms: int = 3
+    # By default, transforms are applied in Torchvision's suggested order (shown below).
+    # Set this to True to apply them in a random order.
+    random_order: bool = False
+    tfs: dict[str, ImageTransformConfig] = field(
+        default_factory=lambda: {
+            "brightness": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"brightness": (0.8, 1.2)},
+            ),
+            "contrast": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"contrast": (0.8, 1.2)},
+            ),
+            "saturation": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"saturation": (0.5, 1.5)},
+            ),
+            "hue": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"hue": (-0.05, 0.05)},
+            ),
+            "sharpness": ImageTransformConfig(
+                weight=1.0,
+                type="SharpnessJitter",
+                kwargs={"sharpness": (0.5, 1.5)},
+            ),
+            "affine": ImageTransformConfig(
+                weight=1.0,
+                type="RandomAffine",
+                kwargs={"degrees": (-5.0, 5.0), "translate": (0.05, 0.05)},
+            ),
+        }
+    )
+
+
+def make_transform_from_config(cfg: ImageTransformConfig):
+    if cfg.type == "SharpnessJitter":
+        return SharpnessJitter(**cfg.kwargs)
+
+    transform_cls = getattr(v2, cfg.type, None)
+    if isinstance(transform_cls, type) and issubclass(transform_cls, Transform):
+        return transform_cls(**cfg.kwargs)
+
+    raise ValueError(
+        f"Transform '{cfg.type}' is not valid. It must be a class in "
+        f"torchvision.transforms.v2 or 'SharpnessJitter'."
+    )
+
+
+class ImageTransforms(Transform):
+    """A class to compose image transforms based on configuration."""
+
+    def __init__(self, cfg: ImageTransformsConfig) -> None:
+        super().__init__()
+        self._cfg = cfg
+
+        self.weights = []
+        self.transforms = {}
+        for tf_name, tf_cfg in cfg.tfs.items():
+            if tf_cfg.weight <= 0.0:
+                continue
+
+            self.transforms[tf_name] = make_transform_from_config(tf_cfg)
+            self.weights.append(tf_cfg.weight)
+
+        n_subset = min(len(self.transforms), cfg.max_num_transforms)
+        if n_subset == 0 or not cfg.enable:
+            self.tf = v2.Identity()
+        else:
+            self.tf = RandomSubsetApply(
+                transforms=list(self.transforms.values()),
+                p=self.weights,
+                n_subset=n_subset,
+                random_order=cfg.random_order,
+            )
+
+    def forward(self, *inputs: Any) -> Any:
+        return self.tf(*inputs)
diff --git a/lerobot/src/lerobot/datasets/utils.py b/lerobot/src/lerobot/datasets/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..2e1d360f90ab269100817ca860d54c003c42deaa
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/utils.py
@@ -0,0 +1,430 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import contextlib
+import importlib.resources
+import json
+import logging
+from collections.abc import Iterator
+from typing import Any
+
+import datasets
+import numpy as np
+import packaging.version
+import torch
+from huggingface_hub import DatasetCard, DatasetCardData, HfApi
+from huggingface_hub.errors import RevisionNotFoundError
+
+V30_MESSAGE = """
+The dataset you requested ({repo_id}) is in {version} format.
+
+We introduced a new format since v3.0 which is not backward compatible with v2.1.
+Please, update your dataset to the new format using this command:
+```
+python -m lerobot.scripts.convert_dataset_v21_to_v30 --repo-id={repo_id}
+```
+
+If you already have a converted version uploaded to the hub, then this error might be because of
+an older version in your local cache. Consider deleting the cached version and retrying.
+
+If you encounter a problem, contact LeRobot maintainers on [Discord](https://discord.com/invite/s3KuuzsPFb)
+or open an [issue on GitHub](https://github.com/huggingface/lerobot/issues/new/choose).
+"""
+
+FUTURE_MESSAGE = """
+The dataset you requested ({repo_id}) is only available in {version} format.
+As we cannot ensure forward compatibility with it, please update your current version of lerobot.
+"""
+
+
+class CompatibilityError(Exception): ...
+
+
+class BackwardCompatibilityError(CompatibilityError):
+    def __init__(self, repo_id: str, version: packaging.version.Version):
+        if version.major == 2 and version.minor == 1:
+            message = V30_MESSAGE.format(repo_id=repo_id, version=version)
+        else:
+            raise NotImplementedError(
+                "Contact the maintainer on [Discord](https://discord.com/invite/s3KuuzsPFb)."
+            )
+        super().__init__(message)
+
+
+class ForwardCompatibilityError(CompatibilityError):
+    def __init__(self, repo_id: str, version: packaging.version.Version):
+        message = FUTURE_MESSAGE.format(repo_id=repo_id, version=version)
+        super().__init__(message)
+
+
+DEFAULT_CHUNK_SIZE = 1000  # Max number of files per chunk
+DEFAULT_DATA_FILE_SIZE_IN_MB = 100  # Max size per file
+DEFAULT_VIDEO_FILE_SIZE_IN_MB = 200  # Max size per file
+
+INFO_PATH = "meta/info.json"
+STATS_PATH = "meta/stats.json"
+
+EPISODES_DIR = "meta/episodes"
+DATA_DIR = "data"
+VIDEO_DIR = "videos"
+
+CHUNK_FILE_PATTERN = "chunk-{chunk_index:03d}/file-{file_index:03d}"
+DEFAULT_TASKS_PATH = "meta/tasks.parquet"
+DEFAULT_SUBTASKS_PATH = "meta/subtasks.parquet"
+DEFAULT_EPISODES_PATH = EPISODES_DIR + "/" + CHUNK_FILE_PATTERN + ".parquet"
+DEFAULT_DATA_PATH = DATA_DIR + "/" + CHUNK_FILE_PATTERN + ".parquet"
+DEFAULT_VIDEO_PATH = VIDEO_DIR + "/{video_key}/" + CHUNK_FILE_PATTERN + ".mp4"
+DEFAULT_IMAGE_PATH = "images/{image_key}/episode-{episode_index:06d}/frame-{frame_index:06d}.png"
+
+LEGACY_EPISODES_PATH = "meta/episodes.jsonl"
+LEGACY_EPISODES_STATS_PATH = "meta/episodes_stats.jsonl"
+LEGACY_TASKS_PATH = "meta/tasks.jsonl"
+
+DEFAULT_FEATURES = {
+    "timestamp": {"dtype": "float32", "shape": (1,), "names": None},
+    "frame_index": {"dtype": "int64", "shape": (1,), "names": None},
+    "episode_index": {"dtype": "int64", "shape": (1,), "names": None},
+    "index": {"dtype": "int64", "shape": (1,), "names": None},
+    "task_index": {"dtype": "int64", "shape": (1,), "names": None},
+}
+
+
+def update_chunk_file_indices(chunk_idx: int, file_idx: int, chunks_size: int) -> tuple[int, int]:
+    if file_idx == chunks_size - 1:
+        file_idx = 0
+        chunk_idx += 1
+    else:
+        file_idx += 1
+    return chunk_idx, file_idx
+
+
+def flatten_dict(d: dict, parent_key: str = "", sep: str = "/") -> dict:
+    """Flatten a nested dictionary by joining keys with a separator.
+
+    Example:
+        >>> dct = {"a": {"b": 1, "c": {"d": 2}}, "e": 3}
+        >>> print(flatten_dict(dct))
+        {'a/b': 1, 'a/c/d': 2, 'e': 3}
+
+    Args:
+        d (dict): The dictionary to flatten.
+        parent_key (str): The base key to prepend to the keys in this level.
+        sep (str): The separator to use between keys.
+
+    Returns:
+        dict: A flattened dictionary.
+    """
+    items = []
+    for k, v in d.items():
+        new_key = f"{parent_key}{sep}{k}" if parent_key else k
+        if isinstance(v, dict):
+            items.extend(flatten_dict(v, new_key, sep=sep).items())
+        else:
+            items.append((new_key, v))
+    return dict(items)
+
+
+def unflatten_dict(d: dict, sep: str = "/") -> dict:
+    """Unflatten a dictionary with delimited keys into a nested dictionary.
+
+    Example:
+        >>> flat_dct = {"a/b": 1, "a/c/d": 2, "e": 3}
+        >>> print(unflatten_dict(flat_dct))
+        {'a': {'b': 1, 'c': {'d': 2}}, 'e': 3}
+
+    Args:
+        d (dict): A dictionary with flattened keys.
+        sep (str): The separator used in the keys.
+
+    Returns:
+        dict: A nested dictionary.
+    """
+    outdict = {}
+    for key, value in d.items():
+        parts = key.split(sep)
+        d = outdict
+        for part in parts[:-1]:
+            if part not in d:
+                d[part] = {}
+            d = d[part]
+        d[parts[-1]] = value
+    return outdict
+
+
+def serialize_dict(stats: dict[str, torch.Tensor | np.ndarray | dict]) -> dict:
+    """Serialize a dictionary containing tensors or numpy arrays to be JSON-compatible.
+
+    Converts torch.Tensor, np.ndarray, and np.generic types to lists or native Python types.
+
+    Args:
+        stats (dict): A dictionary that may contain non-serializable numeric types.
+
+    Returns:
+        dict: A dictionary with all values converted to JSON-serializable types.
+
+    Raises:
+        NotImplementedError: If a value has an unsupported type.
+    """
+    serialized_dict = {}
+    for key, value in flatten_dict(stats).items():
+        if isinstance(value, (torch.Tensor | np.ndarray)):
+            serialized_dict[key] = value.tolist()
+        elif isinstance(value, list) and isinstance(value[0], (int | float | list)):
+            serialized_dict[key] = value
+        elif isinstance(value, np.generic):
+            serialized_dict[key] = value.item()
+        elif isinstance(value, (int | float)):
+            serialized_dict[key] = value
+        else:
+            raise NotImplementedError(f"The value '{value}' of type '{type(value)}' is not supported.")
+    return unflatten_dict(serialized_dict)
+
+
+def is_valid_version(version: str) -> bool:
+    """Check if a string is a valid PEP 440 version.
+
+    Args:
+        version (str): The version string to check.
+
+    Returns:
+        bool: True if the version string is valid, False otherwise.
+    """
+    try:
+        packaging.version.parse(version)
+        return True
+    except packaging.version.InvalidVersion:
+        return False
+
+
+def check_version_compatibility(
+    repo_id: str,
+    version_to_check: str | packaging.version.Version,
+    current_version: str | packaging.version.Version,
+    enforce_breaking_major: bool = True,
+) -> None:
+    """Check for version compatibility between a dataset and the current codebase.
+
+    Args:
+        repo_id (str): The repository ID for logging purposes.
+        version_to_check (str | packaging.version.Version): The version of the dataset.
+        current_version (str | packaging.version.Version): The current version of the codebase.
+        enforce_breaking_major (bool): If True, raise an error on major version mismatch.
+
+    Raises:
+        BackwardCompatibilityError: If the dataset version is from a newer, incompatible
+            major version of the codebase.
+    """
+    v_check = (
+        packaging.version.parse(version_to_check)
+        if not isinstance(version_to_check, packaging.version.Version)
+        else version_to_check
+    )
+    v_current = (
+        packaging.version.parse(current_version)
+        if not isinstance(current_version, packaging.version.Version)
+        else current_version
+    )
+    if v_check.major < v_current.major and enforce_breaking_major:
+        raise BackwardCompatibilityError(repo_id, v_check)
+    elif v_check.minor < v_current.minor:
+        logging.warning(FUTURE_MESSAGE.format(repo_id=repo_id, version=v_check))
+
+
+def get_repo_versions(repo_id: str) -> list[packaging.version.Version]:
+    """Return available valid versions (branches and tags) on a given Hub repo.
+
+    Args:
+        repo_id (str): The repository ID on the Hugging Face Hub.
+
+    Returns:
+        list[packaging.version.Version]: A list of valid versions found.
+    """
+    api = HfApi()
+    repo_refs = api.list_repo_refs(repo_id, repo_type="dataset")
+    repo_refs = [b.name for b in repo_refs.branches + repo_refs.tags]
+    repo_versions = []
+    for ref in repo_refs:
+        with contextlib.suppress(packaging.version.InvalidVersion):
+            repo_versions.append(packaging.version.parse(ref))
+
+    return repo_versions
+
+
+def get_safe_version(repo_id: str, version: str | packaging.version.Version) -> str:
+    """Return the specified version if available on repo, or the latest compatible one.
+
+    If the exact version is not found, it looks for the latest version with the
+    same major version number that is less than or equal to the target minor version.
+
+    Args:
+        repo_id (str): The repository ID on the Hugging Face Hub.
+        version (str | packaging.version.Version): The target version.
+
+    Returns:
+        str: The safe version string (e.g., "v1.2.3") to use as a revision.
+
+    Raises:
+        RevisionNotFoundError: If the repo has no version tags.
+        BackwardCompatibilityError: If only older major versions are available.
+        ForwardCompatibilityError: If only newer major versions are available.
+    """
+    target_version = (
+        packaging.version.parse(version) if not isinstance(version, packaging.version.Version) else version
+    )
+    hub_versions = get_repo_versions(repo_id)
+
+    if not hub_versions:
+        raise RevisionNotFoundError(
+            f"""Your dataset must be tagged with a codebase version.
+            Assuming _version_ is the codebase_version value in the info.json, you can run this:
+            ```python
+            from huggingface_hub import HfApi
+
+            hub_api = HfApi()
+            hub_api.create_tag("{repo_id}", tag="_version_", repo_type="dataset")
+            ```
+            """
+        )
+
+    if target_version in hub_versions:
+        return f"v{target_version}"
+
+    compatibles = [
+        v for v in hub_versions if v.major == target_version.major and v.minor <= target_version.minor
+    ]
+    if compatibles:
+        return_version = max(compatibles)
+        if return_version < target_version:
+            logging.warning(f"Revision {version} for {repo_id} not found, using version v{return_version}")
+        return f"v{return_version}"
+
+    lower_major = [v for v in hub_versions if v.major < target_version.major]
+    if lower_major:
+        raise BackwardCompatibilityError(repo_id, max(lower_major))
+
+    upper_versions = [v for v in hub_versions if v > target_version]
+    assert len(upper_versions) > 0
+    raise ForwardCompatibilityError(repo_id, min(upper_versions))
+
+
+def cycle(iterable: Any) -> Iterator[Any]:
+    """Create a dataloader-safe cyclical iterator.
+
+    This is an equivalent of `itertools.cycle` but is safe for use with
+    PyTorch DataLoaders with multiple workers.
+    See https://github.com/pytorch/pytorch/issues/23900 for details.
+
+    Args:
+        iterable: The iterable to cycle over.
+
+    Yields:
+        Items from the iterable, restarting from the beginning when exhausted.
+    """
+    iterator = iter(iterable)
+    while True:
+        try:
+            yield next(iterator)
+        except StopIteration:
+            iterator = iter(iterable)
+
+
+def create_branch(repo_id: str, *, branch: str, repo_type: str | None = None) -> None:
+    """Create a branch on an existing Hugging Face repo.
+
+    Deletes the branch if it already exists before creating it.
+
+    Args:
+        repo_id (str): The ID of the repository.
+        branch (str): The name of the branch to create.
+        repo_type (str | None): The type of the repository (e.g., "dataset").
+    """
+    api = HfApi()
+
+    branches = api.list_repo_refs(repo_id, repo_type=repo_type).branches
+    refs = [branch.ref for branch in branches]
+    ref = f"refs/heads/{branch}"
+    if ref in refs:
+        api.delete_branch(repo_id, repo_type=repo_type, branch=branch)
+
+    api.create_branch(repo_id, repo_type=repo_type, branch=branch)
+
+
+def create_lerobot_dataset_card(
+    tags: list | None = None,
+    dataset_info: dict | None = None,
+    **kwargs,
+) -> DatasetCard:
+    """Create a `DatasetCard` for a LeRobot dataset.
+
+    Keyword arguments are used to replace values in the card template.
+    Note: If specified, `license` must be a valid license identifier from
+    https://huggingface.co/docs/hub/repositories-licenses.
+
+    Args:
+        tags (list | None): A list of tags to add to the dataset card.
+        dataset_info (dict | None): The dataset's info dictionary, which will
+            be displayed on the card.
+        **kwargs: Additional keyword arguments to populate the card template.
+
+    Returns:
+        DatasetCard: The generated dataset card object.
+    """
+    card_tags = ["LeRobot"]
+
+    if tags:
+        card_tags += tags
+    if dataset_info:
+        dataset_structure = "[meta/info.json](meta/info.json):\n"
+        dataset_structure += f"```json\n{json.dumps(dataset_info, indent=4)}\n```\n"
+        kwargs = {**kwargs, "dataset_structure": dataset_structure}
+    card_data = DatasetCardData(
+        license=kwargs.get("license"),
+        tags=card_tags,
+        task_categories=["robotics"],
+        configs=[
+            {
+                "config_name": "default",
+                "data_files": "data/*/*.parquet",
+            }
+        ],
+    )
+
+    card_template = (importlib.resources.files("lerobot.datasets") / "card_template.md").read_text()
+
+    return DatasetCard.from_template(
+        card_data=card_data,
+        template_str=card_template,
+        **kwargs,
+    )
+
+
+def is_float_in_list(target, float_list, threshold=1e-6):
+    return any(abs(target - x) <= threshold for x in float_list)
+
+
+def find_float_index(target, float_list, threshold=1e-6):
+    for i, x in enumerate(float_list):
+        if abs(target - x) <= threshold:
+            return i
+    return -1
+
+
+def safe_shard(dataset: datasets.IterableDataset, index: int, num_shards: int) -> datasets.Dataset:
+    """
+    Safe shards the dataset.
+    """
+    shard_idx = min(dataset.num_shards, index + 1) - 1
+
+    return dataset.shard(num_shards, index=shard_idx)
diff --git a/lerobot/src/lerobot/datasets/video_utils.py b/lerobot/src/lerobot/datasets/video_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..e465b79b43fc926acdb55c8ff6ff686ac0bedcc5
--- /dev/null
+++ b/lerobot/src/lerobot/datasets/video_utils.py
@@ -0,0 +1,1114 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import contextlib
+import glob
+import importlib
+import logging
+import queue
+import shutil
+import tempfile
+import threading
+import warnings
+from dataclasses import dataclass, field
+from fractions import Fraction
+from pathlib import Path
+from threading import Lock
+from typing import Any, ClassVar
+
+import av
+import fsspec
+import numpy as np
+import pyarrow as pa
+import torch
+import torchvision
+from datasets.features.features import register_feature
+from PIL import Image
+
+logger = logging.getLogger(__name__)
+
+# List of hardware encoders to probe for auto-selection. Availability depends on the platform and FFmpeg build.
+# Determines the order of preference for auto-selection when vcodec="auto" is used.
+HW_ENCODERS = [
+    "h264_videotoolbox",  # macOS
+    "hevc_videotoolbox",  # macOS
+    "h264_nvenc",  # NVIDIA GPU
+    "hevc_nvenc",  # NVIDIA GPU
+    "h264_vaapi",  # Linux Intel/AMD
+    "h264_qsv",  # Intel Quick Sync
+]
+
+VALID_VIDEO_CODECS = {"h264", "hevc", "libsvtav1", "auto"} | set(HW_ENCODERS)
+
+
+def _get_codec_options(
+    vcodec: str,
+    g: int | None = 2,
+    crf: int | None = 30,
+    preset: int | None = None,
+) -> dict:
+    """Build codec-specific options dict for video encoding."""
+    options = {}
+
+    # GOP size (keyframe interval) - supported by VideoToolbox and software encoders
+    if g is not None and (vcodec in ("h264_videotoolbox", "hevc_videotoolbox") or vcodec not in HW_ENCODERS):
+        options["g"] = str(g)
+
+    # Quality control (codec-specific parameter names)
+    if crf is not None:
+        if vcodec in ("h264", "hevc", "libsvtav1"):
+            options["crf"] = str(crf)
+        elif vcodec in ("h264_videotoolbox", "hevc_videotoolbox"):
+            quality = max(1, min(100, int(100 - crf * 2)))
+            options["q:v"] = str(quality)
+        elif vcodec in ("h264_nvenc", "hevc_nvenc"):
+            options["rc"] = "constqp"
+            options["qp"] = str(crf)
+        elif vcodec in ("h264_vaapi",):
+            options["qp"] = str(crf)
+        elif vcodec in ("h264_qsv",):
+            options["global_quality"] = str(crf)
+
+    # Preset (only for libsvtav1)
+    if vcodec == "libsvtav1":
+        options["preset"] = str(preset) if preset is not None else "12"
+
+    return options
+
+
+def detect_available_hw_encoders() -> list[str]:
+    """Probe PyAV/FFmpeg for available hardware video encoders."""
+    available = []
+    for codec_name in HW_ENCODERS:
+        try:
+            av.codec.Codec(codec_name, "w")
+            available.append(codec_name)
+        except Exception:  # nosec B110
+            logger.debug("HW encoder '%s' not available", codec_name)  # nosec B110
+    return available
+
+
+def resolve_vcodec(vcodec: str) -> str:
+    """Validate vcodec and resolve 'auto' to best available HW encoder, fallback to libsvtav1."""
+    if vcodec not in VALID_VIDEO_CODECS:
+        raise ValueError(f"Invalid vcodec '{vcodec}'. Must be one of: {sorted(VALID_VIDEO_CODECS)}")
+    if vcodec != "auto":
+        logger.info(f"Using video codec: {vcodec}")
+        return vcodec
+    available = detect_available_hw_encoders()
+    for encoder in HW_ENCODERS:
+        if encoder in available:
+            logger.info(f"Auto-selected video codec: {encoder}")
+            return encoder
+    logger.info("No hardware encoder available, falling back to software encoder 'libsvtav1'")
+    return "libsvtav1"
+
+
+def get_safe_default_codec():
+    if importlib.util.find_spec("torchcodec"):
+        return "torchcodec"
+    else:
+        logger.warning(
+            "'torchcodec' is not available in your platform, falling back to 'pyav' as a default decoder"
+        )
+        return "pyav"
+
+
+def decode_video_frames(
+    video_path: Path | str,
+    timestamps: list[float],
+    tolerance_s: float,
+    backend: str | None = None,
+) -> torch.Tensor:
+    """
+    Decodes video frames using the specified backend.
+
+    Args:
+        video_path (Path): Path to the video file.
+        timestamps (list[float]): List of timestamps to extract frames.
+        tolerance_s (float): Allowed deviation in seconds for frame retrieval.
+        backend (str, optional): Backend to use for decoding. Defaults to "torchcodec" when available in the platform; otherwise, defaults to "pyav"..
+
+    Returns:
+        torch.Tensor: Decoded frames.
+
+    Currently supports torchcodec on cpu and pyav.
+    """
+    if backend is None:
+        backend = get_safe_default_codec()
+    if backend == "torchcodec":
+        return decode_video_frames_torchcodec(video_path, timestamps, tolerance_s)
+    elif backend in ["pyav", "video_reader"]:
+        return decode_video_frames_torchvision(video_path, timestamps, tolerance_s, backend)
+    else:
+        raise ValueError(f"Unsupported video backend: {backend}")
+
+
+def decode_video_frames_torchvision(
+    video_path: Path | str,
+    timestamps: list[float],
+    tolerance_s: float,
+    backend: str = "pyav",
+    log_loaded_timestamps: bool = False,
+) -> torch.Tensor:
+    """Loads frames associated to the requested timestamps of a video
+
+    The backend can be either "pyav" (default) or "video_reader".
+    "video_reader" requires installing torchvision from source, see:
+    https://github.com/pytorch/vision/blob/main/torchvision/csrc/io/decoder/gpu/README.rst
+    (note that you need to compile against ffmpeg<4.3)
+
+    While both use cpu, "video_reader" is supposedly faster than "pyav" but requires additional setup.
+    For more info on video decoding, see `benchmark/video/README.md`
+
+    See torchvision doc for more info on these two backends:
+    https://pytorch.org/vision/0.18/index.html?highlight=backend#torchvision.set_video_backend
+
+    Note: Video benefits from inter-frame compression. Instead of storing every frame individually,
+    the encoder stores a reference frame (or a key frame) and subsequent frames as differences relative to
+    that key frame. As a consequence, to access a requested frame, we need to load the preceding key frame,
+    and all subsequent frames until reaching the requested frame. The number of key frames in a video
+    can be adjusted during encoding to take into account decoding time and video size in bytes.
+    """
+    video_path = str(video_path)
+
+    # set backend
+    keyframes_only = False
+    torchvision.set_video_backend(backend)
+    if backend == "pyav":
+        keyframes_only = True  # pyav doesn't support accurate seek
+
+    # set a video stream reader
+    # TODO(rcadene): also load audio stream at the same time
+    reader = torchvision.io.VideoReader(video_path, "video")
+
+    # set the first and last requested timestamps
+    # Note: previous timestamps are usually loaded, since we need to access the previous key frame
+    first_ts = min(timestamps)
+    last_ts = max(timestamps)
+
+    # access closest key frame of the first requested frame
+    # Note: closest key frame timestamp is usually smaller than `first_ts` (e.g. key frame can be the first frame of the video)
+    # for details on what `seek` is doing see: https://pyav.basswood-io.com/docs/stable/api/container.html?highlight=inputcontainer#av.container.InputContainer.seek
+    reader.seek(first_ts, keyframes_only=keyframes_only)
+
+    # load all frames until last requested frame
+    loaded_frames = []
+    loaded_ts = []
+    for frame in reader:
+        current_ts = frame["pts"]
+        if log_loaded_timestamps:
+            logger.info(f"frame loaded at timestamp={current_ts:.4f}")
+        loaded_frames.append(frame["data"])
+        loaded_ts.append(current_ts)
+        if current_ts >= last_ts:
+            break
+
+    if backend == "pyav":
+        reader.container.close()
+
+    reader = None
+
+    query_ts = torch.tensor(timestamps)
+    loaded_ts = torch.tensor(loaded_ts)
+
+    # compute distances between each query timestamp and timestamps of all loaded frames
+    dist = torch.cdist(query_ts[:, None], loaded_ts[:, None], p=1)
+    min_, argmin_ = dist.min(1)
+
+    is_within_tol = min_ < tolerance_s
+    if not is_within_tol.all():
+        raise FrameTimestampError(
+            f"One or several query timestamps unexpectedly violate the tolerance ({min_[~is_within_tol]} > {tolerance_s=})."
+            " It means that the closest frame that can be loaded from the video is too far away in time."
+            " This might be due to synchronization issues with timestamps during data collection."
+            " To be safe, we advise to ignore this item during training."
+            f"\nqueried timestamps: {query_ts}"
+            f"\nloaded timestamps: {loaded_ts}"
+            f"\nvideo: {video_path}"
+            f"\nbackend: {backend}"
+        )
+
+    # get closest frames to the query timestamps
+    closest_frames = torch.stack([loaded_frames[idx] for idx in argmin_])
+    closest_ts = loaded_ts[argmin_]
+
+    if log_loaded_timestamps:
+        logger.info(f"{closest_ts=}")
+
+    # convert to the pytorch format which is float32 in [0,1] range (and channel first)
+    closest_frames = closest_frames.type(torch.float32) / 255
+
+    if len(timestamps) != len(closest_frames):
+        raise FrameTimestampError(
+            f"Number of retrieved frames ({len(closest_frames)}) does not match "
+            f"number of queried timestamps ({len(timestamps)})"
+        )
+    return closest_frames
+
+
+class VideoDecoderCache:
+    """Thread-safe cache for video decoders to avoid expensive re-initialization."""
+
+    def __init__(self):
+        self._cache: dict[str, tuple[Any, Any]] = {}
+        self._lock = Lock()
+
+    def get_decoder(self, video_path: str):
+        """Get a cached decoder or create a new one."""
+        if importlib.util.find_spec("torchcodec"):
+            from torchcodec.decoders import VideoDecoder
+        else:
+            raise ImportError("torchcodec is required but not available.")
+
+        video_path = str(video_path)
+
+        with self._lock:
+            if video_path not in self._cache:
+                file_handle = fsspec.open(video_path).__enter__()
+                decoder = VideoDecoder(file_handle, seek_mode="approximate")
+                self._cache[video_path] = (decoder, file_handle)
+
+            return self._cache[video_path][0]
+
+    def clear(self):
+        """Clear the cache and close file handles."""
+        with self._lock:
+            for _, file_handle in self._cache.values():
+                file_handle.close()
+            self._cache.clear()
+
+    def size(self) -> int:
+        """Return the number of cached decoders."""
+        with self._lock:
+            return len(self._cache)
+
+
+class FrameTimestampError(ValueError):
+    """Helper error to indicate the retrieved timestamps exceed the queried ones"""
+
+    pass
+
+
+_default_decoder_cache = VideoDecoderCache()
+
+
+def decode_video_frames_torchcodec(
+    video_path: Path | str,
+    timestamps: list[float],
+    tolerance_s: float,
+    log_loaded_timestamps: bool = False,
+    decoder_cache: VideoDecoderCache | None = None,
+) -> torch.Tensor:
+    """Loads frames associated with the requested timestamps of a video using torchcodec.
+
+    Args:
+        video_path: Path to the video file.
+        timestamps: List of timestamps to extract frames.
+        tolerance_s: Allowed deviation in seconds for frame retrieval.
+        log_loaded_timestamps: Whether to log loaded timestamps.
+        decoder_cache: Optional decoder cache instance. Uses default if None.
+
+    Note: Setting device="cuda" outside the main process, e.g. in data loader workers, will lead to CUDA initialization errors.
+
+    Note: Video benefits from inter-frame compression. Instead of storing every frame individually,
+    the encoder stores a reference frame (or a key frame) and subsequent frames as differences relative to
+    that key frame. As a consequence, to access a requested frame, we need to load the preceding key frame,
+    and all subsequent frames until reaching the requested frame. The number of key frames in a video
+    can be adjusted during encoding to take into account decoding time and video size in bytes.
+    """
+    if decoder_cache is None:
+        decoder_cache = _default_decoder_cache
+
+    # Use cached decoder instead of creating new one each time
+    decoder = decoder_cache.get_decoder(str(video_path))
+
+    loaded_ts = []
+    loaded_frames = []
+
+    # get metadata for frame information
+    metadata = decoder.metadata
+    average_fps = metadata.average_fps
+    # convert timestamps to frame indices
+    frame_indices = [round(ts * average_fps) for ts in timestamps]
+    # retrieve frames based on indices
+    frames_batch = decoder.get_frames_at(indices=frame_indices)
+
+    for frame, pts in zip(frames_batch.data, frames_batch.pts_seconds, strict=True):
+        loaded_frames.append(frame)
+        loaded_ts.append(pts.item())
+        if log_loaded_timestamps:
+            logger.info(f"Frame loaded at timestamp={pts:.4f}")
+
+    query_ts = torch.tensor(timestamps)
+    loaded_ts = torch.tensor(loaded_ts)
+
+    # compute distances between each query timestamp and loaded timestamps
+    dist = torch.cdist(query_ts[:, None], loaded_ts[:, None], p=1)
+    min_, argmin_ = dist.min(1)
+
+    is_within_tol = min_ < tolerance_s
+    if not is_within_tol.all():
+        raise FrameTimestampError(
+            f"One or several query timestamps unexpectedly violate the tolerance ({min_[~is_within_tol]} > {tolerance_s=})."
+            " It means that the closest frame that can be loaded from the video is too far away in time."
+            " This might be due to synchronization issues with timestamps during data collection."
+            " To be safe, we advise to ignore this item during training."
+            f"\nqueried timestamps: {query_ts}"
+            f"\nloaded timestamps: {loaded_ts}"
+            f"\nvideo: {video_path}"
+        )
+
+    # get closest frames to the query timestamps
+    closest_frames = torch.stack([loaded_frames[idx] for idx in argmin_])
+    closest_ts = loaded_ts[argmin_]
+
+    if log_loaded_timestamps:
+        logger.info(f"{closest_ts=}")
+
+    # convert to float32 in [0,1] range
+    closest_frames = (closest_frames / 255.0).type(torch.float32)
+
+    if not len(timestamps) == len(closest_frames):
+        raise FrameTimestampError(
+            f"Retrieved timestamps differ from queried {set(closest_frames) - set(timestamps)}"
+        )
+
+    return closest_frames
+
+
+def encode_video_frames(
+    imgs_dir: Path | str,
+    video_path: Path | str,
+    fps: int,
+    vcodec: str = "libsvtav1",
+    pix_fmt: str = "yuv420p",
+    g: int | None = 2,
+    crf: int | None = 30,
+    fast_decode: int = 0,
+    log_level: int | None = av.logging.WARNING,
+    overwrite: bool = False,
+    preset: int | None = None,
+    encoder_threads: int | None = None,
+) -> None:
+    """More info on ffmpeg arguments tuning on `benchmark/video/README.md`"""
+    vcodec = resolve_vcodec(vcodec)
+
+    video_path = Path(video_path)
+    imgs_dir = Path(imgs_dir)
+
+    if video_path.exists() and not overwrite:
+        logger.warning(f"Video file already exists: {video_path}. Skipping encoding.")
+        return
+
+    video_path.parent.mkdir(parents=True, exist_ok=True)
+
+    # Encoders/pixel formats incompatibility check
+    if (vcodec == "libsvtav1" or vcodec == "hevc") and pix_fmt == "yuv444p":
+        logger.warning(
+            f"Incompatible pixel format 'yuv444p' for codec {vcodec}, auto-selecting format 'yuv420p'"
+        )
+        pix_fmt = "yuv420p"
+
+    # Get input frames
+    template = "frame-" + ("[0-9]" * 6) + ".png"
+    input_list = sorted(
+        glob.glob(str(imgs_dir / template)), key=lambda x: int(x.split("-")[-1].split(".")[0])
+    )
+
+    # Define video output frame size (assuming all input frames are the same size)
+    if len(input_list) == 0:
+        raise FileNotFoundError(f"No images found in {imgs_dir}.")
+    with Image.open(input_list[0]) as dummy_image:
+        width, height = dummy_image.size
+
+    # Define video codec options
+    video_options = _get_codec_options(vcodec, g, crf, preset)
+
+    if fast_decode:
+        key = "svtav1-params" if vcodec == "libsvtav1" else "tune"
+        value = f"fast-decode={fast_decode}" if vcodec == "libsvtav1" else "fastdecode"
+        video_options[key] = value
+
+    if encoder_threads is not None:
+        if vcodec == "libsvtav1":
+            lp_param = f"lp={encoder_threads}"
+            if "svtav1-params" in video_options:
+                video_options["svtav1-params"] += f":{lp_param}"
+            else:
+                video_options["svtav1-params"] = lp_param
+        else:
+            video_options["threads"] = str(encoder_threads)
+
+    # Set logging level
+    if log_level is not None:
+        # "While less efficient, it is generally preferable to modify logging with Python's logging"
+        logging.getLogger("libav").setLevel(log_level)
+
+    # Create and open output file (overwrite by default)
+    with av.open(str(video_path), "w") as output:
+        output_stream = output.add_stream(vcodec, fps, options=video_options)
+        output_stream.pix_fmt = pix_fmt
+        output_stream.width = width
+        output_stream.height = height
+
+        # Loop through input frames and encode them
+        for input_data in input_list:
+            with Image.open(input_data) as input_image:
+                input_image = input_image.convert("RGB")
+                input_frame = av.VideoFrame.from_image(input_image)
+                packet = output_stream.encode(input_frame)
+                if packet:
+                    output.mux(packet)
+
+        # Flush the encoder
+        packet = output_stream.encode()
+        if packet:
+            output.mux(packet)
+
+    # Reset logging level
+    if log_level is not None:
+        av.logging.restore_default_callback()
+
+    if not video_path.exists():
+        raise OSError(f"Video encoding did not work. File not found: {video_path}.")
+
+
+def concatenate_video_files(
+    input_video_paths: list[Path | str], output_video_path: Path, overwrite: bool = True
+):
+    """
+    Concatenate multiple video files into a single video file using pyav.
+
+    This function takes a list of video input file paths and concatenates them into a single
+    output video file. It uses ffmpeg's concat demuxer with stream copy mode for fast
+    concatenation without re-encoding.
+
+    Args:
+        input_video_paths: Ordered list of input video file paths to concatenate.
+        output_video_path: Path to the output video file.
+        overwrite: Whether to overwrite the output video file if it already exists. Default is True.
+
+    Note:
+        - Creates a temporary directory for intermediate files that is cleaned up after use.
+        - Uses ffmpeg's concat demuxer which requires all input videos to have the same
+          codec, resolution, and frame rate for proper concatenation.
+    """
+
+    output_video_path = Path(output_video_path)
+
+    if output_video_path.exists() and not overwrite:
+        logger.warning(f"Video file already exists: {output_video_path}. Skipping concatenation.")
+        return
+
+    output_video_path.parent.mkdir(parents=True, exist_ok=True)
+
+    if len(input_video_paths) == 0:
+        raise FileNotFoundError("No input video paths provided.")
+
+    # Create a temporary .ffconcat file to list the input video paths
+    with tempfile.NamedTemporaryFile(mode="w", suffix=".ffconcat", delete=False) as tmp_concatenate_file:
+        tmp_concatenate_file.write("ffconcat version 1.0\n")
+        for input_path in input_video_paths:
+            tmp_concatenate_file.write(f"file '{str(input_path.resolve())}'\n")
+        tmp_concatenate_file.flush()
+        tmp_concatenate_path = tmp_concatenate_file.name
+
+    # Create input and output containers
+    input_container = av.open(
+        tmp_concatenate_path, mode="r", format="concat", options={"safe": "0"}
+    )  # safe = 0 allows absolute paths as well as relative paths
+
+    with tempfile.NamedTemporaryFile(suffix=".mp4", delete=False) as tmp_named_file:
+        tmp_output_video_path = tmp_named_file.name
+
+    output_container = av.open(
+        tmp_output_video_path, mode="w", options={"movflags": "faststart"}
+    )  # faststart is to move the metadata to the beginning of the file to speed up loading
+
+    # Replicate input streams in output container
+    stream_map = {}
+    for input_stream in input_container.streams:
+        if input_stream.type in ("video", "audio", "subtitle"):  # only copy compatible streams
+            stream_map[input_stream.index] = output_container.add_stream_from_template(
+                template=input_stream, opaque=True
+            )
+
+            # set the time base to the input stream time base (missing in the codec context)
+            stream_map[input_stream.index].time_base = input_stream.time_base
+
+    # Demux + remux packets (no re-encode)
+    for packet in input_container.demux():
+        # Skip packets from un-mapped streams
+        if packet.stream.index not in stream_map:
+            continue
+
+        # Skip demux flushing packets
+        if packet.dts is None:
+            continue
+
+        output_stream = stream_map[packet.stream.index]
+        packet.stream = output_stream
+        output_container.mux(packet)
+
+    input_container.close()
+    output_container.close()
+    shutil.move(tmp_output_video_path, output_video_path)
+    Path(tmp_concatenate_path).unlink()
+
+
+class _CameraEncoderThread(threading.Thread):
+    """A thread that encodes video frames streamed via a queue into an MP4 file.
+
+    One instance is created per camera per episode. Frames are received as numpy arrays
+    from the main thread, encoded in real-time using PyAV (which releases the GIL during
+    encoding), and written to disk. Stats are computed incrementally using
+    RunningQuantileStats and returned via result_queue.
+    """
+
+    def __init__(
+        self,
+        video_path: Path,
+        fps: int,
+        vcodec: str,
+        pix_fmt: str,
+        g: int | None,
+        crf: int | None,
+        preset: int | None,
+        frame_queue: queue.Queue,
+        result_queue: queue.Queue,
+        stop_event: threading.Event,
+        encoder_threads: int | None = None,
+    ):
+        super().__init__(daemon=True)
+        self.video_path = video_path
+        self.fps = fps
+        self.vcodec = vcodec
+        self.pix_fmt = pix_fmt
+        self.g = g
+        self.crf = crf
+        self.preset = preset
+        self.frame_queue = frame_queue
+        self.result_queue = result_queue
+        self.stop_event = stop_event
+        self.encoder_threads = encoder_threads
+
+    def run(self) -> None:
+        from lerobot.datasets.compute_stats import RunningQuantileStats, auto_downsample_height_width
+
+        container = None
+        output_stream = None
+        stats_tracker = RunningQuantileStats()
+        frame_count = 0
+
+        try:
+            logging.getLogger("libav").setLevel(av.logging.WARNING)
+
+            while True:
+                try:
+                    frame_data = self.frame_queue.get(timeout=1)
+                except queue.Empty:
+                    if self.stop_event.is_set():
+                        break
+                    continue
+
+                if frame_data is None:
+                    # Sentinel: flush and close
+                    break
+
+                # Ensure HWC uint8 numpy array
+                if isinstance(frame_data, np.ndarray):
+                    if frame_data.ndim == 3 and frame_data.shape[0] == 3:
+                        # CHW -> HWC
+                        frame_data = frame_data.transpose(1, 2, 0)
+                    if frame_data.dtype != np.uint8:
+                        frame_data = (frame_data * 255).astype(np.uint8)
+
+                # Open container on first frame (to get width/height)
+                if container is None:
+                    height, width = frame_data.shape[:2]
+                    video_options = _get_codec_options(self.vcodec, self.g, self.crf, self.preset)
+                    if self.encoder_threads is not None:
+                        if self.vcodec == "libsvtav1":
+                            lp_param = f"lp={self.encoder_threads}"
+                            if "svtav1-params" in video_options:
+                                video_options["svtav1-params"] += f":{lp_param}"
+                            else:
+                                video_options["svtav1-params"] = lp_param
+                        else:
+                            video_options["threads"] = str(self.encoder_threads)
+                    Path(self.video_path).parent.mkdir(parents=True, exist_ok=True)
+                    container = av.open(str(self.video_path), "w")
+                    output_stream = container.add_stream(self.vcodec, self.fps, options=video_options)
+                    output_stream.pix_fmt = self.pix_fmt
+                    output_stream.width = width
+                    output_stream.height = height
+                    output_stream.time_base = Fraction(1, self.fps)
+
+                # Encode frame with explicit timestamps
+                pil_img = Image.fromarray(frame_data)
+                video_frame = av.VideoFrame.from_image(pil_img)
+                video_frame.pts = frame_count
+                video_frame.time_base = Fraction(1, self.fps)
+                packet = output_stream.encode(video_frame)
+                if packet:
+                    container.mux(packet)
+
+                # Update stats with downsampled frame (per-channel stats like compute_episode_stats)
+                img_chw = frame_data.transpose(2, 0, 1)  # HWC -> CHW
+                img_downsampled = auto_downsample_height_width(img_chw)
+                # Reshape CHW to (H*W, C) for per-channel stats
+                channels = img_downsampled.shape[0]
+                img_for_stats = img_downsampled.transpose(1, 2, 0).reshape(-1, channels)
+                stats_tracker.update(img_for_stats)
+
+                frame_count += 1
+
+            # Flush encoder
+            if output_stream is not None:
+                packet = output_stream.encode()
+                if packet:
+                    container.mux(packet)
+
+            if container is not None:
+                container.close()
+
+            av.logging.restore_default_callback()
+
+            # Get stats and put on result queue
+            if frame_count >= 2:
+                stats = stats_tracker.get_statistics()
+                self.result_queue.put(("ok", stats))
+            else:
+                self.result_queue.put(("ok", None))
+
+        except Exception as e:
+            logger.error(f"Encoder thread error: {e}")
+            if container is not None:
+                with contextlib.suppress(Exception):
+                    container.close()
+            self.result_queue.put(("error", str(e)))
+
+
+class StreamingVideoEncoder:
+    """Manages per-camera encoder threads for real-time video encoding during recording.
+
+    Instead of writing frames as PNG images and then encoding to MP4 at episode end,
+    this class streams frames directly to encoder threads, eliminating the
+    PNG round-trip and making save_episode() near-instant.
+
+    Uses threading instead of multiprocessing to avoid the overhead of pickling large
+    numpy arrays through multiprocessing.Queue. PyAV's encode() releases the GIL,
+    so encoding runs in parallel with the main recording loop.
+    """
+
+    def __init__(
+        self,
+        fps: int,
+        vcodec: str = "libsvtav1",
+        pix_fmt: str = "yuv420p",
+        g: int | None = 2,
+        crf: int | None = 30,
+        preset: int | None = None,
+        queue_maxsize: int = 30,
+        encoder_threads: int | None = None,
+    ):
+        self.fps = fps
+        self.vcodec = resolve_vcodec(vcodec)
+        self.pix_fmt = pix_fmt
+        self.g = g
+        self.crf = crf
+        self.preset = preset
+        self.queue_maxsize = queue_maxsize
+        self.encoder_threads = encoder_threads
+
+        self._frame_queues: dict[str, queue.Queue] = {}
+        self._result_queues: dict[str, queue.Queue] = {}
+        self._threads: dict[str, _CameraEncoderThread] = {}
+        self._stop_events: dict[str, threading.Event] = {}
+        self._video_paths: dict[str, Path] = {}
+        self._dropped_frames: dict[str, int] = {}
+        self._episode_active = False
+
+    def start_episode(self, video_keys: list[str], temp_dir: Path) -> None:
+        """Start encoder threads for a new episode.
+
+        Args:
+            video_keys: List of video feature keys (e.g. ["observation.images.laptop"])
+            temp_dir: Base directory for temporary MP4 files
+        """
+        if self._episode_active:
+            self.cancel_episode()
+
+        self._dropped_frames.clear()
+
+        for video_key in video_keys:
+            frame_queue: queue.Queue = queue.Queue(maxsize=self.queue_maxsize)
+            result_queue: queue.Queue = queue.Queue(maxsize=1)
+            stop_event = threading.Event()
+
+            temp_video_dir = Path(tempfile.mkdtemp(dir=temp_dir))
+            video_path = temp_video_dir / f"{video_key.replace('/', '_')}_streaming.mp4"
+
+            encoder_thread = _CameraEncoderThread(
+                video_path=video_path,
+                fps=self.fps,
+                vcodec=self.vcodec,
+                pix_fmt=self.pix_fmt,
+                g=self.g,
+                crf=self.crf,
+                preset=self.preset,
+                frame_queue=frame_queue,
+                result_queue=result_queue,
+                stop_event=stop_event,
+                encoder_threads=self.encoder_threads,
+            )
+            encoder_thread.start()
+
+            self._frame_queues[video_key] = frame_queue
+            self._result_queues[video_key] = result_queue
+            self._threads[video_key] = encoder_thread
+            self._stop_events[video_key] = stop_event
+            self._video_paths[video_key] = video_path
+
+        self._episode_active = True
+
+    def feed_frame(self, video_key: str, image: np.ndarray) -> None:
+        """Feed a frame to the encoder for a specific camera.
+
+        A copy of the image is made before enqueueing to prevent race conditions
+        with camera drivers that may reuse buffers. If the encoder queue is full
+        (encoder can't keep up), the frame is dropped with a warning instead of
+        crashing the recording session.
+
+        Args:
+            video_key: The video feature key
+            image: numpy array in (H,W,C) or (C,H,W) format, uint8 or float
+
+        Raises:
+            RuntimeError: If the encoder thread has crashed
+        """
+        if not self._episode_active:
+            raise RuntimeError("No active episode. Call start_episode() first.")
+
+        thread = self._threads[video_key]
+        if not thread.is_alive():
+            # Check for error
+            try:
+                status, msg = self._result_queues[video_key].get_nowait()
+                if status == "error":
+                    raise RuntimeError(f"Encoder thread for {video_key} crashed: {msg}")
+            except queue.Empty:
+                pass
+            raise RuntimeError(f"Encoder thread for {video_key} is not alive")
+
+        try:
+            self._frame_queues[video_key].put(image.copy(), timeout=0.1)
+        except queue.Full:
+            self._dropped_frames[video_key] = self._dropped_frames.get(video_key, 0) + 1
+            count = self._dropped_frames[video_key]
+            # Log periodically to avoid spam (1st, then every 10th)
+            if count == 1 or count % 10 == 0:
+                logger.warning(
+                    f"Encoder queue full for {video_key}, dropped {count} frame(s). "
+                    f"Consider using vcodec='auto' for hardware encoding or increasing encoder_queue_maxsize."
+                )
+
+    def finish_episode(self) -> dict[str, tuple[Path, dict | None]]:
+        """Finish encoding the current episode.
+
+        Sends sentinel values, waits for encoder threads to complete,
+        and collects results.
+
+        Returns:
+            Dict mapping video_key to (mp4_path, stats_dict_or_None)
+        """
+        if not self._episode_active:
+            raise RuntimeError("No active episode to finish.")
+
+        results = {}
+
+        # Report dropped frames
+        for video_key, count in self._dropped_frames.items():
+            if count > 0:
+                logger.warning(f"Episode finished with {count} dropped frame(s) for {video_key}.")
+
+        # Send sentinel to all queues
+        for video_key in self._frame_queues:
+            self._frame_queues[video_key].put(None)
+
+        # Wait for all threads and collect results
+        for video_key in self._threads:
+            self._threads[video_key].join(timeout=120)
+            if self._threads[video_key].is_alive():
+                logger.error(f"Encoder thread for {video_key} did not finish in time")
+                self._stop_events[video_key].set()
+                self._threads[video_key].join(timeout=5)
+                results[video_key] = (self._video_paths[video_key], None)
+                continue
+
+            try:
+                status, data = self._result_queues[video_key].get(timeout=5)
+                if status == "error":
+                    raise RuntimeError(f"Encoder thread for {video_key} failed: {data}")
+                results[video_key] = (self._video_paths[video_key], data)
+            except queue.Empty:
+                logger.error(f"No result from encoder thread for {video_key}")
+                results[video_key] = (self._video_paths[video_key], None)
+
+        self._cleanup()
+        self._episode_active = False
+        return results
+
+    def cancel_episode(self) -> None:
+        """Cancel the current episode, stopping encoder threads and cleaning up."""
+        if not self._episode_active:
+            return
+
+        # Signal all threads to stop
+        for video_key in self._stop_events:
+            self._stop_events[video_key].set()
+
+        # Wait for threads to finish
+        for video_key in self._threads:
+            self._threads[video_key].join(timeout=5)
+
+            # Clean up temp MP4 files
+            video_path = self._video_paths.get(video_key)
+            if video_path is not None and video_path.exists():
+                shutil.rmtree(str(video_path.parent), ignore_errors=True)
+
+        self._cleanup()
+        self._episode_active = False
+
+    def close(self) -> None:
+        """Close the encoder, canceling any in-progress episode."""
+        if self._episode_active:
+            self.cancel_episode()
+
+    def _cleanup(self) -> None:
+        """Clean up queues and thread tracking dicts."""
+        for q in self._frame_queues.values():
+            with contextlib.suppress(Exception):
+                while not q.empty():
+                    q.get_nowait()
+        self._frame_queues.clear()
+        self._result_queues.clear()
+        self._threads.clear()
+        self._stop_events.clear()
+        self._video_paths.clear()
+
+
+@dataclass
+class VideoFrame:
+    # TODO(rcadene, lhoestq): move to Hugging Face `datasets` repo
+    """
+    Provides a type for a dataset containing video frames.
+
+    Example:
+
+    ```python
+    data_dict = [{"image": {"path": "videos/episode_0.mp4", "timestamp": 0.3}}]
+    features = {"image": VideoFrame()}
+    Dataset.from_dict(data_dict, features=Features(features))
+    ```
+    """
+
+    pa_type: ClassVar[Any] = pa.struct({"path": pa.string(), "timestamp": pa.float32()})
+    _type: str = field(default="VideoFrame", init=False, repr=False)
+
+    def __call__(self):
+        return self.pa_type
+
+
+with warnings.catch_warnings():
+    warnings.filterwarnings(
+        "ignore",
+        "'register_feature' is experimental and might be subject to breaking changes in the future.",
+        category=UserWarning,
+    )
+    # to make VideoFrame available in HuggingFace `datasets`
+    register_feature(VideoFrame, "VideoFrame")
+
+
+def get_audio_info(video_path: Path | str) -> dict:
+    # Set logging level
+    logging.getLogger("libav").setLevel(av.logging.WARNING)
+
+    # Getting audio stream information
+    audio_info = {}
+    with av.open(str(video_path), "r") as audio_file:
+        try:
+            audio_stream = audio_file.streams.audio[0]
+        except IndexError:
+            # Reset logging level
+            av.logging.restore_default_callback()
+            return {"has_audio": False}
+
+        audio_info["audio.channels"] = audio_stream.channels
+        audio_info["audio.codec"] = audio_stream.codec.canonical_name
+        # In an ideal loseless case : bit depth x sample rate x channels = bit rate.
+        # In an actual compressed case, the bit rate is set according to the compression level : the lower the bit rate, the more compression is applied.
+        audio_info["audio.bit_rate"] = audio_stream.bit_rate
+        audio_info["audio.sample_rate"] = audio_stream.sample_rate  # Number of samples per second
+        # In an ideal loseless case : fixed number of bits per sample.
+        # In an actual compressed case : variable number of bits per sample (often reduced to match a given depth rate).
+        audio_info["audio.bit_depth"] = audio_stream.format.bits
+        audio_info["audio.channel_layout"] = audio_stream.layout.name
+        audio_info["has_audio"] = True
+
+    # Reset logging level
+    av.logging.restore_default_callback()
+
+    return audio_info
+
+
+def get_video_info(video_path: Path | str) -> dict:
+    # Set logging level
+    logging.getLogger("libav").setLevel(av.logging.WARNING)
+
+    # Getting video stream information
+    video_info = {}
+    with av.open(str(video_path), "r") as video_file:
+        try:
+            video_stream = video_file.streams.video[0]
+        except IndexError:
+            # Reset logging level
+            av.logging.restore_default_callback()
+            return {}
+
+        video_info["video.height"] = video_stream.height
+        video_info["video.width"] = video_stream.width
+        video_info["video.codec"] = video_stream.codec.canonical_name
+        video_info["video.pix_fmt"] = video_stream.pix_fmt
+        video_info["video.is_depth_map"] = False
+
+        # Calculate fps from r_frame_rate
+        video_info["video.fps"] = int(video_stream.base_rate)
+
+        pixel_channels = get_video_pixel_channels(video_stream.pix_fmt)
+        video_info["video.channels"] = pixel_channels
+
+    # Reset logging level
+    av.logging.restore_default_callback()
+
+    # Adding audio stream information
+    video_info.update(**get_audio_info(video_path))
+
+    return video_info
+
+
+def get_video_pixel_channels(pix_fmt: str) -> int:
+    if "gray" in pix_fmt or "depth" in pix_fmt or "monochrome" in pix_fmt:
+        return 1
+    elif "rgba" in pix_fmt or "yuva" in pix_fmt:
+        return 4
+    elif "rgb" in pix_fmt or "yuv" in pix_fmt:
+        return 3
+    else:
+        raise ValueError("Unknown format")
+
+
+def get_video_duration_in_s(video_path: Path | str) -> float:
+    """
+    Get the duration of a video file in seconds using PyAV.
+
+    Args:
+        video_path: Path to the video file.
+
+    Returns:
+        Duration of the video in seconds.
+    """
+    with av.open(str(video_path)) as container:
+        # Get the first video stream
+        video_stream = container.streams.video[0]
+        # Calculate duration: stream.duration * stream.time_base gives duration in seconds
+        if video_stream.duration is not None:
+            duration = float(video_stream.duration * video_stream.time_base)
+        else:
+            # Fallback to container duration if stream duration is not available
+            duration = float(container.duration / av.time_base)
+    return duration
+
+
+class VideoEncodingManager:
+    """
+    Context manager that ensures proper video encoding and data cleanup even if exceptions occur.
+
+    This manager handles:
+    - Batch encoding for any remaining episodes when recording interrupted
+    - Cleaning up temporary image files from interrupted episodes
+    - Removing empty image directories
+
+    Args:
+        dataset: The LeRobotDataset instance
+    """
+
+    def __init__(self, dataset):
+        self.dataset = dataset
+
+    def __enter__(self):
+        return self
+
+    def __exit__(self, exc_type, exc_val, exc_tb):
+        streaming_encoder = getattr(self.dataset, "_streaming_encoder", None)
+
+        if streaming_encoder is not None:
+            # Handle streaming encoder cleanup
+            if exc_type is not None:
+                streaming_encoder.cancel_episode()
+            streaming_encoder.close()
+        elif self.dataset.episodes_since_last_encoding > 0:
+            # Handle any remaining episodes that haven't been batch encoded
+            if exc_type is not None:
+                logger.info("Exception occurred. Encoding remaining episodes before exit...")
+            else:
+                logger.info("Recording stopped. Encoding remaining episodes...")
+
+            start_ep = self.dataset.num_episodes - self.dataset.episodes_since_last_encoding
+            end_ep = self.dataset.num_episodes
+            logger.info(
+                f"Encoding remaining {self.dataset.episodes_since_last_encoding} episodes, "
+                f"from episode {start_ep} to {end_ep - 1}"
+            )
+            self.dataset._batch_save_episode_video(start_ep, end_ep)
+
+        # Finalize the dataset to properly close all writers
+        self.dataset.finalize()
+
+        # Clean up episode images if recording was interrupted (only for non-streaming mode)
+        if exc_type is not None and streaming_encoder is None:
+            interrupted_episode_index = self.dataset.num_episodes
+            for key in self.dataset.meta.video_keys:
+                img_dir = self.dataset._get_image_file_path(
+                    episode_index=interrupted_episode_index, image_key=key, frame_index=0
+                ).parent
+                if img_dir.exists():
+                    logger.debug(
+                        f"Cleaning up interrupted episode images for episode {interrupted_episode_index}, camera {key}"
+                    )
+                    shutil.rmtree(img_dir)
+
+        # Clean up any remaining images directory if it's empty
+        img_dir = self.dataset.root / "images"
+        if img_dir.exists():
+            png_files = list(img_dir.rglob("*.png"))
+            if len(png_files) == 0:
+                shutil.rmtree(img_dir)
+                logger.debug("Cleaned up empty images directory")
+            else:
+                logger.debug(f"Images directory is not empty, containing {len(png_files)} PNG files")
+
+        return False  # Don't suppress the original exception
diff --git a/lerobot/src/lerobot/envs/__init__.py b/lerobot/src/lerobot/envs/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..183c123252a67ea7f328456eaad84f92ef5266ca
--- /dev/null
+++ b/lerobot/src/lerobot/envs/__init__.py
@@ -0,0 +1,15 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configs import AlohaEnv, EnvConfig, HubEnvConfig, PushtEnv  # noqa: F401
diff --git a/lerobot/src/lerobot/envs/configs.py b/lerobot/src/lerobot/envs/configs.py
new file mode 100644
index 0000000000000000000000000000000000000000..9c1c083a49041a5726309102ac29249ee105d33c
--- /dev/null
+++ b/lerobot/src/lerobot/envs/configs.py
@@ -0,0 +1,456 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import abc
+from dataclasses import dataclass, field, fields
+from typing import Any
+
+import draccus
+
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.robots import RobotConfig
+from lerobot.teleoperators.config import TeleoperatorConfig
+from lerobot.utils.constants import (
+    ACTION,
+    LIBERO_KEY_EEF_MAT,
+    LIBERO_KEY_EEF_POS,
+    LIBERO_KEY_EEF_QUAT,
+    LIBERO_KEY_GRIPPER_QPOS,
+    LIBERO_KEY_GRIPPER_QVEL,
+    LIBERO_KEY_JOINTS_POS,
+    LIBERO_KEY_JOINTS_VEL,
+    LIBERO_KEY_PIXELS_AGENTVIEW,
+    LIBERO_KEY_PIXELS_EYE_IN_HAND,
+    OBS_ENV_STATE,
+    OBS_IMAGE,
+    OBS_IMAGES,
+    OBS_STATE,
+)
+
+
+@dataclass
+class EnvConfig(draccus.ChoiceRegistry, abc.ABC):
+    task: str | None = None
+    fps: int = 30
+    features: dict[str, PolicyFeature] = field(default_factory=dict)
+    features_map: dict[str, str] = field(default_factory=dict)
+    max_parallel_tasks: int = 1
+    disable_env_checker: bool = True
+
+    @property
+    def type(self) -> str:
+        return self.get_choice_name(self.__class__)
+
+    @property
+    def package_name(self) -> str:
+        """Package name to import if environment not found in gym registry"""
+        return f"gym_{self.type}"
+
+    @property
+    def gym_id(self) -> str:
+        """ID string used in gym.make() to instantiate the environment"""
+        return f"{self.package_name}/{self.task}"
+
+    @property
+    @abc.abstractmethod
+    def gym_kwargs(self) -> dict:
+        raise NotImplementedError()
+
+
+@dataclass
+class HubEnvConfig(EnvConfig):
+    """Base class for environments that delegate creation to a hub-hosted make_env.
+
+    Hub environments download and execute remote code from the HF Hub.
+    The hub_path points to a repository containing an env.py with a make_env function.
+    """
+
+    hub_path: str | None = None  # required: e.g., "username/repo" or "username/repo@branch:file.py"
+
+    @property
+    def gym_kwargs(self) -> dict:
+        # Not used for hub environments - the hub's make_env handles everything
+        return {}
+
+
+@EnvConfig.register_subclass("aloha")
+@dataclass
+class AlohaEnv(EnvConfig):
+    task: str | None = "AlohaInsertion-v0"
+    fps: int = 50
+    episode_length: int = 400
+    obs_type: str = "pixels_agent_pos"
+    observation_height: int = 480
+    observation_width: int = 640
+    render_mode: str = "rgb_array"
+    features: dict[str, PolicyFeature] = field(
+        default_factory=lambda: {
+            ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(14,)),
+        }
+    )
+    features_map: dict[str, str] = field(
+        default_factory=lambda: {
+            ACTION: ACTION,
+            "agent_pos": OBS_STATE,
+            "top": f"{OBS_IMAGE}.top",
+            "pixels/top": f"{OBS_IMAGES}.top",
+        }
+    )
+
+    def __post_init__(self):
+        if self.obs_type == "pixels":
+            self.features["top"] = PolicyFeature(
+                type=FeatureType.VISUAL, shape=(self.observation_height, self.observation_width, 3)
+            )
+        elif self.obs_type == "pixels_agent_pos":
+            self.features["agent_pos"] = PolicyFeature(type=FeatureType.STATE, shape=(14,))
+            self.features["pixels/top"] = PolicyFeature(
+                type=FeatureType.VISUAL, shape=(self.observation_height, self.observation_width, 3)
+            )
+
+    @property
+    def gym_kwargs(self) -> dict:
+        return {
+            "obs_type": self.obs_type,
+            "render_mode": self.render_mode,
+            "max_episode_steps": self.episode_length,
+        }
+
+
+@EnvConfig.register_subclass("pusht")
+@dataclass
+class PushtEnv(EnvConfig):
+    task: str | None = "PushT-v0"
+    fps: int = 10
+    episode_length: int = 300
+    obs_type: str = "pixels_agent_pos"
+    render_mode: str = "rgb_array"
+    visualization_width: int = 384
+    visualization_height: int = 384
+    observation_height: int = 384
+    observation_width: int = 384
+    features: dict[str, PolicyFeature] = field(
+        default_factory=lambda: {
+            ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(2,)),
+            "agent_pos": PolicyFeature(type=FeatureType.STATE, shape=(2,)),
+        }
+    )
+    features_map: dict[str, str] = field(
+        default_factory=lambda: {
+            ACTION: ACTION,
+            "agent_pos": OBS_STATE,
+            "environment_state": OBS_ENV_STATE,
+            "pixels": OBS_IMAGE,
+        }
+    )
+
+    def __post_init__(self):
+        if self.obs_type == "pixels_agent_pos":
+            self.features["pixels"] = PolicyFeature(
+                type=FeatureType.VISUAL, shape=(self.observation_height, self.observation_width, 3)
+            )
+        elif self.obs_type == "environment_state_agent_pos":
+            self.features["environment_state"] = PolicyFeature(type=FeatureType.ENV, shape=(16,))
+
+    @property
+    def gym_kwargs(self) -> dict:
+        return {
+            "obs_type": self.obs_type,
+            "render_mode": self.render_mode,
+            "visualization_width": self.visualization_width,
+            "visualization_height": self.visualization_height,
+            "max_episode_steps": self.episode_length,
+        }
+
+
+@dataclass
+class ImagePreprocessingConfig:
+    crop_params_dict: dict[str, tuple[int, int, int, int]] | None = None
+    resize_size: tuple[int, int] | None = None
+
+
+@dataclass
+class RewardClassifierConfig:
+    """Configuration for reward classification."""
+
+    pretrained_path: str | None = None
+    success_threshold: float = 0.5
+    success_reward: float = 1.0
+
+
+@dataclass
+class InverseKinematicsConfig:
+    """Configuration for inverse kinematics processing."""
+
+    urdf_path: str | None = None
+    target_frame_name: str | None = None
+    end_effector_bounds: dict[str, list[float]] | None = None
+    end_effector_step_sizes: dict[str, float] | None = None
+
+
+@dataclass
+class ObservationConfig:
+    """Configuration for observation processing."""
+
+    add_joint_velocity_to_observation: bool = False
+    add_current_to_observation: bool = False
+    add_ee_pose_to_observation: bool = False
+    display_cameras: bool = False
+
+
+@dataclass
+class GripperConfig:
+    """Configuration for gripper control and penalties."""
+
+    use_gripper: bool = True
+    gripper_penalty: float = 0.0
+
+
+@dataclass
+class ResetConfig:
+    """Configuration for environment reset behavior."""
+
+    fixed_reset_joint_positions: Any | None = None
+    reset_time_s: float = 5.0
+    control_time_s: float = 20.0
+    terminate_on_success: bool = True
+
+
+@dataclass
+class HILSerlProcessorConfig:
+    """Configuration for environment processing pipeline."""
+
+    control_mode: str = "gamepad"
+    observation: ObservationConfig | None = None
+    image_preprocessing: ImagePreprocessingConfig | None = None
+    gripper: GripperConfig | None = None
+    reset: ResetConfig | None = None
+    inverse_kinematics: InverseKinematicsConfig | None = None
+    reward_classifier: RewardClassifierConfig | None = None
+    max_gripper_pos: float | None = 100.0
+
+
+@EnvConfig.register_subclass(name="gym_manipulator")
+@dataclass
+class HILSerlRobotEnvConfig(EnvConfig):
+    """Configuration for the HILSerlRobotEnv environment."""
+
+    robot: RobotConfig | None = None
+    teleop: TeleoperatorConfig | None = None
+    processor: HILSerlProcessorConfig = field(default_factory=HILSerlProcessorConfig)
+
+    name: str = "real_robot"
+
+    @property
+    def gym_kwargs(self) -> dict:
+        return {}
+
+
+@EnvConfig.register_subclass("libero")
+@dataclass
+class LiberoEnv(EnvConfig):
+    task: str = "libero_10"  # can also choose libero_spatial, libero_object, etc.
+    task_ids: list[int] | None = None
+    fps: int = 30
+    episode_length: int | None = None
+    obs_type: str = "pixels_agent_pos"
+    render_mode: str = "rgb_array"
+    camera_name: str = "agentview_image,robot0_eye_in_hand_image"
+    init_states: bool = True
+    camera_name_mapping: dict[str, str] | None = None
+    observation_height: int = 360
+    observation_width: int = 360
+    features: dict[str, PolicyFeature] = field(
+        default_factory=lambda: {
+            ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+        }
+    )
+    features_map: dict[str, str] = field(
+        default_factory=lambda: {
+            ACTION: ACTION,
+            LIBERO_KEY_EEF_POS: f"{OBS_STATE}.eef_pos",
+            LIBERO_KEY_EEF_QUAT: f"{OBS_STATE}.eef_quat",
+            LIBERO_KEY_EEF_MAT: f"{OBS_STATE}.eef_mat",
+            LIBERO_KEY_GRIPPER_QPOS: f"{OBS_STATE}.gripper_qpos",
+            LIBERO_KEY_GRIPPER_QVEL: f"{OBS_STATE}.gripper_qvel",
+            LIBERO_KEY_JOINTS_POS: f"{OBS_STATE}.joint_pos",
+            LIBERO_KEY_JOINTS_VEL: f"{OBS_STATE}.joint_vel",
+            LIBERO_KEY_PIXELS_AGENTVIEW: f"{OBS_IMAGES}.image",
+            LIBERO_KEY_PIXELS_EYE_IN_HAND: f"{OBS_IMAGES}.image2",
+        }
+    )
+    control_mode: str = "relative"  # or "absolute"
+
+    def __post_init__(self):
+        if self.obs_type == "pixels":
+            self.features[LIBERO_KEY_PIXELS_AGENTVIEW] = PolicyFeature(
+                type=FeatureType.VISUAL, shape=(self.observation_height, self.observation_width, 3)
+            )
+            self.features[LIBERO_KEY_PIXELS_EYE_IN_HAND] = PolicyFeature(
+                type=FeatureType.VISUAL, shape=(self.observation_height, self.observation_width, 3)
+            )
+        elif self.obs_type == "pixels_agent_pos":
+            self.features[LIBERO_KEY_PIXELS_AGENTVIEW] = PolicyFeature(
+                type=FeatureType.VISUAL, shape=(self.observation_height, self.observation_width, 3)
+            )
+            self.features[LIBERO_KEY_PIXELS_EYE_IN_HAND] = PolicyFeature(
+                type=FeatureType.VISUAL, shape=(self.observation_height, self.observation_width, 3)
+            )
+            self.features[LIBERO_KEY_EEF_POS] = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(3,),
+            )
+            self.features[LIBERO_KEY_EEF_QUAT] = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(4,),
+            )
+            self.features[LIBERO_KEY_EEF_MAT] = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(3, 3),
+            )
+            self.features[LIBERO_KEY_GRIPPER_QPOS] = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(2,),
+            )
+            self.features[LIBERO_KEY_GRIPPER_QVEL] = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(2,),
+            )
+            self.features[LIBERO_KEY_JOINTS_POS] = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(7,),
+            )
+            self.features[LIBERO_KEY_JOINTS_VEL] = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(7,),
+            )
+        else:
+            raise ValueError(f"Unsupported obs_type: {self.obs_type}")
+
+    @property
+    def gym_kwargs(self) -> dict:
+        kwargs: dict[str, Any] = {"obs_type": self.obs_type, "render_mode": self.render_mode}
+        if self.task_ids is not None:
+            kwargs["task_ids"] = self.task_ids
+        return kwargs
+
+
+@EnvConfig.register_subclass("metaworld")
+@dataclass
+class MetaworldEnv(EnvConfig):
+    task: str = "metaworld-push-v2"  # add all tasks
+    fps: int = 80
+    episode_length: int = 400
+    obs_type: str = "pixels_agent_pos"
+    render_mode: str = "rgb_array"
+    multitask_eval: bool = True
+    features: dict[str, PolicyFeature] = field(
+        default_factory=lambda: {
+            "action": PolicyFeature(type=FeatureType.ACTION, shape=(4,)),
+        }
+    )
+    features_map: dict[str, str] = field(
+        default_factory=lambda: {
+            "action": ACTION,
+            "agent_pos": OBS_STATE,
+            "top": f"{OBS_IMAGE}",
+            "pixels/top": f"{OBS_IMAGE}",
+        }
+    )
+
+    def __post_init__(self):
+        if self.obs_type == "pixels":
+            self.features["top"] = PolicyFeature(type=FeatureType.VISUAL, shape=(480, 480, 3))
+
+        elif self.obs_type == "pixels_agent_pos":
+            self.features["agent_pos"] = PolicyFeature(type=FeatureType.STATE, shape=(4,))
+            self.features["pixels/top"] = PolicyFeature(type=FeatureType.VISUAL, shape=(480, 480, 3))
+
+        else:
+            raise ValueError(f"Unsupported obs_type: {self.obs_type}")
+
+    @property
+    def gym_kwargs(self) -> dict:
+        return {
+            "obs_type": self.obs_type,
+            "render_mode": self.render_mode,
+        }
+
+
+@EnvConfig.register_subclass("isaaclab_arena")
+@dataclass
+class IsaaclabArenaEnv(HubEnvConfig):
+    hub_path: str = "nvidia/isaaclab-arena-envs"
+    episode_length: int = 300
+    num_envs: int = 1
+    embodiment: str | None = "gr1_pink"
+    object: str | None = "power_drill"
+    mimic: bool = False
+    teleop_device: str | None = None
+    seed: int | None = 42
+    device: str | None = "cuda:0"
+    disable_fabric: bool = False
+    enable_cameras: bool = False
+    headless: bool = False
+    enable_pinocchio: bool = True
+    environment: str | None = "gr1_microwave"
+    task: str | None = "Reach out to the microwave and open it."
+    state_dim: int = 54
+    action_dim: int = 36
+    camera_height: int = 512
+    camera_width: int = 512
+    video: bool = False
+    video_length: int = 100
+    video_interval: int = 200
+    # Comma-separated keys, e.g., "robot_joint_pos,left_eef_pos"
+    state_keys: str = "robot_joint_pos"
+    # Comma-separated keys, e.g., "robot_pov_cam_rgb,front_cam_rgb"
+    # Set to None or "" for environments without cameras
+    camera_keys: str | None = None
+    features: dict[str, PolicyFeature] = field(default_factory=dict)
+    features_map: dict[str, str] = field(default_factory=dict)
+    kwargs: dict | None = None
+
+    def __post_init__(self):
+        if self.kwargs:
+            # dynamically convert kwargs to fields in the dataclass
+            # NOTE! the new fields will not bee seen by the dataclass repr
+            field_names = {f.name for f in fields(self)}
+            for key, value in self.kwargs.items():
+                if key not in field_names and key != "kwargs":
+                    setattr(self, key, value)
+            self.kwargs = None
+
+        # Set action feature
+        self.features[ACTION] = PolicyFeature(type=FeatureType.ACTION, shape=(self.action_dim,))
+        self.features_map[ACTION] = ACTION
+
+        # Set state feature
+        self.features[OBS_STATE] = PolicyFeature(type=FeatureType.STATE, shape=(self.state_dim,))
+        self.features_map[OBS_STATE] = OBS_STATE
+
+        # Add camera features for each camera key
+        if self.enable_cameras and self.camera_keys:
+            for cam_key in self.camera_keys.split(","):
+                cam_key = cam_key.strip()
+                if cam_key:
+                    self.features[cam_key] = PolicyFeature(
+                        type=FeatureType.VISUAL,
+                        shape=(self.camera_height, self.camera_width, 3),
+                    )
+                    self.features_map[cam_key] = f"{OBS_IMAGES}.{cam_key}"
+
+    @property
+    def gym_kwargs(self) -> dict:
+        return {}
diff --git a/lerobot/src/lerobot/envs/factory.py b/lerobot/src/lerobot/envs/factory.py
new file mode 100644
index 0000000000000000000000000000000000000000..1c59ccb7dd27f56237b2c6c01d1ef9085de16d60
--- /dev/null
+++ b/lerobot/src/lerobot/envs/factory.py
@@ -0,0 +1,219 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import importlib
+from typing import Any
+
+import gymnasium as gym
+from gymnasium.envs.registration import registry as gym_registry
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.envs.configs import AlohaEnv, EnvConfig, HubEnvConfig, IsaaclabArenaEnv, LiberoEnv, PushtEnv
+from lerobot.envs.utils import _call_make_env, _download_hub_file, _import_hub_module, _normalize_hub_result
+from lerobot.policies.xvla.configuration_xvla import XVLAConfig
+from lerobot.processor import ProcessorStep
+from lerobot.processor.env_processor import IsaaclabArenaProcessorStep, LiberoProcessorStep
+from lerobot.processor.pipeline import PolicyProcessorPipeline
+
+
+def make_env_config(env_type: str, **kwargs) -> EnvConfig:
+    if env_type == "aloha":
+        return AlohaEnv(**kwargs)
+    elif env_type == "pusht":
+        return PushtEnv(**kwargs)
+    elif env_type == "libero":
+        return LiberoEnv(**kwargs)
+    else:
+        raise ValueError(f"Policy type '{env_type}' is not available.")
+
+
+def make_env_pre_post_processors(
+    env_cfg: EnvConfig,
+    policy_cfg: PreTrainedConfig,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+]:
+    """
+    Create preprocessor and postprocessor pipelines for environment observations.
+
+    This function creates processor pipelines that transform raw environment
+    observations and actions. By default, it returns identity processors that do nothing.
+    For specific environments like LIBERO, it adds environment-specific processing steps.
+
+    Args:
+        env_cfg: The configuration of the environment.
+
+    Returns:
+        A tuple containing:
+            - preprocessor: Pipeline that processes environment observations
+            - postprocessor: Pipeline that processes environment outputs (currently identity)
+    """
+    # Preprocessor and Postprocessor steps are Identity for most environments
+    preprocessor_steps: list[ProcessorStep] = []
+    postprocessor_steps: list[ProcessorStep] = []
+    if isinstance(policy_cfg, XVLAConfig):
+        from lerobot.policies.xvla.processor_xvla import make_xvla_libero_pre_post_processors
+
+        return make_xvla_libero_pre_post_processors()
+
+    # For LIBERO environments, add the LiberoProcessorStep to preprocessor
+    if isinstance(env_cfg, LiberoEnv) or "libero" in env_cfg.type:
+        preprocessor_steps.append(LiberoProcessorStep())
+
+    # For Isaaclab Arena environments, add the IsaaclabArenaProcessorStep
+    if isinstance(env_cfg, IsaaclabArenaEnv) or "isaaclab_arena" in env_cfg.type:
+        # Parse comma-separated keys (handle None for state-based policies)
+        if env_cfg.state_keys:
+            state_keys = tuple(k.strip() for k in env_cfg.state_keys.split(",") if k.strip())
+        else:
+            state_keys = ()
+        if env_cfg.camera_keys:
+            camera_keys = tuple(k.strip() for k in env_cfg.camera_keys.split(",") if k.strip())
+        else:
+            camera_keys = ()
+        if not state_keys and not camera_keys:
+            raise ValueError("At least one of state_keys or camera_keys must be specified.")
+        preprocessor_steps.append(
+            IsaaclabArenaProcessorStep(
+                state_keys=state_keys,
+                camera_keys=camera_keys,
+            )
+        )
+
+    preprocessor = PolicyProcessorPipeline(steps=preprocessor_steps)
+    postprocessor = PolicyProcessorPipeline(steps=postprocessor_steps)
+
+    return preprocessor, postprocessor
+
+
+def make_env(
+    cfg: EnvConfig | str,
+    n_envs: int = 1,
+    use_async_envs: bool = False,
+    hub_cache_dir: str | None = None,
+    trust_remote_code: bool = False,
+) -> dict[str, dict[int, gym.vector.VectorEnv]]:
+    """Makes a gym vector environment according to the config or Hub reference.
+
+    Args:
+        cfg (EnvConfig | str): Either an `EnvConfig` object describing the environment to build locally,
+            or a Hugging Face Hub repository identifier (e.g. `"username/repo"`). In the latter case,
+            the repo must include a Python file (usually `env.py`).
+        n_envs (int, optional): The number of parallelized env to return. Defaults to 1.
+        use_async_envs (bool, optional): Whether to return an AsyncVectorEnv or a SyncVectorEnv. Defaults to
+            False.
+        hub_cache_dir (str | None): Optional cache path for downloaded hub files.
+        trust_remote_code (bool): **Explicit consent** to execute remote code from the Hub.
+            Default False — must be set to True to import/exec hub `env.py`.
+    Raises:
+        ValueError: if n_envs < 1
+        ModuleNotFoundError: If the requested env package is not installed
+
+    Returns:
+        dict[str, dict[int, gym.vector.VectorEnv]]:
+            A mapping from suite name to indexed vectorized environments.
+            - For multi-task benchmarks (e.g., LIBERO): one entry per suite, and one vec env per task_id.
+            - For single-task environments: a single suite entry (cfg.type) with task_id=0.
+
+    """
+    # if user passed a hub id string (e.g., "username/repo", "username/repo@main:env.py")
+    # simplified: only support hub-provided `make_env`
+    # TODO: (jadechoghari): deprecate string API and remove this check
+    if isinstance(cfg, str):
+        hub_path: str | None = cfg
+    elif isinstance(cfg, HubEnvConfig):
+        hub_path = cfg.hub_path
+    else:
+        hub_path = None
+
+    # If hub_path is set, download and call hub-provided `make_env`
+    if hub_path:
+        # _download_hub_file will raise the same RuntimeError if trust_remote_code is False
+        repo_id, file_path, local_file, revision = _download_hub_file(
+            hub_path, trust_remote_code, hub_cache_dir
+        )
+
+        # import and surface clear import errors
+        module = _import_hub_module(local_file, repo_id)
+
+        # call the hub-provided make_env
+        env_cfg = None if isinstance(cfg, str) else cfg
+        raw_result = _call_make_env(module, n_envs=n_envs, use_async_envs=use_async_envs, cfg=env_cfg)
+
+        # normalize the return into {suite: {task_id: vec_env}}
+        return _normalize_hub_result(raw_result)
+
+    # At this point, cfg must be an EnvConfig (not a string) since hub_path would have been set otherwise
+    if isinstance(cfg, str):
+        raise TypeError("cfg should be an EnvConfig at this point")
+
+    if n_envs < 1:
+        raise ValueError("`n_envs` must be at least 1")
+
+    env_cls = gym.vector.AsyncVectorEnv if use_async_envs else gym.vector.SyncVectorEnv
+
+    if "libero" in cfg.type:
+        from lerobot.envs.libero import create_libero_envs
+
+        if cfg.task is None:
+            raise ValueError("LiberoEnv requires a task to be specified")
+
+        return create_libero_envs(
+            task=cfg.task,
+            n_envs=n_envs,
+            camera_name=cfg.camera_name,
+            init_states=cfg.init_states,
+            gym_kwargs=cfg.gym_kwargs,
+            env_cls=env_cls,
+            control_mode=cfg.control_mode,
+            episode_length=cfg.episode_length,
+        )
+    elif "metaworld" in cfg.type:
+        from lerobot.envs.metaworld import create_metaworld_envs
+
+        if cfg.task is None:
+            raise ValueError("MetaWorld requires a task to be specified")
+
+        return create_metaworld_envs(
+            task=cfg.task,
+            n_envs=n_envs,
+            gym_kwargs=cfg.gym_kwargs,
+            env_cls=env_cls,
+        )
+
+    if cfg.gym_id not in gym_registry:
+        print(f"gym id '{cfg.gym_id}' not found, attempting to import '{cfg.package_name}'...")
+        try:
+            importlib.import_module(cfg.package_name)
+        except ModuleNotFoundError as e:
+            raise ModuleNotFoundError(
+                f"Package '{cfg.package_name}' required for env '{cfg.type}' not found. "
+                f"Please install it or check PYTHONPATH."
+            ) from e
+
+        if cfg.gym_id not in gym_registry:
+            raise gym.error.NameNotFound(
+                f"Environment '{cfg.gym_id}' not registered even after importing '{cfg.package_name}'."
+            )
+
+    def _make_one():
+        return gym.make(cfg.gym_id, disable_env_checker=cfg.disable_env_checker, **(cfg.gym_kwargs or {}))
+
+    vec = env_cls([_make_one for _ in range(n_envs)], autoreset_mode=gym.vector.AutoresetMode.SAME_STEP)
+
+    # normalize to {suite: {task_id: vec_env}} for consistency
+    suite_name = cfg.type  # e.g., "pusht", "aloha"
+    return {suite_name: {0: vec}}
diff --git a/lerobot/src/lerobot/envs/libero.py b/lerobot/src/lerobot/envs/libero.py
new file mode 100644
index 0000000000000000000000000000000000000000..6d3589fed261859d4eb47157cf53cbc10835ccd8
--- /dev/null
+++ b/lerobot/src/lerobot/envs/libero.py
@@ -0,0 +1,457 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from __future__ import annotations
+
+import os
+from collections import defaultdict
+from collections.abc import Callable, Iterable, Mapping, Sequence
+from functools import partial
+from pathlib import Path
+from typing import Any
+
+import gymnasium as gym
+import numpy as np
+import torch
+from gymnasium import spaces
+from libero.libero import benchmark, get_libero_path
+from libero.libero.envs import OffScreenRenderEnv
+
+from lerobot.types import RobotObservation
+
+
+def _parse_camera_names(camera_name: str | Sequence[str]) -> list[str]:
+    """Normalize camera_name into a non-empty list of strings."""
+    if isinstance(camera_name, str):
+        cams = [c.strip() for c in camera_name.split(",") if c.strip()]
+    elif isinstance(camera_name, (list | tuple)):
+        cams = [str(c).strip() for c in camera_name if str(c).strip()]
+    else:
+        raise TypeError(f"camera_name must be str or sequence[str], got {type(camera_name).__name__}")
+    if not cams:
+        raise ValueError("camera_name resolved to an empty list.")
+    return cams
+
+
+def _get_suite(name: str) -> benchmark.Benchmark:
+    """Instantiate a LIBERO suite by name with clear validation."""
+    bench = benchmark.get_benchmark_dict()
+    if name not in bench:
+        raise ValueError(f"Unknown LIBERO suite '{name}'. Available: {', '.join(sorted(bench.keys()))}")
+    suite = bench[name]()
+    if not getattr(suite, "tasks", None):
+        raise ValueError(f"Suite '{name}' has no tasks.")
+    return suite
+
+
+def _select_task_ids(total_tasks: int, task_ids: Iterable[int] | None) -> list[int]:
+    """Validate/normalize task ids. If None → all tasks."""
+    if task_ids is None:
+        return list(range(total_tasks))
+    ids = sorted({int(t) for t in task_ids})
+    for t in ids:
+        if t < 0 or t >= total_tasks:
+            raise ValueError(f"task_id {t} out of range [0, {total_tasks - 1}].")
+    return ids
+
+
+def get_task_init_states(task_suite: Any, i: int) -> np.ndarray:
+    init_states_path = (
+        Path(get_libero_path("init_states"))
+        / task_suite.tasks[i].problem_folder
+        / task_suite.tasks[i].init_states_file
+    )
+    init_states = torch.load(init_states_path, weights_only=False)  # nosec B614
+    return init_states
+
+
+def get_libero_dummy_action():
+    """Get dummy/no-op action, used to roll out the simulation while the robot does nothing."""
+    return [0, 0, 0, 0, 0, 0, -1]
+
+
+ACTION_DIM = 7
+ACTION_LOW = -1.0
+ACTION_HIGH = 1.0
+TASK_SUITE_MAX_STEPS: dict[str, int] = {
+    "libero_spatial": 280,  # longest training demo has 193 steps
+    "libero_object": 280,  # longest training demo has 254 steps
+    "libero_goal": 300,  # longest training demo has 270 steps
+    "libero_10": 520,  # longest training demo has 505 steps
+    "libero_90": 400,  # longest training demo has 373 steps
+}
+
+
+class LiberoEnv(gym.Env):
+    metadata = {"render_modes": ["rgb_array"], "render_fps": 80}
+
+    def __init__(
+        self,
+        task_suite: Any,
+        task_id: int,
+        task_suite_name: str,
+        episode_length: int | None = None,
+        camera_name: str | Sequence[str] = "agentview_image,robot0_eye_in_hand_image",
+        obs_type: str = "pixels",
+        render_mode: str = "rgb_array",
+        observation_width: int = 256,
+        observation_height: int = 256,
+        visualization_width: int = 640,
+        visualization_height: int = 480,
+        init_states: bool = True,
+        episode_index: int = 0,
+        n_envs: int = 1,
+        camera_name_mapping: dict[str, str] | None = None,
+        num_steps_wait: int = 10,
+        control_mode: str = "relative",
+    ):
+        super().__init__()
+        self.task_id = task_id
+        self.obs_type = obs_type
+        self.render_mode = render_mode
+        self.observation_width = observation_width
+        self.observation_height = observation_height
+        self.visualization_width = visualization_width
+        self.visualization_height = visualization_height
+        self.init_states = init_states
+        self.camera_name = _parse_camera_names(
+            camera_name
+        )  # agentview_image (main) or robot0_eye_in_hand_image (wrist)
+
+        # Map raw camera names to "image1" and "image2".
+        # The preprocessing step `preprocess_observation` will then prefix these with `.images.*`,
+        # following the LeRobot convention (e.g., `observation.images.image`, `observation.images.image2`).
+        # This ensures the policy consistently receives observations in the
+        # expected format regardless of the original camera naming.
+        if camera_name_mapping is None:
+            camera_name_mapping = {
+                "agentview_image": "image",
+                "robot0_eye_in_hand_image": "image2",
+            }
+        self.camera_name_mapping = camera_name_mapping
+        self.num_steps_wait = num_steps_wait
+        self.episode_index = episode_index
+        self.episode_length = episode_length
+        # Load once and keep
+        self._init_states = get_task_init_states(task_suite, self.task_id) if self.init_states else None
+        self._reset_stride = n_envs  # when performing a reset, append `_reset_stride` to `init_state_id`.
+
+        self.init_state_id = self.episode_index  # tie each sub-env to a fixed init state
+
+        self._env = self._make_envs_task(task_suite, self.task_id)
+        default_steps = 500
+        self._max_episode_steps = (
+            TASK_SUITE_MAX_STEPS.get(task_suite_name, default_steps)
+            if self.episode_length is None
+            else self.episode_length
+        )
+        self.control_mode = control_mode
+        images = {}
+        for cam in self.camera_name:
+            images[self.camera_name_mapping[cam]] = spaces.Box(
+                low=0,
+                high=255,
+                shape=(self.observation_height, self.observation_width, 3),
+                dtype=np.uint8,
+            )
+
+        if self.obs_type == "state":
+            raise NotImplementedError(
+                "The 'state' observation type is not supported in LiberoEnv. "
+                "Please switch to an image-based obs_type (e.g. 'pixels', 'pixels_agent_pos')."
+            )
+
+        elif self.obs_type == "pixels":
+            self.observation_space = spaces.Dict(
+                {
+                    "pixels": spaces.Dict(images),
+                }
+            )
+        elif self.obs_type == "pixels_agent_pos":
+            self.observation_space = spaces.Dict(
+                {
+                    "pixels": spaces.Dict(images),
+                    "robot_state": spaces.Dict(
+                        {
+                            "eef": spaces.Dict(
+                                {
+                                    "pos": spaces.Box(low=-np.inf, high=np.inf, shape=(3,), dtype=np.float64),
+                                    "quat": spaces.Box(
+                                        low=-np.inf, high=np.inf, shape=(4,), dtype=np.float64
+                                    ),
+                                    "mat": spaces.Box(
+                                        low=-np.inf, high=np.inf, shape=(3, 3), dtype=np.float64
+                                    ),
+                                }
+                            ),
+                            "gripper": spaces.Dict(
+                                {
+                                    "qpos": spaces.Box(
+                                        low=-np.inf, high=np.inf, shape=(2,), dtype=np.float64
+                                    ),
+                                    "qvel": spaces.Box(
+                                        low=-np.inf, high=np.inf, shape=(2,), dtype=np.float64
+                                    ),
+                                }
+                            ),
+                            "joints": spaces.Dict(
+                                {
+                                    "pos": spaces.Box(low=-np.inf, high=np.inf, shape=(7,), dtype=np.float64),
+                                    "vel": spaces.Box(low=-np.inf, high=np.inf, shape=(7,), dtype=np.float64),
+                                }
+                            ),
+                        }
+                    ),
+                }
+            )
+
+        self.action_space = spaces.Box(
+            low=ACTION_LOW, high=ACTION_HIGH, shape=(ACTION_DIM,), dtype=np.float32
+        )
+
+    def render(self):
+        raw_obs = self._env.env._get_observations()
+        image = self._format_raw_obs(raw_obs)["pixels"]["image"]
+        image = image[::-1, ::-1]  # flip both H and W for visualization
+        return image
+
+    def _make_envs_task(self, task_suite: Any, task_id: int = 0):
+        task = task_suite.get_task(task_id)
+        self.task = task.name
+        self.task_description = task.language
+        task_bddl_file = os.path.join(get_libero_path("bddl_files"), task.problem_folder, task.bddl_file)
+
+        env_args = {
+            "bddl_file_name": task_bddl_file,
+            "camera_heights": self.observation_height,
+            "camera_widths": self.observation_width,
+        }
+        env = OffScreenRenderEnv(**env_args)
+        env.reset()
+        return env
+
+    def _format_raw_obs(self, raw_obs: RobotObservation) -> RobotObservation:
+        images = {}
+        for camera_name in self.camera_name:
+            image = raw_obs[camera_name]
+            images[self.camera_name_mapping[camera_name]] = image
+
+        eef_pos = raw_obs.get("robot0_eef_pos")
+        eef_quat = raw_obs.get("robot0_eef_quat")
+
+        # rotation matrix from controller
+        eef_mat = self._env.robots[0].controller.ee_ori_mat if eef_pos is not None else None
+        gripper_qpos = raw_obs.get("robot0_gripper_qpos")
+        gripper_qvel = raw_obs.get("robot0_gripper_qvel")
+        joint_pos = raw_obs.get("robot0_joint_pos")
+        joint_vel = raw_obs.get("robot0_joint_vel")
+        obs = {
+            "pixels": images,
+            "robot_state": {
+                "eef": {
+                    "pos": eef_pos,  # (3,)
+                    "quat": eef_quat,  # (4,)
+                    "mat": eef_mat,  # (3, 3)
+                },
+                "gripper": {
+                    "qpos": gripper_qpos,  # (2,)
+                    "qvel": gripper_qvel,  # (2,)
+                },
+                "joints": {
+                    "pos": joint_pos,  # (7,)
+                    "vel": joint_vel,  # (7,)
+                },
+            },
+        }
+        if self.obs_type == "pixels":
+            return {"pixels": images.copy()}
+
+        if self.obs_type == "pixels_agent_pos":
+            # Validate required fields are present
+            if eef_pos is None or eef_quat is None or gripper_qpos is None:
+                raise ValueError(
+                    f"Missing required robot state fields in raw observation. "
+                    f"Got eef_pos={eef_pos is not None}, eef_quat={eef_quat is not None}, "
+                    f"gripper_qpos={gripper_qpos is not None}"
+                )
+            return obs
+
+        raise NotImplementedError(
+            f"The observation type '{self.obs_type}' is not supported in LiberoEnv. "
+            "Please switch to an image-based obs_type (e.g. 'pixels', 'pixels_agent_pos')."
+        )
+
+    def reset(self, seed=None, **kwargs):
+        super().reset(seed=seed)
+        self._env.seed(seed)
+        raw_obs = self._env.reset()
+        if self.init_states and self._init_states is not None:
+            raw_obs = self._env.set_init_state(self._init_states[self.init_state_id % len(self._init_states)])
+            self.init_state_id += self._reset_stride  # Change init_state_id when reset
+
+        # After reset, objects may be unstable (slightly floating, intersecting, etc.).
+        # Step the simulator with a no-op action for a few frames so everything settles.
+        # Increasing this value can improve determinism and reproducibility across resets.
+        for _ in range(self.num_steps_wait):
+            raw_obs, _, _, _ = self._env.step(get_libero_dummy_action())
+
+        if self.control_mode == "absolute":
+            for robot in self._env.robots:
+                robot.controller.use_delta = False
+        elif self.control_mode == "relative":
+            for robot in self._env.robots:
+                robot.controller.use_delta = True
+        else:
+            raise ValueError(f"Invalid control mode: {self.control_mode}")
+        observation = self._format_raw_obs(raw_obs)
+        info = {"is_success": False}
+        return observation, info
+
+    def step(self, action: np.ndarray) -> tuple[RobotObservation, float, bool, bool, dict[str, Any]]:
+        if action.ndim != 1:
+            raise ValueError(
+                f"Expected action to be 1-D (shape (action_dim,)), "
+                f"but got shape {action.shape} with ndim={action.ndim}"
+            )
+        raw_obs, reward, done, info = self._env.step(action)
+
+        is_success = self._env.check_success()
+        terminated = done or is_success
+        info.update(
+            {
+                "task": self.task,
+                "task_id": self.task_id,
+                "done": done,
+                "is_success": is_success,
+            }
+        )
+        observation = self._format_raw_obs(raw_obs)
+        if terminated:
+            info["final_info"] = {
+                "task": self.task,
+                "task_id": self.task_id,
+                "done": bool(done),
+                "is_success": bool(is_success),
+            }
+            self.reset()
+        truncated = False
+        return observation, reward, terminated, truncated, info
+
+    def close(self):
+        self._env.close()
+
+
+def _make_env_fns(
+    *,
+    suite,
+    suite_name: str,
+    task_id: int,
+    n_envs: int,
+    camera_names: list[str],
+    episode_length: int | None,
+    init_states: bool,
+    gym_kwargs: Mapping[str, Any],
+    control_mode: str,
+) -> list[Callable[[], LiberoEnv]]:
+    """Build n_envs factory callables for a single (suite, task_id)."""
+
+    def _make_env(episode_index: int, **kwargs) -> LiberoEnv:
+        local_kwargs = dict(kwargs)
+        return LiberoEnv(
+            task_suite=suite,
+            task_id=task_id,
+            task_suite_name=suite_name,
+            camera_name=camera_names,
+            init_states=init_states,
+            episode_length=episode_length,
+            episode_index=episode_index,
+            n_envs=n_envs,
+            control_mode=control_mode,
+            **local_kwargs,
+        )
+
+    fns: list[Callable[[], LiberoEnv]] = []
+    for episode_index in range(n_envs):
+        fns.append(partial(_make_env, episode_index, **gym_kwargs))
+    return fns
+
+
+# ---- Main API ----------------------------------------------------------------
+
+
+def create_libero_envs(
+    task: str,
+    n_envs: int,
+    gym_kwargs: dict[str, Any] | None = None,
+    camera_name: str | Sequence[str] = "agentview_image,robot0_eye_in_hand_image",
+    init_states: bool = True,
+    env_cls: Callable[[Sequence[Callable[[], Any]]], Any] | None = None,
+    control_mode: str = "relative",
+    episode_length: int | None = None,
+) -> dict[str, dict[int, Any]]:
+    """
+    Create vectorized LIBERO environments with a consistent return shape.
+
+    Returns:
+        dict[suite_name][task_id] -> vec_env (env_cls([...]) with exactly n_envs factories)
+    Notes:
+        - n_envs is the number of rollouts *per task* (episode_index = 0..n_envs-1).
+        - `task` can be a single suite or a comma-separated list of suites.
+        - You may pass `task_ids` (list[int]) inside `gym_kwargs` to restrict tasks per suite.
+    """
+    if env_cls is None or not callable(env_cls):
+        raise ValueError("env_cls must be a callable that wraps a list of environment factory callables.")
+    if not isinstance(n_envs, int) or n_envs <= 0:
+        raise ValueError(f"n_envs must be a positive int; got {n_envs}.")
+
+    gym_kwargs = dict(gym_kwargs or {})
+    task_ids_filter = gym_kwargs.pop("task_ids", None)  # optional: limit to specific tasks
+
+    camera_names = _parse_camera_names(camera_name)
+    suite_names = [s.strip() for s in str(task).split(",") if s.strip()]
+    if not suite_names:
+        raise ValueError("`task` must contain at least one LIBERO suite name.")
+
+    print(
+        f"Creating LIBERO envs | suites={suite_names} | n_envs(per task)={n_envs} | init_states={init_states}"
+    )
+    if task_ids_filter is not None:
+        print(f"Restricting to task_ids={task_ids_filter}")
+
+    out: dict[str, dict[int, Any]] = defaultdict(dict)
+    for suite_name in suite_names:
+        suite = _get_suite(suite_name)
+        total = len(suite.tasks)
+        selected = _select_task_ids(total, task_ids_filter)
+        if not selected:
+            raise ValueError(f"No tasks selected for suite '{suite_name}' (available: {total}).")
+
+        for tid in selected:
+            fns = _make_env_fns(
+                suite=suite,
+                episode_length=episode_length,
+                suite_name=suite_name,
+                task_id=tid,
+                n_envs=n_envs,
+                camera_names=camera_names,
+                init_states=init_states,
+                gym_kwargs=gym_kwargs,
+                control_mode=control_mode,
+            )
+            out[suite_name][tid] = env_cls(fns)
+            print(f"Built vec env | suite={suite_name} | task_id={tid} | n_envs={n_envs}")
+
+    # return plain dicts for predictability
+    return {suite: dict(task_map) for suite, task_map in out.items()}
diff --git a/lerobot/src/lerobot/envs/metaworld.py b/lerobot/src/lerobot/envs/metaworld.py
new file mode 100644
index 0000000000000000000000000000000000000000..e9e29f304953f8e6c679b296f1e61d7841027135
--- /dev/null
+++ b/lerobot/src/lerobot/envs/metaworld.py
@@ -0,0 +1,315 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import json
+from collections import defaultdict
+from collections.abc import Callable, Sequence
+from pathlib import Path
+from typing import Any
+
+import gymnasium as gym
+import metaworld
+import metaworld.policies as policies
+import numpy as np
+from gymnasium import spaces
+
+from lerobot.types import RobotObservation
+
+# ---- Load configuration data from the external JSON file ----
+CONFIG_PATH = Path(__file__).parent / "metaworld_config.json"
+try:
+    with open(CONFIG_PATH) as f:
+        data = json.load(f)
+except FileNotFoundError as err:
+    raise FileNotFoundError(
+        "Could not find 'metaworld_config.json'. "
+        "Please ensure the configuration file is in the same directory as the script."
+    ) from err
+except json.JSONDecodeError as err:
+    raise ValueError(
+        "Failed to decode 'metaworld_config.json'. Please ensure it is a valid JSON file."
+    ) from err
+
+# ---- Process the loaded data ----
+
+# extract and type-check top-level dicts
+task_descriptions_obj = data.get("TASK_DESCRIPTIONS")
+if not isinstance(task_descriptions_obj, dict):
+    raise TypeError("Expected TASK_DESCRIPTIONS to be a dict[str, str]")
+TASK_DESCRIPTIONS: dict[str, str] = task_descriptions_obj
+
+task_name_to_id_obj = data.get("TASK_NAME_TO_ID")
+if not isinstance(task_name_to_id_obj, dict):
+    raise TypeError("Expected TASK_NAME_TO_ID to be a dict[str, int]")
+TASK_NAME_TO_ID: dict[str, int] = task_name_to_id_obj
+
+# difficulty -> tasks mapping
+difficulty_to_tasks = data.get("DIFFICULTY_TO_TASKS")
+if not isinstance(difficulty_to_tasks, dict):
+    raise TypeError("Expected 'DIFFICULTY_TO_TASKS' to be a dict[str, list[str]]")
+DIFFICULTY_TO_TASKS: dict[str, list[str]] = difficulty_to_tasks
+
+# convert policy strings -> actual policy classes
+task_policy_mapping = data.get("TASK_POLICY_MAPPING")
+if not isinstance(task_policy_mapping, dict):
+    raise TypeError("Expected 'TASK_POLICY_MAPPING' to be a dict[str, str]")
+TASK_POLICY_MAPPING: dict[str, Any] = {
+    task_name: getattr(policies, policy_class_name)
+    for task_name, policy_class_name in task_policy_mapping.items()
+}
+ACTION_DIM = 4
+OBS_DIM = 4
+
+
+class MetaworldEnv(gym.Env):
+    metadata = {"render_modes": ["rgb_array"], "render_fps": 80}
+
+    def __init__(
+        self,
+        task,
+        camera_name="corner2",
+        obs_type="pixels",
+        render_mode="rgb_array",
+        observation_width=480,
+        observation_height=480,
+        visualization_width=640,
+        visualization_height=480,
+    ):
+        super().__init__()
+        self.task = task.replace("metaworld-", "")
+        self.obs_type = obs_type
+        self.render_mode = render_mode
+        self.observation_width = observation_width
+        self.observation_height = observation_height
+        self.visualization_width = visualization_width
+        self.visualization_height = visualization_height
+        self.camera_name = camera_name
+
+        self._env = self._make_envs_task(self.task)
+        self._max_episode_steps = self._env.max_path_length
+        self.task_description = TASK_DESCRIPTIONS[self.task]
+
+        self.expert_policy = TASK_POLICY_MAPPING[self.task]()
+
+        if self.obs_type == "state":
+            raise NotImplementedError()
+        elif self.obs_type == "pixels":
+            self.observation_space = spaces.Dict(
+                {
+                    "pixels": spaces.Box(
+                        low=0,
+                        high=255,
+                        shape=(self.observation_height, self.observation_width, 3),
+                        dtype=np.uint8,
+                    )
+                }
+            )
+        elif self.obs_type == "pixels_agent_pos":
+            self.observation_space = spaces.Dict(
+                {
+                    "pixels": spaces.Box(
+                        low=0,
+                        high=255,
+                        shape=(self.observation_height, self.observation_width, 3),
+                        dtype=np.uint8,
+                    ),
+                    "agent_pos": spaces.Box(
+                        low=-1000.0,
+                        high=1000.0,
+                        shape=(OBS_DIM,),
+                        dtype=np.float64,
+                    ),
+                }
+            )
+
+        self.action_space = spaces.Box(low=-1, high=1, shape=(ACTION_DIM,), dtype=np.float32)
+
+    def render(self) -> np.ndarray:
+        """
+        Render the current environment frame.
+
+        Returns:
+            np.ndarray: The rendered RGB image from the environment.
+        """
+        image = self._env.render()
+        if self.camera_name == "corner2":
+            # Images from this camera are flipped — correct them
+            image = np.flip(image, (0, 1))
+        return image
+
+    def _make_envs_task(self, env_name: str):
+        mt1 = metaworld.MT1(env_name, seed=42)
+        env = mt1.train_classes[env_name](render_mode="rgb_array", camera_name=self.camera_name)
+        env.set_task(mt1.train_tasks[0])
+        if self.camera_name == "corner2":
+            env.model.cam_pos[2] = [
+                0.75,
+                0.075,
+                0.7,
+            ]  # corner2 position, similar to https://arxiv.org/pdf/2206.14244
+        env.reset()
+        env._freeze_rand_vec = False  # otherwise no randomization
+        return env
+
+    def _format_raw_obs(self, raw_obs: np.ndarray) -> RobotObservation:
+        image = None
+        if self._env is not None:
+            image = self._env.render()
+            if self.camera_name == "corner2":
+                # NOTE: The "corner2" camera in MetaWorld environments outputs images with both axes inverted.
+                image = np.flip(image, (0, 1))
+        agent_pos = raw_obs[:4]
+        if self.obs_type == "state":
+            raise NotImplementedError(
+                "'state' obs_type not implemented for MetaWorld. Use pixel modes instead."
+            )
+
+        elif self.obs_type in ("pixels", "pixels_agent_pos"):
+            assert image is not None, (
+                "Expected `image` to be rendered before constructing pixel-based observations. "
+                "This likely means `env.render()` returned None or the environment was not provided."
+            )
+
+            if self.obs_type == "pixels":
+                obs = {"pixels": image.copy()}
+
+            else:  # pixels_agent_pos
+                obs = {
+                    "pixels": image.copy(),
+                    "agent_pos": agent_pos,
+                }
+        else:
+            raise ValueError(f"Unknown obs_type: {self.obs_type}")
+        return obs
+
+    def reset(
+        self,
+        seed: int | None = None,
+        **kwargs,
+    ) -> tuple[RobotObservation, dict[str, Any]]:
+        """
+        Reset the environment to its initial state.
+
+        Args:
+            seed (Optional[int]): Random seed for environment initialization.
+
+        Returns:
+            observation (RobotObservation): The initial formatted observation.
+            info (Dict[str, Any]): Additional info about the reset state.
+        """
+        super().reset(seed=seed)
+
+        raw_obs, info = self._env.reset(seed=seed)
+
+        observation = self._format_raw_obs(raw_obs)
+
+        info = {"is_success": False}
+        return observation, info
+
+    def step(self, action: np.ndarray) -> tuple[RobotObservation, float, bool, bool, dict[str, Any]]:
+        """
+        Perform one environment step.
+
+        Args:
+            action (np.ndarray): The action to execute, must be 1-D with shape (action_dim,).
+
+        Returns:
+            observation (RobotObservation): The formatted observation after the step.
+            reward (float): The scalar reward for this step.
+            terminated (bool): Whether the episode terminated successfully.
+            truncated (bool): Whether the episode was truncated due to a time limit.
+            info (Dict[str, Any]): Additional environment info.
+        """
+        if action.ndim != 1:
+            raise ValueError(
+                f"Expected action to be 1-D (shape (action_dim,)), "
+                f"but got shape {action.shape} with ndim={action.ndim}"
+            )
+        raw_obs, reward, done, truncated, info = self._env.step(action)
+
+        # Determine whether the task was successful
+        is_success = bool(info.get("success", 0))
+        terminated = done or is_success
+        info.update(
+            {
+                "task": self.task,
+                "done": done,
+                "is_success": is_success,
+            }
+        )
+
+        # Format the raw observation into the expected structure
+        observation = self._format_raw_obs(raw_obs)
+        if terminated:
+            info["final_info"] = {
+                "task": self.task,
+                "done": bool(done),
+                "is_success": bool(is_success),
+            }
+            self.reset()
+
+        return observation, reward, terminated, truncated, info
+
+    def close(self):
+        self._env.close()
+
+
+# ---- Main API ----------------------------------------------------------------
+
+
+def create_metaworld_envs(
+    task: str,
+    n_envs: int,
+    gym_kwargs: dict[str, Any] | None = None,
+    env_cls: Callable[[Sequence[Callable[[], Any]]], Any] | None = None,
+) -> dict[str, dict[int, Any]]:
+    """
+    Create vectorized Meta-World environments with a consistent return shape.
+
+    Returns:
+        dict[task_group][task_id] -> vec_env (env_cls([...]) with exactly n_envs factories)
+    Notes:
+        - n_envs is the number of rollouts *per task* (episode_index = 0..n_envs-1).
+        - `task` can be a single difficulty group (e.g., "easy", "medium", "hard") or a comma-separated list.
+        - If a task name is not in DIFFICULTY_TO_TASKS, we treat it as a single custom task.
+    """
+    if env_cls is None or not callable(env_cls):
+        raise ValueError("env_cls must be a callable that wraps a list of environment factory callables.")
+    if not isinstance(n_envs, int) or n_envs <= 0:
+        raise ValueError(f"n_envs must be a positive int; got {n_envs}.")
+
+    gym_kwargs = dict(gym_kwargs or {})
+    task_groups = [t.strip() for t in task.split(",") if t.strip()]
+    if not task_groups:
+        raise ValueError("`task` must contain at least one Meta-World task or difficulty group.")
+
+    print(f"Creating Meta-World envs | task_groups={task_groups} | n_envs(per task)={n_envs}")
+
+    out: dict[str, dict[int, Any]] = defaultdict(dict)
+
+    for group in task_groups:
+        # if not in difficulty presets, treat it as a single custom task
+        tasks = DIFFICULTY_TO_TASKS.get(group, [group])
+
+        for tid, task_name in enumerate(tasks):
+            print(f"Building vec env | group={group} | task_id={tid} | task={task_name}")
+
+            # build n_envs factories
+            fns = [(lambda tn=task_name: MetaworldEnv(task=tn, **gym_kwargs)) for _ in range(n_envs)]
+
+            out[group][tid] = env_cls(fns)
+
+    # return a plain dict for consistency
+    return {group: dict(task_map) for group, task_map in out.items()}
diff --git a/lerobot/src/lerobot/envs/metaworld_config.json b/lerobot/src/lerobot/envs/metaworld_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..41a417fef8b1429b4bed1fb382e3bd2408962400
--- /dev/null
+++ b/lerobot/src/lerobot/envs/metaworld_config.json
@@ -0,0 +1,121 @@
+{
+  "TASK_DESCRIPTIONS": {
+    "assembly-v3": "Pick up a nut and place it onto a peg",
+    "basketball-v3": "Dunk the basketball into the basket",
+    "bin-picking-v3": "Grasp the puck from one bin and place it into another bin",
+    "box-close-v3": "Grasp the cover and close the box with it",
+    "button-press-topdown-v3": "Press a button from the top",
+    "button-press-topdown-wall-v3": "Bypass a wall and press a button from the top",
+    "button-press-v3": "Press a button",
+    "button-press-wall-v3": "Bypass a wall and press a button",
+    "coffee-button-v3": "Push a button on the coffee machine",
+    "coffee-pull-v3": "Pull a mug from a coffee machine",
+    "coffee-push-v3": "Push a mug under a coffee machine",
+    "dial-turn-v3": "Rotate a dial 180 degrees",
+    "disassemble-v3": "Pick a nut out of a peg",
+    "door-close-v3": "Close a door with a revolving joint",
+    "door-lock-v3": "Lock the door by rotating the lock clockwise",
+    "door-open-v3": "Open a door with a revolving joint",
+    "door-unlock-v3": "Unlock the door by rotating the lock counter-clockwise",
+    "hand-insert-v3": "Insert the gripper into a hole",
+    "drawer-close-v3": "Push and close a drawer",
+    "drawer-open-v3": "Open a drawer",
+    "faucet-open-v3": "Rotate the faucet counter-clockwise",
+    "faucet-close-v3": "Rotate the faucet clockwise",
+    "hammer-v3": "Hammer a screw on the wall",
+    "handle-press-side-v3": "Press a handle down sideways",
+    "handle-press-v3": "Press a handle down",
+    "handle-pull-side-v3": "Pull a handle up sideways",
+    "handle-pull-v3": "Pull a handle up",
+    "lever-pull-v3": "Pull a lever down 90 degrees",
+    "peg-insert-side-v3": "Insert a peg sideways",
+    "pick-place-wall-v3": "Pick a puck, bypass a wall and place the puck",
+    "pick-out-of-hole-v3": "Pick up a puck from a hole",
+    "reach-v3": "Reach a goal position",
+    "push-back-v3": "Push the puck to a goal",
+    "push-v3": "Push the puck to a goal",
+    "pick-place-v3": "Pick and place a puck to a goal",
+    "plate-slide-v3": "Slide a plate into a cabinet",
+    "plate-slide-side-v3": "Slide a plate into a cabinet sideways",
+    "plate-slide-back-v3": "Get a plate from the cabinet",
+    "plate-slide-back-side-v3": "Get a plate from the cabinet sideways",
+    "peg-unplug-side-v3": "Unplug a peg sideways",
+    "soccer-v3": "Kick a soccer into the goal",
+    "stick-push-v3": "Grasp a stick and push a box using the stick",
+    "stick-pull-v3": "Grasp a stick and pull a box with the stick",
+    "push-wall-v3": "Bypass a wall and push a puck to a goal",
+    "reach-wall-v3": "Bypass a wall and reach a goal",
+    "shelf-place-v3": "Pick and place a puck onto a shelf",
+    "sweep-into-v3": "Sweep a puck into a hole",
+    "sweep-v3": "Sweep a puck off the table",
+    "window-open-v3": "Push and open a window",
+    "window-close-v3": "Push and close a window"
+  },
+  "TASK_NAME_TO_ID": {
+    "assembly-v3": 0, "basketball-v3": 1, "bin-picking-v3": 2, "box-close-v3": 3,
+    "button-press-topdown-v3": 4, "button-press-topdown-wall-v3": 5, "button-press-v3": 6,
+    "button-press-wall-v3": 7, "coffee-button-v3": 8, "coffee-pull-v3": 9, "coffee-push-v3": 10,
+    "dial-turn-v3": 11, "disassemble-v3": 12, "door-close-v3": 13, "door-lock-v3": 14,
+    "door-open-v3": 15, "door-unlock-v3": 16, "drawer-close-v3": 17, "drawer-open-v3": 18,
+    "faucet-close-v3": 19, "faucet-open-v3": 20, "hammer-v3": 21, "hand-insert-v3": 22,
+    "handle-press-side-v3": 23, "handle-press-v3": 24, "handle-pull-side-v3": 25,
+    "handle-pull-v3": 26, "lever-pull-v3": 27, "peg-insert-side-v3": 28, "peg-unplug-side-v3": 29,
+    "pick-out-of-hole-v3": 30, "pick-place-v3": 31, "pick-place-wall-v3": 32,
+    "plate-slide-back-side-v3": 33, "plate-slide-back-v3": 34, "plate-slide-side-v3": 35,
+    "plate-slide-v3": 36, "push-back-v3": 37, "push-v3": 38, "push-wall-v3": 39, "reach-v3": 40,
+    "reach-wall-v3": 41, "shelf-place-v3": 42, "soccer-v3": 43, "stick-pull-v3": 44,
+    "stick-push-v3": 45, "sweep-into-v3": 46, "sweep-v3": 47, "window-open-v3": 48,
+    "window-close-v3": 49
+  },
+  "DIFFICULTY_TO_TASKS": {
+    "easy": [
+      "button-press-v3", "button-press-topdown-v3", "button-press-topdown-wall-v3",
+      "button-press-wall-v3", "coffee-button-v3", "dial-turn-v3", "door-close-v3",
+      "door-lock-v3", "door-open-v3", "door-unlock-v3", "drawer-close-v3", "drawer-open-v3",
+      "faucet-close-v3", "faucet-open-v3", "handle-press-v3", "handle-press-side-v3",
+      "handle-pull-v3", "handle-pull-side-v3", "lever-pull-v3", "plate-slide-v3",
+      "plate-slide-back-v3", "plate-slide-back-side-v3", "plate-slide-side-v3", "reach-v3",
+      "reach-wall-v3", "window-close-v3", "window-open-v3", "peg-unplug-side-v3"
+    ],
+    "medium": [
+      "basketball-v3", "bin-picking-v3", "box-close-v3", "coffee-pull-v3", "coffee-push-v3",
+      "hammer-v3", "peg-insert-side-v3", "push-wall-v3", "soccer-v3", "sweep-v3", "sweep-into-v3"
+    ],
+    "hard": [
+      "assembly-v3", "hand-insert-v3", "pick-out-of-hole-v3", "pick-place-v3", "push-v3", "push-back-v3"
+    ],
+    "very_hard": [
+      "shelf-place-v3", "disassemble-v3", "stick-pull-v3", "stick-push-v3", "pick-place-wall-v3"
+    ]
+  },
+  "TASK_POLICY_MAPPING": {
+    "assembly-v3": "SawyerAssemblyV3Policy", "basketball-v3": "SawyerBasketballV3Policy",
+    "bin-picking-v3": "SawyerBinPickingV3Policy", "box-close-v3": "SawyerBoxCloseV3Policy",
+    "button-press-topdown-v3": "SawyerButtonPressTopdownV3Policy",
+    "button-press-topdown-wall-v3": "SawyerButtonPressTopdownWallV3Policy",
+    "button-press-v3": "SawyerButtonPressV3Policy", "button-press-wall-v3": "SawyerButtonPressWallV3Policy",
+    "coffee-button-v3": "SawyerCoffeeButtonV3Policy", "coffee-pull-v3": "SawyerCoffeePullV3Policy",
+    "coffee-push-v3": "SawyerCoffeePushV3Policy", "dial-turn-v3": "SawyerDialTurnV3Policy",
+    "disassemble-v3": "SawyerDisassembleV3Policy", "door-close-v3": "SawyerDoorCloseV3Policy",
+    "door-lock-v3": "SawyerDoorLockV3Policy", "door-open-v3": "SawyerDoorOpenV3Policy",
+    "door-unlock-v3": "SawyerDoorUnlockV3Policy", "drawer-close-v3": "SawyerDrawerCloseV3Policy",
+    "drawer-open-v3": "SawyerDrawerOpenV3Policy", "faucet-close-v3": "SawyerFaucetCloseV3Policy",
+    "faucet-open-v3": "SawyerFaucetOpenV3Policy", "hammer-v3": "SawyerHammerV3Policy",
+    "hand-insert-v3": "SawyerHandInsertV3Policy", "handle-press-side-v3": "SawyerHandlePressSideV3Policy",
+    "handle-press-v3": "SawyerHandlePressV3Policy", "handle-pull-side-v3": "SawyerHandlePullSideV3Policy",
+    "handle-pull-v3": "SawyerHandlePullV3Policy", "lever-pull-v3": "SawyerLeverPullV3Policy",
+    "peg-insert-side-v3": "SawyerPegInsertionSideV3Policy", "peg-unplug-side-v3": "SawyerPegUnplugSideV3Policy",
+    "pick-out-of-hole-v3": "SawyerPickOutOfHoleV3Policy", "pick-place-v3": "SawyerPickPlaceV3Policy",
+    "pick-place-wall-v3": "SawyerPickPlaceWallV3Policy",
+    "plate-slide-back-side-v3": "SawyerPlateSlideBackSideV3Policy",
+    "plate-slide-back-v3": "SawyerPlateSlideBackV3Policy",
+    "plate-slide-side-v3": "SawyerPlateSlideSideV3Policy", "plate-slide-v3": "SawyerPlateSlideV3Policy",
+    "push-back-v3": "SawyerPushBackV3Policy", "push-v3": "SawyerPushV3Policy",
+    "push-wall-v3": "SawyerPushWallV3Policy", "reach-v3": "SawyerReachV3Policy",
+    "reach-wall-v3": "SawyerReachWallV3Policy", "shelf-place-v3": "SawyerShelfPlaceV3Policy",
+    "soccer-v3": "SawyerSoccerV3Policy", "stick-pull-v3": "SawyerStickPullV3Policy",
+    "stick-push-v3": "SawyerStickPushV3Policy", "sweep-into-v3": "SawyerSweepIntoV3Policy",
+    "sweep-v3": "SawyerSweepV3Policy", "window-open-v3": "SawyerWindowOpenV3Policy",
+    "window-close-v3": "SawyerWindowCloseV3Policy"
+  }
+}
diff --git a/lerobot/src/lerobot/envs/utils.py b/lerobot/src/lerobot/envs/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..fd17a67621370d08a21bd0acb66b9d5569ce610d
--- /dev/null
+++ b/lerobot/src/lerobot/envs/utils.py
@@ -0,0 +1,356 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import importlib.util
+import os
+import warnings
+from collections.abc import Mapping, Sequence
+from functools import singledispatch
+from typing import Any
+
+import einops
+import gymnasium as gym
+import numpy as np
+import torch
+from huggingface_hub import hf_hub_download, snapshot_download
+from torch import Tensor
+
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.envs.configs import EnvConfig
+from lerobot.types import RobotObservation
+from lerobot.utils.constants import OBS_ENV_STATE, OBS_IMAGE, OBS_IMAGES, OBS_STATE, OBS_STR
+from lerobot.utils.utils import get_channel_first_image_shape
+
+
+def _convert_nested_dict(d):
+    result = {}
+    for k, v in d.items():
+        if isinstance(v, dict):
+            result[k] = _convert_nested_dict(v)
+        elif isinstance(v, np.ndarray):
+            result[k] = torch.from_numpy(v)
+        else:
+            result[k] = v
+    return result
+
+
+def preprocess_observation(observations: dict[str, np.ndarray]) -> dict[str, Tensor]:
+    # TODO(jadechoghari, imstevenpmwork): refactor this to use features from the environment (no hardcoding)
+    """Convert environment observation to LeRobot format observation.
+    Args:
+        observation: Dictionary of observation batches from a Gym vector environment.
+    Returns:
+        Dictionary of observation batches with keys renamed to LeRobot format and values as tensors.
+    """
+    # map to expected inputs for the policy
+    return_observations = {}
+    if "pixels" in observations:
+        if isinstance(observations["pixels"], dict):
+            imgs = {f"{OBS_IMAGES}.{key}": img for key, img in observations["pixels"].items()}
+        else:
+            imgs = {OBS_IMAGE: observations["pixels"]}
+
+        for imgkey, img in imgs.items():
+            # TODO(aliberts, rcadene): use transforms.ToTensor()?
+            img_tensor = torch.from_numpy(img)
+
+            # When preprocessing observations in a non-vectorized environment, we need to add a batch dimension.
+            # This is the case for human-in-the-loop RL where there is only one environment.
+            if img_tensor.ndim == 3:
+                img_tensor = img_tensor.unsqueeze(0)
+            # sanity check that images are channel last
+            _, h, w, c = img_tensor.shape
+            assert c < h and c < w, f"expect channel last images, but instead got {img_tensor.shape=}"
+
+            # sanity check that images are uint8
+            assert img_tensor.dtype == torch.uint8, f"expect torch.uint8, but instead {img_tensor.dtype=}"
+
+            # convert to channel first of type float32 in range [0,1]
+            img_tensor = einops.rearrange(img_tensor, "b h w c -> b c h w").contiguous()
+            img_tensor = img_tensor.type(torch.float32)
+            img_tensor /= 255
+
+            return_observations[imgkey] = img_tensor
+
+    if "environment_state" in observations:
+        env_state = torch.from_numpy(observations["environment_state"]).float()
+        if env_state.dim() == 1:
+            env_state = env_state.unsqueeze(0)
+
+        return_observations[OBS_ENV_STATE] = env_state
+
+    if "agent_pos" in observations:
+        agent_pos = torch.from_numpy(observations["agent_pos"]).float()
+        if agent_pos.dim() == 1:
+            agent_pos = agent_pos.unsqueeze(0)
+        return_observations[OBS_STATE] = agent_pos
+
+    if "robot_state" in observations:
+        return_observations[f"{OBS_STR}.robot_state"] = _convert_nested_dict(observations["robot_state"])
+
+    # Handle IsaacLab Arena format: observations have 'policy' and 'camera_obs' keys
+    if "policy" in observations:
+        return_observations[f"{OBS_STR}.policy"] = observations["policy"]
+
+    if "camera_obs" in observations:
+        return_observations[f"{OBS_STR}.camera_obs"] = observations["camera_obs"]
+
+    return return_observations
+
+
+def env_to_policy_features(env_cfg: EnvConfig) -> dict[str, PolicyFeature]:
+    # TODO(jadechoghari, imstevenpmwork): remove this hardcoding of keys and just use the nested keys as is
+    # (need to also refactor preprocess_observation and externalize normalization from policies)
+    policy_features = {}
+    for key, ft in env_cfg.features.items():
+        if ft.type is FeatureType.VISUAL:
+            if len(ft.shape) != 3:
+                raise ValueError(f"Number of dimensions of {key} != 3 (shape={ft.shape})")
+
+            shape = get_channel_first_image_shape(ft.shape)
+            feature = PolicyFeature(type=ft.type, shape=shape)
+        else:
+            feature = ft
+
+        policy_key = env_cfg.features_map[key]
+        policy_features[policy_key] = feature
+
+    return policy_features
+
+
+def are_all_envs_same_type(env: gym.vector.VectorEnv) -> bool:
+    first_type = type(env.envs[0])  # Get type of first env
+    return all(type(e) is first_type for e in env.envs)  # Fast type check
+
+
+def check_env_attributes_and_types(env: gym.vector.VectorEnv) -> None:
+    with warnings.catch_warnings():
+        warnings.simplefilter("once", UserWarning)  # Apply filter only in this function
+
+        if not (hasattr(env.envs[0], "task_description") and hasattr(env.envs[0], "task")):
+            warnings.warn(
+                "The environment does not have 'task_description' and 'task'. Some policies require these features.",
+                UserWarning,
+                stacklevel=2,
+            )
+        if not are_all_envs_same_type(env):
+            warnings.warn(
+                "The environments have different types. Make sure you infer the right task from each environment. Empty task will be passed instead.",
+                UserWarning,
+                stacklevel=2,
+            )
+
+
+def add_envs_task(env: gym.vector.VectorEnv, observation: RobotObservation) -> RobotObservation:
+    """Adds task feature to the observation dict with respect to the first environment attribute."""
+    if hasattr(env.envs[0], "task_description"):
+        task_result = env.call("task_description")
+
+        if isinstance(task_result, tuple):
+            task_result = list(task_result)
+
+        if not isinstance(task_result, list):
+            raise TypeError(f"Expected task_description to return a list, got {type(task_result)}")
+        if not all(isinstance(item, str) for item in task_result):
+            raise TypeError("All items in task_description result must be strings")
+
+        observation["task"] = task_result
+    elif hasattr(env.envs[0], "task"):
+        task_result = env.call("task")
+
+        if isinstance(task_result, tuple):
+            task_result = list(task_result)
+
+        if not isinstance(task_result, list):
+            raise TypeError(f"Expected task to return a list, got {type(task_result)}")
+        if not all(isinstance(item, str) for item in task_result):
+            raise TypeError("All items in task result must be strings")
+
+        observation["task"] = task_result
+    else:  #  For envs without language instructions, e.g. aloha transfer cube and etc.
+        num_envs = observation[list(observation.keys())[0]].shape[0]
+        observation["task"] = ["" for _ in range(num_envs)]
+    return observation
+
+
+def _close_single_env(env: Any) -> None:
+    try:
+        env.close()
+    except Exception as exc:
+        print(f"Exception while closing env {env}: {exc}")
+
+
+@singledispatch
+def close_envs(obj: Any) -> None:
+    """Default: raise if the type is not recognized."""
+    raise NotImplementedError(f"close_envs not implemented for type {type(obj).__name__}")
+
+
+@close_envs.register
+def _(env: Mapping) -> None:
+    for v in env.values():
+        if isinstance(v, Mapping):
+            close_envs(v)
+        elif hasattr(v, "close"):
+            _close_single_env(v)
+
+
+@close_envs.register
+def _(envs: Sequence) -> None:
+    if isinstance(envs, (str | bytes)):
+        return
+    for v in envs:
+        if isinstance(v, Mapping) or isinstance(v, Sequence) and not isinstance(v, (str | bytes)):
+            close_envs(v)
+        elif hasattr(v, "close"):
+            _close_single_env(v)
+
+
+@close_envs.register
+def _(env: gym.Env) -> None:
+    _close_single_env(env)
+
+
+# helper to safely load a python file as a module
+def _load_module_from_path(path: str, module_name: str | None = None):
+    module_name = module_name or f"hub_env_{os.path.basename(path).replace('.', '_')}"
+    spec = importlib.util.spec_from_file_location(module_name, path)
+    if spec is None:
+        raise ImportError(f"Could not load module spec for {module_name} from {path}")
+    module = importlib.util.module_from_spec(spec)
+    spec.loader.exec_module(module)  # type: ignore
+    return module
+
+
+# helper to parse hub string (supports "user/repo", "user/repo@rev", optional path)
+# examples:
+#   "user/repo" -> will look for env.py at repo root
+#   "user/repo@main:envs/my_env.py" -> explicit revision and path
+def _parse_hub_url(hub_uri: str):
+    # very small parser: [repo_id][@revision][:path]
+    # repo_id is required (user/repo or org/repo)
+    revision = None
+    file_path = "env.py"
+    if "@" in hub_uri:
+        repo_and_rev, *rest = hub_uri.split(":", 1)
+        repo_id, rev = repo_and_rev.split("@", 1)
+        revision = rev
+        if rest:
+            file_path = rest[0]
+    else:
+        repo_id, *rest = hub_uri.split(":", 1)
+        if rest:
+            file_path = rest[0]
+    return repo_id, revision, file_path
+
+
+def _download_hub_file(
+    cfg_str: str,
+    trust_remote_code: bool,
+    hub_cache_dir: str | None,
+) -> tuple[str, str, str, str]:
+    """
+    Parse `cfg_str` (hub URL), enforce `trust_remote_code`, and return
+    (repo_id, file_path, local_file, revision).
+    """
+    if not trust_remote_code:
+        raise RuntimeError(
+            f"Refusing to execute remote code from the Hub for '{cfg_str}'. "
+            "Executing hub env modules runs arbitrary Python code from third-party repositories. "
+            "If you trust this repo and understand the risks, call `make_env(..., trust_remote_code=True)` "
+            "and prefer pinning to a specific revision: 'user/repo@<commit-hash>:env.py'."
+        )
+
+    repo_id, revision, file_path = _parse_hub_url(cfg_str)
+
+    try:
+        local_file = hf_hub_download(
+            repo_id=repo_id, filename=file_path, revision=revision, cache_dir=hub_cache_dir
+        )
+    except Exception as e:
+        # fallback to snapshot download
+        snapshot_dir = snapshot_download(repo_id=repo_id, revision=revision, cache_dir=hub_cache_dir)
+        local_file = os.path.join(snapshot_dir, file_path)
+        if not os.path.exists(local_file):
+            raise FileNotFoundError(
+                f"Could not find {file_path} in repository {repo_id}@{revision or 'main'}"
+            ) from e
+
+    return repo_id, file_path, local_file, revision
+
+
+def _import_hub_module(local_file: str, repo_id: str) -> Any:
+    """
+    Import the downloaded file as a module and surface helpful import error messages.
+    """
+    module_name = f"hub_env_{repo_id.replace('/', '_')}"
+    try:
+        module = _load_module_from_path(local_file, module_name=module_name)
+    except ModuleNotFoundError as e:
+        missing = getattr(e, "name", None) or str(e)
+        raise ModuleNotFoundError(
+            f"Hub env '{repo_id}:{os.path.basename(local_file)}' failed to import because the dependency "
+            f"'{missing}' is not installed locally.\n\n"
+        ) from e
+    except ImportError as e:
+        raise ImportError(
+            f"Failed to load hub env module '{repo_id}:{os.path.basename(local_file)}'. Import error: {e}\n\n"
+        ) from e
+    return module
+
+
+def _call_make_env(module: Any, n_envs: int, use_async_envs: bool, cfg: EnvConfig | None) -> Any:
+    """
+    Ensure module exposes make_env and call it.
+    """
+    if not hasattr(module, "make_env"):
+        raise AttributeError(
+            f"The hub module {getattr(module, '__name__', 'hub_module')} must expose `make_env(n_envs=int, use_async_envs=bool)`."
+        )
+    entry_fn = module.make_env
+    # Only pass cfg if it's not None (i.e., when an EnvConfig was provided, not a string hub ID)
+    if cfg is not None:
+        return entry_fn(n_envs=n_envs, use_async_envs=use_async_envs, cfg=cfg)
+    else:
+        return entry_fn(n_envs=n_envs, use_async_envs=use_async_envs)
+
+
+def _normalize_hub_result(result: Any) -> dict[str, dict[int, gym.vector.VectorEnv]]:
+    """
+    Normalize possible return types from hub `make_env` into the mapping:
+      { suite_name: { task_id: vector_env } }
+    Accepts:
+      - dict (assumed already correct)
+      - gym.vector.VectorEnv
+      - gym.Env (will be wrapped into SyncVectorEnv)
+    """
+    if isinstance(result, dict):
+        return result
+
+    # VectorEnv: use its spec.id if available
+    if isinstance(result, gym.vector.VectorEnv):
+        suite_name = getattr(result, "spec", None) and getattr(result.spec, "id", None) or "hub_env"
+        return {suite_name: {0: result}}
+
+    # Single Env: wrap into SyncVectorEnv
+    if isinstance(result, gym.Env):
+        vec = gym.vector.SyncVectorEnv([lambda: result])
+        suite_name = getattr(result, "spec", None) and getattr(result.spec, "id", None) or "hub_env"
+        return {suite_name: {0: vec}}
+
+    raise ValueError(
+        "Hub `make_env` must return either a mapping {suite: {task_id: vec_env}}, "
+        "a gym.vector.VectorEnv, or a single gym.Env."
+    )
diff --git a/lerobot/src/lerobot/model/kinematics.py b/lerobot/src/lerobot/model/kinematics.py
new file mode 100644
index 0000000000000000000000000000000000000000..95d3b235c642c9dafa51009d96a5af014295b3f0
--- /dev/null
+++ b/lerobot/src/lerobot/model/kinematics.py
@@ -0,0 +1,132 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import numpy as np
+
+
+class RobotKinematics:
+    """Robot kinematics using placo library for forward and inverse kinematics."""
+
+    def __init__(
+        self,
+        urdf_path: str,
+        target_frame_name: str = "gripper_frame_link",
+        joint_names: list[str] | None = None,
+    ):
+        """
+        Initialize placo-based kinematics solver.
+
+        Args:
+            urdf_path (str): Path to the robot URDF file
+            target_frame_name (str): Name of the end-effector frame in the URDF
+            joint_names (list[str] | None): List of joint names to use for the kinematics solver
+        """
+        try:
+            import placo  # type: ignore[import-not-found] # C++ library with Python bindings, no type stubs available. TODO: Create stub file or request upstream typing support.
+        except ImportError as e:
+            raise ImportError(
+                "placo is required for RobotKinematics. "
+                "Please install the optional dependencies of `kinematics` in the package."
+            ) from e
+
+        self.robot = placo.RobotWrapper(urdf_path)
+        self.solver = placo.KinematicsSolver(self.robot)
+        self.solver.mask_fbase(True)  # Fix the base
+
+        self.target_frame_name = target_frame_name
+
+        # Set joint names
+        self.joint_names = list(self.robot.joint_names()) if joint_names is None else joint_names
+
+        # Initialize frame task for IK
+        self.tip_frame = self.solver.add_frame_task(self.target_frame_name, np.eye(4))
+
+    def forward_kinematics(self, joint_pos_deg: np.ndarray) -> np.ndarray:
+        """
+        Compute forward kinematics for given joint configuration given the target frame name in the constructor.
+
+        Args:
+            joint_pos_deg: Joint positions in degrees (numpy array)
+
+        Returns:
+            4x4 transformation matrix of the end-effector pose
+        """
+
+        # Convert degrees to radians
+        joint_pos_rad = np.deg2rad(joint_pos_deg[: len(self.joint_names)])
+
+        # Update joint positions in placo robot
+        for i, joint_name in enumerate(self.joint_names):
+            self.robot.set_joint(joint_name, joint_pos_rad[i])
+
+        # Update kinematics
+        self.robot.update_kinematics()
+
+        # Get the transformation matrix
+        return self.robot.get_T_world_frame(self.target_frame_name)
+
+    def inverse_kinematics(
+        self,
+        current_joint_pos: np.ndarray,
+        desired_ee_pose: np.ndarray,
+        position_weight: float = 1.0,
+        orientation_weight: float = 0.01,
+    ) -> np.ndarray:
+        """
+        Compute inverse kinematics using placo solver.
+
+        Args:
+            current_joint_pos: Current joint positions in degrees (used as initial guess)
+            desired_ee_pose: Target end-effector pose as a 4x4 transformation matrix
+            position_weight: Weight for position constraint in IK
+            orientation_weight: Weight for orientation constraint in IK, set to 0.0 to only constrain position
+
+        Returns:
+            Joint positions in degrees that achieve the desired end-effector pose
+        """
+
+        # Convert current joint positions to radians for initial guess
+        current_joint_rad = np.deg2rad(current_joint_pos[: len(self.joint_names)])
+
+        # Set current joint positions as initial guess
+        for i, joint_name in enumerate(self.joint_names):
+            self.robot.set_joint(joint_name, current_joint_rad[i])
+
+        # Update the target pose for the frame task
+        self.tip_frame.T_world_frame = desired_ee_pose
+
+        # Configure the task based on position_only flag
+        self.tip_frame.configure(self.target_frame_name, "soft", position_weight, orientation_weight)
+
+        # Solve IK
+        self.solver.solve(True)
+        self.robot.update_kinematics()
+
+        # Extract joint positions
+        joint_pos_rad = []
+        for joint_name in self.joint_names:
+            joint = self.robot.get_joint(joint_name)
+            joint_pos_rad.append(joint)
+
+        # Convert back to degrees
+        joint_pos_deg = np.rad2deg(joint_pos_rad)
+
+        # Preserve gripper position if present in current_joint_pos
+        if len(current_joint_pos) > len(self.joint_names):
+            result = np.zeros_like(current_joint_pos)
+            result[: len(self.joint_names)] = joint_pos_deg
+            result[len(self.joint_names) :] = current_joint_pos[len(self.joint_names) :]
+            return result
+        else:
+            return joint_pos_deg
diff --git a/lerobot/src/lerobot/motors/__init__.py b/lerobot/src/lerobot/motors/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..5df80d5ba3498e436339e627c9ff610bac1d0401
--- /dev/null
+++ b/lerobot/src/lerobot/motors/__init__.py
@@ -0,0 +1,21 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .motors_bus import (
+    Motor,
+    MotorCalibration,
+    MotorNormMode,
+)
diff --git a/lerobot/src/lerobot/motors/calibration_gui.py b/lerobot/src/lerobot/motors/calibration_gui.py
new file mode 100644
index 0000000000000000000000000000000000000000..3410cb28ad06cce2ee547c71345958ccf34ab6f6
--- /dev/null
+++ b/lerobot/src/lerobot/motors/calibration_gui.py
@@ -0,0 +1,403 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import math
+import os
+from dataclasses import dataclass
+
+os.environ["PYGAME_HIDE_SUPPORT_PROMPT"] = "1"
+
+from .motors_bus import MotorCalibration, MotorsBus
+
+BAR_LEN, BAR_THICKNESS = 450, 8
+HANDLE_R = 10
+BRACKET_W, BRACKET_H = 6, 14
+TRI_W, TRI_H = 12, 14
+
+BTN_W, BTN_H = 60, 22
+SAVE_W, SAVE_H = 80, 28
+LOAD_W = 80
+DD_W, DD_H = 160, 28
+
+TOP_GAP = 50
+PADDING_Y, TOP_OFFSET = 70, 60
+FONT_SIZE, FPS = 20, 60
+
+BG_COLOR = (30, 30, 30)
+BAR_RED, BAR_GREEN = (200, 60, 60), (60, 200, 60)
+HANDLE_COLOR, TEXT_COLOR = (240, 240, 240), (250, 250, 250)
+TICK_COLOR = (250, 220, 40)
+BTN_COLOR, BTN_COLOR_HL = (80, 80, 80), (110, 110, 110)
+DD_COLOR, DD_COLOR_HL = (70, 70, 70), (100, 100, 100)
+
+
+def dist(a, b):
+    return math.hypot(a[0] - b[0], a[1] - b[1])
+
+
+@dataclass
+class RangeValues:
+    min_v: int
+    pos_v: int
+    max_v: int
+
+
+class RangeSlider:
+    """One motor = one slider row"""
+
+    def __init__(self, motor, idx, res, calibration, present, label_pad, base_y):
+        import pygame
+
+        self.motor = motor
+        self.res = res
+        self.x0 = 40 + label_pad
+        self.x1 = self.x0 + BAR_LEN
+        self.y = base_y + idx * PADDING_Y
+
+        self.min_v = calibration.range_min
+        self.max_v = calibration.range_max
+        self.pos_v = max(self.min_v, min(present, self.max_v))
+
+        self.min_x = self._pos_from_val(self.min_v)
+        self.max_x = self._pos_from_val(self.max_v)
+        self.pos_x = self._pos_from_val(self.pos_v)
+
+        self.min_btn = pygame.Rect(self.x0 - BTN_W - 6, self.y - BTN_H // 2, BTN_W, BTN_H)
+        self.max_btn = pygame.Rect(self.x1 + 6, self.y - BTN_H // 2, BTN_W, BTN_H)
+
+        self.drag_min = self.drag_max = self.drag_pos = False
+        self.tick_val = present
+        self.font = pygame.font.Font(None, FONT_SIZE)
+
+    def _val_from_pos(self, x):
+        return round((x - self.x0) / BAR_LEN * self.res)
+
+    def _pos_from_val(self, v):
+        return self.x0 + (v / self.res) * BAR_LEN
+
+    def set_tick(self, v):
+        self.tick_val = max(0, min(v, self.res))
+
+    def _triangle_hit(self, pos):
+        import pygame
+
+        tri_top = self.y - BAR_THICKNESS // 2 - 2
+        return pygame.Rect(self.pos_x - TRI_W // 2, tri_top - TRI_H, TRI_W, TRI_H).collidepoint(pos)
+
+    def handle_event(self, e):
+        import pygame
+
+        if e.type == pygame.MOUSEBUTTONDOWN and e.button == 1:
+            if self.min_btn.collidepoint(e.pos):
+                self.min_x, self.min_v = self.pos_x, self.pos_v
+                return
+            if self.max_btn.collidepoint(e.pos):
+                self.max_x, self.max_v = self.pos_x, self.pos_v
+                return
+            if dist(e.pos, (self.min_x, self.y)) <= HANDLE_R:
+                self.drag_min = True
+            elif dist(e.pos, (self.max_x, self.y)) <= HANDLE_R:
+                self.drag_max = True
+            elif self._triangle_hit(e.pos):
+                self.drag_pos = True
+
+        elif e.type == pygame.MOUSEBUTTONUP and e.button == 1:
+            self.drag_min = self.drag_max = self.drag_pos = False
+
+        elif e.type == pygame.MOUSEMOTION:
+            x = e.pos[0]
+            if self.drag_min:
+                self.min_x = max(self.x0, min(x, self.pos_x))
+            elif self.drag_max:
+                self.max_x = min(self.x1, max(x, self.pos_x))
+            elif self.drag_pos:
+                self.pos_x = max(self.min_x, min(x, self.max_x))
+
+            self.min_v = self._val_from_pos(self.min_x)
+            self.max_v = self._val_from_pos(self.max_x)
+            self.pos_v = self._val_from_pos(self.pos_x)
+
+    def _draw_button(self, surf, rect, text):
+        import pygame
+
+        clr = BTN_COLOR_HL if rect.collidepoint(pygame.mouse.get_pos()) else BTN_COLOR
+        pygame.draw.rect(surf, clr, rect, border_radius=4)
+        t = self.font.render(text, True, TEXT_COLOR)
+        surf.blit(t, (rect.centerx - t.get_width() // 2, rect.centery - t.get_height() // 2))
+
+    def draw(self, surf):
+        import pygame
+
+        # motor name above set-min button (right-aligned)
+        name_surf = self.font.render(self.motor, True, TEXT_COLOR)
+        surf.blit(
+            name_surf,
+            (self.min_btn.right - name_surf.get_width(), self.min_btn.y - name_surf.get_height() - 4),
+        )
+
+        # bar + active section
+        pygame.draw.rect(surf, BAR_RED, (self.x0, self.y - BAR_THICKNESS // 2, BAR_LEN, BAR_THICKNESS))
+        pygame.draw.rect(
+            surf, BAR_GREEN, (self.min_x, self.y - BAR_THICKNESS // 2, self.max_x - self.min_x, BAR_THICKNESS)
+        )
+
+        # tick
+        tick_x = self._pos_from_val(self.tick_val)
+        pygame.draw.line(
+            surf,
+            TICK_COLOR,
+            (tick_x, self.y - BAR_THICKNESS // 2 - 4),
+            (tick_x, self.y + BAR_THICKNESS // 2 + 4),
+            2,
+        )
+
+        # brackets
+        for x, sign in ((self.min_x, +1), (self.max_x, -1)):
+            pygame.draw.line(
+                surf, HANDLE_COLOR, (x, self.y - BRACKET_H // 2), (x, self.y + BRACKET_H // 2), 2
+            )
+            pygame.draw.line(
+                surf,
+                HANDLE_COLOR,
+                (x, self.y - BRACKET_H // 2),
+                (x + sign * BRACKET_W, self.y - BRACKET_H // 2),
+                2,
+            )
+            pygame.draw.line(
+                surf,
+                HANDLE_COLOR,
+                (x, self.y + BRACKET_H // 2),
+                (x + sign * BRACKET_W, self.y + BRACKET_H // 2),
+                2,
+            )
+
+        # triangle ▼
+        tri_top = self.y - BAR_THICKNESS // 2 - 2
+        pygame.draw.polygon(
+            surf,
+            HANDLE_COLOR,
+            [
+                (self.pos_x, tri_top),
+                (self.pos_x - TRI_W // 2, tri_top - TRI_H),
+                (self.pos_x + TRI_W // 2, tri_top - TRI_H),
+            ],
+        )
+
+        # numeric labels
+        fh = self.font.get_height()
+        pos_y = tri_top - TRI_H - 4 - fh
+        txts = [
+            (self.min_v, self.min_x, self.y - BRACKET_H // 2 - 4 - fh),
+            (self.max_v, self.max_x, self.y - BRACKET_H // 2 - 4 - fh),
+            (self.pos_v, self.pos_x, pos_y),
+        ]
+        for v, x, y in txts:
+            s = self.font.render(str(v), True, TEXT_COLOR)
+            surf.blit(s, (x - s.get_width() // 2, y))
+
+        # buttons
+        self._draw_button(surf, self.min_btn, "set min")
+        self._draw_button(surf, self.max_btn, "set max")
+
+    # external
+    def values(self) -> RangeValues:
+        return RangeValues(self.min_v, self.pos_v, self.max_v)
+
+
+class RangeFinderGUI:
+    def __init__(self, bus: MotorsBus, groups: dict[str, list[str]] | None = None):
+        import pygame
+
+        self.bus = bus
+        self.groups = groups if groups is not None else {"all": list(bus.motors)}
+        self.group_names = list(self.groups)
+        self.current_group = self.group_names[0]
+
+        if not bus.is_connected:
+            bus.connect()
+
+        self.calibration = bus.read_calibration()
+        self.res_table = bus.model_resolution_table
+        self.present_cache = {
+            m: bus.read("Present_Position", m, normalize=False)
+            for motors in self.groups.values()
+            for m in motors
+        }
+
+        pygame.init()
+        self.font = pygame.font.Font(None, FONT_SIZE)
+
+        label_pad = max(self.font.size(m)[0] for ms in self.groups.values() for m in ms)
+        self.label_pad = label_pad
+        width = 40 + label_pad + BAR_LEN + 6 + BTN_W + 10 + SAVE_W + 10
+        self.controls_bottom = 10 + SAVE_H
+        self.base_y = self.controls_bottom + TOP_GAP
+        height = self.base_y + PADDING_Y * len(self.groups[self.current_group]) + 40
+
+        self.screen = pygame.display.set_mode((width, height))
+        pygame.display.set_caption("Motors range finder")
+
+        # ui rects
+        self.save_btn = pygame.Rect(width - SAVE_W - 10, 10, SAVE_W, SAVE_H)
+        self.load_btn = pygame.Rect(self.save_btn.left - LOAD_W - 10, 10, LOAD_W, SAVE_H)
+        self.dd_btn = pygame.Rect(width // 2 - DD_W // 2, 10, DD_W, DD_H)
+        self.dd_open = False  # dropdown expanded?
+
+        self.clock = pygame.time.Clock()
+        self._build_sliders()
+        self._adjust_height()
+
+    def _adjust_height(self):
+        import pygame
+
+        motors = self.groups[self.current_group]
+        new_h = self.base_y + PADDING_Y * len(motors) + 40
+        if new_h != self.screen.get_height():
+            w = self.screen.get_width()
+            self.screen = pygame.display.set_mode((w, new_h))
+
+    def _build_sliders(self):
+        self.sliders: list[RangeSlider] = []
+        motors = self.groups[self.current_group]
+        for i, m in enumerate(motors):
+            self.sliders.append(
+                RangeSlider(
+                    motor=m,
+                    idx=i,
+                    res=self.res_table[self.bus.motors[m].model] - 1,
+                    calibration=self.calibration[m],
+                    present=self.present_cache[m],
+                    label_pad=self.label_pad,
+                    base_y=self.base_y,
+                )
+            )
+
+    def _draw_dropdown(self):
+        import pygame
+
+        # collapsed box
+        hover = self.dd_btn.collidepoint(pygame.mouse.get_pos())
+        pygame.draw.rect(self.screen, DD_COLOR_HL if hover else DD_COLOR, self.dd_btn, border_radius=6)
+
+        txt = self.font.render(self.current_group, True, TEXT_COLOR)
+        self.screen.blit(
+            txt, (self.dd_btn.centerx - txt.get_width() // 2, self.dd_btn.centery - txt.get_height() // 2)
+        )
+
+        tri_w, tri_h = 12, 6
+        cx = self.dd_btn.right - 14
+        cy = self.dd_btn.centery + 1
+        pygame.draw.polygon(
+            self.screen,
+            TEXT_COLOR,
+            [(cx - tri_w // 2, cy - tri_h // 2), (cx + tri_w // 2, cy - tri_h // 2), (cx, cy + tri_h // 2)],
+        )
+
+        if not self.dd_open:
+            return
+
+        # expanded list
+        for i, name in enumerate(self.group_names):
+            item_rect = pygame.Rect(self.dd_btn.left, self.dd_btn.bottom + i * DD_H, DD_W, DD_H)
+            clr = DD_COLOR_HL if item_rect.collidepoint(pygame.mouse.get_pos()) else DD_COLOR
+            pygame.draw.rect(self.screen, clr, item_rect)
+            t = self.font.render(name, True, TEXT_COLOR)
+            self.screen.blit(
+                t, (item_rect.centerx - t.get_width() // 2, item_rect.centery - t.get_height() // 2)
+            )
+
+    def _handle_dropdown_event(self, e):
+        import pygame
+
+        if e.type == pygame.MOUSEBUTTONDOWN and e.button == 1:
+            if self.dd_btn.collidepoint(e.pos):
+                self.dd_open = not self.dd_open
+                return True
+            if self.dd_open:
+                for i, name in enumerate(self.group_names):
+                    item_rect = pygame.Rect(self.dd_btn.left, self.dd_btn.bottom + i * DD_H, DD_W, DD_H)
+                    if item_rect.collidepoint(e.pos):
+                        if name != self.current_group:
+                            self.current_group = name
+                            self._build_sliders()
+                            self._adjust_height()
+                        self.dd_open = False
+                        return True
+                self.dd_open = False
+        return False
+
+    def _save_current(self):
+        for s in self.sliders:
+            self.calibration[s.motor].range_min = s.min_v
+            self.calibration[s.motor].range_max = s.max_v
+
+        with self.bus.torque_disabled():
+            self.bus.write_calibration(self.calibration)
+
+    def _load_current(self):
+        self.calibration = self.bus.read_calibration()
+        for s in self.sliders:
+            s.min_v = self.calibration[s.motor].range_min
+            s.max_v = self.calibration[s.motor].range_max
+            s.min_x = s._pos_from_val(s.min_v)
+            s.max_x = s._pos_from_val(s.max_v)
+
+    def run(self) -> dict[str, MotorCalibration]:
+        import pygame
+
+        while True:
+            for e in pygame.event.get():
+                if e.type == pygame.QUIT:
+                    pygame.quit()
+                    return self.calibration
+
+                if self._handle_dropdown_event(e):
+                    continue
+
+                if e.type == pygame.MOUSEBUTTONDOWN and e.button == 1:
+                    if self.save_btn.collidepoint(e.pos):
+                        self._save_current()
+                    elif self.load_btn.collidepoint(e.pos):
+                        self._load_current()
+
+                for s in self.sliders:
+                    s.handle_event(e)
+
+            # live goal write while dragging
+            for s in self.sliders:
+                if s.drag_pos:
+                    self.bus.write("Goal_Position", s.motor, s.pos_v, normalize=False)
+
+            # tick update
+            for s in self.sliders:
+                pos = self.bus.read("Present_Position", s.motor, normalize=False)
+                s.set_tick(pos)
+                self.present_cache[s.motor] = pos
+
+            # ─ drawing
+            self.screen.fill(BG_COLOR)
+            for s in self.sliders:
+                s.draw(self.screen)
+
+            self._draw_dropdown()
+
+            # load / save buttons
+            for rect, text in ((self.load_btn, "LOAD"), (self.save_btn, "SAVE")):
+                clr = BTN_COLOR_HL if rect.collidepoint(pygame.mouse.get_pos()) else BTN_COLOR
+                pygame.draw.rect(self.screen, clr, rect, border_radius=6)
+                t = self.font.render(text, True, TEXT_COLOR)
+                self.screen.blit(t, (rect.centerx - t.get_width() // 2, rect.centery - t.get_height() // 2))
+
+            pygame.display.flip()
+            self.clock.tick(FPS)
diff --git a/lerobot/src/lerobot/motors/damiao/__init__.py b/lerobot/src/lerobot/motors/damiao/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..8240138cf242fbee5a2d92880e60a5ef010381d3
--- /dev/null
+++ b/lerobot/src/lerobot/motors/damiao/__init__.py
@@ -0,0 +1,18 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .damiao import DamiaoMotorsBus
+from .tables import *
diff --git a/lerobot/src/lerobot/motors/damiao/damiao.py b/lerobot/src/lerobot/motors/damiao/damiao.py
new file mode 100644
index 0000000000000000000000000000000000000000..ae619f159a06b07345443d3744a6372acf47eaa3
--- /dev/null
+++ b/lerobot/src/lerobot/motors/damiao/damiao.py
@@ -0,0 +1,859 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# Portions of this file are derived from DM_Control_Python by cmjang.
+# Licensed under the MIT License; see `LICENSE` for the full text:
+# https://github.com/cmjang/DM_Control_Python
+
+import logging
+import time
+from contextlib import contextmanager
+from copy import deepcopy
+from functools import cached_property
+from typing import TYPE_CHECKING, Any, TypedDict
+
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.import_utils import _can_available
+
+if TYPE_CHECKING or _can_available:
+    import can
+else:
+
+    class can:  # noqa: N801
+        Message = object
+        interface = None
+
+
+import numpy as np
+
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import enter_pressed, move_cursor_up
+
+from ..motors_bus import Motor, MotorCalibration, MotorsBusBase, NameOrID, Value
+from .tables import (
+    AVAILABLE_BAUDRATES,
+    CAN_CMD_DISABLE,
+    CAN_CMD_ENABLE,
+    CAN_CMD_REFRESH,
+    CAN_CMD_SET_ZERO,
+    CAN_PARAM_ID,
+    DEFAULT_BAUDRATE,
+    DEFAULT_TIMEOUT_MS,
+    MIT_KD_RANGE,
+    MIT_KP_RANGE,
+    MOTOR_LIMIT_PARAMS,
+    MotorType,
+)
+
+logger = logging.getLogger(__name__)
+
+
+LONG_TIMEOUT_SEC = 0.1
+MEDIUM_TIMEOUT_SEC = 0.01
+SHORT_TIMEOUT_SEC = 0.001
+PRECISE_TIMEOUT_SEC = 0.0001
+
+
+class MotorState(TypedDict):
+    position: float
+    velocity: float
+    torque: float
+    temp_mos: float
+    temp_rotor: float
+
+
+class DamiaoMotorsBus(MotorsBusBase):
+    """
+    The Damiao implementation for a MotorsBus using CAN bus communication.
+
+    This class uses python-can for CAN bus communication with Damiao motors.
+    For more info, see:
+    - python-can documentation: https://python-can.readthedocs.io/en/stable/
+    - Seedstudio documentation: https://wiki.seeedstudio.com/damiao_series/
+    - DM_Control_Python repo: https://github.com/cmjang/DM_Control_Python
+    """
+
+    # CAN-specific settings
+    available_baudrates = deepcopy(AVAILABLE_BAUDRATES)
+    default_baudrate = DEFAULT_BAUDRATE
+    default_timeout = DEFAULT_TIMEOUT_MS
+
+    def __init__(
+        self,
+        port: str,
+        motors: dict[str, Motor],
+        calibration: dict[str, MotorCalibration] | None = None,
+        can_interface: str = "auto",
+        use_can_fd: bool = True,
+        bitrate: int = 1000000,
+        data_bitrate: int | None = 5000000,
+    ):
+        """
+        Initialize the Damiao motors bus.
+
+        Args:
+            port: CAN interface name (e.g., "can0" for Linux, "/dev/cu.usbmodem*" for macOS)
+            motors: Dictionary mapping motor names to Motor objects
+            calibration: Optional calibration data
+            can_interface: CAN interface type - "auto" (default), "socketcan" (Linux), or "slcan" (macOS/serial)
+            use_can_fd: Whether to use CAN FD mode (default: True for OpenArms)
+            bitrate: Nominal bitrate in bps (default: 1000000 = 1 Mbps)
+            data_bitrate: Data bitrate for CAN FD in bps (default: 5000000 = 5 Mbps), ignored if use_can_fd is False
+        """
+        super().__init__(port, motors, calibration)
+        self.port = port
+        self.can_interface = can_interface
+        self.use_can_fd = use_can_fd
+        self.bitrate = bitrate
+        self.data_bitrate = data_bitrate
+        self.canbus: can.interface.Bus | None = None
+        self._is_connected = False
+
+        # Map motor names to CAN IDs
+        self._motor_can_ids: dict[str, int] = {}
+        self._recv_id_to_motor: dict[int, str] = {}
+        self._motor_types: dict[str, MotorType] = {}
+
+        for name, motor in self.motors.items():
+            if motor.motor_type_str is None:
+                raise ValueError(f"Motor '{name}' is missing required 'motor_type'")
+            self._motor_types[name] = getattr(MotorType, motor.motor_type_str.upper().replace("-", "_"))
+
+            # Map recv_id to motor name for filtering responses
+            if motor.recv_id is not None:
+                self._recv_id_to_motor[motor.recv_id] = name
+
+        # State cache for handling packet drops safely
+        self._last_known_states: dict[str, MotorState] = {
+            name: {
+                "position": 0.0,
+                "velocity": 0.0,
+                "torque": 0.0,
+                "temp_mos": 0.0,
+                "temp_rotor": 0.0,
+            }
+            for name in self.motors
+        }
+
+        # Dynamic gains storage
+        # Defaults: Kp=10.0 (Stiffness), Kd=0.5 (Damping)
+        self._gains: dict[str, dict[str, float]] = {name: {"kp": 10.0, "kd": 0.5} for name in self.motors}
+
+    @property
+    def is_connected(self) -> bool:
+        """Check if the CAN bus is connected."""
+        return self._is_connected and self.canbus is not None
+
+    @check_if_already_connected
+    def connect(self, handshake: bool = True) -> None:
+        """
+        Open the CAN bus and initialize communication.
+
+        Args:
+            handshake: If True, ping all motors to verify they're present
+        """
+
+        try:
+            # Auto-detect interface type based on port name
+            if self.can_interface == "auto":
+                if self.port.startswith("/dev/"):
+                    self.can_interface = "slcan"
+                    logger.info(f"Auto-detected slcan interface for port {self.port}")
+                else:
+                    self.can_interface = "socketcan"
+                    logger.info(f"Auto-detected socketcan interface for port {self.port}")
+
+            # Connect to CAN bus
+            kwargs = {
+                "channel": self.port,
+                "bitrate": self.bitrate,
+                "interface": self.can_interface,
+            }
+
+            if self.can_interface == "socketcan" and self.use_can_fd and self.data_bitrate is not None:
+                kwargs.update({"data_bitrate": self.data_bitrate, "fd": True})
+                logger.info(
+                    f"Connected to {self.port} with CAN FD (bitrate={self.bitrate}, data_bitrate={self.data_bitrate})"
+                )
+            else:
+                logger.info(f"Connected to {self.port} with {self.can_interface} (bitrate={self.bitrate})")
+
+            self.canbus = can.interface.Bus(**kwargs)
+            self._is_connected = True
+
+            if handshake:
+                self._handshake()
+
+            logger.debug(f"{self.__class__.__name__} connected via {self.can_interface}.")
+        except Exception as e:
+            self._is_connected = False
+            raise ConnectionError(f"Failed to connect to CAN bus: {e}") from e
+
+    def _handshake(self) -> None:
+        """
+        Verify all motors are present and populate initial state cache.
+        Raises ConnectionError if any motor fails to respond.
+        """
+        logger.info("Starting handshake with motors...")
+
+        # Drain any pending messages
+        if self.canbus is None:
+            raise RuntimeError("CAN bus is not initialized.")
+
+        while self.canbus.recv(timeout=0.01):
+            pass
+
+        missing_motors = []
+        for motor_name in self.motors:
+            motor_id = self._get_motor_id(motor_name)
+            recv_id = self._get_motor_recv_id(motor_name)
+
+            # Send enable command
+            data = [0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, CAN_CMD_ENABLE]
+            msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False, is_fd=self.use_can_fd)
+            self.canbus.send(msg)
+
+            # Wait for response with longer timeout
+            response = None
+            start_time = time.time()
+            while time.time() - start_time < 0.1:
+                response = self.canbus.recv(timeout=0.1)
+                if response and response.arbitration_id == recv_id:
+                    break
+                response = None
+
+            if response is None:
+                missing_motors.append(motor_name)
+            else:
+                self._process_response(motor_name, msg)
+            time.sleep(MEDIUM_TIMEOUT_SEC)
+
+        if missing_motors:
+            raise ConnectionError(
+                f"Handshake failed. The following motors did not respond: {missing_motors}. "
+                "Check power (24V) and CAN wiring."
+            )
+        logger.info("Handshake successful. All motors ready.")
+
+    @check_if_not_connected
+    def disconnect(self, disable_torque: bool = True) -> None:
+        """
+        Close the CAN bus connection.
+
+        Args:
+            disable_torque: If True, disable torque on all motors before disconnecting
+        """
+
+        if disable_torque:
+            try:
+                self.disable_torque()
+            except Exception as e:
+                logger.warning(f"Failed to disable torque during disconnect: {e}")
+
+        if self.canbus:
+            self.canbus.shutdown()
+            self.canbus = None
+        self._is_connected = False
+        logger.debug(f"{self.__class__.__name__} disconnected.")
+
+    def configure_motors(self) -> None:
+        """Configure all motors with default settings."""
+        # Damiao motors don't require much configuration in MIT mode
+        # Just ensure they're enabled
+        for motor in self.motors:
+            self._send_simple_command(motor, CAN_CMD_ENABLE)
+            time.sleep(MEDIUM_TIMEOUT_SEC)
+
+    def _send_simple_command(self, motor: NameOrID, command_byte: int) -> None:
+        """Helper to send simple 8-byte commands (Enable, Disable, Zero)."""
+        motor_id = self._get_motor_id(motor)
+        motor_name = self._get_motor_name(motor)
+        recv_id = self._get_motor_recv_id(motor)
+        data = [0xFF] * 7 + [command_byte]
+        msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False, is_fd=self.use_can_fd)
+
+        if self.canbus is None:
+            raise RuntimeError("CAN bus is not initialized.")
+
+        self.canbus.send(msg)
+        if msg := self._recv_motor_response(expected_recv_id=recv_id):
+            self._process_response(motor_name, msg)
+        else:
+            logger.debug(f"No response from {motor_name} after command 0x{command_byte:02X}")
+
+    def enable_torque(self, motors: str | list[str] | None = None, num_retry: int = 0) -> None:
+        """Enable torque on selected motors."""
+        target_motors = self._get_motors_list(motors)
+        for motor in target_motors:
+            for _ in range(num_retry + 1):
+                try:
+                    self._send_simple_command(motor, CAN_CMD_ENABLE)
+                    break
+                except Exception as e:
+                    if _ == num_retry:
+                        raise e
+                    time.sleep(MEDIUM_TIMEOUT_SEC)
+
+    def disable_torque(self, motors: str | list[str] | None = None, num_retry: int = 0) -> None:
+        """Disable torque on selected motors."""
+        target_motors = self._get_motors_list(motors)
+        for motor in target_motors:
+            for _ in range(num_retry + 1):
+                try:
+                    self._send_simple_command(motor, CAN_CMD_DISABLE)
+                    break
+                except Exception as e:
+                    if _ == num_retry:
+                        raise e
+                    time.sleep(MEDIUM_TIMEOUT_SEC)
+
+    @contextmanager
+    def torque_disabled(self, motors: str | list[str] | None = None):
+        """
+        Context manager that guarantees torque is re-enabled.
+
+        This helper is useful to temporarily disable torque when configuring motors.
+        """
+        self.disable_torque(motors)
+        try:
+            yield
+        finally:
+            self.enable_torque(motors)
+
+    def set_zero_position(self, motors: str | list[str] | None = None) -> None:
+        """Set current position as zero for selected motors."""
+        target_motors = self._get_motors_list(motors)
+        for motor in target_motors:
+            self._send_simple_command(motor, CAN_CMD_SET_ZERO)
+            time.sleep(MEDIUM_TIMEOUT_SEC)
+
+    def _refresh_motor(self, motor: NameOrID) -> can.Message | None:
+        """Refresh motor status and return the response."""
+        motor_id = self._get_motor_id(motor)
+        recv_id = self._get_motor_recv_id(motor)
+        data = [motor_id & 0xFF, (motor_id >> 8) & 0xFF, CAN_CMD_REFRESH, 0, 0, 0, 0, 0]
+        msg = can.Message(arbitration_id=CAN_PARAM_ID, data=data, is_extended_id=False, is_fd=self.use_can_fd)
+
+        if self.canbus is None:
+            raise RuntimeError("CAN bus is not initialized.")
+
+        self.canbus.send(msg)
+        return self._recv_motor_response(expected_recv_id=recv_id)
+
+    def _recv_motor_response(
+        self, expected_recv_id: int | None = None, timeout: float = 0.001
+    ) -> can.Message | None:
+        """
+        Receive a response from a motor.
+
+        Args:
+            expected_recv_id: If provided, only return messages from this CAN ID
+            timeout: Timeout in seconds (default: 1ms for high-speed operation)
+        Returns:
+            CAN message if received, None otherwise
+        """
+
+        if self.canbus is None:
+            raise RuntimeError("CAN bus is not initialized.")
+
+        try:
+            start_time = time.time()
+            messages_seen = []
+            while time.time() - start_time < timeout:
+                msg = self.canbus.recv(timeout=PRECISE_TIMEOUT_SEC)
+                if msg:
+                    messages_seen.append(f"0x{msg.arbitration_id:02X}")
+                    if expected_recv_id is None or msg.arbitration_id == expected_recv_id:
+                        return msg
+                    logger.debug(
+                        f"Ignoring message from 0x{msg.arbitration_id:02X}, expected 0x{expected_recv_id:02X}"
+                    )
+
+            if logger.isEnabledFor(logging.DEBUG):
+                if messages_seen:
+                    logger.debug(
+                        f"Received {len(messages_seen)} msgs from {set(messages_seen)}, expected 0x{expected_recv_id:02X}"
+                    )
+                else:
+                    logger.debug(f"No CAN messages received (expected 0x{expected_recv_id:02X})")
+        except Exception as e:
+            logger.debug(f"Failed to receive CAN message: {e}")
+        return None
+
+    def _recv_all_responses(
+        self, expected_recv_ids: list[int], timeout: float = 0.002
+    ) -> dict[int, can.Message]:
+        """
+        Efficiently receive responses from multiple motors at once.
+        Uses the OpenArms pattern: collect all available messages within timeout.
+
+        Args:
+            expected_recv_ids: List of CAN IDs we expect responses from
+            timeout: Total timeout in seconds (default: 2ms)
+
+        Returns:
+            Dictionary mapping recv_id to CAN message
+        """
+        responses: dict[int, can.Message] = {}
+        expected_set = set(expected_recv_ids)
+        start_time = time.time()
+
+        if self.canbus is None:
+            raise RuntimeError("CAN bus is not initialized.")
+
+        try:
+            while len(responses) < len(expected_recv_ids) and (time.time() - start_time) < timeout:
+                # 100us poll timeout
+                msg = self.canbus.recv(timeout=PRECISE_TIMEOUT_SEC)
+                if msg and msg.arbitration_id in expected_set:
+                    responses[msg.arbitration_id] = msg
+                    if len(responses) == len(expected_recv_ids):
+                        break
+        except Exception as e:
+            logger.debug(f"Error receiving responses: {e}")
+
+        return responses
+
+    def _encode_mit_packet(
+        self,
+        motor_type: MotorType,
+        kp: float,
+        kd: float,
+        position_degrees: float,
+        velocity_deg_per_sec: float,
+        torque: float,
+    ) -> list[int]:
+        """Helper to encode control parameters into 8 bytes for MIT mode."""
+        # Convert degrees to radians
+        position_rad = np.radians(position_degrees)
+        velocity_rad_per_sec = np.radians(velocity_deg_per_sec)
+
+        # Get motor limits
+        pmax, vmax, tmax = MOTOR_LIMIT_PARAMS[motor_type]
+
+        # Encode parameters
+        kp_uint = self._float_to_uint(kp, *MIT_KP_RANGE, 12)
+        kd_uint = self._float_to_uint(kd, *MIT_KD_RANGE, 12)
+        q_uint = self._float_to_uint(position_rad, -pmax, pmax, 16)
+        dq_uint = self._float_to_uint(velocity_rad_per_sec, -vmax, vmax, 12)
+        tau_uint = self._float_to_uint(torque, -tmax, tmax, 12)
+
+        # Pack data
+        data = [0] * 8
+        data[0] = (q_uint >> 8) & 0xFF
+        data[1] = q_uint & 0xFF
+        data[2] = dq_uint >> 4
+        data[3] = ((dq_uint & 0xF) << 4) | ((kp_uint >> 8) & 0xF)
+        data[4] = kp_uint & 0xFF
+        data[5] = kd_uint >> 4
+        data[6] = ((kd_uint & 0xF) << 4) | ((tau_uint >> 8) & 0xF)
+        data[7] = tau_uint & 0xFF
+        return data
+
+    def _mit_control(
+        self,
+        motor: NameOrID,
+        kp: float,
+        kd: float,
+        position_degrees: float,
+        velocity_deg_per_sec: float,
+        torque: float,
+    ) -> None:
+        """Send MIT control command to a motor."""
+        motor_id = self._get_motor_id(motor)
+        motor_name = self._get_motor_name(motor)
+        motor_type = self._motor_types[motor_name]
+
+        if self.canbus is None:
+            raise RuntimeError("CAN bus is not initialized.")
+
+        data = self._encode_mit_packet(motor_type, kp, kd, position_degrees, velocity_deg_per_sec, torque)
+        msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False, is_fd=self.use_can_fd)
+        self.canbus.send(msg)
+
+        recv_id = self._get_motor_recv_id(motor)
+        if msg := self._recv_motor_response(expected_recv_id=recv_id):
+            self._process_response(motor_name, msg)
+        else:
+            logger.debug(f"No response from {motor_name} after MIT control command")
+
+    def _mit_control_batch(
+        self,
+        commands: dict[NameOrID, tuple[float, float, float, float, float]],
+    ) -> None:
+        """
+        Send MIT control commands to multiple motors in batch.
+        Sends all commands first, then collects responses.
+
+        Args:
+            commands: Dict mapping motor name/ID to (kp, kd, position_deg, velocity_deg/s, torque)
+                     Example: {'joint_1': (10.0, 0.5, 45.0, 0.0, 0.0), ...}
+        """
+        if not commands:
+            return
+
+        recv_id_to_motor: dict[int, str] = {}
+
+        if self.canbus is None:
+            raise RuntimeError("CAN bus is not initialized.")
+
+        # Step 1: Send all MIT control commands
+        for motor, (kp, kd, position_degrees, velocity_deg_per_sec, torque) in commands.items():
+            motor_id = self._get_motor_id(motor)
+            motor_name = self._get_motor_name(motor)
+            motor_type = self._motor_types[motor_name]
+
+            data = self._encode_mit_packet(motor_type, kp, kd, position_degrees, velocity_deg_per_sec, torque)
+            msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False, is_fd=self.use_can_fd)
+            self.canbus.send(msg)
+
+            recv_id_to_motor[self._get_motor_recv_id(motor)] = motor_name
+
+        # Step 2: Collect responses and update state cache
+        responses = self._recv_all_responses(list(recv_id_to_motor.keys()), timeout=SHORT_TIMEOUT_SEC)
+        for recv_id, motor_name in recv_id_to_motor.items():
+            if msg := responses.get(recv_id):
+                self._process_response(motor_name, msg)
+
+    def _float_to_uint(self, x: float, x_min: float, x_max: float, bits: int) -> int:
+        """Convert float to unsigned integer for CAN transmission."""
+        x = max(x_min, min(x_max, x))  # Clamp to range
+        span = x_max - x_min
+        data_norm = (x - x_min) / span
+        return int(data_norm * ((1 << bits) - 1))
+
+    def _uint_to_float(self, x: int, x_min: float, x_max: float, bits: int) -> float:
+        """Convert unsigned integer from CAN to float."""
+        span = x_max - x_min
+        data_norm = float(x) / ((1 << bits) - 1)
+        return data_norm * span + x_min
+
+    def _decode_motor_state(
+        self, data: bytearray | bytes, motor_type: MotorType
+    ) -> tuple[float, float, float, int, int]:
+        """
+        Decode motor state from CAN data.
+        Returns: (position_deg, velocity_deg_s, torque, temp_mos, temp_rotor)
+        """
+        if len(data) < 8:
+            raise ValueError("Invalid motor state data")
+
+        # Extract encoded values
+        q_uint = (data[1] << 8) | data[2]
+        dq_uint = (data[3] << 4) | (data[4] >> 4)
+        tau_uint = ((data[4] & 0x0F) << 8) | data[5]
+        t_mos = data[6]
+        t_rotor = data[7]
+
+        # Get motor limits
+        pmax, vmax, tmax = MOTOR_LIMIT_PARAMS[motor_type]
+
+        # Decode to physical values
+        position_rad = self._uint_to_float(q_uint, -pmax, pmax, 16)
+        velocity_rad_per_sec = self._uint_to_float(dq_uint, -vmax, vmax, 12)
+        torque = self._uint_to_float(tau_uint, -tmax, tmax, 12)
+
+        return np.degrees(position_rad), np.degrees(velocity_rad_per_sec), torque, t_mos, t_rotor
+
+    def _process_response(self, motor: str, msg: can.Message) -> None:
+        """Decode a message and update the motor state cache."""
+        try:
+            motor_type = self._motor_types[motor]
+            pos, vel, torque, t_mos, t_rotor = self._decode_motor_state(msg.data, motor_type)
+
+            self._last_known_states[motor] = {
+                "position": pos,
+                "velocity": vel,
+                "torque": torque,
+                "temp_mos": float(t_mos),
+                "temp_rotor": float(t_rotor),
+            }
+        except Exception as e:
+            logger.warning(f"Failed to decode response from {motor}: {e}")
+
+    @check_if_not_connected
+    def read(self, data_name: str, motor: str) -> Value:
+        """Read a value from a single motor. Positions are always in degrees."""
+
+        # Refresh motor to get latest state
+        msg = self._refresh_motor(motor)
+        if msg is None:
+            motor_id = self._get_motor_id(motor)
+            recv_id = self._get_motor_recv_id(motor)
+            raise ConnectionError(
+                f"No response from motor '{motor}' (send ID: 0x{motor_id:02X}, recv ID: 0x{recv_id:02X}). "
+                f"Check that: 1) Motor is powered (24V), 2) CAN wiring is correct, "
+                f"3) Motor IDs are configured correctly using Damiao Debugging Tools"
+            )
+
+        self._process_response(motor, msg)
+        return self._get_cached_value(motor, data_name)
+
+    def _get_cached_value(self, motor: str, data_name: str) -> Value:
+        """Retrieve a specific value from the cache."""
+        state = self._last_known_states[motor]
+        mapping: dict[str, Any] = {
+            "Present_Position": state["position"],
+            "Present_Velocity": state["velocity"],
+            "Present_Torque": state["torque"],
+            "Temperature_MOS": state["temp_mos"],
+            "Temperature_Rotor": state["temp_rotor"],
+        }
+        if data_name not in mapping:
+            raise ValueError(f"Unknown data_name: {data_name}")
+        return mapping[data_name]
+
+    @check_if_not_connected
+    def write(
+        self,
+        data_name: str,
+        motor: str,
+        value: Value,
+    ) -> None:
+        """
+        Write a value to a single motor. Positions are always in degrees.
+        Can write 'Goal_Position', 'Kp', or 'Kd'.
+        """
+
+        if data_name in ("Kp", "Kd"):
+            self._gains[motor][data_name.lower()] = float(value)
+        elif data_name == "Goal_Position":
+            kp = self._gains[motor]["kp"]
+            kd = self._gains[motor]["kd"]
+            self._mit_control(motor, kp, kd, float(value), 0.0, 0.0)
+        else:
+            raise ValueError(f"Writing {data_name} not supported in MIT mode")
+
+    def sync_read(
+        self,
+        data_name: str,
+        motors: str | list[str] | None = None,
+    ) -> dict[str, Value]:
+        """
+        Read the same value from multiple motors simultaneously.
+        """
+        target_motors = self._get_motors_list(motors)
+        self._batch_refresh(target_motors)
+
+        result = {}
+        for motor in target_motors:
+            result[motor] = self._get_cached_value(motor, data_name)
+        return result
+
+    def sync_read_all_states(
+        self,
+        motors: str | list[str] | None = None,
+        *,
+        num_retry: int = 0,
+    ) -> dict[str, MotorState]:
+        """
+        Read ALL motor states (position, velocity, torque) from multiple motors in ONE refresh cycle.
+
+        Returns:
+            Dictionary mapping motor names to state dicts with keys: 'position', 'velocity', 'torque'
+            Example: {'joint_1': {'position': 45.2, 'velocity': 1.3, 'torque': 0.5}, ...}
+        """
+        target_motors = self._get_motors_list(motors)
+        self._batch_refresh(target_motors)
+
+        result = {}
+        for motor in target_motors:
+            result[motor] = self._last_known_states[motor].copy()
+        return result
+
+    def _batch_refresh(self, motors: list[str]) -> None:
+        """Internal helper to refresh a list of motors and update cache."""
+
+        if self.canbus is None:
+            raise RuntimeError("CAN bus is not initialized.")
+
+        # Send refresh commands
+        for motor in motors:
+            motor_id = self._get_motor_id(motor)
+            data = [motor_id & 0xFF, (motor_id >> 8) & 0xFF, CAN_CMD_REFRESH, 0, 0, 0, 0, 0]
+            msg = can.Message(
+                arbitration_id=CAN_PARAM_ID, data=data, is_extended_id=False, is_fd=self.use_can_fd
+            )
+            self.canbus.send(msg)
+
+        # Collect responses
+        expected_recv_ids = [self._get_motor_recv_id(m) for m in motors]
+        responses = self._recv_all_responses(expected_recv_ids, timeout=MEDIUM_TIMEOUT_SEC)
+
+        # Update cache
+        for motor in motors:
+            recv_id = self._get_motor_recv_id(motor)
+            msg = responses.get(recv_id)
+            if msg:
+                self._process_response(motor, msg)
+            else:
+                logger.warning(f"Packet drop: {motor} (ID: 0x{recv_id:02X}). Using last known state.")
+
+    @check_if_not_connected
+    def sync_write(self, data_name: str, values: dict[str, Value]) -> None:
+        """
+        Write values to multiple motors simultaneously. Positions are always in degrees.
+        """
+
+        if data_name in ("Kp", "Kd"):
+            key = data_name.lower()
+            for motor, val in values.items():
+                self._gains[motor][key] = float(val)
+
+        elif data_name == "Goal_Position":
+            # Step 1: Send all MIT control commands
+            recv_id_to_motor: dict[int, str] = {}
+            if self.canbus is None:
+                raise RuntimeError("CAN bus is not initialized.")
+            for motor, value_degrees in values.items():
+                motor_id = self._get_motor_id(motor)
+                motor_name = self._get_motor_name(motor)
+                motor_type = self._motor_types[motor_name]
+
+                kp = self._gains[motor]["kp"]
+                kd = self._gains[motor]["kd"]
+
+                data = self._encode_mit_packet(motor_type, kp, kd, float(value_degrees), 0.0, 0.0)
+                msg = can.Message(
+                    arbitration_id=motor_id, data=data, is_extended_id=False, is_fd=self.use_can_fd
+                )
+                self.canbus.send(msg)
+                precise_sleep(PRECISE_TIMEOUT_SEC)
+
+                recv_id_to_motor[self._get_motor_recv_id(motor)] = motor_name
+
+            # Step 2: Collect responses and update state cache
+            responses = self._recv_all_responses(list(recv_id_to_motor.keys()), timeout=MEDIUM_TIMEOUT_SEC)
+            for recv_id, motor_name in recv_id_to_motor.items():
+                if msg := responses.get(recv_id):
+                    self._process_response(motor_name, msg)
+        else:
+            # Fall back to individual writes
+            for motor, value in values.items():
+                self.write(data_name, motor, value)
+
+    def read_calibration(self) -> dict[str, MotorCalibration]:
+        """Read calibration data from motors."""
+        # Damiao motors don't store calibration internally
+        # Return existing calibration or empty dict
+        return self.calibration if self.calibration else {}
+
+    def write_calibration(self, calibration_dict: dict[str, MotorCalibration], cache: bool = True) -> None:
+        """Write calibration data to motors."""
+        # Damiao motors don't store calibration internally
+        # Just cache it in memory
+        if cache:
+            self.calibration = calibration_dict
+
+    def record_ranges_of_motion(
+        self,
+        motors: str | list[str] | None = None,
+        display_values: bool = True,
+    ) -> tuple[dict[str, Value], dict[str, Value]]:
+        """
+        Interactively record the min/max values of each motor in degrees.
+
+        Move the joints by hand (with torque disabled) while the method streams live positions.
+        Press Enter to finish.
+        """
+        target_motors = self._get_motors_list(motors)
+
+        self.disable_torque(target_motors)
+        time.sleep(LONG_TIMEOUT_SEC)
+
+        start_positions = self.sync_read("Present_Position", target_motors)
+        mins = start_positions.copy()
+        maxes = start_positions.copy()
+
+        print("\nMove joints through their full range of motion. Press ENTER when done.")
+        user_pressed_enter = False
+
+        while not user_pressed_enter:
+            positions = self.sync_read("Present_Position", target_motors)
+
+            for motor in target_motors:
+                if motor in positions:
+                    mins[motor] = min(positions[motor], mins.get(motor, positions[motor]))
+                    maxes[motor] = max(positions[motor], maxes.get(motor, positions[motor]))
+
+            if display_values:
+                print("\n" + "=" * 50)
+                print(f"{'MOTOR':<20} | {'MIN (deg)':>12} | {'POS (deg)':>12} | {'MAX (deg)':>12}")
+                print("-" * 50)
+                for motor in target_motors:
+                    if motor in positions:
+                        print(
+                            f"{motor:<20} | {mins[motor]:>12.1f} | {positions[motor]:>12.1f} | {maxes[motor]:>12.1f}"
+                        )
+
+            if enter_pressed():
+                user_pressed_enter = True
+
+            if display_values and not user_pressed_enter:
+                move_cursor_up(len(target_motors) + 4)
+
+            time.sleep(LONG_TIMEOUT_SEC)
+
+        self.enable_torque(target_motors)
+
+        for motor in target_motors:
+            if (motor in mins) and (motor in maxes) and (int(abs(maxes[motor] - mins[motor])) < 5):
+                raise ValueError(f"Motor {motor} has insufficient range of motion (< 5 degrees)")
+
+        return mins, maxes
+
+    def _get_motors_list(self, motors: str | list[str] | None) -> list[str]:
+        """Convert motor specification to list of motor names."""
+        if motors is None:
+            return list(self.motors.keys())
+        elif isinstance(motors, str):
+            return [motors]
+        elif isinstance(motors, list):
+            return motors
+        else:
+            raise TypeError(f"Invalid motors type: {type(motors)}")
+
+    def _get_motor_id(self, motor: NameOrID) -> int:
+        """Get CAN ID for a motor."""
+        if isinstance(motor, str):
+            if motor in self.motors:
+                return self.motors[motor].id
+            else:
+                raise ValueError(f"Unknown motor: {motor}")
+        else:
+            return motor
+
+    def _get_motor_name(self, motor: NameOrID) -> str:
+        """Get motor name from name or ID."""
+        if isinstance(motor, str):
+            return motor
+        else:
+            for name, m in self.motors.items():
+                if m.id == motor:
+                    return name
+            raise ValueError(f"Unknown motor ID: {motor}")
+
+    def _get_motor_recv_id(self, motor: NameOrID) -> int:
+        """Get motor recv_id from name or ID."""
+        motor_name = self._get_motor_name(motor)
+        motor_obj = self.motors.get(motor_name)
+        if motor_obj and motor_obj.recv_id is not None:
+            return motor_obj.recv_id
+        else:
+            raise ValueError(f"Motor {motor_obj} doesn't have a valid recv_id (None).")
+
+    @cached_property
+    def is_calibrated(self) -> bool:
+        """Check if motors are calibrated."""
+        return bool(self.calibration)
diff --git a/lerobot/src/lerobot/motors/damiao/tables.py b/lerobot/src/lerobot/motors/damiao/tables.py
new file mode 100644
index 0000000000000000000000000000000000000000..22d1624fae95bb98f1a8d587402ce44d1bf1bad7
--- /dev/null
+++ b/lerobot/src/lerobot/motors/damiao/tables.py
@@ -0,0 +1,209 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Configuration tables for Damiao motors."""
+
+from enum import IntEnum
+
+
+# Motor type definitions
+class MotorType(IntEnum):
+    DM3507 = 0
+    DM4310 = 1
+    DM4310_48V = 2
+    DM4340 = 3
+    DM4340_48V = 4
+    DM6006 = 5
+    DM8006 = 6
+    DM8009 = 7
+    DM10010L = 8
+    DM10010 = 9
+    DMH3510 = 10
+    DMH6215 = 11
+    DMG6220 = 12
+
+
+# Control modes
+class ControlMode(IntEnum):
+    MIT = 1
+    POS_VEL = 2
+    VEL = 3
+    TORQUE_POS = 4
+
+
+# Motor variable IDs (RID)
+class MotorVariable(IntEnum):
+    UV_VALUE = 0
+    KT_VALUE = 1
+    OT_VALUE = 2
+    OC_VALUE = 3
+    ACC = 4
+    DEC = 5
+    MAX_SPD = 6
+    MST_ID = 7
+    ESC_ID = 8
+    TIMEOUT = 9
+    CTRL_MODE = 10
+    DAMP = 11
+    INERTIA = 12
+    HW_VER = 13
+    SW_VER = 14
+    SN = 15
+    NPP = 16
+    RS = 17
+    LS = 18
+    FLUX = 19
+    GR = 20
+    PMAX = 21
+    VMAX = 22
+    TMAX = 23
+    I_BW = 24
+    KP_ASR = 25
+    KI_ASR = 26
+    KP_APR = 27
+    KI_APR = 28
+    OV_VALUE = 29
+    GREF = 30
+    DETA = 31
+    V_BW = 32
+    IQ_C1 = 33
+    VL_C1 = 34
+    CAN_BR = 35
+    SUB_VER = 36
+    U_OFF = 50
+    V_OFF = 51
+    K1 = 52
+    K2 = 53
+    M_OFF = 54
+    DIR = 55
+    P_M = 80
+    XOUT = 81
+
+
+# Motor limit parameters [PMAX, VMAX, TMAX]
+# PMAX: Maximum position (rad)
+# VMAX: Maximum velocity (rad/s)
+# TMAX: Maximum torque (N·m)
+MOTOR_LIMIT_PARAMS = {
+    MotorType.DM3507: (12.5, 30, 10),
+    MotorType.DM4310: (12.5, 30, 10),
+    MotorType.DM4310_48V: (12.5, 50, 10),
+    MotorType.DM4340: (12.5, 8, 28),
+    MotorType.DM4340_48V: (12.5, 10, 28),
+    MotorType.DM6006: (12.5, 45, 20),
+    MotorType.DM8006: (12.5, 45, 40),
+    MotorType.DM8009: (12.5, 45, 54),
+    MotorType.DM10010L: (12.5, 25, 200),
+    MotorType.DM10010: (12.5, 20, 200),
+    MotorType.DMH3510: (12.5, 280, 1),
+    MotorType.DMH6215: (12.5, 45, 10),
+    MotorType.DMG6220: (12.5, 45, 10),
+}
+
+# Motor model names
+MODEL_NAMES = {
+    MotorType.DM3507: "dm3507",
+    MotorType.DM4310: "dm4310",
+    MotorType.DM4310_48V: "dm4310_48v",
+    MotorType.DM4340: "dm4340",
+    MotorType.DM4340_48V: "dm4340_48v",
+    MotorType.DM6006: "dm6006",
+    MotorType.DM8006: "dm8006",
+    MotorType.DM8009: "dm8009",
+    MotorType.DM10010L: "dm10010l",
+    MotorType.DM10010: "dm10010",
+    MotorType.DMH3510: "dmh3510",
+    MotorType.DMH6215: "dmh6215",
+    MotorType.DMG6220: "dmg6220",
+}
+
+# Motor resolution table (encoder counts per revolution)
+MODEL_RESOLUTION = {
+    "dm3507": 65536,
+    "dm4310": 65536,
+    "dm4310_48v": 65536,
+    "dm4340": 65536,
+    "dm4340_48v": 65536,
+    "dm6006": 65536,
+    "dm8006": 65536,
+    "dm8009": 65536,
+    "dm10010l": 65536,
+    "dm10010": 65536,
+    "dmh3510": 65536,
+    "dmh6215": 65536,
+    "dmg6220": 65536,
+}
+
+# CAN baudrates supported by Damiao motors
+AVAILABLE_BAUDRATES = [
+    125000,  # 0: 125 kbps
+    200000,  # 1: 200 kbps
+    250000,  # 2: 250 kbps
+    500000,  # 3: 500 kbps
+    1000000,  # 4: 1 mbps (default for OpenArms)
+    2000000,  # 5: 2 mbps
+    2500000,  # 6: 2.5 mbps
+    3200000,  # 7: 3.2 mbps
+    4000000,  # 8: 4 mbps
+    5000000,  # 9: 5 mbps
+]
+DEFAULT_BAUDRATE = 1000000  # 1 Mbps is standard for OpenArms
+
+# Default timeout in milliseconds
+DEFAULT_TIMEOUT_MS = 1000
+
+# OpenArms specific configurations
+# Based on: https://docs.openarm.dev/software/setup/configure-test
+# OpenArms has 7 DOF per arm (14 total for dual arm)
+OPENARMS_ARM_MOTOR_IDS = {
+    "joint_1": {"send": 0x01, "recv": 0x11},  # J1 - Shoulder pan
+    "joint_2": {"send": 0x02, "recv": 0x12},  # J2 - Shoulder lift
+    "joint_3": {"send": 0x03, "recv": 0x13},  # J3 - Elbow flex
+    "joint_4": {"send": 0x04, "recv": 0x14},  # J4 - Wrist flex
+    "joint_5": {"send": 0x05, "recv": 0x15},  # J5 - Wrist roll
+    "joint_6": {"send": 0x06, "recv": 0x16},  # J6 - Wrist pitch
+    "joint_7": {"send": 0x07, "recv": 0x17},  # J7 - Wrist rotation
+}
+
+OPENARMS_GRIPPER_MOTOR_IDS = {
+    "gripper": {"send": 0x08, "recv": 0x18},  # J8 - Gripper
+}
+
+# Default motor types for OpenArms
+OPENARMS_DEFAULT_MOTOR_TYPES = {
+    "joint_1": MotorType.DM8009,  # Shoulder pan - high torque
+    "joint_2": MotorType.DM8009,  # Shoulder lift - high torque
+    "joint_3": MotorType.DM4340,  # Shoulder rotation
+    "joint_4": MotorType.DM4340,  # Elbow flex
+    "joint_5": MotorType.DM4310,  # Wrist roll
+    "joint_6": MotorType.DM4310,  # Wrist pitch
+    "joint_7": MotorType.DM4310,  # Wrist rotation
+    "gripper": MotorType.DM4310,  # Gripper
+}
+
+# MIT control parameter ranges
+MIT_KP_RANGE = (0.0, 500.0)
+MIT_KD_RANGE = (0.0, 5.0)
+
+# CAN frame command IDs
+CAN_CMD_ENABLE = 0xFC
+CAN_CMD_DISABLE = 0xFD
+CAN_CMD_SET_ZERO = 0xFE
+CAN_CMD_REFRESH = 0xCC
+CAN_CMD_QUERY_PARAM = 0x33
+CAN_CMD_WRITE_PARAM = 0x55
+CAN_CMD_SAVE_PARAM = 0xAA
+
+# CAN ID for parameter operations
+CAN_PARAM_ID = 0x7FF
diff --git a/lerobot/src/lerobot/motors/dynamixel/__init__.py b/lerobot/src/lerobot/motors/dynamixel/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..425f8538ab263bef7185265ee21fa9fda89d8636
--- /dev/null
+++ b/lerobot/src/lerobot/motors/dynamixel/__init__.py
@@ -0,0 +1,18 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .dynamixel import DriveMode, DynamixelMotorsBus, OperatingMode, TorqueMode
+from .tables import *
diff --git a/lerobot/src/lerobot/motors/dynamixel/dynamixel.py b/lerobot/src/lerobot/motors/dynamixel/dynamixel.py
new file mode 100644
index 0000000000000000000000000000000000000000..bca455dc5ad669e9d6f6f5e0aac922113a994039
--- /dev/null
+++ b/lerobot/src/lerobot/motors/dynamixel/dynamixel.py
@@ -0,0 +1,263 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# TODO(aliberts): Should we implement FastSyncRead/Write?
+# https://github.com/ROBOTIS-GIT/DynamixelSDK/pull/643
+# https://github.com/ROBOTIS-GIT/DynamixelSDK/releases/tag/3.8.2
+# https://emanual.robotis.com/docs/en/dxl/protocol2/#fast-sync-read-0x8a
+# -> Need to check compatibility across models
+
+import logging
+from copy import deepcopy
+from enum import Enum
+
+from ..encoding_utils import decode_twos_complement, encode_twos_complement
+from ..motors_bus import Motor, MotorCalibration, NameOrID, SerialMotorsBus, Value, get_address
+from .tables import (
+    AVAILABLE_BAUDRATES,
+    MODEL_BAUDRATE_TABLE,
+    MODEL_CONTROL_TABLE,
+    MODEL_ENCODING_TABLE,
+    MODEL_NUMBER_TABLE,
+    MODEL_RESOLUTION,
+)
+
+PROTOCOL_VERSION = 2.0
+DEFAULT_BAUDRATE = 1_000_000
+DEFAULT_TIMEOUT_MS = 1000
+
+NORMALIZED_DATA = ["Goal_Position", "Present_Position"]
+
+logger = logging.getLogger(__name__)
+
+
+class OperatingMode(Enum):
+    # DYNAMIXEL only controls current(torque) regardless of speed and position. This mode is ideal for a
+    # gripper or a system that only uses current(torque) control or a system that has additional
+    # velocity/position controllers.
+    CURRENT = 0
+
+    # This mode controls velocity. This mode is identical to the Wheel Mode(endless) from existing DYNAMIXEL.
+    # This mode is ideal for wheel-type robots.
+    VELOCITY = 1
+
+    # This mode controls position. This mode is identical to the Joint Mode from existing DYNAMIXEL. Operating
+    # position range is limited by the Max Position Limit(48) and the Min Position Limit(52). This mode is
+    # ideal for articulated robots that each joint rotates less than 360 degrees.
+    POSITION = 3
+
+    # This mode controls position. This mode is identical to the Multi-turn Position Control from existing
+    # DYNAMIXEL. 512 turns are supported(-256[rev] ~ 256[rev]). This mode is ideal for multi-turn wrists or
+    # conveyor systems or a system that requires an additional reduction gear. Note that Max Position
+    # Limit(48), Min Position Limit(52) are not used on Extended Position Control Mode.
+    EXTENDED_POSITION = 4
+
+    # This mode controls both position and current(torque). Up to 512 turns are supported (-256[rev] ~
+    # 256[rev]). This mode is ideal for a system that requires both position and current control such as
+    # articulated robots or grippers.
+    CURRENT_POSITION = 5
+
+    # This mode directly controls PWM output. (Voltage Control Mode)
+    PWM = 16
+
+
+class DriveMode(Enum):
+    NON_INVERTED = 0
+    INVERTED = 1
+
+
+class TorqueMode(Enum):
+    ENABLED = 1
+    DISABLED = 0
+
+
+def _split_into_byte_chunks(value: int, length: int) -> list[int]:
+    import dynamixel_sdk as dxl
+
+    if length == 1:
+        data = [value]
+    elif length == 2:
+        data = [dxl.DXL_LOBYTE(value), dxl.DXL_HIBYTE(value)]
+    elif length == 4:
+        data = [
+            dxl.DXL_LOBYTE(dxl.DXL_LOWORD(value)),
+            dxl.DXL_HIBYTE(dxl.DXL_LOWORD(value)),
+            dxl.DXL_LOBYTE(dxl.DXL_HIWORD(value)),
+            dxl.DXL_HIBYTE(dxl.DXL_HIWORD(value)),
+        ]
+    return data
+
+
+class DynamixelMotorsBus(SerialMotorsBus):
+    """
+    The Dynamixel implementation for a MotorsBus. It relies on the python dynamixel sdk to communicate with
+    the motors. For more info, see the Dynamixel SDK Documentation:
+    https://emanual.robotis.com/docs/en/software/dynamixel/dynamixel_sdk/sample_code/python_read_write_protocol_2_0/#python-read-write-protocol-20
+    """
+
+    apply_drive_mode = False
+    available_baudrates = deepcopy(AVAILABLE_BAUDRATES)
+    default_baudrate = DEFAULT_BAUDRATE
+    default_timeout = DEFAULT_TIMEOUT_MS
+    model_baudrate_table = deepcopy(MODEL_BAUDRATE_TABLE)
+    model_ctrl_table = deepcopy(MODEL_CONTROL_TABLE)
+    model_encoding_table = deepcopy(MODEL_ENCODING_TABLE)
+    model_number_table = deepcopy(MODEL_NUMBER_TABLE)
+    model_resolution_table = deepcopy(MODEL_RESOLUTION)
+    normalized_data = deepcopy(NORMALIZED_DATA)
+
+    def __init__(
+        self,
+        port: str,
+        motors: dict[str, Motor],
+        calibration: dict[str, MotorCalibration] | None = None,
+    ):
+        super().__init__(port, motors, calibration)
+        import dynamixel_sdk as dxl
+
+        self.port_handler = dxl.PortHandler(self.port)
+        self.packet_handler = dxl.PacketHandler(PROTOCOL_VERSION)
+        self.sync_reader = dxl.GroupSyncRead(self.port_handler, self.packet_handler, 0, 0)
+        self.sync_writer = dxl.GroupSyncWrite(self.port_handler, self.packet_handler, 0, 0)
+        self._comm_success = dxl.COMM_SUCCESS
+        self._no_error = 0x00
+
+    def _assert_protocol_is_compatible(self, instruction_name: str) -> None:
+        pass
+
+    def _handshake(self) -> None:
+        self._assert_motors_exist()
+
+    def _find_single_motor(self, motor: str, initial_baudrate: int | None = None) -> tuple[int, int]:
+        model = self.motors[motor].model
+        search_baudrates = (
+            [initial_baudrate] if initial_baudrate is not None else self.model_baudrate_table[model]
+        )
+
+        for baudrate in search_baudrates:
+            self.set_baudrate(baudrate)
+            id_model = self.broadcast_ping()
+            if id_model:
+                found_id, found_model = next(iter(id_model.items()))
+                expected_model_nb = self.model_number_table[model]
+                if found_model != expected_model_nb:
+                    raise RuntimeError(
+                        f"Found one motor on {baudrate=} with id={found_id} but it has a "
+                        f"model number '{found_model}' different than the one expected: '{expected_model_nb}'. "
+                        f"Make sure you are connected only connected to the '{motor}' motor (model '{model}')."
+                    )
+                return baudrate, found_id
+
+        raise RuntimeError(f"Motor '{motor}' (model '{model}') was not found. Make sure it is connected.")
+
+    def configure_motors(self, return_delay_time=0) -> None:
+        # By default, Dynamixel motors have a 500µs delay response time (corresponding to a value of 250 on
+        # the 'Return_Delay_Time' address). We ensure this is reduced to the minimum of 2µs (value of 0).
+        for motor in self.motors:
+            self.write("Return_Delay_Time", motor, return_delay_time)
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.calibration == self.read_calibration()
+
+    def read_calibration(self) -> dict[str, MotorCalibration]:
+        offsets = self.sync_read("Homing_Offset", normalize=False)
+        mins = self.sync_read("Min_Position_Limit", normalize=False)
+        maxes = self.sync_read("Max_Position_Limit", normalize=False)
+        drive_modes = self.sync_read("Drive_Mode", normalize=False)
+
+        calibration = {}
+        for motor, m in self.motors.items():
+            calibration[motor] = MotorCalibration(
+                id=m.id,
+                drive_mode=int(drive_modes[motor]),
+                homing_offset=int(offsets[motor]),
+                range_min=int(mins[motor]),
+                range_max=int(maxes[motor]),
+            )
+
+        return calibration
+
+    def write_calibration(self, calibration_dict: dict[str, MotorCalibration], cache: bool = True) -> None:
+        for motor, calibration in calibration_dict.items():
+            self.write("Homing_Offset", motor, calibration.homing_offset)
+            self.write("Min_Position_Limit", motor, calibration.range_min)
+            self.write("Max_Position_Limit", motor, calibration.range_max)
+
+        if cache:
+            self.calibration = calibration_dict
+
+    def disable_torque(self, motors: int | str | list[str] | None = None, num_retry: int = 0) -> None:
+        for motor in self._get_motors_list(motors):
+            self.write("Torque_Enable", motor, TorqueMode.DISABLED.value, num_retry=num_retry)
+
+    def _disable_torque(self, motor: int, model: str, num_retry: int = 0) -> None:
+        addr, length = get_address(self.model_ctrl_table, model, "Torque_Enable")
+        self._write(addr, length, motor, TorqueMode.DISABLED.value, num_retry=num_retry)
+
+    def enable_torque(self, motors: int | str | list[str] | None = None, num_retry: int = 0) -> None:
+        for motor in self._get_motors_list(motors):
+            self.write("Torque_Enable", motor, TorqueMode.ENABLED.value, num_retry=num_retry)
+
+    def _encode_sign(self, data_name: str, ids_values: dict[int, int]) -> dict[int, int]:
+        for id_ in ids_values:
+            model = self._id_to_model(id_)
+            encoding_table = self.model_encoding_table.get(model)
+            if encoding_table and data_name in encoding_table:
+                n_bytes = encoding_table[data_name]
+                ids_values[id_] = encode_twos_complement(ids_values[id_], n_bytes)
+
+        return ids_values
+
+    def _decode_sign(self, data_name: str, ids_values: dict[int, int]) -> dict[int, int]:
+        for id_ in ids_values:
+            model = self._id_to_model(id_)
+            encoding_table = self.model_encoding_table.get(model)
+            if encoding_table and data_name in encoding_table:
+                n_bytes = encoding_table[data_name]
+                ids_values[id_] = decode_twos_complement(ids_values[id_], n_bytes)
+
+        return ids_values
+
+    def _get_half_turn_homings(self, positions: dict[NameOrID, Value]) -> dict[NameOrID, Value]:
+        """
+        On Dynamixel Motors:
+        Present_Position = Actual_Position + Homing_Offset
+        """
+        half_turn_homings: dict[NameOrID, Value] = {}
+        for motor, pos in positions.items():
+            model = self._get_motor_model(motor)
+            max_res = self.model_resolution_table[model] - 1
+            half_turn_homings[motor] = int(max_res / 2) - pos
+
+        return half_turn_homings
+
+    def _split_into_byte_chunks(self, value: int, length: int) -> list[int]:
+        return _split_into_byte_chunks(value, length)
+
+    def broadcast_ping(self, num_retry: int = 0, raise_on_error: bool = False) -> dict[int, int] | None:
+        for n_try in range(1 + num_retry):
+            data_list, comm = self.packet_handler.broadcastPing(self.port_handler)
+            if self._is_comm_success(comm):
+                break
+            logger.debug(f"Broadcast ping failed on port '{self.port}' ({n_try=})")
+            logger.debug(self.packet_handler.getTxRxResult(comm))
+
+        if not self._is_comm_success(comm):
+            if raise_on_error:
+                raise ConnectionError(self.packet_handler.getTxRxResult(comm))
+
+            return None
+
+        return {id_: data[0] for id_, data in data_list.items()}
diff --git a/lerobot/src/lerobot/motors/dynamixel/tables.py b/lerobot/src/lerobot/motors/dynamixel/tables.py
new file mode 100644
index 0000000000000000000000000000000000000000..5417d8cee712e5abe056525badd75cc9651400dd
--- /dev/null
+++ b/lerobot/src/lerobot/motors/dynamixel/tables.py
@@ -0,0 +1,199 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# TODO(Steven): Consider doing the following:
+# from enum import Enum
+# class MyControlTableKey(Enum):
+#   ID = "ID"
+#   GOAL_SPEED = "Goal_Speed"
+#   ...
+#
+# MY_CONTROL_TABLE ={
+#   MyControlTableKey.ID.value: (5,1)
+#   MyControlTableKey.GOAL_SPEED.value: (46, 2)
+#   ...
+# }
+# This allows me do to:
+# bus.write(MyControlTableKey.GOAL_SPEED, ...)
+# Instead of:
+# bus.write("Goal_Speed", ...)
+# This is important for two reasons:
+# 1. The linter will tell me if I'm trying to use an invalid key, instead of me realizing when I get the RunTimeError
+# 2. We can change the value of the MyControlTableKey enums without impacting the client code
+
+
+# {data_name: (address, size_byte)}
+# https://emanual.robotis.com/docs/en/dxl/x/{MODEL}/#control-table
+X_SERIES_CONTROL_TABLE = {
+    "Model_Number": (0, 2),
+    "Model_Information": (2, 4),
+    "Firmware_Version": (6, 1),
+    "ID": (7, 1),
+    "Baud_Rate": (8, 1),
+    "Return_Delay_Time": (9, 1),
+    "Drive_Mode": (10, 1),
+    "Operating_Mode": (11, 1),
+    "Secondary_ID": (12, 1),
+    "Protocol_Type": (13, 1),
+    "Homing_Offset": (20, 4),
+    "Moving_Threshold": (24, 4),
+    "Temperature_Limit": (31, 1),
+    "Max_Voltage_Limit": (32, 2),
+    "Min_Voltage_Limit": (34, 2),
+    "PWM_Limit": (36, 2),
+    "Current_Limit": (38, 2),
+    "Acceleration_Limit": (40, 4),
+    "Velocity_Limit": (44, 4),
+    "Max_Position_Limit": (48, 4),
+    "Min_Position_Limit": (52, 4),
+    "Shutdown": (63, 1),
+    "Torque_Enable": (64, 1),
+    "LED": (65, 1),
+    "Status_Return_Level": (68, 1),
+    "Registered_Instruction": (69, 1),
+    "Hardware_Error_Status": (70, 1),
+    "Velocity_I_Gain": (76, 2),
+    "Velocity_P_Gain": (78, 2),
+    "Position_D_Gain": (80, 2),
+    "Position_I_Gain": (82, 2),
+    "Position_P_Gain": (84, 2),
+    "Feedforward_2nd_Gain": (88, 2),
+    "Feedforward_1st_Gain": (90, 2),
+    "Bus_Watchdog": (98, 1),
+    "Goal_PWM": (100, 2),
+    "Goal_Current": (102, 2),
+    "Goal_Velocity": (104, 4),
+    "Profile_Acceleration": (108, 4),
+    "Profile_Velocity": (112, 4),
+    "Goal_Position": (116, 4),
+    "Realtime_Tick": (120, 2),
+    "Moving": (122, 1),
+    "Moving_Status": (123, 1),
+    "Present_PWM": (124, 2),
+    "Present_Current": (126, 2),
+    "Present_Velocity": (128, 4),
+    "Present_Position": (132, 4),
+    "Velocity_Trajectory": (136, 4),
+    "Position_Trajectory": (140, 4),
+    "Present_Input_Voltage": (144, 2),
+    "Present_Temperature": (146, 1),
+}
+
+# https://emanual.robotis.com/docs/en/dxl/x/{MODEL}/#baud-rate8
+X_SERIES_BAUDRATE_TABLE = {
+    9_600: 0,
+    57_600: 1,
+    115_200: 2,
+    1_000_000: 3,
+    2_000_000: 4,
+    3_000_000: 5,
+    4_000_000: 6,
+}
+
+# {data_name: size_byte}
+X_SERIES_ENCODINGS_TABLE = {
+    "Homing_Offset": X_SERIES_CONTROL_TABLE["Homing_Offset"][1],
+    "Goal_PWM": X_SERIES_CONTROL_TABLE["Goal_PWM"][1],
+    "Goal_Current": X_SERIES_CONTROL_TABLE["Goal_Current"][1],
+    "Goal_Velocity": X_SERIES_CONTROL_TABLE["Goal_Velocity"][1],
+    "Goal_Position": X_SERIES_CONTROL_TABLE["Goal_Position"][1],
+    "Present_Position": X_SERIES_CONTROL_TABLE["Present_Position"][1],
+    "Present_PWM": X_SERIES_CONTROL_TABLE["Present_PWM"][1],
+    "Present_Current": X_SERIES_CONTROL_TABLE["Present_Current"][1],
+    "Present_Velocity": X_SERIES_CONTROL_TABLE["Present_Velocity"][1],
+}
+
+MODEL_ENCODING_TABLE = {
+    "x_series": X_SERIES_ENCODINGS_TABLE,
+    "xl330-m077": X_SERIES_ENCODINGS_TABLE,
+    "xl330-m288": X_SERIES_ENCODINGS_TABLE,
+    "xl430-w250": X_SERIES_ENCODINGS_TABLE,
+    "xm430-w350": X_SERIES_ENCODINGS_TABLE,
+    "xm540-w270": X_SERIES_ENCODINGS_TABLE,
+    "xc430-w150": X_SERIES_ENCODINGS_TABLE,
+}
+
+# {model: model_resolution}
+# https://emanual.robotis.com/docs/en/dxl/x/{MODEL}/#specifications
+MODEL_RESOLUTION = {
+    "x_series": 4096,
+    "xl330-m077": 4096,
+    "xl330-m288": 4096,
+    "xl430-w250": 4096,
+    "xm430-w350": 4096,
+    "xm540-w270": 4096,
+    "xc430-w150": 4096,
+}
+
+# {model: model_number}
+# https://emanual.robotis.com/docs/en/dxl/x/{MODEL}/#control-table-of-eeprom-area
+MODEL_NUMBER_TABLE = {
+    "xl330-m077": 1190,
+    "xl330-m288": 1200,
+    "xl430-w250": 1060,
+    "xm430-w350": 1020,
+    "xm540-w270": 1120,
+    "xc430-w150": 1070,
+}
+
+# {model: available_operating_modes}
+# https://emanual.robotis.com/docs/en/dxl/x/{MODEL}/#operating-mode11
+MODEL_OPERATING_MODES = {
+    "xl330-m077": [0, 1, 3, 4, 5, 16],
+    "xl330-m288": [0, 1, 3, 4, 5, 16],
+    "xl430-w250": [1, 3, 4, 16],
+    "xm430-w350": [0, 1, 3, 4, 5, 16],
+    "xm540-w270": [0, 1, 3, 4, 5, 16],
+    "xc430-w150": [1, 3, 4, 16],
+}
+
+MODEL_CONTROL_TABLE = {
+    "x_series": X_SERIES_CONTROL_TABLE,
+    "xl330-m077": X_SERIES_CONTROL_TABLE,
+    "xl330-m288": X_SERIES_CONTROL_TABLE,
+    "xl430-w250": X_SERIES_CONTROL_TABLE,
+    "xm430-w350": X_SERIES_CONTROL_TABLE,
+    "xm540-w270": X_SERIES_CONTROL_TABLE,
+    "xc430-w150": X_SERIES_CONTROL_TABLE,
+}
+
+MODEL_BAUDRATE_TABLE = {
+    "x_series": X_SERIES_BAUDRATE_TABLE,
+    "xl330-m077": X_SERIES_BAUDRATE_TABLE,
+    "xl330-m288": X_SERIES_BAUDRATE_TABLE,
+    "xl430-w250": X_SERIES_BAUDRATE_TABLE,
+    "xm430-w350": X_SERIES_BAUDRATE_TABLE,
+    "xm540-w270": X_SERIES_BAUDRATE_TABLE,
+    "xc430-w150": X_SERIES_BAUDRATE_TABLE,
+}
+
+AVAILABLE_BAUDRATES = [
+    9_600,
+    19_200,
+    38_400,
+    57_600,
+    115_200,
+    230_400,
+    460_800,
+    500_000,
+    576_000,
+    921_600,
+    1_000_000,
+    1_152_000,
+    2_000_000,
+    2_500_000,
+    3_000_000,
+    3_500_000,
+    4_000_000,
+]
diff --git a/lerobot/src/lerobot/motors/encoding_utils.py b/lerobot/src/lerobot/motors/encoding_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..195cdbe2c068b020c0b47cb81aa23bc5c3f9d82f
--- /dev/null
+++ b/lerobot/src/lerobot/motors/encoding_utils.py
@@ -0,0 +1,67 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+
+def encode_sign_magnitude(value: int, sign_bit_index: int):
+    """
+    https://en.wikipedia.org/wiki/Signed_number_representations#Sign%E2%80%93magnitude
+    """
+    max_magnitude = (1 << sign_bit_index) - 1
+    magnitude = abs(value)
+    if magnitude > max_magnitude:
+        raise ValueError(f"Magnitude {magnitude} exceeds {max_magnitude} (max for {sign_bit_index=})")
+
+    direction_bit = 1 if value < 0 else 0
+    return (direction_bit << sign_bit_index) | magnitude
+
+
+def decode_sign_magnitude(encoded_value: int, sign_bit_index: int):
+    """
+    https://en.wikipedia.org/wiki/Signed_number_representations#Sign%E2%80%93magnitude
+    """
+    direction_bit = (encoded_value >> sign_bit_index) & 1
+    magnitude_mask = (1 << sign_bit_index) - 1
+    magnitude = encoded_value & magnitude_mask
+    return -magnitude if direction_bit else magnitude
+
+
+def encode_twos_complement(value: int, n_bytes: int):
+    """
+    https://en.wikipedia.org/wiki/Signed_number_representations#Two%27s_complement
+    """
+
+    bit_width = n_bytes * 8
+    min_val = -(1 << (bit_width - 1))
+    max_val = (1 << (bit_width - 1)) - 1
+
+    if not (min_val <= value <= max_val):
+        raise ValueError(
+            f"Value {value} out of range for {n_bytes}-byte two's complement: [{min_val}, {max_val}]"
+        )
+
+    if value >= 0:
+        return value
+
+    return (1 << bit_width) + value
+
+
+def decode_twos_complement(value: int, n_bytes: int) -> int:
+    """
+    https://en.wikipedia.org/wiki/Signed_number_representations#Two%27s_complement
+    """
+    bits = n_bytes * 8
+    sign_bit = 1 << (bits - 1)
+    if value & sign_bit:
+        value -= 1 << bits
+    return value
diff --git a/lerobot/src/lerobot/motors/feetech/__init__.py b/lerobot/src/lerobot/motors/feetech/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..75da2d221270cdb5a845fbca8eed8489756e9fd6
--- /dev/null
+++ b/lerobot/src/lerobot/motors/feetech/__init__.py
@@ -0,0 +1,18 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .feetech import DriveMode, FeetechMotorsBus, OperatingMode, TorqueMode
+from .tables import *
diff --git a/lerobot/src/lerobot/motors/feetech/feetech.py b/lerobot/src/lerobot/motors/feetech/feetech.py
new file mode 100644
index 0000000000000000000000000000000000000000..58a65310deac0776f571ebefefa0838c92d23c08
--- /dev/null
+++ b/lerobot/src/lerobot/motors/feetech/feetech.py
@@ -0,0 +1,454 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+from copy import deepcopy
+from enum import Enum
+from pprint import pformat
+
+from ..encoding_utils import decode_sign_magnitude, encode_sign_magnitude
+from ..motors_bus import Motor, MotorCalibration, NameOrID, SerialMotorsBus, Value, get_address
+from .tables import (
+    FIRMWARE_MAJOR_VERSION,
+    FIRMWARE_MINOR_VERSION,
+    MODEL_BAUDRATE_TABLE,
+    MODEL_CONTROL_TABLE,
+    MODEL_ENCODING_TABLE,
+    MODEL_NUMBER,
+    MODEL_NUMBER_TABLE,
+    MODEL_PROTOCOL,
+    MODEL_RESOLUTION,
+    SCAN_BAUDRATES,
+)
+
+DEFAULT_PROTOCOL_VERSION = 0
+DEFAULT_BAUDRATE = 1_000_000
+DEFAULT_TIMEOUT_MS = 1000
+
+NORMALIZED_DATA = ["Goal_Position", "Present_Position"]
+
+logger = logging.getLogger(__name__)
+
+
+class OperatingMode(Enum):
+    # position servo mode
+    POSITION = 0
+    # The motor is in constant speed mode, which is controlled by parameter 0x2e, and the highest bit 15 is
+    # the direction bit
+    VELOCITY = 1
+    # PWM open-loop speed regulation mode, with parameter 0x2c running time parameter control, bit11 as
+    # direction bit
+    PWM = 2
+    # In step servo mode, the number of step progress is represented by parameter 0x2a, and the highest bit 15
+    # is the direction bit
+    STEP = 3
+
+
+class DriveMode(Enum):
+    NON_INVERTED = 0
+    INVERTED = 1
+
+
+class TorqueMode(Enum):
+    ENABLED = 1
+    DISABLED = 0
+
+
+def _split_into_byte_chunks(value: int, length: int) -> list[int]:
+    import scservo_sdk as scs
+
+    if length == 1:
+        data = [value]
+    elif length == 2:
+        data = [scs.SCS_LOBYTE(value), scs.SCS_HIBYTE(value)]
+    elif length == 4:
+        data = [
+            scs.SCS_LOBYTE(scs.SCS_LOWORD(value)),
+            scs.SCS_HIBYTE(scs.SCS_LOWORD(value)),
+            scs.SCS_LOBYTE(scs.SCS_HIWORD(value)),
+            scs.SCS_HIBYTE(scs.SCS_HIWORD(value)),
+        ]
+    return data
+
+
+def patch_setPacketTimeout(self, packet_length):  # noqa: N802
+    """
+    HACK: This patches the PortHandler behavior to set the correct packet timeouts.
+
+    It fixes https://gitee.com/ftservo/SCServoSDK/issues/IBY2S6
+    The bug is fixed on the official Feetech SDK repo (https://gitee.com/ftservo/FTServo_Python)
+    but because that version is not published on PyPI, we rely on the (unofficial) on that is, which needs
+    patching.
+    """
+    self.packet_start_time = self.getCurrentTime()
+    self.packet_timeout = (self.tx_time_per_byte * packet_length) + (self.tx_time_per_byte * 3.0) + 50
+
+
+class FeetechMotorsBus(SerialMotorsBus):
+    """
+    The FeetechMotorsBus class allows to efficiently read and write to the attached motors. It relies on the
+    python feetech sdk to communicate with the motors, which is itself based on the dynamixel sdk.
+    """
+
+    apply_drive_mode = True
+    available_baudrates = deepcopy(SCAN_BAUDRATES)
+    default_baudrate = DEFAULT_BAUDRATE
+    default_timeout = DEFAULT_TIMEOUT_MS
+    model_baudrate_table = deepcopy(MODEL_BAUDRATE_TABLE)
+    model_ctrl_table = deepcopy(MODEL_CONTROL_TABLE)
+    model_encoding_table = deepcopy(MODEL_ENCODING_TABLE)
+    model_number_table = deepcopy(MODEL_NUMBER_TABLE)
+    model_resolution_table = deepcopy(MODEL_RESOLUTION)
+    normalized_data = deepcopy(NORMALIZED_DATA)
+
+    def __init__(
+        self,
+        port: str,
+        motors: dict[str, Motor],
+        calibration: dict[str, MotorCalibration] | None = None,
+        protocol_version: int = DEFAULT_PROTOCOL_VERSION,
+    ):
+        super().__init__(port, motors, calibration)
+        self.protocol_version = protocol_version
+        self._assert_same_protocol()
+        import scservo_sdk as scs
+
+        self.port_handler = scs.PortHandler(self.port)
+        # HACK: monkeypatch
+        self.port_handler.setPacketTimeout = patch_setPacketTimeout.__get__(  # type: ignore[method-assign]
+            self.port_handler, scs.PortHandler
+        )
+        self.packet_handler = scs.PacketHandler(protocol_version)
+        self.sync_reader = scs.GroupSyncRead(self.port_handler, self.packet_handler, 0, 0)
+        self.sync_writer = scs.GroupSyncWrite(self.port_handler, self.packet_handler, 0, 0)
+        self._comm_success = scs.COMM_SUCCESS
+        self._no_error = 0x00
+
+        if any(MODEL_PROTOCOL[model] != self.protocol_version for model in self.models):
+            raise ValueError(f"Some motors are incompatible with protocol_version={self.protocol_version}")
+
+    def _assert_same_protocol(self) -> None:
+        if any(MODEL_PROTOCOL[model] != self.protocol_version for model in self.models):
+            raise RuntimeError("Some motors use an incompatible protocol.")
+
+    def _assert_protocol_is_compatible(self, instruction_name: str) -> None:
+        if instruction_name == "sync_read" and self.protocol_version == 1:
+            raise NotImplementedError(
+                "'Sync Read' is not available with Feetech motors using Protocol 1. Use 'Read' sequentially instead."
+            )
+        if instruction_name == "broadcast_ping" and self.protocol_version == 1:
+            raise NotImplementedError(
+                "'Broadcast Ping' is not available with Feetech motors using Protocol 1. Use 'Ping' sequentially instead."
+            )
+
+    def _assert_same_firmware(self) -> None:
+        firmware_versions = self._read_firmware_version(self.ids, raise_on_error=True)
+        if len(set(firmware_versions.values())) != 1:
+            raise RuntimeError(
+                "Some Motors use different firmware versions:"
+                f"\n{pformat(firmware_versions)}\n"
+                "Update their firmware first using Feetech's software. "
+                "Visit https://www.feetechrc.com/software."
+            )
+
+    def _handshake(self) -> None:
+        self._assert_motors_exist()
+        self._assert_same_firmware()
+
+    def _find_single_motor(self, motor: str, initial_baudrate: int | None = None) -> tuple[int, int]:
+        if self.protocol_version == 0:
+            return self._find_single_motor_p0(motor, initial_baudrate)
+        else:
+            return self._find_single_motor_p1(motor, initial_baudrate)
+
+    def _find_single_motor_p0(self, motor: str, initial_baudrate: int | None = None) -> tuple[int, int]:
+        model = self.motors[motor].model
+        search_baudrates = (
+            [initial_baudrate] if initial_baudrate is not None else self.model_baudrate_table[model]
+        )
+        expected_model_nb = self.model_number_table[model]
+
+        for baudrate in search_baudrates:
+            self.set_baudrate(baudrate)
+            id_model = self.broadcast_ping()
+            if id_model:
+                found_id, found_model = next(iter(id_model.items()))
+                if found_model != expected_model_nb:
+                    raise RuntimeError(
+                        f"Found one motor on {baudrate=} with id={found_id} but it has a "
+                        f"model number '{found_model}' different than the one expected: '{expected_model_nb}'. "
+                        f"Make sure you are connected only connected to the '{motor}' motor (model '{model}')."
+                    )
+                return baudrate, found_id
+
+        raise RuntimeError(f"Motor '{motor}' (model '{model}') was not found. Make sure it is connected.")
+
+    def _find_single_motor_p1(self, motor: str, initial_baudrate: int | None = None) -> tuple[int, int]:
+        import scservo_sdk as scs
+
+        model = self.motors[motor].model
+        search_baudrates = (
+            [initial_baudrate] if initial_baudrate is not None else self.model_baudrate_table[model]
+        )
+        expected_model_nb = self.model_number_table[model]
+
+        for baudrate in search_baudrates:
+            self.set_baudrate(baudrate)
+            for id_ in range(scs.MAX_ID + 1):
+                found_model = self.ping(id_)
+                if found_model is not None:
+                    if found_model != expected_model_nb:
+                        raise RuntimeError(
+                            f"Found one motor on {baudrate=} with id={id_} but it has a "
+                            f"model number '{found_model}' different than the one expected: '{expected_model_nb}'. "
+                            f"Make sure you are connected only connected to the '{motor}' motor (model '{model}')."
+                        )
+                    return baudrate, id_
+
+        raise RuntimeError(f"Motor '{motor}' (model '{model}') was not found. Make sure it is connected.")
+
+    def configure_motors(self, return_delay_time=0, maximum_acceleration=254, acceleration=254) -> None:
+        for motor in self.motors:
+            # By default, Feetech motors have a 500µs delay response time (corresponding to a value of 250 on
+            # the 'Return_Delay_Time' address). We ensure this is reduced to the minimum of 2µs (value of 0).
+            self.write("Return_Delay_Time", motor, return_delay_time)
+            # Set 'Maximum_Acceleration' to 254 to speedup acceleration and deceleration of the motors.
+            if self.protocol_version == 0:
+                self.write("Maximum_Acceleration", motor, maximum_acceleration)
+            self.write("Acceleration", motor, acceleration)
+
+    @property
+    def is_calibrated(self) -> bool:
+        motors_calibration = self.read_calibration()
+        if set(motors_calibration) != set(self.calibration):
+            return False
+
+        same_ranges = all(
+            self.calibration[motor].range_min == cal.range_min
+            and self.calibration[motor].range_max == cal.range_max
+            for motor, cal in motors_calibration.items()
+        )
+        if self.protocol_version == 1:
+            return same_ranges
+
+        same_offsets = all(
+            self.calibration[motor].homing_offset == cal.homing_offset
+            for motor, cal in motors_calibration.items()
+        )
+        return same_ranges and same_offsets
+
+    def read_calibration(self) -> dict[str, MotorCalibration]:
+        offsets, mins, maxes = {}, {}, {}
+        for motor in self.motors:
+            mins[motor] = self.read("Min_Position_Limit", motor, normalize=False)
+            maxes[motor] = self.read("Max_Position_Limit", motor, normalize=False)
+            offsets[motor] = (
+                self.read("Homing_Offset", motor, normalize=False) if self.protocol_version == 0 else 0
+            )
+
+        calibration = {}
+        for motor, m in self.motors.items():
+            calibration[motor] = MotorCalibration(
+                id=m.id,
+                drive_mode=0,
+                homing_offset=int(offsets[motor]),
+                range_min=int(mins[motor]),
+                range_max=int(maxes[motor]),
+            )
+
+        return calibration
+
+    def write_calibration(self, calibration_dict: dict[str, MotorCalibration], cache: bool = True) -> None:
+        for motor, calibration in calibration_dict.items():
+            if self.protocol_version == 0:
+                self.write("Homing_Offset", motor, calibration.homing_offset)
+            self.write("Min_Position_Limit", motor, calibration.range_min)
+            self.write("Max_Position_Limit", motor, calibration.range_max)
+
+        if cache:
+            self.calibration = calibration_dict
+
+    def _get_half_turn_homings(self, positions: dict[NameOrID, Value]) -> dict[NameOrID, Value]:
+        """
+        On Feetech Motors:
+        Present_Position = Actual_Position - Homing_Offset
+        """
+        half_turn_homings: dict[NameOrID, Value] = {}
+        for motor, pos in positions.items():
+            model = self._get_motor_model(motor)
+            max_res = self.model_resolution_table[model] - 1
+            half_turn_homings[motor] = pos - int(max_res / 2)
+
+        return half_turn_homings
+
+    def disable_torque(self, motors: int | str | list[str] | None = None, num_retry: int = 0) -> None:
+        for motor in self._get_motors_list(motors):
+            self.write("Torque_Enable", motor, TorqueMode.DISABLED.value, num_retry=num_retry)
+            self.write("Lock", motor, 0, num_retry=num_retry)
+
+    def _disable_torque(self, motor: int, model: str, num_retry: int = 0) -> None:
+        addr, length = get_address(self.model_ctrl_table, model, "Torque_Enable")
+        self._write(addr, length, motor, TorqueMode.DISABLED.value, num_retry=num_retry)
+        addr, length = get_address(self.model_ctrl_table, model, "Lock")
+        self._write(addr, length, motor, 0, num_retry=num_retry)
+
+    def enable_torque(self, motors: int | str | list[str] | None = None, num_retry: int = 0) -> None:
+        for motor in self._get_motors_list(motors):
+            self.write("Torque_Enable", motor, TorqueMode.ENABLED.value, num_retry=num_retry)
+            self.write("Lock", motor, 1, num_retry=num_retry)
+
+    def _encode_sign(self, data_name: str, ids_values: dict[int, int]) -> dict[int, int]:
+        for id_ in ids_values:
+            model = self._id_to_model(id_)
+            encoding_table = self.model_encoding_table.get(model)
+            if encoding_table and data_name in encoding_table:
+                sign_bit = encoding_table[data_name]
+                ids_values[id_] = encode_sign_magnitude(ids_values[id_], sign_bit)
+
+        return ids_values
+
+    def _decode_sign(self, data_name: str, ids_values: dict[int, int]) -> dict[int, int]:
+        for id_ in ids_values:
+            model = self._id_to_model(id_)
+            encoding_table = self.model_encoding_table.get(model)
+            if encoding_table and data_name in encoding_table:
+                sign_bit = encoding_table[data_name]
+                ids_values[id_] = decode_sign_magnitude(ids_values[id_], sign_bit)
+
+        return ids_values
+
+    def _split_into_byte_chunks(self, value: int, length: int) -> list[int]:
+        return _split_into_byte_chunks(value, length)
+
+    def _broadcast_ping(self) -> tuple[dict[int, int], int]:
+        import scservo_sdk as scs
+
+        data_list: dict[int, int] = {}
+
+        status_length = 6
+
+        rx_length = 0
+        wait_length = status_length * scs.MAX_ID
+
+        txpacket = [0] * 6
+
+        tx_time_per_byte = (1000.0 / self.port_handler.getBaudRate()) * 10.0
+
+        txpacket[scs.PKT_ID] = scs.BROADCAST_ID
+        txpacket[scs.PKT_LENGTH] = 2
+        txpacket[scs.PKT_INSTRUCTION] = scs.INST_PING
+
+        result = self.packet_handler.txPacket(self.port_handler, txpacket)
+        if result != scs.COMM_SUCCESS:
+            self.port_handler.is_using = False
+            return data_list, result
+
+        # set rx timeout
+        self.port_handler.setPacketTimeoutMillis((wait_length * tx_time_per_byte) + (3.0 * scs.MAX_ID) + 16.0)
+
+        rxpacket = []
+        while not self.port_handler.isPacketTimeout() and rx_length < wait_length:
+            rxpacket += self.port_handler.readPort(wait_length - rx_length)
+            rx_length = len(rxpacket)
+
+        self.port_handler.is_using = False
+
+        if rx_length == 0:
+            return data_list, scs.COMM_RX_TIMEOUT
+
+        while True:
+            if rx_length < status_length:
+                return data_list, scs.COMM_RX_CORRUPT
+
+            # find packet header
+            for idx in range(0, (rx_length - 1)):
+                if (rxpacket[idx] == 0xFF) and (rxpacket[idx + 1] == 0xFF):
+                    break
+
+            if idx == 0:  # found at the beginning of the packet
+                # calculate checksum
+                checksum = 0
+                for idx in range(2, status_length - 1):  # except header & checksum
+                    checksum += rxpacket[idx]
+
+                checksum = ~checksum & 0xFF
+                if rxpacket[status_length - 1] == checksum:
+                    result = scs.COMM_SUCCESS
+                    data_list[rxpacket[scs.PKT_ID]] = rxpacket[scs.PKT_ERROR]
+
+                    del rxpacket[0:status_length]
+                    rx_length = rx_length - status_length
+
+                    if rx_length == 0:
+                        return data_list, result
+                else:
+                    result = scs.COMM_RX_CORRUPT
+                    # remove header (0xFF 0xFF)
+                    del rxpacket[0:2]
+                    rx_length = rx_length - 2
+            else:
+                # remove unnecessary packets
+                del rxpacket[0:idx]
+                rx_length = rx_length - idx
+
+    def broadcast_ping(self, num_retry: int = 0, raise_on_error: bool = False) -> dict[int, int] | None:
+        self._assert_protocol_is_compatible("broadcast_ping")
+        for n_try in range(1 + num_retry):
+            ids_status, comm = self._broadcast_ping()
+            if self._is_comm_success(comm):
+                break
+            logger.debug(f"Broadcast ping failed on port '{self.port}' ({n_try=})")
+            logger.debug(self.packet_handler.getTxRxResult(comm))
+
+        if not self._is_comm_success(comm):
+            if raise_on_error:
+                raise ConnectionError(self.packet_handler.getTxRxResult(comm))
+            return None
+
+        ids_errors = {id_: status for id_, status in ids_status.items() if self._is_error(status)}
+        if ids_errors:
+            display_dict = {id_: self.packet_handler.getRxPacketError(err) for id_, err in ids_errors.items()}
+            logger.error(f"Some motors found returned an error status:\n{pformat(display_dict, indent=4)}")
+
+        return self._read_model_number(list(ids_status), raise_on_error)
+
+    def _read_firmware_version(self, motor_ids: list[int], raise_on_error: bool = False) -> dict[int, str]:
+        firmware_versions = {}
+        for id_ in motor_ids:
+            firm_ver_major, comm, error = self._read(
+                *FIRMWARE_MAJOR_VERSION, id_, raise_on_error=raise_on_error
+            )
+            if not self._is_comm_success(comm) or self._is_error(error):
+                continue
+
+            firm_ver_minor, comm, error = self._read(
+                *FIRMWARE_MINOR_VERSION, id_, raise_on_error=raise_on_error
+            )
+            if not self._is_comm_success(comm) or self._is_error(error):
+                continue
+
+            firmware_versions[id_] = f"{firm_ver_major}.{firm_ver_minor}"
+
+        return firmware_versions
+
+    def _read_model_number(self, motor_ids: list[int], raise_on_error: bool = False) -> dict[int, int]:
+        model_numbers = {}
+        for id_ in motor_ids:
+            model_nb, comm, error = self._read(*MODEL_NUMBER, id_, raise_on_error=raise_on_error)
+            if not self._is_comm_success(comm) or self._is_error(error):
+                continue
+
+            model_numbers[id_] = model_nb
+
+        return model_numbers
diff --git a/lerobot/src/lerobot/motors/feetech/tables.py b/lerobot/src/lerobot/motors/feetech/tables.py
new file mode 100644
index 0000000000000000000000000000000000000000..56500e527d310af1463523056f99cfb6b8de675c
--- /dev/null
+++ b/lerobot/src/lerobot/motors/feetech/tables.py
@@ -0,0 +1,257 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+FIRMWARE_MAJOR_VERSION = (0, 1)
+FIRMWARE_MINOR_VERSION = (1, 1)
+MODEL_NUMBER = (3, 2)
+
+# TODO(Steven): Consider doing the following:
+# from enum import Enum
+# class MyControlTableKey(Enum):
+#   ID = "ID"
+#   GOAL_SPEED = "Goal_Speed"
+#   ...
+#
+# MY_CONTROL_TABLE ={
+#   MyControlTableKey.ID.value: (5,1)
+#   MyControlTableKey.GOAL_SPEED.value: (46, 2)
+#   ...
+# }
+# This allows me do to:
+# bus.write(MyControlTableKey.GOAL_SPEED, ...)
+# Instead of:
+# bus.write("Goal_Speed", ...)
+# This is important for two reasons:
+# 1. The linter will tell me if I'm trying to use an invalid key, instead of me realizing when I get the RunTimeError
+# 2. We can change the value of the MyControlTableKey enums without impacting the client code
+
+# data_name: (address, size_byte)
+# http://doc.feetech.cn/#/prodinfodownload?srcType=FT-SMS-STS-emanual-229f4476422d4059abfb1cb0
+STS_SMS_SERIES_CONTROL_TABLE = {
+    # EPROM
+    "Firmware_Major_Version": FIRMWARE_MAJOR_VERSION,  # read-only
+    "Firmware_Minor_Version": FIRMWARE_MINOR_VERSION,  # read-only
+    "Model_Number": MODEL_NUMBER,  # read-only
+    "ID": (5, 1),
+    "Baud_Rate": (6, 1),
+    "Return_Delay_Time": (7, 1),
+    "Response_Status_Level": (8, 1),
+    "Min_Position_Limit": (9, 2),
+    "Max_Position_Limit": (11, 2),
+    "Max_Temperature_Limit": (13, 1),
+    "Max_Voltage_Limit": (14, 1),
+    "Min_Voltage_Limit": (15, 1),
+    "Max_Torque_Limit": (16, 2),
+    "Phase": (18, 1),
+    "Unloading_Condition": (19, 1),
+    "LED_Alarm_Condition": (20, 1),
+    "P_Coefficient": (21, 1),
+    "D_Coefficient": (22, 1),
+    "I_Coefficient": (23, 1),
+    "Minimum_Startup_Force": (24, 2),
+    "CW_Dead_Zone": (26, 1),
+    "CCW_Dead_Zone": (27, 1),
+    "Protection_Current": (28, 2),
+    "Angular_Resolution": (30, 1),
+    "Homing_Offset": (31, 2),
+    "Operating_Mode": (33, 1),
+    "Protective_Torque": (34, 1),
+    "Protection_Time": (35, 1),
+    "Overload_Torque": (36, 1),
+    "Velocity_closed_loop_P_proportional_coefficient": (37, 1),
+    "Over_Current_Protection_Time": (38, 1),
+    "Velocity_closed_loop_I_integral_coefficient": (39, 1),
+    # SRAM
+    "Torque_Enable": (40, 1),
+    "Acceleration": (41, 1),
+    "Goal_Position": (42, 2),
+    "Goal_Time": (44, 2),
+    "Goal_Velocity": (46, 2),
+    "Torque_Limit": (48, 2),
+    "Lock": (55, 1),
+    "Present_Position": (56, 2),  # read-only
+    "Present_Velocity": (58, 2),  # read-only
+    "Present_Load": (60, 2),  # read-only
+    "Present_Voltage": (62, 1),  # read-only
+    "Present_Temperature": (63, 1),  # read-only
+    "Status": (65, 1),  # read-only
+    "Moving": (66, 1),  # read-only
+    "Present_Current": (69, 2),  # read-only
+    "Goal_Position_2": (71, 2),  # read-only
+    # Factory
+    "Moving_Velocity": (80, 1),
+    "Moving_Velocity_Threshold": (80, 1),
+    "DTs": (81, 1),  # (ms)
+    "Velocity_Unit_factor": (82, 1),
+    "Hts": (83, 1),  # (ns) valid for firmware >= 2.54, other versions keep 0
+    "Maximum_Velocity_Limit": (84, 1),
+    "Maximum_Acceleration": (85, 1),
+    "Acceleration_Multiplier ": (86, 1),  # Acceleration multiplier in effect when acceleration is 0
+}
+
+# http://doc.feetech.cn/#/prodinfodownload?srcType=FT-SCSCL-emanual-cbcc8ab2e3384282a01d4bf3
+SCS_SERIES_CONTROL_TABLE = {
+    # EPROM
+    "Firmware_Major_Version": FIRMWARE_MAJOR_VERSION,  # read-only
+    "Firmware_Minor_Version": FIRMWARE_MINOR_VERSION,  # read-only
+    "Model_Number": MODEL_NUMBER,  # read-only
+    "ID": (5, 1),
+    "Baud_Rate": (6, 1),
+    "Return_Delay_Time": (7, 1),
+    "Response_Status_Level": (8, 1),
+    "Min_Position_Limit": (9, 2),
+    "Max_Position_Limit": (11, 2),
+    "Max_Temperature_Limit": (13, 1),
+    "Max_Voltage_Limit": (14, 1),
+    "Min_Voltage_Limit": (15, 1),
+    "Max_Torque_Limit": (16, 2),
+    "Phase": (18, 1),
+    "Unloading_Condition": (19, 1),
+    "LED_Alarm_Condition": (20, 1),
+    "P_Coefficient": (21, 1),
+    "D_Coefficient": (22, 1),
+    "I_Coefficient": (23, 1),
+    "Minimum_Startup_Force": (24, 2),
+    "CW_Dead_Zone": (26, 1),
+    "CCW_Dead_Zone": (27, 1),
+    "Protective_Torque": (37, 1),
+    "Protection_Time": (38, 1),
+    # SRAM
+    "Torque_Enable": (40, 1),
+    "Acceleration": (41, 1),
+    "Goal_Position": (42, 2),
+    "Running_Time": (44, 2),
+    "Goal_Velocity": (46, 2),
+    "Lock": (48, 1),
+    "Present_Position": (56, 2),  # read-only
+    "Present_Velocity": (58, 2),  # read-only
+    "Present_Load": (60, 2),  # read-only
+    "Present_Voltage": (62, 1),  # read-only
+    "Present_Temperature": (63, 1),  # read-only
+    "Sync_Write_Flag": (64, 1),  # read-only
+    "Status": (65, 1),  # read-only
+    "Moving": (66, 1),  # read-only
+    # Factory
+    "PWM_Maximum_Step": (78, 1),
+    "Moving_Velocity_Threshold*50": (79, 1),
+    "DTs": (80, 1),  # (ms)
+    "Minimum_Velocity_Limit*50": (81, 1),
+    "Maximum_Velocity_Limit*50": (82, 1),
+    "Acceleration_2": (83, 1),  # don't know what that is
+}
+
+STS_SMS_SERIES_BAUDRATE_TABLE = {
+    1_000_000: 0,
+    500_000: 1,
+    250_000: 2,
+    128_000: 3,
+    115_200: 4,
+    57_600: 5,
+    38_400: 6,
+    19_200: 7,
+}
+
+SCS_SERIES_BAUDRATE_TABLE = {
+    1_000_000: 0,
+    500_000: 1,
+    250_000: 2,
+    128_000: 3,
+    115_200: 4,
+    57_600: 5,
+    38_400: 6,
+    19_200: 7,
+}
+
+MODEL_CONTROL_TABLE = {
+    "sts_series": STS_SMS_SERIES_CONTROL_TABLE,
+    "scs_series": SCS_SERIES_CONTROL_TABLE,
+    "sms_series": STS_SMS_SERIES_CONTROL_TABLE,
+    "sts3215": STS_SMS_SERIES_CONTROL_TABLE,
+    "sts3250": STS_SMS_SERIES_CONTROL_TABLE,
+    "scs0009": SCS_SERIES_CONTROL_TABLE,
+    "sm8512bl": STS_SMS_SERIES_CONTROL_TABLE,
+}
+
+MODEL_RESOLUTION = {
+    "sts_series": 4096,
+    "sms_series": 4096,
+    "scs_series": 1024,
+    "sts3215": 4096,
+    "sts3250": 4096,
+    "sm8512bl": 4096,
+    "scs0009": 1024,
+}
+
+MODEL_BAUDRATE_TABLE = {
+    "sts_series": STS_SMS_SERIES_BAUDRATE_TABLE,
+    "sms_series": STS_SMS_SERIES_BAUDRATE_TABLE,
+    "scs_series": SCS_SERIES_BAUDRATE_TABLE,
+    "sm8512bl": STS_SMS_SERIES_BAUDRATE_TABLE,
+    "sts3215": STS_SMS_SERIES_BAUDRATE_TABLE,
+    "sts3250": STS_SMS_SERIES_BAUDRATE_TABLE,
+    "scs0009": SCS_SERIES_BAUDRATE_TABLE,
+}
+
+# Sign-Magnitude encoding bits
+STS_SMS_SERIES_ENCODINGS_TABLE = {
+    "Present_Load": 10,
+    "Homing_Offset": 11,
+    "Goal_Position": 15,
+    "Goal_Velocity": 15,
+    "Goal_Speed": 15,
+    "Present_Position": 15,
+    "Present_Velocity": 15,
+    "Present_Speed": 15,
+}
+
+MODEL_ENCODING_TABLE = {
+    "sts_series": STS_SMS_SERIES_ENCODINGS_TABLE,
+    "sms_series": STS_SMS_SERIES_ENCODINGS_TABLE,
+    "scs_series": {},
+    "sts3215": STS_SMS_SERIES_ENCODINGS_TABLE,
+    "sts3250": STS_SMS_SERIES_ENCODINGS_TABLE,
+    "sm8512bl": STS_SMS_SERIES_ENCODINGS_TABLE,
+    "scs0009": {},
+}
+
+SCAN_BAUDRATES = [
+    4_800,
+    9_600,
+    14_400,
+    19_200,
+    38_400,
+    57_600,
+    115_200,
+    128_000,
+    250_000,
+    500_000,
+    1_000_000,
+]
+
+MODEL_NUMBER_TABLE = {
+    "sts3215": 777,
+    "sts3250": 2825,
+    "sm8512bl": 11272,
+    "scs0009": 1284,
+}
+
+MODEL_PROTOCOL = {
+    "sts_series": 0,
+    "sms_series": 0,
+    "scs_series": 1,
+    "sts3215": 0,
+    "sts3250": 0,
+    "sm8512bl": 0,
+    "scs0009": 1,
+}
diff --git a/lerobot/src/lerobot/motors/motors_bus.py b/lerobot/src/lerobot/motors/motors_bus.py
new file mode 100644
index 0000000000000000000000000000000000000000..509f5e95fcc064fb57c9aedb0348637ef068bbd7
--- /dev/null
+++ b/lerobot/src/lerobot/motors/motors_bus.py
@@ -0,0 +1,1280 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# ruff: noqa: N802
+# This noqa is for the Protocols classes: PortHandler, PacketHandler GroupSyncRead/Write
+# TODO(aliberts): Add block noqa when feature below is available
+# https://github.com/astral-sh/ruff/issues/3711
+
+from __future__ import annotations
+
+import abc
+import logging
+from collections.abc import Sequence
+from contextlib import contextmanager
+from dataclasses import dataclass
+from enum import Enum
+from functools import cached_property
+from pprint import pformat
+from typing import Protocol
+
+import serial
+from deepdiff import DeepDiff
+from tqdm import tqdm
+
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.utils import enter_pressed, move_cursor_up
+
+type NameOrID = str | int
+type Value = int | float
+
+logger = logging.getLogger(__name__)
+
+
+class MotorsBusBase(abc.ABC):
+    """
+    Base class for all motor bus implementations.
+
+    This is a minimal interface that all motor buses must implement, regardless of their
+    communication protocol (serial, CAN, etc.).
+    """
+
+    def __init__(
+        self,
+        port: str,
+        motors: dict[str, Motor],
+        calibration: dict[str, MotorCalibration] | None = None,
+    ):
+        self.port = port
+        self.motors = motors
+        self.calibration = calibration if calibration else {}
+
+    @abc.abstractmethod
+    def connect(self, handshake: bool = True) -> None:
+        """Establish connection to the motors."""
+        pass
+
+    @abc.abstractmethod
+    def disconnect(self, disable_torque: bool = True) -> None:
+        """Disconnect from the motors."""
+        pass
+
+    @property
+    @abc.abstractmethod
+    def is_connected(self) -> bool:
+        """Check if connected to the motors."""
+        pass
+
+    @abc.abstractmethod
+    def read(self, data_name: str, motor: str) -> Value:
+        """Read a value from a single motor."""
+        pass
+
+    @abc.abstractmethod
+    def write(self, data_name: str, motor: str, value: Value) -> None:
+        """Write a value to a single motor."""
+        pass
+
+    @abc.abstractmethod
+    def sync_read(self, data_name: str, motors: str | list[str] | None = None) -> dict[str, Value]:
+        """Read a value from multiple motors."""
+        pass
+
+    @abc.abstractmethod
+    def sync_write(self, data_name: str, values: dict[str, Value]) -> None:
+        """Write values to multiple motors."""
+        pass
+
+    @abc.abstractmethod
+    def enable_torque(self, motors: str | list[str] | None = None, num_retry: int = 0) -> None:
+        """Enable torque on selected motors."""
+        pass
+
+    @abc.abstractmethod
+    def disable_torque(self, motors: str | list[str] | None = None, num_retry: int = 0) -> None:
+        """Disable torque on selected motors."""
+        pass
+
+    @abc.abstractmethod
+    def read_calibration(self) -> dict[str, MotorCalibration]:
+        """Read calibration parameters from the motors."""
+        pass
+
+    @abc.abstractmethod
+    def write_calibration(self, calibration_dict: dict[str, MotorCalibration], cache: bool = True) -> None:
+        """Write calibration parameters to the motors."""
+        pass
+
+
+def get_ctrl_table(model_ctrl_table: dict[str, dict], model: str) -> dict[str, tuple[int, int]]:
+    ctrl_table = model_ctrl_table.get(model)
+    if ctrl_table is None:
+        raise KeyError(f"Control table for {model=} not found.")
+    return ctrl_table
+
+
+def get_address(model_ctrl_table: dict[str, dict], model: str, data_name: str) -> tuple[int, int]:
+    ctrl_table = get_ctrl_table(model_ctrl_table, model)
+    addr_bytes = ctrl_table.get(data_name)
+    if addr_bytes is None:
+        raise KeyError(f"Address for '{data_name}' not found in {model} control table.")
+    return addr_bytes
+
+
+def assert_same_address(model_ctrl_table: dict[str, dict], motor_models: list[str], data_name: str) -> None:
+    all_addr = []
+    all_bytes = []
+    for model in motor_models:
+        addr, bytes = get_address(model_ctrl_table, model, data_name)
+        all_addr.append(addr)
+        all_bytes.append(bytes)
+
+    if len(set(all_addr)) != 1:
+        raise NotImplementedError(
+            f"At least two motor models use a different address for `data_name`='{data_name}'"
+            f"({list(zip(motor_models, all_addr, strict=False))})."
+        )
+
+    if len(set(all_bytes)) != 1:
+        raise NotImplementedError(
+            f"At least two motor models use a different bytes representation for `data_name`='{data_name}'"
+            f"({list(zip(motor_models, all_bytes, strict=False))})."
+        )
+
+
+class MotorNormMode(str, Enum):
+    RANGE_0_100 = "range_0_100"
+    RANGE_M100_100 = "range_m100_100"
+    DEGREES = "degrees"
+
+
+@dataclass
+class MotorCalibration:
+    id: int
+    drive_mode: int
+    homing_offset: int
+    range_min: int
+    range_max: int
+
+
+@dataclass
+class Motor:
+    id: int
+    model: str
+    norm_mode: MotorNormMode
+    motor_type_str: str | None = None
+    recv_id: int | None = None
+
+
+class PortHandler(Protocol):
+    is_open: bool
+    baudrate: int
+    packet_start_time: float
+    packet_timeout: float
+    tx_time_per_byte: float
+    is_using: bool
+    port_name: str
+    ser: serial.Serial
+
+    def __init__(self, port_name: str) -> None: ...
+
+    def openPort(self): ...
+    def closePort(self): ...
+    def clearPort(self): ...
+    def setPortName(self, port_name): ...
+    def getPortName(self): ...
+    def setBaudRate(self, baudrate): ...
+    def getBaudRate(self): ...
+    def getBytesAvailable(self): ...
+    def readPort(self, length): ...
+    def writePort(self, packet): ...
+    def setPacketTimeout(self, packet_length): ...
+    def setPacketTimeoutMillis(self, msec): ...
+    def isPacketTimeout(self): ...
+    def getCurrentTime(self): ...
+    def getTimeSinceStart(self): ...
+    def setupPort(self, cflag_baud): ...
+    def getCFlagBaud(self, baudrate): ...
+
+
+class PacketHandler(Protocol):
+    def getTxRxResult(self, result): ...
+    def getRxPacketError(self, error): ...
+    def txPacket(self, port, txpacket): ...
+    def rxPacket(self, port): ...
+    def txRxPacket(self, port, txpacket): ...
+    def ping(self, port, id): ...
+    def action(self, port, id): ...
+    def readTx(self, port, id, address, length): ...
+    def readRx(self, port, id, length): ...
+    def readTxRx(self, port, id, address, length): ...
+    def read1ByteTx(self, port, id, address): ...
+    def read1ByteRx(self, port, id): ...
+    def read1ByteTxRx(self, port, id, address): ...
+    def read2ByteTx(self, port, id, address): ...
+    def read2ByteRx(self, port, id): ...
+    def read2ByteTxRx(self, port, id, address): ...
+    def read4ByteTx(self, port, id, address): ...
+    def read4ByteRx(self, port, id): ...
+    def read4ByteTxRx(self, port, id, address): ...
+    def writeTxOnly(self, port, id, address, length, data): ...
+    def writeTxRx(self, port, id, address, length, data): ...
+    def write1ByteTxOnly(self, port, id, address, data): ...
+    def write1ByteTxRx(self, port, id, address, data): ...
+    def write2ByteTxOnly(self, port, id, address, data): ...
+    def write2ByteTxRx(self, port, id, address, data): ...
+    def write4ByteTxOnly(self, port, id, address, data): ...
+    def write4ByteTxRx(self, port, id, address, data): ...
+    def regWriteTxOnly(self, port, id, address, length, data): ...
+    def regWriteTxRx(self, port, id, address, length, data): ...
+    def syncReadTx(self, port, start_address, data_length, param, param_length): ...
+    def syncWriteTxOnly(self, port, start_address, data_length, param, param_length): ...
+    def broadcastPing(self, port): ...
+
+
+class GroupSyncRead(Protocol):
+    port: str
+    ph: PortHandler
+    start_address: int
+    data_length: int
+    last_result: bool
+    is_param_changed: bool
+    param: list
+    data_dict: dict
+
+    def __init__(
+        self, port: PortHandler, ph: PacketHandler, start_address: int, data_length: int
+    ) -> None: ...
+    def makeParam(self): ...
+    def addParam(self, id): ...
+    def removeParam(self, id): ...
+    def clearParam(self): ...
+    def txPacket(self): ...
+    def rxPacket(self): ...
+    def txRxPacket(self): ...
+    def isAvailable(self, id, address, data_length): ...
+    def getData(self, id, address, data_length): ...
+
+
+class GroupSyncWrite(Protocol):
+    port: str
+    ph: PortHandler
+    start_address: int
+    data_length: int
+    is_param_changed: bool
+    param: list
+    data_dict: dict
+
+    def __init__(
+        self, port: PortHandler, ph: PacketHandler, start_address: int, data_length: int
+    ) -> None: ...
+    def makeParam(self): ...
+    def addParam(self, id, data): ...
+    def removeParam(self, id): ...
+    def changeParam(self, id, data): ...
+    def clearParam(self): ...
+    def txPacket(self): ...
+
+
+class SerialMotorsBus(MotorsBusBase):
+    """
+    A SerialMotorsBus allows to efficiently read and write to motors connected via serial communication.
+    It represents several motors daisy-chained together and connected through a serial port.
+    There are currently two implementations of this class:
+        - DynamixelMotorsBus
+        - FeetechMotorsBus
+
+    This class is specifically for serial-based motor protocols (Dynamixel, Feetech, etc.).
+
+    A MotorsBus subclass instance requires a port (e.g. `FeetechMotorsBus(port="/dev/tty.usbmodem575E0031751"`)).
+    To find the port, you can run our utility script:
+    ```bash
+    lerobot-find-port.py
+    >>> Finding all available ports for the MotorsBus.
+    >>> ["/dev/tty.usbmodem575E0032081", "/dev/tty.usbmodem575E0031751"]
+    >>> Remove the usb cable from your MotorsBus and press Enter when done.
+    >>> The port of this MotorsBus is /dev/tty.usbmodem575E0031751.
+    >>> Reconnect the usb cable.
+    ```
+
+    Example of usage for 1 Feetech sts3215 motor connected to the bus:
+    ```python
+    bus = FeetechMotorsBus(
+        port="/dev/tty.usbmodem575E0031751",
+        motors={"my_motor": (1, "sts3215")},
+    )
+    bus.connect()
+
+    position = bus.read("Present_Position", "my_motor", normalize=False)
+
+    # Move from a few motor steps as an example
+    few_steps = 30
+    bus.write("Goal_Position", "my_motor", position + few_steps, normalize=False)
+
+    # When done, properly disconnect the port using
+    bus.disconnect()
+    ```
+    """
+
+    apply_drive_mode: bool
+    available_baudrates: list[int]
+    default_baudrate: int
+    default_timeout: int
+    model_baudrate_table: dict[str, dict]
+    model_ctrl_table: dict[str, dict]
+    model_encoding_table: dict[str, dict]
+    model_number_table: dict[str, int]
+    model_resolution_table: dict[str, int]
+    normalized_data: list[str]
+
+    def __init__(
+        self,
+        port: str,
+        motors: dict[str, Motor],
+        calibration: dict[str, MotorCalibration] | None = None,
+    ):
+        super().__init__(port, motors, calibration)
+
+        self.port_handler: PortHandler
+        self.packet_handler: PacketHandler
+        self.sync_reader: GroupSyncRead
+        self.sync_writer: GroupSyncWrite
+        self._comm_success: int
+        self._no_error: int
+
+        self._id_to_model_dict = {m.id: m.model for m in self.motors.values()}
+        self._id_to_name_dict = {m.id: motor for motor, m in self.motors.items()}
+        self._model_nb_to_model_dict = {v: k for k, v in self.model_number_table.items()}
+
+        self._validate_motors()
+
+    def __len__(self):
+        return len(self.motors)
+
+    def __repr__(self):
+        return (
+            f"{self.__class__.__name__}(\n"
+            f"    Port: '{self.port}',\n"
+            f"    Motors: \n{pformat(self.motors, indent=8, sort_dicts=False)},\n"
+            ")',\n"
+        )
+
+    @cached_property
+    def _has_different_ctrl_tables(self) -> bool:
+        if len(self.models) < 2:
+            return False
+
+        first_table = self.model_ctrl_table[self.models[0]]
+        return any(
+            DeepDiff(first_table, get_ctrl_table(self.model_ctrl_table, model)) for model in self.models[1:]
+        )
+
+    @cached_property
+    def models(self) -> list[str]:
+        return [m.model for m in self.motors.values()]
+
+    @cached_property
+    def ids(self) -> list[int]:
+        return [m.id for m in self.motors.values()]
+
+    def _model_nb_to_model(self, motor_nb: int) -> str:
+        return self._model_nb_to_model_dict[motor_nb]
+
+    def _id_to_model(self, motor_id: int) -> str:
+        return self._id_to_model_dict[motor_id]
+
+    def _id_to_name(self, motor_id: int) -> str:
+        return self._id_to_name_dict[motor_id]
+
+    def _get_motor_id(self, motor: NameOrID) -> int:
+        if isinstance(motor, str):
+            return self.motors[motor].id
+        elif isinstance(motor, int):
+            return motor
+        else:
+            raise TypeError(f"'{motor}' should be int, str.")
+
+    def _get_motor_model(self, motor: NameOrID) -> str:
+        if isinstance(motor, str):
+            return self.motors[motor].model
+        elif isinstance(motor, int):
+            return self._id_to_model_dict[motor]
+        else:
+            raise TypeError(f"'{motor}' should be int, str.")
+
+    def _get_motors_list(self, motors: NameOrID | Sequence[NameOrID] | None) -> list[str]:
+        if motors is None:
+            return list(self.motors)
+        elif isinstance(motors, str):
+            return [motors]
+        elif isinstance(motors, int):
+            return [self._id_to_name(motors)]
+        elif isinstance(motors, Sequence):
+            return [m if isinstance(m, str) else self._id_to_name(m) for m in motors]
+        else:
+            raise TypeError(motors)
+
+    def _get_ids_values_dict(self, values: Value | dict[str, Value] | None) -> dict[int, Value]:
+        if isinstance(values, (int | float)):
+            return dict.fromkeys(self.ids, values)
+        elif isinstance(values, dict):
+            return {self.motors[motor].id: val for motor, val in values.items()}
+        else:
+            raise TypeError(f"'values' is expected to be a single value or a dict. Got {values}")
+
+    def _validate_motors(self) -> None:
+        if len(self.ids) != len(set(self.ids)):
+            raise ValueError(f"Some motors have the same id!\n{self}")
+
+        # Ensure ctrl table available for all models
+        for model in self.models:
+            get_ctrl_table(self.model_ctrl_table, model)
+
+    def _is_comm_success(self, comm: int) -> bool:
+        return comm == self._comm_success
+
+    def _is_error(self, error: int) -> bool:
+        return error != self._no_error
+
+    def _assert_motors_exist(self) -> None:
+        expected_models = {m.id: self.model_number_table[m.model] for m in self.motors.values()}
+
+        found_models = {}
+        for id_ in self.ids:
+            model_nb = self.ping(id_)
+            if model_nb is not None:
+                found_models[id_] = model_nb
+
+        missing_ids = [id_ for id_ in self.ids if id_ not in found_models]
+        wrong_models = {
+            id_: (expected_models[id_], found_models[id_])
+            for id_ in found_models
+            if expected_models.get(id_) != found_models[id_]
+        }
+
+        if missing_ids or wrong_models:
+            error_lines = [f"{self.__class__.__name__} motor check failed on port '{self.port}':"]
+
+            if missing_ids:
+                error_lines.append("\nMissing motor IDs:")
+                error_lines.extend(
+                    f"  - {id_} (expected model: {expected_models[id_]})" for id_ in missing_ids
+                )
+
+            if wrong_models:
+                error_lines.append("\nMotors with incorrect model numbers:")
+                error_lines.extend(
+                    f"  - {id_} ({self._id_to_name(id_)}): expected {expected}, found {found}"
+                    for id_, (expected, found) in wrong_models.items()
+                )
+
+            error_lines.append("\nFull expected motor list (id: model_number):")
+            error_lines.append(pformat(expected_models, indent=4, sort_dicts=False))
+            error_lines.append("\nFull found motor list (id: model_number):")
+            error_lines.append(pformat(found_models, indent=4, sort_dicts=False))
+
+            raise RuntimeError("\n".join(error_lines))
+
+    @abc.abstractmethod
+    def _assert_protocol_is_compatible(self, instruction_name: str) -> None:
+        pass
+
+    @property
+    def is_connected(self) -> bool:
+        """bool: `True` if the underlying serial port is open."""
+        return self.port_handler.is_open
+
+    @check_if_already_connected
+    def connect(self, handshake: bool = True) -> None:
+        """Open the serial port and initialise communication.
+
+        Args:
+            handshake (bool, optional): Pings every expected motor and performs additional
+                integrity checks specific to the implementation. Defaults to `True`.
+
+        Raises:
+            DeviceAlreadyConnectedError: The port is already open.
+            ConnectionError: The underlying SDK failed to open the port or the handshake did not succeed.
+        """
+
+        self._connect(handshake)
+        self.set_timeout()
+        logger.debug(f"{self.__class__.__name__} connected.")
+
+    def _connect(self, handshake: bool = True) -> None:
+        try:
+            if not self.port_handler.openPort():
+                raise OSError(f"Failed to open port '{self.port}'.")
+            elif handshake:
+                self._handshake()
+        except (FileNotFoundError, OSError, serial.SerialException) as e:
+            raise ConnectionError(
+                f"\nCould not connect on port '{self.port}'. Make sure you are using the correct port."
+                "\nTry running `lerobot-find-port`\n"
+            ) from e
+
+    @abc.abstractmethod
+    def _handshake(self) -> None:
+        pass
+
+    @check_if_not_connected
+    def disconnect(self, disable_torque: bool = True) -> None:
+        """Close the serial port (optionally disabling torque first).
+
+        Args:
+            disable_torque (bool, optional): If `True` (default) torque is disabled on every motor before
+                closing the port. This can prevent damaging motors if they are left applying resisting torque
+                after disconnect.
+        """
+
+        if disable_torque:
+            self.port_handler.clearPort()
+            self.port_handler.is_using = False
+            self.disable_torque(num_retry=5)
+
+        self.port_handler.closePort()
+        logger.debug(f"{self.__class__.__name__} disconnected.")
+
+    @classmethod
+    def scan_port(cls, port: str, *args, **kwargs) -> dict[int, list[int]]:
+        """Probe *port* at every supported baud-rate and list responding IDs.
+
+        Args:
+            port (str): Serial/USB port to scan (e.g. ``"/dev/ttyUSB0"``).
+            *args, **kwargs: Forwarded to the subclass constructor.
+
+        Returns:
+            dict[int, list[int]]: Mapping *baud-rate → list of motor IDs*
+            for every baud-rate that produced at least one response.
+        """
+        bus = cls(port, {}, *args, **kwargs)
+        bus._connect(handshake=False)
+        baudrate_ids = {}
+        for baudrate in tqdm(bus.available_baudrates, desc="Scanning port"):
+            bus.set_baudrate(baudrate)
+            ids_models = bus.broadcast_ping()
+            if ids_models:
+                tqdm.write(f"Motors found for {baudrate=}: {pformat(ids_models, indent=4)}")
+                baudrate_ids[baudrate] = list(ids_models)
+
+        bus.port_handler.closePort()
+        return baudrate_ids
+
+    def setup_motor(
+        self, motor: str, initial_baudrate: int | None = None, initial_id: int | None = None
+    ) -> None:
+        """Assign the correct ID and baud-rate to a single motor.
+
+        This helper temporarily switches to the motor's current settings, disables torque, sets the desired
+        ID, and finally programs the bus' default baud-rate.
+
+        Args:
+            motor (str): Key of the motor in :pyattr:`motors`.
+            initial_baudrate (int | None, optional): Current baud-rate (skips scanning when provided).
+                Defaults to None.
+            initial_id (int | None, optional): Current ID (skips scanning when provided). Defaults to None.
+
+        Raises:
+            RuntimeError: The motor could not be found or its model number
+                does not match the expected one.
+            ConnectionError: Communication with the motor failed.
+        """
+        if not self.is_connected:
+            self._connect(handshake=False)
+
+        if initial_baudrate is None:
+            initial_baudrate, initial_id = self._find_single_motor(motor)
+
+        if initial_id is None:
+            _, initial_id = self._find_single_motor(motor, initial_baudrate)
+
+        model = self.motors[motor].model
+        target_id = self.motors[motor].id
+        self.set_baudrate(initial_baudrate)
+        self._disable_torque(initial_id, model)
+
+        # Set ID
+        addr, length = get_address(self.model_ctrl_table, model, "ID")
+        self._write(addr, length, initial_id, target_id)
+
+        # Set Baudrate
+        addr, length = get_address(self.model_ctrl_table, model, "Baud_Rate")
+        baudrate_value = self.model_baudrate_table[model][self.default_baudrate]
+        self._write(addr, length, target_id, baudrate_value)
+
+        self.set_baudrate(self.default_baudrate)
+
+    @abc.abstractmethod
+    def _find_single_motor(self, motor: str, initial_baudrate: int | None = None) -> tuple[int, int]:
+        pass
+
+    @abc.abstractmethod
+    def configure_motors(self) -> None:
+        """Write implementation-specific recommended settings to every motor.
+
+        Typical changes include shortening the return delay, increasing
+        acceleration limits or disabling safety locks.
+        """
+        pass
+
+    @abc.abstractmethod
+    def disable_torque(self, motors: str | list[str] | None = None, num_retry: int = 0) -> None:
+        """Disable torque on selected motors.
+
+        Disabling Torque allows to write to the motors' permanent memory area (EPROM/EEPROM).
+
+        Args:
+            motors ( str | list[str] | None, optional): Target motors.  Accepts a motor name, an ID, a
+                list of names or `None` to affect every registered motor.  Defaults to `None`.
+            num_retry (int, optional): Number of additional retry attempts on communication failure.
+                Defaults to 0.
+        """
+        pass
+
+    @abc.abstractmethod
+    def _disable_torque(self, motor: int, model: str, num_retry: int = 0) -> None:
+        pass
+
+    @abc.abstractmethod
+    def enable_torque(self, motors: int | str | list[str] | None = None, num_retry: int = 0) -> None:
+        """Enable torque on selected motors.
+
+        Args:
+            motors (int | str | list[str] | None, optional): Same semantics as :pymeth:`disable_torque`.
+                Defaults to `None`.
+            num_retry (int, optional): Number of additional retry attempts on communication failure.
+                Defaults to 0.
+        """
+        pass
+
+    @contextmanager
+    def torque_disabled(self, motors: str | list[str] | None = None):
+        """Context-manager that guarantees torque is re-enabled.
+
+        This helper is useful to temporarily disable torque when configuring motors.
+
+        Examples:
+            >>> with bus.torque_disabled():
+            ...     # Safe operations here
+            ...     pass
+        """
+        self.disable_torque(motors)
+        try:
+            yield
+        finally:
+            self.enable_torque(motors)
+
+    def set_timeout(self, timeout_ms: int | None = None):
+        """Change the packet timeout used by the SDK.
+
+        Args:
+            timeout_ms (int | None, optional): Timeout in *milliseconds*. If `None` (default) the method falls
+                back to :pyattr:`default_timeout`.
+        """
+        timeout_ms = timeout_ms if timeout_ms is not None else self.default_timeout
+        self.port_handler.setPacketTimeoutMillis(timeout_ms)
+
+    def get_baudrate(self) -> int:
+        """Return the current baud-rate configured on the port.
+
+        Returns:
+            int: Baud-rate in bits / second.
+        """
+        return self.port_handler.getBaudRate()
+
+    def set_baudrate(self, baudrate: int) -> None:
+        """Set a new UART baud-rate on the port.
+
+        Args:
+            baudrate (int): Desired baud-rate in bits / second.
+
+        Raises:
+            RuntimeError: The SDK failed to apply the change.
+        """
+        present_bus_baudrate = self.port_handler.getBaudRate()
+        if present_bus_baudrate != baudrate:
+            logger.info(f"Setting bus baud rate to {baudrate}. Previously {present_bus_baudrate}.")
+            self.port_handler.setBaudRate(baudrate)
+
+            if self.port_handler.getBaudRate() != baudrate:
+                raise RuntimeError("Failed to write bus baud rate.")
+
+    @property
+    @abc.abstractmethod
+    def is_calibrated(self) -> bool:
+        """bool: ``True`` if the cached calibration matches the motors."""
+        pass
+
+    @abc.abstractmethod
+    def read_calibration(self) -> dict[str, MotorCalibration]:
+        """Read calibration parameters from the motors.
+
+        Returns:
+            dict[str, MotorCalibration]: Mapping *motor name → calibration*.
+        """
+        pass
+
+    @abc.abstractmethod
+    def write_calibration(self, calibration_dict: dict[str, MotorCalibration], cache: bool = True) -> None:
+        """Write calibration parameters to the motors and optionally cache them.
+
+        Args:
+            calibration_dict (dict[str, MotorCalibration]): Calibration obtained from
+                :pymeth:`read_calibration` or crafted by the user.
+            cache (bool, optional): Save the calibration to :pyattr:`calibration`. Defaults to True.
+        """
+        pass
+
+    def reset_calibration(self, motors: NameOrID | Sequence[NameOrID] | None = None) -> None:
+        """Restore factory calibration for the selected motors.
+
+        Homing offset is set to ``0`` and min/max position limits are set to the full usable range.
+        The in-memory :pyattr:`calibration` is cleared.
+
+        Args:
+            motors (NameOrID | Sequence[NameOrID] | None, optional): Selection of motors. `None` (default)
+                resets every motor.
+        """
+        motor_names = self._get_motors_list(motors)
+
+        for motor in motor_names:
+            model = self._get_motor_model(motor)
+            max_res = self.model_resolution_table[model] - 1
+            self.write("Homing_Offset", motor, 0, normalize=False)
+            self.write("Min_Position_Limit", motor, 0, normalize=False)
+            self.write("Max_Position_Limit", motor, max_res, normalize=False)
+
+        self.calibration = {}
+
+    def set_half_turn_homings(
+        self, motors: NameOrID | Sequence[NameOrID] | None = None
+    ) -> dict[NameOrID, Value]:
+        """Centre each motor range around its current position.
+
+        The function computes and writes a homing offset such that the present position becomes exactly one
+        half-turn (e.g. `2047` on a 12-bit encoder).
+
+        Args:
+            motors (NameOrID | list[NameOrID] | None, optional): Motors to adjust. Defaults to all motors (`None`).
+
+        Returns:
+            dict[str, Value]: Mapping *motor name → written homing offset*.
+        """
+        motor_names = self._get_motors_list(motors)
+
+        self.reset_calibration(motor_names)
+        actual_positions = self.sync_read("Present_Position", motor_names, normalize=False)
+        homing_offsets = self._get_half_turn_homings(actual_positions)
+        for motor, offset in homing_offsets.items():
+            self.write("Homing_Offset", motor, offset)
+
+        return homing_offsets
+
+    @abc.abstractmethod
+    def _get_half_turn_homings(self, positions: dict[NameOrID, Value]) -> dict[NameOrID, Value]:
+        pass
+
+    def record_ranges_of_motion(
+        self, motors: NameOrID | Sequence[NameOrID] | None = None, display_values: bool = True
+    ) -> tuple[dict[str, Value], dict[str, Value]]:
+        """Interactively record the min/max encoder values of each motor.
+
+        Move the joints by hand (with torque disabled) while the method streams live positions. Press
+        :kbd:`Enter` to finish.
+
+        Args:
+            motors (NameOrID | list[NameOrID] | None, optional): Motors to record.
+                Defaults to every motor (`None`).
+            display_values (bool, optional): When `True` (default) a live table is printed to the console.
+
+        Returns:
+            tuple[dict[str, Value], dict[str, Value]]: Two dictionaries *mins* and *maxes* with the
+                extreme values observed for each motor.
+        """
+        motor_names = self._get_motors_list(motors)
+
+        start_positions = self.sync_read("Present_Position", motor_names, normalize=False)
+        mins = start_positions.copy()
+        maxes = start_positions.copy()
+
+        user_pressed_enter = False
+        while not user_pressed_enter:
+            positions = self.sync_read("Present_Position", motor_names, normalize=False)
+            mins = {motor: min(positions[motor], min_) for motor, min_ in mins.items()}
+            maxes = {motor: max(positions[motor], max_) for motor, max_ in maxes.items()}
+
+            if display_values:
+                print("\n-------------------------------------------")
+                print(f"{'NAME':<15} | {'MIN':>6} | {'POS':>6} | {'MAX':>6}")
+                for motor in motor_names:
+                    print(f"{motor:<15} | {mins[motor]:>6} | {positions[motor]:>6} | {maxes[motor]:>6}")
+
+            if enter_pressed():
+                user_pressed_enter = True
+
+            if display_values and not user_pressed_enter:
+                # Move cursor up to overwrite the previous output
+                move_cursor_up(len(motor_names) + 3)
+
+        same_min_max = [motor for motor in motor_names if mins[motor] == maxes[motor]]
+        if same_min_max:
+            raise ValueError(f"Some motors have the same min and max values:\n{pformat(same_min_max)}")
+
+        return mins, maxes
+
+    def _normalize(self, ids_values: dict[int, int]) -> dict[int, float]:
+        if not self.calibration:
+            raise RuntimeError(f"{self} has no calibration registered.")
+
+        normalized_values = {}
+        for id_, val in ids_values.items():
+            motor = self._id_to_name(id_)
+            min_ = self.calibration[motor].range_min
+            max_ = self.calibration[motor].range_max
+            drive_mode = self.apply_drive_mode and self.calibration[motor].drive_mode
+            if max_ == min_:
+                raise ValueError(f"Invalid calibration for motor '{motor}': min and max are equal.")
+
+            bounded_val = min(max_, max(min_, val))
+            if self.motors[motor].norm_mode is MotorNormMode.RANGE_M100_100:
+                norm = (((bounded_val - min_) / (max_ - min_)) * 200) - 100
+                normalized_values[id_] = -norm if drive_mode else norm
+            elif self.motors[motor].norm_mode is MotorNormMode.RANGE_0_100:
+                norm = ((bounded_val - min_) / (max_ - min_)) * 100
+                normalized_values[id_] = 100 - norm if drive_mode else norm
+            elif self.motors[motor].norm_mode is MotorNormMode.DEGREES:
+                mid = (min_ + max_) / 2
+                max_res = self.model_resolution_table[self._id_to_model(id_)] - 1
+                normalized_values[id_] = (val - mid) * 360 / max_res
+            else:
+                raise NotImplementedError
+
+        return normalized_values
+
+    def _unnormalize(self, ids_values: dict[int, float]) -> dict[int, int]:
+        if not self.calibration:
+            raise RuntimeError(f"{self} has no calibration registered.")
+
+        unnormalized_values = {}
+        for id_, val in ids_values.items():
+            motor = self._id_to_name(id_)
+            min_ = self.calibration[motor].range_min
+            max_ = self.calibration[motor].range_max
+            drive_mode = self.apply_drive_mode and self.calibration[motor].drive_mode
+            if max_ == min_:
+                raise ValueError(f"Invalid calibration for motor '{motor}': min and max are equal.")
+
+            if self.motors[motor].norm_mode is MotorNormMode.RANGE_M100_100:
+                val = -val if drive_mode else val
+                bounded_val = min(100.0, max(-100.0, val))
+                unnormalized_values[id_] = int(((bounded_val + 100) / 200) * (max_ - min_) + min_)
+            elif self.motors[motor].norm_mode is MotorNormMode.RANGE_0_100:
+                val = 100 - val if drive_mode else val
+                bounded_val = min(100.0, max(0.0, val))
+                unnormalized_values[id_] = int((bounded_val / 100) * (max_ - min_) + min_)
+            elif self.motors[motor].norm_mode is MotorNormMode.DEGREES:
+                mid = (min_ + max_) / 2
+                max_res = self.model_resolution_table[self._id_to_model(id_)] - 1
+                unnormalized_values[id_] = int((val * max_res / 360) + mid)
+            else:
+                raise NotImplementedError
+
+        return unnormalized_values
+
+    @abc.abstractmethod
+    def _encode_sign(self, data_name: str, ids_values: dict[int, int]) -> dict[int, int]:
+        pass
+
+    @abc.abstractmethod
+    def _decode_sign(self, data_name: str, ids_values: dict[int, int]) -> dict[int, int]:
+        pass
+
+    def _serialize_data(self, value: int, length: int) -> list[int]:
+        """
+        Converts an unsigned integer value into a list of byte-sized integers to be sent via a communication
+        protocol. Depending on the protocol, split values can be in big-endian or little-endian order.
+
+        Supported data length for both Feetech and Dynamixel:
+            - 1 (for values 0 to 255)
+            - 2 (for values 0 to 65,535)
+            - 4 (for values 0 to 4,294,967,295)
+        """
+        if value < 0:
+            raise ValueError(f"Negative values are not allowed: {value}")
+
+        max_value = {1: 0xFF, 2: 0xFFFF, 4: 0xFFFFFFFF}.get(length)
+        if max_value is None:
+            raise NotImplementedError(f"Unsupported byte size: {length}. Expected [1, 2, 4].")
+
+        if value > max_value:
+            raise ValueError(f"Value {value} exceeds the maximum for {length} bytes ({max_value}).")
+
+        return self._split_into_byte_chunks(value, length)
+
+    @abc.abstractmethod
+    def _split_into_byte_chunks(self, value: int, length: int) -> list[int]:
+        """Convert an integer into a list of byte-sized integers."""
+        pass
+
+    def ping(self, motor: NameOrID, num_retry: int = 0, raise_on_error: bool = False) -> int | None:
+        """Ping a single motor and return its model number.
+
+        Args:
+            motor (NameOrID): Target motor (name or ID).
+            num_retry (int, optional): Extra attempts before giving up. Defaults to `0`.
+            raise_on_error (bool, optional): If `True` communication errors raise exceptions instead of
+                returning `None`. Defaults to `False`.
+
+        Returns:
+            int | None: Motor model number or `None` on failure.
+        """
+        id_ = self._get_motor_id(motor)
+        for n_try in range(1 + num_retry):
+            model_number, comm, error = self.packet_handler.ping(self.port_handler, id_)
+            if self._is_comm_success(comm):
+                break
+            logger.debug(f"ping failed for {id_=}: {n_try=} got {comm=} {error=}")
+
+        if not self._is_comm_success(comm):
+            if raise_on_error:
+                raise ConnectionError(self.packet_handler.getTxRxResult(comm))
+            else:
+                return None
+        if self._is_error(error):
+            if raise_on_error:
+                raise RuntimeError(self.packet_handler.getRxPacketError(error))
+            else:
+                return None
+
+        return model_number
+
+    @abc.abstractmethod
+    def broadcast_ping(self, num_retry: int = 0, raise_on_error: bool = False) -> dict[int, int] | None:
+        """Ping every ID on the bus using the broadcast address.
+
+        Args:
+            num_retry (int, optional): Retry attempts.  Defaults to `0`.
+            raise_on_error (bool, optional): When `True` failures raise an exception instead of returning
+                `None`. Defaults to `False`.
+
+        Returns:
+            dict[int, int] | None: Mapping *id → model number* or `None` if the call failed.
+        """
+        pass
+
+    @check_if_not_connected
+    def read(
+        self,
+        data_name: str,
+        motor: str,
+        *,
+        normalize: bool = True,
+        num_retry: int = 0,
+    ) -> Value:
+        """Read a register from a motor.
+
+        Args:
+            data_name (str): Control-table key (e.g. `"Present_Position"`).
+            motor (str): Motor name.
+            normalize (bool, optional): When `True` (default) scale the value to a user-friendly range as
+                defined by the calibration.
+            num_retry (int, optional): Retry attempts.  Defaults to `0`.
+
+        Returns:
+            Value: Raw or normalised value depending on *normalize*.
+        """
+
+        id_ = self.motors[motor].id
+        model = self.motors[motor].model
+        addr, length = get_address(self.model_ctrl_table, model, data_name)
+
+        err_msg = f"Failed to read '{data_name}' on {id_=} after {num_retry + 1} tries."
+        value, _, _ = self._read(addr, length, id_, num_retry=num_retry, raise_on_error=True, err_msg=err_msg)
+
+        decoded = self._decode_sign(data_name, {id_: value})
+
+        if normalize and data_name in self.normalized_data:
+            normalized = self._normalize(decoded)
+            return normalized[id_]
+
+        return decoded[id_]
+
+    def _read(
+        self,
+        address: int,
+        length: int,
+        motor_id: int,
+        *,
+        num_retry: int = 0,
+        raise_on_error: bool = True,
+        err_msg: str = "",
+    ) -> tuple[int, int, int]:
+        if length == 1:
+            read_fn = self.packet_handler.read1ByteTxRx
+        elif length == 2:
+            read_fn = self.packet_handler.read2ByteTxRx
+        elif length == 4:
+            read_fn = self.packet_handler.read4ByteTxRx
+        else:
+            raise ValueError(length)
+
+        for n_try in range(1 + num_retry):
+            value, comm, error = read_fn(self.port_handler, motor_id, address)
+            if self._is_comm_success(comm):
+                break
+            logger.debug(
+                f"Failed to read @{address=} ({length=}) on {motor_id=} ({n_try=}): "
+                + self.packet_handler.getTxRxResult(comm)
+            )
+
+        if not self._is_comm_success(comm) and raise_on_error:
+            raise ConnectionError(f"{err_msg} {self.packet_handler.getTxRxResult(comm)}")
+        elif self._is_error(error) and raise_on_error:
+            raise RuntimeError(f"{err_msg} {self.packet_handler.getRxPacketError(error)}")
+
+        return value, comm, error
+
+    @check_if_not_connected
+    def write(
+        self, data_name: str, motor: str, value: Value, *, normalize: bool = True, num_retry: int = 0
+    ) -> None:
+        """Write a value to a single motor's register.
+
+        Contrary to :pymeth:`sync_write`, this expects a response status packet emitted by the motor, which
+        provides a guarantee that the value was written to the register successfully. In consequence, it is
+        slower than :pymeth:`sync_write` but it is more reliable. It should typically be used when configuring
+        motors.
+
+        Args:
+            data_name (str): Register name.
+            motor (str): Motor name.
+            value (Value): Value to write.  If *normalize* is `True` the value is first converted to raw
+                units.
+            normalize (bool, optional): Enable or disable normalisation. Defaults to `True`.
+            num_retry (int, optional): Retry attempts.  Defaults to `0`.
+        """
+
+        id_ = self.motors[motor].id
+        model = self.motors[motor].model
+        addr, length = get_address(self.model_ctrl_table, model, data_name)
+
+        int_value = int(value)
+        if normalize and data_name in self.normalized_data:
+            int_value = self._unnormalize({id_: value})[id_]
+
+        int_value = self._encode_sign(data_name, {id_: int_value})[id_]
+
+        err_msg = f"Failed to write '{data_name}' on {id_=} with '{int_value}' after {num_retry + 1} tries."
+        self._write(addr, length, id_, int_value, num_retry=num_retry, raise_on_error=True, err_msg=err_msg)
+
+    def _write(
+        self,
+        addr: int,
+        length: int,
+        motor_id: int,
+        value: int,
+        *,
+        num_retry: int = 0,
+        raise_on_error: bool = True,
+        err_msg: str = "",
+    ) -> tuple[int, int]:
+        data = self._serialize_data(value, length)
+        for n_try in range(1 + num_retry):
+            comm, error = self.packet_handler.writeTxRx(self.port_handler, motor_id, addr, length, data)
+            if self._is_comm_success(comm):
+                break
+            logger.debug(
+                f"Failed to sync write @{addr=} ({length=}) on id={motor_id} with {value=} ({n_try=}): "
+                + self.packet_handler.getTxRxResult(comm)
+            )
+
+        if not self._is_comm_success(comm) and raise_on_error:
+            raise ConnectionError(f"{err_msg} {self.packet_handler.getTxRxResult(comm)}")
+        elif self._is_error(error) and raise_on_error:
+            raise RuntimeError(f"{err_msg} {self.packet_handler.getRxPacketError(error)}")
+
+        return comm, error
+
+    @check_if_not_connected
+    def sync_read(
+        self,
+        data_name: str,
+        motors: NameOrID | Sequence[NameOrID] | None = None,
+        *,
+        normalize: bool = True,
+        num_retry: int = 0,
+    ) -> dict[str, Value]:
+        """Read the same register from several motors at once.
+
+        Args:
+            data_name (str): Register name.
+            motors (NameOrID | Sequence[NameOrID] | None, optional): Motors to query. `None` (default) reads every motor.
+            normalize (bool, optional): Normalisation flag.  Defaults to `True`.
+            num_retry (int, optional): Retry attempts.  Defaults to `0`.
+
+        Returns:
+            dict[str, Value]: Mapping *motor name → value*.
+        """
+
+        self._assert_protocol_is_compatible("sync_read")
+
+        names = self._get_motors_list(motors)
+        ids = [self.motors[motor].id for motor in names]
+        models = [self.motors[motor].model for motor in names]
+
+        if self._has_different_ctrl_tables:
+            assert_same_address(self.model_ctrl_table, models, data_name)
+
+        model = next(iter(models))
+        addr, length = get_address(self.model_ctrl_table, model, data_name)
+
+        err_msg = f"Failed to sync read '{data_name}' on {ids=} after {num_retry + 1} tries."
+        raw_ids_values, _ = self._sync_read(
+            addr, length, ids, num_retry=num_retry, raise_on_error=True, err_msg=err_msg
+        )
+
+        decoded = self._decode_sign(data_name, raw_ids_values)
+
+        if normalize and data_name in self.normalized_data:
+            normalized = self._normalize(decoded)
+            return {self._id_to_name(id_): value for id_, value in normalized.items()}
+
+        return {self._id_to_name(id_): value for id_, value in decoded.items()}
+
+    def _sync_read(
+        self,
+        addr: int,
+        length: int,
+        motor_ids: list[int],
+        *,
+        num_retry: int = 0,
+        raise_on_error: bool = True,
+        err_msg: str = "",
+    ) -> tuple[dict[int, int], int]:
+        self._setup_sync_reader(motor_ids, addr, length)
+        for n_try in range(1 + num_retry):
+            comm = self.sync_reader.txRxPacket()
+            if self._is_comm_success(comm):
+                break
+            logger.debug(
+                f"Failed to sync read @{addr=} ({length=}) on {motor_ids=} ({n_try=}): "
+                + self.packet_handler.getTxRxResult(comm)
+            )
+
+        if not self._is_comm_success(comm) and raise_on_error:
+            raise ConnectionError(f"{err_msg} {self.packet_handler.getTxRxResult(comm)}")
+
+        values = {id_: self.sync_reader.getData(id_, addr, length) for id_ in motor_ids}
+        return values, comm
+
+    def _setup_sync_reader(self, motor_ids: list[int], addr: int, length: int) -> None:
+        self.sync_reader.clearParam()
+        self.sync_reader.start_address = addr
+        self.sync_reader.data_length = length
+        for id_ in motor_ids:
+            self.sync_reader.addParam(id_)
+
+    # TODO(aliberts, pkooij): Implementing something like this could get even much faster read times if need be.
+    # Would have to handle the logic of checking if a packet has been sent previously though but doable.
+    # This could be at the cost of increase latency between the moment the data is produced by the motors and
+    # the moment it is used by a policy.
+    # def _async_read(self, motor_ids: list[int], address: int, length: int):
+    #     if self.sync_reader.start_address != address or self.sync_reader.data_length != length or ...:
+    #         self._setup_sync_reader(motor_ids, address, length)
+    #     else:
+    #         self.sync_reader.rxPacket()
+    #         self.sync_reader.txPacket()
+
+    #     for id_ in motor_ids:
+    #         value = self.sync_reader.getData(id_, address, length)
+
+    @check_if_not_connected
+    def sync_write(
+        self,
+        data_name: str,
+        values: Value | dict[str, Value],
+        *,
+        normalize: bool = True,
+        num_retry: int = 0,
+    ) -> None:
+        """Write the same register on multiple motors.
+
+        Contrary to :pymeth:`write`, this *does not* expects a response status packet emitted by the motor, which
+        can allow for lost packets. It is faster than :pymeth:`write` and should typically be used when
+        frequency matters and losing some packets is acceptable (e.g. teleoperation loops).
+
+        Args:
+            data_name (str): Register name.
+            values (Value | dict[str, Value]): Either a single value (applied to every motor) or a mapping
+                *motor name → value*.
+            normalize (bool, optional): If `True` (default) convert values from the user range to raw units.
+            num_retry (int, optional): Retry attempts.  Defaults to `0`.
+        """
+
+        raw_ids_values = self._get_ids_values_dict(values)
+        models = [self._id_to_model(id_) for id_ in raw_ids_values]
+        if self._has_different_ctrl_tables:
+            assert_same_address(self.model_ctrl_table, models, data_name)
+
+        model = next(iter(models))
+        addr, length = get_address(self.model_ctrl_table, model, data_name)
+
+        int_ids_values = {id_: int(val) for id_, val in raw_ids_values.items()}
+        if normalize and data_name in self.normalized_data:
+            int_ids_values = self._unnormalize(raw_ids_values)
+
+        int_ids_values = self._encode_sign(data_name, int_ids_values)
+
+        err_msg = f"Failed to sync write '{data_name}' with ids_values={int_ids_values} after {num_retry + 1} tries."
+        self._sync_write(
+            addr, length, int_ids_values, num_retry=num_retry, raise_on_error=True, err_msg=err_msg
+        )
+
+    def _sync_write(
+        self,
+        addr: int,
+        length: int,
+        ids_values: dict[int, int],
+        num_retry: int = 0,
+        raise_on_error: bool = True,
+        err_msg: str = "",
+    ) -> int:
+        self._setup_sync_writer(ids_values, addr, length)
+        for n_try in range(1 + num_retry):
+            comm = self.sync_writer.txPacket()
+            if self._is_comm_success(comm):
+                break
+            logger.debug(
+                f"Failed to sync write @{addr=} ({length=}) with {ids_values=} ({n_try=}): "
+                + self.packet_handler.getTxRxResult(comm)
+            )
+
+        if not self._is_comm_success(comm) and raise_on_error:
+            raise ConnectionError(f"{err_msg} {self.packet_handler.getTxRxResult(comm)}")
+
+        return comm
+
+    def _setup_sync_writer(self, ids_values: dict[int, int], addr: int, length: int) -> None:
+        self.sync_writer.clearParam()
+        self.sync_writer.start_address = addr
+        self.sync_writer.data_length = length
+        for id_, value in ids_values.items():
+            data = self._serialize_data(value, length)
+            self.sync_writer.addParam(id_, data)
+
+
+# Backward compatibility alias
+MotorsBus = SerialMotorsBus
diff --git a/lerobot/src/lerobot/motors/robstride/__init__.py b/lerobot/src/lerobot/motors/robstride/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..7933ac6fa7f8ee935a176ef97eba82e7fea44f8b
--- /dev/null
+++ b/lerobot/src/lerobot/motors/robstride/__init__.py
@@ -0,0 +1,18 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .robstride import RobstrideMotorsBus
+from .tables import *
diff --git a/lerobot/src/lerobot/motors/robstride/robstride.py b/lerobot/src/lerobot/motors/robstride/robstride.py
new file mode 100644
index 0000000000000000000000000000000000000000..f47e41509bfcbe23e69b5fb91d911750c54514fd
--- /dev/null
+++ b/lerobot/src/lerobot/motors/robstride/robstride.py
@@ -0,0 +1,1003 @@
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# TODO(Virgile) : Robustify mode control , only the MIT protocole is implemented for now
+
+import logging
+import time
+from contextlib import contextmanager
+from copy import deepcopy
+from functools import cached_property
+from types import SimpleNamespace
+from typing import TYPE_CHECKING, Any, TypedDict
+
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.import_utils import _can_available
+
+if TYPE_CHECKING or _can_available:
+    import can
+else:
+    can = SimpleNamespace(Message=object, interface=None)
+import numpy as np
+
+from lerobot.utils.errors import DeviceNotConnectedError
+from lerobot.utils.utils import enter_pressed, move_cursor_up
+
+from ..motors_bus import Motor, MotorCalibration, MotorsBusBase, NameOrID, Value
+from .tables import (
+    AVAILABLE_BAUDRATES,
+    CAN_CMD_CLEAR_FAULT,
+    CAN_CMD_DISABLE,
+    CAN_CMD_ENABLE,
+    CAN_CMD_SET_ZERO,
+    DEFAULT_BAUDRATE,
+    DEFAULT_TIMEOUT_MS,
+    MODEL_RESOLUTION,
+    MOTOR_LIMIT_PARAMS,
+    NORMALIZED_DATA,
+    PARAM_TIMEOUT,
+    RUNNING_TIMEOUT,
+    STATE_CACHE_TTL_S,
+    ControlMode,
+    MotorType,
+)
+
+logger = logging.getLogger(__name__)
+
+
+class MotorState(TypedDict):
+    position: float
+    velocity: float
+    torque: float
+    temp_mos: float
+    temp_rotor: float
+
+
+class RobstrideMotorsBus(MotorsBusBase):
+    """
+    The Robstride implementation for a MotorsBus using CAN bus communication.
+
+    This class uses python-can for CAN bus communication with Robstride motors.
+    The motors need to be switched to MIT control mode to be compatible with this implementation.
+    More details on the protocol can be found in the documentation links below:
+    - python-can documentation: https://python-can.readthedocs.io/en/stable/
+    - Robstride CAN protocol: https://github.com/RobStride/MotorStudio
+    """
+
+    # CAN-specific settings
+    available_baudrates = deepcopy(AVAILABLE_BAUDRATES)
+    default_baudrate = DEFAULT_BAUDRATE
+    default_timeout = DEFAULT_TIMEOUT_MS
+
+    # Motor configuration
+    model_resolution_table = deepcopy(MODEL_RESOLUTION)
+    normalized_data = deepcopy(NORMALIZED_DATA)
+
+    def __init__(
+        self,
+        port: str,
+        motors: dict[str, Motor],
+        calibration: dict[str, MotorCalibration] | None = None,
+        can_interface: str = "auto",
+        use_can_fd: bool = True,
+        bitrate: int = 1000000,
+        data_bitrate: int | None = 5000000,
+    ):
+        """
+        Initialize the Robstride motors bus.
+
+        Args:
+            port: CAN interface name (e.g., "can0" for Linux, "/dev/cu.usbmodem*" for macOS)
+            motors: Dictionary mapping motor names to Motor objects
+            calibration: Optional calibration data
+            can_interface: CAN interface type - "auto" (default), "socketcan" (Linux), or "slcan" (macOS/serial)
+            use_can_fd: Whether to use CAN FD mode (default: True for OpenArms)
+            bitrate: Nominal bitrate in bps (default: 1000000 = 1 Mbps)
+            data_bitrate: Data bitrate for CAN FD in bps (default: 5000000 = 5 Mbps), ignored if use_can_fd is False
+        """
+        super().__init__(port, motors, calibration)
+        self.port = port
+        self.can_interface = can_interface
+        self.use_can_fd = use_can_fd
+        self.bitrate = bitrate
+        self.data_bitrate = data_bitrate
+        self.canbus: can.BusABC | None = None
+        self._is_connected = False
+
+        # Map motor names to CAN IDs
+        self._motor_can_ids: dict[str, int] = {}
+        self._recv_id_to_motor: dict[int, str] = {}
+
+        # Store motor types and recv IDs
+        self._motor_types: dict[str, MotorType] = {}
+        # Dynamic gains storage (Damiao-style update path via write/sync_write)
+        self._gains: dict[str, dict[str, float]] = {}
+        for name, motor in self.motors.items():
+            if motor.motor_type_str is not None:
+                self._motor_types[name] = getattr(MotorType, motor.motor_type_str.upper())
+            else:
+                # Default to O0if not specified
+                self._motor_types[name] = MotorType.O0
+
+            # Damiao-style defaults: fixed gains at startup for every motor.
+            self._gains[name] = {"kp": 10.0, "kd": 0.5}
+
+            # Map recv_id to motor name for filtering responses
+            if motor.recv_id is not None:
+                self._recv_id_to_motor[motor.recv_id] = name
+        # Motor Mode
+        self.enabled: dict[str, bool] = {}
+        self.operation_mode: dict[str, ControlMode] = {}
+        self._last_known_states: dict[str, MotorState] = {
+            name: {
+                "position": 0.0,
+                "velocity": 0.0,
+                "torque": 0.0,
+                "temp_mos": 0.0,
+                "temp_rotor": 0.0,
+            }
+            for name in self.motors
+        }
+        self.last_feedback_time: dict[str, float | None] = {}
+        self._id_to_name: dict[int, str] = {}
+        for name in self.motors:
+            self.enabled[name] = False
+            self.operation_mode[name] = ControlMode.MIT  # default mode
+            self.last_feedback_time[name] = None
+
+        for name, motor in self.motors.items():
+            key = motor.recv_id if motor.recv_id is not None else motor.id
+            self._id_to_name[key] = name
+
+    @property
+    def is_connected(self) -> bool:
+        """Check if the CAN bus is connected."""
+        return self._is_connected and self.canbus is not None
+
+    def _bus(self) -> can.BusABC:
+        if self.canbus is None:
+            raise DeviceNotConnectedError(f"{self.__class__.__name__}('{self.port}') is not connected.")
+        return self.canbus
+
+    @check_if_already_connected
+    def connect(self, handshake: bool = True) -> None:
+        """
+        Open the CAN bus and initialize communication.
+
+        Args:
+            handshake: If True, ping all motors to verify they're present
+        """
+        try:
+            # Auto-detect interface type based on port name
+            if self.can_interface == "auto":
+                if self.port.startswith("/dev/"):
+                    self.can_interface = "slcan"
+                    logger.info(f"Auto-detected slcan interface for port {self.port}")
+                else:
+                    self.can_interface = "socketcan"
+                    logger.info(f"Auto-detected socketcan interface for port {self.port}")
+
+            kwargs = {
+                "channel": self.port,
+                "bitrate": self.bitrate,
+                "interface": self.can_interface,
+            }
+
+            if self.can_interface == "socketcan" and self.use_can_fd and self.data_bitrate is not None:
+                kwargs.update({"data_bitrate": self.data_bitrate, "fd": True})
+                logger.info(
+                    f"Connected to {self.port} with CAN FD (bitrate={self.bitrate}, data_bitrate={self.data_bitrate})"
+                )
+            else:
+                logger.info(f"Connected to {self.port} with {self.can_interface} (bitrate={self.bitrate})")
+
+            self.canbus = can.interface.Bus(**kwargs)
+
+            self._is_connected = True
+
+            if handshake:
+                self._handshake()
+
+            logger.debug(f"{self.__class__.__name__} connected via {self.can_interface}.")
+        except Exception as e:
+            self._is_connected = False
+            raise ConnectionError(f"Failed to connect to CAN bus: {e}") from e
+
+    def _query_status_via_clear_fault(self, motor: NameOrID) -> tuple[bool, can.Message | None]:
+        motor_name = self._get_motor_name(motor)
+        motor_id = self._get_motor_id(motor_name)
+        recv_id = self._get_motor_recv_id(motor_name)
+        data = [0xFF] * 7 + [CAN_CMD_CLEAR_FAULT]
+        msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False)
+        self._bus().send(msg)
+        return self._recv_status_via_clear_fault(expected_recv_id=recv_id)
+
+    def _recv_status_via_clear_fault(
+        self, expected_recv_id: int | None = None, timeout: float = RUNNING_TIMEOUT
+    ) -> tuple[bool, can.Message | None]:
+        """
+        Poll the bus for a response to a fault-clear request.
+
+        Args:
+            expected_recv_id: Only accept frames from this CAN ID when provided.
+            timeout: Maximum time spent polling the bus in seconds.
+
+        Returns:
+            Tuple where the first element is True if a fault frame was received,
+            and the second element is the CAN message (or None on timeout).
+        """
+        start_time = time.time()
+
+        while time.time() - start_time < timeout:
+            msg = self._bus().recv(timeout=RUNNING_TIMEOUT / 10)
+            if not msg:
+                continue
+
+            if expected_recv_id is not None and msg.data[0] != expected_recv_id:
+                continue
+
+            # Fault-status frame heuristic (doc-based)
+            fault_bits = int.from_bytes(msg.data[1:5], "little")
+            if fault_bits != 0 and msg.data[5] == msg.data[6] == msg.data[7] == 0:
+                logger.error(
+                    f"Motor fault received from CAN ID 0x{msg.arbitration_id:02X}: "
+                    f"fault_bits=0x{fault_bits:08X}"
+                )
+                return True, msg
+
+            # Otherwise: valid normal response
+            return False, msg
+
+        return False, None
+
+    def update_motor_state(self, motor: NameOrID) -> bool:
+        has_fault, msg = self._query_status_via_clear_fault(motor)
+        if msg is None:
+            logger.warning(f"No response received from motor '{motor}' during state update.")
+            raise ConnectionError(f"No response received from motor '{motor}' during state update.")
+        if has_fault:
+            logger.error(f"Fault reported by motor '{motor}' during state update. msg={msg.data.hex()}")
+            raise RuntimeError(f"Fault reported by motor '{motor}' during state update.")
+
+        self._decode_motor_state(msg.data)  # updates cache
+        return True
+
+    def _handshake(self) -> None:
+        logger.info("Starting handshake with motors...")
+        missing_motors = []
+        faulted_motors = []
+
+        for motor_name in self.motors:
+            has_fault, msg = self._query_status_via_clear_fault(motor_name)
+            if msg is None:
+                missing_motors.append(motor_name)
+            elif has_fault:
+                faulted_motors.append(motor_name)
+            else:
+                # CLEAR_FAULT responses are not guaranteed to always match the MIT feedback layout
+                # on all firmware versions. Handshake should not fail just because cache warm-up fails.
+                try:
+                    self._decode_motor_state(msg.data)
+                except Exception as e:
+                    logger.debug(
+                        "Handshake cache warm-up decode failed for motor '%s': %s",
+                        motor_name,
+                        e,
+                    )
+            time.sleep(0.01)
+
+        if missing_motors or faulted_motors:
+            details = []
+            if missing_motors:
+                details.append(f"did not respond: {missing_motors}")
+            if faulted_motors:
+                details.append(f"reported fault: {faulted_motors}")
+            raise ConnectionError("Handshake failed. " + "; ".join(details))
+
+        logger.info("Handshake successful. All motors ready.")
+
+    def _switch_operation_mode(self, motor: NameOrID, mode: ControlMode) -> None:
+        """Switch the operation mode of a motor."""
+        motor_name = self._get_motor_name(motor)
+        motor_id = self._get_motor_id(motor_name)
+        recv_id = self._get_motor_recv_id(motor_name)
+        data = [0xFF] * 8
+        data[6] = mode.value
+        data[7] = 0xFC
+        msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False)
+        self._bus().send(msg)
+        msg = self._recv_motor_response(expected_recv_id=recv_id, timeout=PARAM_TIMEOUT)
+        if msg is not None:
+            self.operation_mode[motor_name] = mode
+
+    @check_if_not_connected
+    def disconnect(self, disable_torque: bool = True) -> None:
+        """
+        Close the CAN bus connection.
+
+        Args:
+            disable_torque: If True, disable torque on all motors before disconnecting
+        """
+        if disable_torque:
+            try:
+                self.disable_torque()
+            except Exception as e:
+                logger.warning(f"Failed to disable torque during disconnect: {e}")
+
+        if self.canbus:
+            self.canbus.shutdown()
+            self.canbus = None
+        self._is_connected = False
+        logger.debug(f"{self.__class__.__name__} disconnected.")
+
+    def configure_motors(self) -> None:
+        """Configure all motors with default settings."""
+        # Robstride motors don't require much configuration in MIT mode
+        # Just ensure they're enabled
+        for motor in self.motors:
+            self._enable_motor(self._get_motor_name(motor))
+            self._switch_operation_mode(motor, ControlMode.MIT)
+            time.sleep(0.01)
+
+    def switch_to_mode(self, mode: ControlMode) -> None:
+        """Switch operation mode on selected motors."""
+        for motor in self.motors:
+            self._switch_operation_mode(motor, mode)
+            time.sleep(0.01)
+
+    def _enable_motor(self, motor: NameOrID) -> None:
+        """Enable a single motor."""
+        motor_id = self._get_motor_id(motor)
+        recv_id = self._get_motor_recv_id(motor)
+        data = [0xFF] * 7 + [CAN_CMD_ENABLE]
+        msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False)
+        self._bus().send(msg)
+        self._recv_motor_response(expected_recv_id=recv_id, timeout=PARAM_TIMEOUT)
+
+    def _disable_motor(self, motor: NameOrID) -> None:
+        """Disable a single motor."""
+        motor_id = self._get_motor_id(motor)
+        recv_id = self._get_motor_recv_id(motor)
+        data = [0xFF] * 7 + [CAN_CMD_DISABLE]
+        msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False)
+        self._bus().send(msg)
+        self._recv_motor_response(expected_recv_id=recv_id)
+
+    def enable_torque(self, motors: str | list[str] | None = None, num_retry: int = 0) -> None:
+        """Enable torque on selected motors."""
+        motors = self._get_motors_list(motors)
+        for motor in motors:
+            for _ in range(num_retry + 1):
+                try:
+                    self._get_motor_name(motor)
+                    self._enable_motor(self._get_motor_name(motor))
+                    break
+                except Exception as e:
+                    if _ == num_retry:
+                        raise e
+                    time.sleep(0.01)
+
+    def disable_torque(self, motors: str | list[str] | None = None, num_retry: int = 0) -> None:
+        """Disable torque on selected motors."""
+        motors = self._get_motors_list(motors)
+        for motor in motors:
+            for _ in range(num_retry + 1):
+                try:
+                    self._disable_motor(self._get_motor_name(motor))
+                    break
+                except Exception as e:
+                    if _ == num_retry:
+                        raise e
+                    time.sleep(0.01)
+
+    @contextmanager
+    def torque_disabled(self, motors: str | list[str] | None = None):
+        """
+        Context manager that guarantees torque is re-enabled.
+
+        This helper is useful to temporarily disable torque when configuring motors.
+
+        Examples:
+            >>> with bus.torque_disabled():
+            ...     # Safe operations here with torque disabled
+            ...     pass
+        """
+        self.disable_torque(motors)
+        try:
+            yield
+        finally:
+            self.enable_torque(motors)
+
+    def set_zero_position(self, motors: str | list[str] | None = None) -> None:
+        """Set current position as zero for selected motors."""
+        motors = self._get_motors_list(motors)
+        for motor in motors:
+            motor_id = self._get_motor_id(motor)
+            recv_id = self._get_motor_recv_id(motor)
+            data = [0xFF] * 7 + [CAN_CMD_SET_ZERO]
+            msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False)
+            self._bus().send(msg)
+            self._recv_motor_response(expected_recv_id=recv_id)
+            time.sleep(0.01)
+
+    def _recv_motor_response(
+        self, expected_recv_id: int | None = None, timeout: float = 0.001
+    ) -> can.Message | None:
+        """
+        Receive a response from a motor.
+
+        Args:
+            expected_recv_id: If provided, only return messages from this CAN ID
+            timeout: Timeout in seconds (default: 1ms for high-speed operation)
+
+        Returns:
+            CAN message if received, None otherwise
+        """
+        try:
+            start_time = time.time()
+            messages_seen = []
+            while time.time() - start_time < timeout:
+                msg = self._bus().recv(timeout=RUNNING_TIMEOUT / 10)  # 100us timeout for fast polling
+                if msg:
+                    messages_seen.append(f"0x{msg.arbitration_id:02X}")
+                    # If no filter specified, return any message
+                    if expected_recv_id is None:
+                        return msg
+                    # Otherwise, only return if it matches the expected recv_id
+                    if msg.data[0] == expected_recv_id:
+                        return msg
+                    else:
+                        logger.debug(
+                            f"Ignoring message from CAN ID 0x{msg.arbitration_id:02X}, expected 0x{expected_recv_id:02X}"
+                        )
+
+            # Only log warnings if we're in debug mode to reduce overhead
+            if logger.isEnabledFor(logging.DEBUG):
+                if messages_seen:
+                    logger.debug(
+                        f"Received {len(messages_seen)} message(s) from IDs {set(messages_seen)}, but expected 0x{expected_recv_id:02X}"
+                    )
+                else:
+                    logger.debug(f"No CAN messages received (expected from 0x{expected_recv_id:02X})")
+        except Exception as e:
+            logger.debug(f"Failed to receive CAN message: {e}")
+        return None
+
+    def _recv_all_responses(
+        self, expected_recv_ids: list[int], timeout: float = 0.002
+    ) -> dict[int, can.Message]:
+        """
+        Efficiently receive responses from multiple motors at once.
+        Uses the OpenArms pattern: collect all available messages within timeout.
+
+        Args:
+            expected_recv_ids: List of CAN IDs we expect responses from
+            timeout: Total timeout in seconds (default: 2ms)
+
+        Returns:
+            Dictionary mapping recv_id to CAN message
+        """
+        responses: dict[int, can.Message] = {}
+        expected_set = set(expected_recv_ids)
+        start_time = time.time()
+
+        try:
+            while len(responses) < len(expected_recv_ids) and (time.time() - start_time) < timeout:
+                msg = self._bus().recv(timeout=RUNNING_TIMEOUT / 10)  # 100us poll timeout
+                if msg and msg.data[0] in expected_set:
+                    responses[msg.data[0]] = msg
+                    if len(responses) == len(expected_recv_ids):
+                        break  # Got all responses, exit early
+        except Exception as e:
+            logger.debug(f"Error receiving responses: {e}")
+
+        return responses
+
+    def _speed_control(
+        self,
+        motor: NameOrID,
+        velocity_deg_per_sec: float,
+        current_limit_a: float,
+    ) -> None:
+        """
+        Send a Velocity Mode Control Command (Command 11) to a single motor.
+
+        Args:
+            motor: Motor name or CAN ID.
+            velocity_rad_per_sec: Target speed in rad/s (32-bit float).
+            current_limit_a: Current limit in A (32-bit float).
+        """
+        if not self.is_connected:
+            raise DeviceNotConnectedError(f"{self} is not connected.")
+
+        motor_id = self._get_motor_id(motor)
+        motor_name = self._get_motor_name(motor)
+        # Optional: ensure the motor is in velocity control mode
+
+        if self.operation_mode[motor_name] != ControlMode.VEL:
+            raise RuntimeError(f"Motor '{motor_name}' is not in velocity control mode.")
+        # Convert to rad/s to match protocol specification
+
+        velocity_rad_per_sec = np.radians(velocity_deg_per_sec)
+
+        # Encode float32 little-endian without struct (byte list)
+        def _float32_to_le_bytes(x: float) -> list[int]:
+            b = np.float32(x).tobytes()  # 4 bytes, little-endian
+            return [b[0], b[1], b[2], b[3]]
+
+        speed_bytes = _float32_to_le_bytes(velocity_rad_per_sec)
+        limit_bytes = _float32_to_le_bytes(current_limit_a)
+
+        data = speed_bytes + limit_bytes  # 8 octets : [0–3]=speed, [4–7]=current limit
+
+        msg = can.Message(
+            arbitration_id=motor_id,
+            data=data,
+            is_extended_id=False,
+        )
+        self._bus().send(msg)
+
+        # Si le proto renvoie une réponse type état, on peut la décoder comme pour MIT
+        recv_id = self._get_motor_recv_id(motor)
+        if recv_id is not None:
+            resp = self._recv_motor_response(expected_recv_id=recv_id)
+            if resp:
+                self._decode_motor_state(resp.data)
+
+    def _mit_control(
+        self,
+        motor: NameOrID,
+        kp: float,
+        kd: float,
+        position_degrees: float,
+        velocity_deg_per_sec: float,
+        torque: float,
+        *,
+        wait_for_response: bool = True,
+    ) -> None:
+        """
+        Send MIT control command to a motor.
+
+        Args:
+            motor: Motor name or ID
+            kp: Position gain
+            kd: Velocity gain
+            position_degrees: Target position (degrees)
+            velocity_deg_per_sec: Target velocity (degrees/s)
+            torque: Target torque (N·m)
+        """
+        motor_name = self._get_motor_name(motor)
+        motor_type = self._motor_types[motor_name]
+        if self.operation_mode[motor_name] != ControlMode.MIT:
+            raise RuntimeError(f"Motor '{motor_name}' is not in MIT control mode.")
+        motor_id = self._get_motor_id(motor)
+        data = self._encode_mit_packet(motor_type, kp, kd, position_degrees, velocity_deg_per_sec, torque)
+        msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False)
+        self._bus().send(msg)
+
+        if wait_for_response:
+            recv_id = self._get_motor_recv_id(motor)
+            msg = self._recv_motor_response(expected_recv_id=recv_id)
+            if msg:
+                self._process_response(motor_name, msg)
+
+    def _encode_mit_packet(
+        self,
+        motor_type: MotorType,
+        kp: float,
+        kd: float,
+        position_degrees: float,
+        velocity_deg_per_sec: float,
+        torque: float,
+    ) -> list[int]:
+        """Encode an MIT control command payload from physical units."""
+        position_rad = np.radians(position_degrees)
+        velocity_rad_per_sec = np.radians(velocity_deg_per_sec)
+        pmax, vmax, tmax = MOTOR_LIMIT_PARAMS[motor_type]
+
+        kp_uint = self._float_to_uint(kp, 0, 500, 12)
+        kd_uint = self._float_to_uint(kd, 0, 5, 12)
+        q_uint = self._float_to_uint(position_rad, -pmax, pmax, 16)
+        dq_uint = self._float_to_uint(velocity_rad_per_sec, -vmax, vmax, 12)
+        tau_uint = self._float_to_uint(torque, -tmax, tmax, 12)
+
+        data = [0] * 8
+        data[0] = (q_uint >> 8) & 0xFF
+        data[1] = q_uint & 0xFF
+        data[2] = dq_uint >> 4
+        data[3] = ((dq_uint & 0xF) << 4) | ((kp_uint >> 8) & 0xF)
+        data[4] = kp_uint & 0xFF
+        data[5] = kd_uint >> 4
+        data[6] = ((kd_uint & 0xF) << 4) | ((tau_uint >> 8) & 0xF)
+        data[7] = tau_uint & 0xFF
+        return data
+
+    def _mit_control_batch(
+        self,
+        commands: dict[NameOrID, tuple[float, float, float, float, float]],
+    ) -> None:
+        """Send MIT commands in batch and update cache from collected responses."""
+        if not commands:
+            return
+
+        recv_id_to_motor: dict[int, str] = {}
+        for motor, (kp, kd, position_degrees, velocity_deg_per_sec, torque) in commands.items():
+            motor_name = self._get_motor_name(motor)
+            if self.operation_mode[motor_name] != ControlMode.MIT:
+                raise RuntimeError(f"Motor '{motor_name}' is not in MIT control mode.")
+
+            motor_id = self._get_motor_id(motor)
+            motor_type = self._motor_types[motor_name]
+            data = self._encode_mit_packet(motor_type, kp, kd, position_degrees, velocity_deg_per_sec, torque)
+            msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False)
+            self._bus().send(msg)
+            recv_id_to_motor[self._get_motor_recv_id(motor)] = motor_name
+
+        responses = self._recv_all_responses(list(recv_id_to_motor.keys()), timeout=RUNNING_TIMEOUT)
+        for recv_id, motor_name in recv_id_to_motor.items():
+            if msg := responses.get(recv_id):
+                self._process_response(motor_name, msg)
+
+    def _float_to_uint(self, x: float, x_min: float, x_max: float, bits: int) -> int:
+        """Convert float to unsigned integer for CAN transmission."""
+        x = max(x_min, min(x_max, x))  # Clamp to range
+        span = x_max - x_min
+        data_norm = (x - x_min) / span
+        return int(data_norm * ((1 << bits) - 1))
+
+    def _uint_to_float(self, x: int, x_min: float, x_max: float, bits: int) -> float:
+        """Convert unsigned integer from CAN to float."""
+        span = x_max - x_min
+        data_norm = float(x) / ((1 << bits) - 1)
+        return data_norm * span + x_min
+
+    def _decode_motor_state(self, data: bytearray | bytes) -> tuple[float, float, float, float]:
+        """
+        Decode motor state from CAN data.
+
+        Returns:
+            Tuple of (position_degrees, velocity_deg_per_sec, torque, temp_mos)
+        """
+        if len(data) < 8:
+            raise ValueError("Invalid motor state data")
+
+        # Extract encoded values
+        motor_id = data[0]
+        motor_name = self._id_to_name[motor_id]
+        q_uint = (data[1] << 8) | data[2]
+        dq_uint = (data[3] << 4) | (data[4] >> 4)
+        tau_uint = ((data[4] & 0x0F) << 8) | data[5]
+        t_mos = (data[6] << 8) | data[7]
+
+        motor_type = self._motor_types[motor_name]
+        # Get motor limits
+        pmax, vmax, tmax = MOTOR_LIMIT_PARAMS[motor_type]
+
+        # Decode to physical values (radians)
+        position_rad = self._uint_to_float(q_uint, -pmax, pmax, 16)
+        velocity_rad_per_sec = self._uint_to_float(dq_uint, -vmax, vmax, 12)
+        torque = self._uint_to_float(tau_uint, -tmax, tmax, 12)
+
+        # Convert to degrees
+        position_degrees = np.degrees(position_rad)
+        velocity_deg_per_sec = np.degrees(velocity_rad_per_sec)
+
+        # Update cached state
+        self.last_feedback_time[motor_name] = time.time()
+        self._last_known_states[motor_name] = {
+            "position": position_degrees,
+            "velocity": velocity_deg_per_sec,
+            "torque": torque,
+            "temp_mos": t_mos / 10,
+            # Not available in Robstride MIT feedback.
+            "temp_rotor": 0.0,
+        }
+        return position_degrees, velocity_deg_per_sec, torque, t_mos / 10
+
+    def _process_response(self, motor: str, msg: can.Message) -> None:
+        """Decode a feedback frame and update the cache for one motor."""
+        try:
+            self._decode_motor_state(msg.data)
+        except Exception as e:
+            logger.warning(f"Failed to decode response from {motor}: {e}")
+
+    def _get_cached_value(self, motor: str, data_name: str) -> Value:
+        """Retrieve a specific value from the state cache."""
+        state = self._last_known_states[motor]
+        mapping: dict[str, Any] = {
+            "Present_Position": state["position"],
+            "Present_Velocity": state["velocity"],
+            "Present_Torque": state["torque"],
+            "Temperature_MOS": state["temp_mos"],
+        }
+        if data_name == "Temperature_Rotor":
+            raise NotImplementedError("Rotor temperature reading not accessible.")
+        if data_name not in mapping:
+            raise ValueError(f"Unknown data_name: {data_name}")
+        return mapping[data_name]
+
+    @check_if_not_connected
+    def read(
+        self,
+        data_name: str,
+        motor: str,
+    ) -> Value:
+        """Read a value from a single motor. Positions are always in degrees."""
+
+        # Refresh motor to get latest state
+        t_init = time.time()
+        if (
+            self.last_feedback_time[motor] is None
+            or t_init - (self.last_feedback_time[motor] or 0) > STATE_CACHE_TTL_S
+        ):
+            self.update_motor_state(motor)
+
+        return self._get_cached_value(motor, data_name)
+
+    @check_if_not_connected
+    def write(
+        self,
+        data_name: str,
+        motor: str,
+        value: Value,
+    ) -> None:
+        """Write a value to a single motor. Positions are always in degrees."""
+        motor_name = self._get_motor_name(motor)
+
+        if data_name in ("Kp", "Kd"):
+            self._gains[motor_name][data_name.lower()] = float(value)
+        elif data_name == "Goal_Position":
+            # Use MIT control with position in degrees
+            kp = self._gains[motor_name]["kp"]
+            kd = self._gains[motor_name]["kd"]
+            self._mit_control(motor, kp, kd, value, 0, 0)
+        elif data_name == "Goal_Velocity":
+            # Use Velocity control mode
+            if self.operation_mode[motor_name] != ControlMode.VEL:
+                raise RuntimeError(f"Motor '{motor_name}' is not in velocity control mode.")
+            current_limit_a = 5.0  # Example current limit / not specified in doc. This mode is rarely used and primarily intended for diagnostics
+            self._speed_control(motor, value, current_limit_a)
+        else:
+            raise ValueError(f"Writing {data_name} not supported in MIT mode")
+
+    def sync_read(
+        self,
+        data_name: str,
+        motors: str | list[str] | None = None,
+    ) -> dict[str, Value]:
+        """
+        Read the same value from multiple motors simultaneously.
+        Uses batched operations: sends all refresh commands, then collects all responses.
+        This is MUCH faster than sequential reads (OpenArms pattern).
+        """
+        target_motors = self._get_motors_list(motors)
+        self._batch_refresh(target_motors)
+        return {motor: self._get_cached_value(motor, data_name) for motor in target_motors}
+
+    @check_if_not_connected
+    def sync_write(
+        self,
+        data_name: str,
+        values: dict[str, Value],
+    ) -> None:
+        """
+        Write different values to multiple motors simultaneously. Positions are always in degrees.
+        Uses batched operations: sends all commands first, then collects responses when MIT mode is used, otherwise send cmd and wait for response for each motor).
+        """
+        if data_name in ("Kp", "Kd"):
+            key = data_name.lower()
+            for motor, val in values.items():
+                motor_name = self._get_motor_name(motor)
+                self._gains[motor_name][key] = float(val)
+        elif data_name == "Goal_Position":
+            commands: dict[NameOrID, tuple[float, float, float, float, float]] = {}
+            for motor, value_degrees in values.items():
+                motor_name = self._get_motor_name(motor)
+                commands[motor] = (
+                    self._gains[motor_name]["kp"],
+                    self._gains[motor_name]["kd"],
+                    float(value_degrees),
+                    0.0,
+                    0.0,
+                )
+            self._mit_control_batch(commands)
+        else:
+            # Fall back to individual writes for other data types
+            for motor, value in values.items():
+                self.write(data_name, motor, value)
+
+    def sync_read_all_states(
+        self,
+        motors: str | list[str] | None = None,
+        *,
+        num_retry: int = 0,
+    ) -> dict[str, MotorState]:
+        """
+        Read ALL motor states (position, velocity, torque) with Robstride TTL refresh policy.
+        """
+        target_motors = self._get_motors_list(motors)
+        self._batch_refresh(target_motors)
+        return {motor: self._last_known_states[motor].copy() for motor in target_motors}
+
+    def _batch_refresh(self, motors: list[str]) -> None:
+        """Refresh a set of motors and update the feedback cache."""
+        init_time = time.time()
+        updated_motors: list[str] = []
+
+        for motor in motors:
+            if (
+                self.last_feedback_time[motor] is not None
+                and (init_time - (self.last_feedback_time[motor] or 0)) < STATE_CACHE_TTL_S
+            ):
+                continue
+            motor_id = self._get_motor_id(motor)
+            data = [0xFF] * 7 + [CAN_CMD_CLEAR_FAULT]
+            msg = can.Message(arbitration_id=motor_id, data=data, is_extended_id=False)
+            self._bus().send(msg)
+            updated_motors.append(motor)
+
+        expected_recv_ids = [self._get_motor_recv_id(motor) for motor in updated_motors]
+        responses = self._recv_all_responses(expected_recv_ids, timeout=RUNNING_TIMEOUT)
+
+        for response in responses.values():
+            payload_motor_name = self._recv_id_to_motor.get(response.data[0])
+            if payload_motor_name is not None:
+                self._process_response(payload_motor_name, response)
+            else:
+                # Fallback: still attempt to decode based on payload byte0 mapping.
+                self._decode_motor_state(response.data)
+
+        for motor in updated_motors:
+            recv_id = self._get_motor_recv_id(motor)
+            if recv_id not in responses:
+                logger.warning(f"Packet drop: {motor} (ID: 0x{recv_id:02X}). Using last known state.")
+
+    def read_calibration(self) -> dict[str, MotorCalibration]:
+        """Read calibration data from motors."""
+        # Robstride motors don't store calibration internally
+        # Return existing calibration or empty dict
+        return self.calibration if self.calibration else {}
+
+    def write_calibration(self, calibration_dict: dict[str, MotorCalibration], cache: bool = True) -> None:
+        """Write calibration data to motors."""
+        # Robstride motors don't store calibration internally
+        # Just cache it in memory
+        if cache:
+            self.calibration = calibration_dict
+
+    def record_ranges_of_motion(
+        self, motors: str | list[str] | None = None, display_values: bool = True
+    ) -> tuple[dict[str, Value], dict[str, Value]]:
+        """
+        Interactively record the min/max values of each motor in degrees.
+
+        Move the joints by hand (with torque disabled) while the method streams live positions.
+        Press Enter to finish.
+        """
+        target_motors = self._get_motors_list(motors)
+
+        # Disable torque for manual movement
+        self.disable_torque(target_motors)
+        time.sleep(0.1)
+
+        # Get initial positions (already in degrees)
+        start_positions = self.sync_read("Present_Position", target_motors)
+        mins = start_positions.copy()
+        maxes = start_positions.copy()
+
+        print("\nMove joints through their full range of motion. Press ENTER when done.")
+        user_pressed_enter = False
+
+        while not user_pressed_enter:
+            positions = self.sync_read("Present_Position", target_motors)
+
+            for motor in target_motors:
+                if motor in positions:
+                    mins[motor] = int(
+                        min(
+                            positions[motor],
+                            mins.get(motor, positions[motor]),
+                        )
+                    )
+                    maxes[motor] = int(
+                        max(
+                            positions[motor],
+                            maxes.get(motor, positions[motor]),
+                        )
+                    )
+
+            if display_values:
+                print("\n" + "=" * 50)
+                print(f"{'MOTOR':<20} | {'MIN (deg)':>12} | {'POS (deg)':>12} | {'MAX (deg)':>12}")
+                print("-" * 50)
+                for motor in target_motors:
+                    if motor in positions:
+                        print(
+                            f"{motor:<20} | {mins[motor]:>12.1f} | {positions[motor]:>12.1f} | {maxes[motor]:>12.1f}"
+                        )
+
+            if enter_pressed():
+                user_pressed_enter = True
+
+            if display_values and not user_pressed_enter:
+                # Move cursor up to overwrite the previous output
+                move_cursor_up(len(target_motors) + 4)
+
+            time.sleep(0.05)
+
+        # Re-enable torque
+        self.enable_torque(target_motors)
+
+        # Validate ranges
+        for motor in target_motors:
+            if (motor in mins) and (motor in maxes) and (abs(maxes[motor] - mins[motor]) < 5.0):
+                raise ValueError(f"Motor {motor} has insufficient range of motion (< 5 degrees)")
+
+        return mins, maxes
+
+    def _get_motors_list(self, motors: str | list[str] | None) -> list[str]:
+        """Convert motor specification to list of motor names."""
+        if motors is None:
+            return list(self.motors.keys())
+        elif isinstance(motors, str):
+            return [motors]
+        elif isinstance(motors, list):
+            return motors
+        else:
+            raise TypeError(f"Invalid motors type: {type(motors)}")
+
+    def _get_motor_id(self, motor: NameOrID) -> int:
+        """Get CAN ID for a motor."""
+        if isinstance(motor, str):
+            if motor in self.motors:
+                return self.motors[motor].id
+            else:
+                raise ValueError(f"Unknown motor: {motor}")
+        else:
+            return motor
+
+    def _get_motor_name(self, motor: NameOrID) -> str:
+        """Get motor name from name or ID."""
+        if isinstance(motor, str):
+            return motor
+        else:
+            for name, m in self.motors.items():
+                if m.id == motor:
+                    return name
+            raise ValueError(f"Unknown motor ID: {motor}")
+
+    def _get_motor_recv_id(self, motor: NameOrID) -> int:
+        """Return the expected ID found in feedback payload byte0 for this motor.
+
+        Robstride MIT feedback frames encode an ID in data[0]. Some setups expose it as
+        `motor.recv_id`; otherwise we fall back to the configured `motor.id`.
+        """
+        motor_name = self._get_motor_name(motor)
+        motor_obj = self.motors[motor_name]
+
+        recv_id = getattr(motor_obj, "recv_id", None)
+        if recv_id is None:
+            logger.debug(
+                "Motor '%s' has no recv_id; falling back to motor.id=%s for feedback demux.",
+                motor_name,
+                motor_obj.id,
+            )
+            return motor_obj.id
+
+        return recv_id
+
+    @cached_property
+    def is_calibrated(self) -> bool:
+        """Check if motors are calibrated."""
+        return bool(self.calibration)
diff --git a/lerobot/src/lerobot/motors/robstride/tables.py b/lerobot/src/lerobot/motors/robstride/tables.py
new file mode 100644
index 0000000000000000000000000000000000000000..2fc1a97b07bfd5fa39a827bb2a3438cc6659cc57
--- /dev/null
+++ b/lerobot/src/lerobot/motors/robstride/tables.py
@@ -0,0 +1,120 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Configuration tables for Damiao motors."""
+
+from enum import IntEnum
+
+
+# Motor type definitions
+class MotorType(IntEnum):
+    O0 = 0
+    O1 = 1
+    O2 = 2
+    O3 = 3
+    O4 = 4
+    O5 = 5
+    ELO5 = 6
+    O6 = 7
+
+
+class CommMode(IntEnum):
+    PrivateProtocole = 0
+    CANopen = 1
+    MIT = 2
+
+
+# Control modes
+class ControlMode(IntEnum):
+    MIT = 0
+    POS_VEL = 1
+    VEL = 2
+
+
+# Motor limit parameters [PMAX, VMAX, TMAX]
+# PMAX: Maximum position (rad)
+# VMAX: Maximum velocity (rad/s)
+# TMAX: Maximum torque (N·m)
+MOTOR_LIMIT_PARAMS: dict[MotorType, tuple[float, float, float]] = {
+    MotorType.O0: (12.57, 33, 14),
+    MotorType.O1: (12.57, 44, 17),
+    MotorType.O2: (12.57, 33, 20),
+    MotorType.O3: (12.57, 33, 60),
+    MotorType.O4: (12.57, 33, 120),
+    MotorType.O5: (12.57, 50, 5.5),
+    MotorType.ELO5: (12.57, 50, 6),
+    MotorType.O6: (112.5, 50, 36),
+}
+
+# Motor model names
+MODEL_NAMES = {
+    MotorType.O0: "O0",
+    MotorType.O1: "O1",
+    MotorType.O2: "O2",
+    MotorType.O3: "O3",
+    MotorType.O4: "O4",
+    MotorType.O5: "O5",
+    MotorType.ELO5: "ELO5",
+    MotorType.O6: "O6",
+}
+
+# Motor resolution table (encoder counts per revolution)
+MODEL_RESOLUTION = {
+    "O0": 65536,
+    "O1": 65536,
+    "O2": 65536,
+    "O3": 65536,
+    "O4": 65536,
+    "O5": 65536,
+    "ELO5": 65536,
+    "O6": 65536,
+}
+
+# CAN baudrates supported by Robstride motors
+AVAILABLE_BAUDRATES = [
+    1000000,  # 4: 1 mbps (default)
+]
+DEFAULT_BAUDRATE = 1000000
+
+# Default timeout in milliseconds
+DEFAULT_TIMEOUT_MS = 0  # disabled by default, otherwise 20000 is 1s
+
+
+# Data that should be normalized
+NORMALIZED_DATA = ["Present_Position", "Goal_Position"]
+
+
+# MIT control parameter ranges
+MIT_KP_RANGE = (0.0, 500.0)
+MIT_KD_RANGE = (0.0, 5.0)
+
+# CAN frame command IDs
+CAN_CMD_ENABLE = 0xFC
+CAN_CMD_DISABLE = 0xFD
+CAN_CMD_SET_ZERO = 0xFE
+CAN_CMD_CLEAR_FAULT = 0xFB
+
+
+CAN_CMD_QUERY_PARAM = 0x33
+CAN_CMD_WRITE_PARAM = 0x55
+CAN_CMD_SAVE_PARAM = 0xAA
+
+# CAN ID for parameter operations
+CAN_PARAM_ID = 0x7FF
+
+
+RUNNING_TIMEOUT = 0.001
+PARAM_TIMEOUT = 0.01
+
+STATE_CACHE_TTL_S = 0.02
diff --git a/lerobot/src/lerobot/optim/__init__.py b/lerobot/src/lerobot/optim/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..de2c4c99651ba9c01137026bd35ccb155670c22c
--- /dev/null
+++ b/lerobot/src/lerobot/optim/__init__.py
@@ -0,0 +1,15 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .optimizers import OptimizerConfig as OptimizerConfig
diff --git a/lerobot/src/lerobot/optim/factory.py b/lerobot/src/lerobot/optim/factory.py
new file mode 100644
index 0000000000000000000000000000000000000000..69928999307979f4a335b035f0a33bb95183fc8d
--- /dev/null
+++ b/lerobot/src/lerobot/optim/factory.py
@@ -0,0 +1,42 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+
+from torch.optim import Optimizer
+from torch.optim.lr_scheduler import LRScheduler
+
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.policies.pretrained import PreTrainedPolicy
+
+
+def make_optimizer_and_scheduler(
+    cfg: TrainPipelineConfig, policy: PreTrainedPolicy
+) -> tuple[Optimizer, LRScheduler | None]:
+    """Generates the optimizer and scheduler based on configs.
+
+    Args:
+        cfg (TrainPipelineConfig): The training config that contains optimizer and scheduler configs
+        policy (PreTrainedPolicy): The policy config from which parameters and presets must be taken from.
+
+    Returns:
+        tuple[Optimizer, LRScheduler | None]: The couple (Optimizer, Scheduler). Scheduler can be `None`.
+    """
+    params = policy.get_optim_params() if cfg.use_policy_training_preset else policy.parameters()
+    if cfg.optimizer is None:
+        raise ValueError("Optimizer config is required but not provided in TrainPipelineConfig")
+    optimizer = cfg.optimizer.build(params)
+    lr_scheduler = cfg.scheduler.build(optimizer, cfg.steps) if cfg.scheduler is not None else None
+    return optimizer, lr_scheduler
diff --git a/lerobot/src/lerobot/optim/optimizers.py b/lerobot/src/lerobot/optim/optimizers.py
new file mode 100644
index 0000000000000000000000000000000000000000..e2e3d8937363a01ef331ebad0cda60ff3abf6741
--- /dev/null
+++ b/lerobot/src/lerobot/optim/optimizers.py
@@ -0,0 +1,359 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import abc
+from collections.abc import Iterable
+from dataclasses import asdict, dataclass, field
+from pathlib import Path
+from typing import Any
+
+import draccus
+import torch
+from safetensors.torch import load_file, save_file
+
+from lerobot.datasets.io_utils import write_json
+from lerobot.datasets.utils import flatten_dict, unflatten_dict
+from lerobot.utils.constants import (
+    OPTIMIZER_PARAM_GROUPS,
+    OPTIMIZER_STATE,
+)
+from lerobot.utils.io_utils import deserialize_json_into_object
+
+# Type alias for parameters accepted by optimizer build() methods.
+# This matches PyTorch's optimizer signature while also supporting:
+# - dict[str, Parameter]: Named parameters for differential LR by name (e.g., XVLA)
+# - dict[str, Iterable]: Multiple parameter groups for multi-optimizer configs (e.g., SAC)
+OptimizerParams = (
+    Iterable[torch.nn.Parameter]  # From model.parameters()
+    | Iterable[dict[str, Any]]  # List of param groups with lr/weight_decay overrides
+    | dict[str, torch.nn.Parameter]  # From dict(model.named_parameters()) for name-based LR
+    | dict[str, Any]  # For multi-optimizer configs (SAC) with multiple param groups
+)
+
+
+@dataclass
+class OptimizerConfig(draccus.ChoiceRegistry, abc.ABC):
+    lr: float
+    weight_decay: float
+    grad_clip_norm: float
+
+    @property
+    def type(self) -> str:
+        return self.get_choice_name(self.__class__)
+
+    @classmethod
+    def default_choice_name(cls) -> str | None:
+        return "adam"
+
+    @abc.abstractmethod
+    def build(self, params: OptimizerParams) -> torch.optim.Optimizer | dict[str, torch.optim.Optimizer]:
+        """
+        Build the optimizer. It can be a single optimizer or a dictionary of optimizers.
+
+        NOTE: Multiple optimizers are useful when you have different models to optimize.
+        For example, you can have one optimizer for the policy and another one for the value function
+        in reinforcement learning settings.
+
+        Args:
+            params: Parameters to optimize. Accepts multiple formats depending on the optimizer:
+                - Iterable[Parameter]: From model.parameters() - standard PyTorch usage
+                - Iterable[dict]: List of param groups with 'params' key and optional
+                  'lr', 'weight_decay' overrides (e.g., ACT, VQBeT policies)
+                - dict[str, Parameter]: From dict(model.named_parameters()) for optimizers
+                  that apply differential learning rates by parameter name (e.g., XVLA)
+                - dict[str, Iterable]: For multi-optimizer configs where each key maps to
+                  a separate optimizer's parameters (e.g., SAC with actor/critic/temperature)
+
+        Returns:
+            The optimizer or a dictionary of optimizers.
+        """
+        raise NotImplementedError
+
+
+@OptimizerConfig.register_subclass("adam")
+@dataclass
+class AdamConfig(OptimizerConfig):
+    lr: float = 1e-3
+    betas: tuple[float, float] = (0.9, 0.999)
+    eps: float = 1e-8
+    weight_decay: float = 0.0
+    grad_clip_norm: float = 10.0
+
+    def build(self, params: OptimizerParams) -> torch.optim.Optimizer:
+        kwargs = asdict(self)
+        kwargs.pop("grad_clip_norm")
+        return torch.optim.Adam(params, **kwargs)
+
+
+@OptimizerConfig.register_subclass("adamw")
+@dataclass
+class AdamWConfig(OptimizerConfig):
+    lr: float = 1e-3
+    betas: tuple[float, float] = (0.9, 0.999)
+    eps: float = 1e-8
+    weight_decay: float = 1e-2
+    grad_clip_norm: float = 10.0
+
+    def build(self, params: OptimizerParams) -> torch.optim.Optimizer:
+        kwargs = asdict(self)
+        kwargs.pop("grad_clip_norm")
+        return torch.optim.AdamW(params, **kwargs)
+
+
+@OptimizerConfig.register_subclass("sgd")
+@dataclass
+class SGDConfig(OptimizerConfig):
+    lr: float = 1e-3
+    momentum: float = 0.0
+    dampening: float = 0.0
+    nesterov: bool = False
+    weight_decay: float = 0.0
+    grad_clip_norm: float = 10.0
+
+    def build(self, params: OptimizerParams) -> torch.optim.Optimizer:
+        kwargs = asdict(self)
+        kwargs.pop("grad_clip_norm")
+        return torch.optim.SGD(params, **kwargs)
+
+
+@OptimizerConfig.register_subclass("xvla-adamw")
+@dataclass
+class XVLAAdamWConfig(OptimizerConfig):
+    """Custom AdamW optimizer for XVLA with differential learning rates.
+
+    The Vision-Language Model (VLM) is trained with 1/10 of the base learning rate
+    for stable optimization, while all other components use the full LR.
+
+    This LR ratio is crucial for achieving strong and stable finetuning performance.
+
+    Soft-prompts can optionally use a separate learning rate with warm-up support.
+    Set `soft_prompt_lr_scale` to a value < 1.0 (e.g., 0.1) to start soft-prompts
+    at a lower LR. Combine with a warmup scheduler for optimal results.
+
+    Note:
+        Completely matching official reported performance may require an additional
+        warm-up LR schedule for soft-prompts, which can bring minor improvements.
+        When `soft_prompt_warmup_lr_scale` is set, soft-prompts start at
+        `lr * soft_prompt_warmup_lr_scale` and should be warmed up via the scheduler.
+
+    Parameter Groups:
+        - Group 0 (vlm): VLM parameters at lr * 0.1, weight_decay * 0.1
+        - Group 1 (soft_prompts): Soft-prompt parameters at lr * soft_prompt_lr_scale
+        - Group 2 (other): All other parameters at full lr
+    """
+
+    lr: float = 1e-4
+    betas: tuple[float, float] = (0.9, 0.99)
+    eps: float = 1e-8
+    weight_decay: float = 0.0
+    grad_clip_norm: float = 10.0
+    # Soft-prompt specific settings
+    soft_prompt_lr_scale: float = 1.0  # Scale factor for soft-prompt LR (1.0 = same as base LR)
+    soft_prompt_warmup_lr_scale: float | None = None  # If set, start soft-prompts at this scale (e.g., 0.01)
+
+    def build(self, params: OptimizerParams) -> torch.optim.Optimizer:
+        """
+        Build AdamW optimizer with differential learning rates.
+
+        Args:
+            params: Must be a dict[str, Parameter] from dict(model.named_parameters())
+                or equivalent.
+
+        Returns:
+            AdamW optimizer with parameter groups for VLM, soft-prompts, and other components
+
+        Raises:
+            AssertionError: If params is not a dict (e.g., from model.parameters())
+        """
+        assert isinstance(params, dict), "Custom LR optimizer requires `named_parameters()` as inputs."
+
+        vlm_group, soft_prompt_group, other_group = [], [], []
+        for name, p in params.items():
+            if not p.requires_grad:
+                continue
+            if "vlm" in name.lower():
+                vlm_group.append(p)
+            elif "soft_prompt" in name.lower():
+                soft_prompt_group.append(p)
+            else:
+                other_group.append(p)
+
+        # Determine soft-prompt LR
+        soft_prompt_lr = self.lr * self.soft_prompt_lr_scale
+        if self.soft_prompt_warmup_lr_scale is not None:
+            # Start at warmup scale, scheduler will warm up to soft_prompt_lr
+            soft_prompt_lr = self.lr * self.soft_prompt_warmup_lr_scale
+
+        param_groups: list[dict[str, Any]] = [
+            {
+                "params": vlm_group,
+                "lr": self.lr * 0.1,
+                "weight_decay": self.weight_decay * 0.1,
+                "name": "vlm",
+            },
+            {
+                "params": soft_prompt_group,
+                "lr": soft_prompt_lr,
+                "weight_decay": self.weight_decay,
+                "name": "soft_prompts",
+            },
+            {
+                "params": other_group,
+                "lr": self.lr,
+                "weight_decay": self.weight_decay,
+                "name": "other",
+            },
+        ]
+
+        # Filter out empty groups
+        param_groups = [g for g in param_groups if len(g["params"]) > 0]
+
+        return torch.optim.AdamW(
+            param_groups,
+            betas=self.betas,
+            eps=self.eps,
+        )
+
+
+@OptimizerConfig.register_subclass("multi_adam")
+@dataclass
+class MultiAdamConfig(OptimizerConfig):
+    """Configuration for multiple Adam optimizers with different parameter groups.
+
+    This creates a dictionary of Adam optimizers, each with its own hyperparameters.
+
+    Args:
+        lr: Default learning rate (used if not specified for a group)
+        weight_decay: Default weight decay (used if not specified for a group)
+        optimizer_groups: Dictionary mapping parameter group names to their hyperparameters
+        grad_clip_norm: Gradient clipping norm
+    """
+
+    lr: float = 1e-3
+    weight_decay: float = 0.0
+    grad_clip_norm: float = 10.0
+    optimizer_groups: dict[str, dict[str, Any]] = field(default_factory=dict)
+
+    def build(self, params: OptimizerParams) -> dict[str, torch.optim.Optimizer]:
+        """Build multiple Adam optimizers.
+
+        Args:
+            params: Must be a dict[str, Iterable[Parameter]] mapping parameter group names
+                to iterables of parameters. The keys should match the keys in optimizer_groups.
+                Typically from policies that need separate optimizers (e.g., SAC with
+                actor/critic/temperature).
+
+        Returns:
+            Dictionary mapping parameter group names to their optimizers
+
+        Raises:
+            AssertionError: If params is not a dict
+        """
+        assert isinstance(params, dict), "MultiAdamConfig requires a dict of parameter groups as inputs."
+        optimizers = {}
+
+        for name, group_params in params.items():
+            # Get group-specific hyperparameters or use defaults
+            group_config = self.optimizer_groups.get(name, {})
+
+            # Create optimizer with merged parameters (defaults + group-specific)
+            optimizer_kwargs = {
+                "lr": group_config.get("lr", self.lr),
+                "betas": group_config.get("betas", (0.9, 0.999)),
+                "eps": group_config.get("eps", 1e-5),
+                "weight_decay": group_config.get("weight_decay", self.weight_decay),
+            }
+
+            optimizers[name] = torch.optim.Adam(group_params, **optimizer_kwargs)
+
+        return optimizers
+
+
+def save_optimizer_state(
+    optimizer: torch.optim.Optimizer | dict[str, torch.optim.Optimizer], save_dir: Path
+) -> None:
+    """Save optimizer state to disk.
+
+    Args:
+        optimizer: Either a single optimizer or a dictionary of optimizers.
+        save_dir: Directory to save the optimizer state.
+    """
+    if isinstance(optimizer, dict):
+        # Handle dictionary of optimizers
+        for name, opt in optimizer.items():
+            optimizer_dir = save_dir / name
+            optimizer_dir.mkdir(exist_ok=True, parents=True)
+            _save_single_optimizer_state(opt, optimizer_dir)
+    else:
+        # Handle single optimizer
+        _save_single_optimizer_state(optimizer, save_dir)
+
+
+def _save_single_optimizer_state(optimizer: torch.optim.Optimizer, save_dir: Path) -> None:
+    """Save a single optimizer's state to disk."""
+    state = optimizer.state_dict()
+    param_groups = state.pop("param_groups")
+    flat_state = flatten_dict(state)
+    save_file(flat_state, save_dir / OPTIMIZER_STATE)
+    write_json(param_groups, save_dir / OPTIMIZER_PARAM_GROUPS)
+
+
+def load_optimizer_state(
+    optimizer: torch.optim.Optimizer | dict[str, torch.optim.Optimizer], save_dir: Path
+) -> torch.optim.Optimizer | dict[str, torch.optim.Optimizer]:
+    """Load optimizer state from disk.
+
+    Args:
+        optimizer: Either a single optimizer or a dictionary of optimizers.
+        save_dir: Directory to load the optimizer state from.
+
+    Returns:
+        The updated optimizer(s) with loaded state.
+    """
+    if isinstance(optimizer, dict):
+        # Handle dictionary of optimizers
+        loaded_optimizers = {}
+        for name, opt in optimizer.items():
+            optimizer_dir = save_dir / name
+            if optimizer_dir.exists():
+                loaded_optimizers[name] = _load_single_optimizer_state(opt, optimizer_dir)
+            else:
+                loaded_optimizers[name] = opt
+        return loaded_optimizers
+    else:
+        # Handle single optimizer
+        return _load_single_optimizer_state(optimizer, save_dir)
+
+
+def _load_single_optimizer_state(optimizer: torch.optim.Optimizer, save_dir: Path) -> torch.optim.Optimizer:
+    """Load a single optimizer's state from disk."""
+    current_state_dict = optimizer.state_dict()
+    flat_state = load_file(save_dir / OPTIMIZER_STATE)
+    state = unflatten_dict(flat_state)
+
+    # Handle case where 'state' key might not exist (for newly created optimizers)
+    if "state" in state:
+        loaded_state_dict = {"state": {int(k): v for k, v in state["state"].items()}}
+    else:
+        loaded_state_dict = {"state": {}}
+
+    if "param_groups" in current_state_dict:
+        param_groups = deserialize_json_into_object(
+            save_dir / OPTIMIZER_PARAM_GROUPS, current_state_dict["param_groups"]
+        )
+        loaded_state_dict["param_groups"] = param_groups
+
+    optimizer.load_state_dict(loaded_state_dict)
+    return optimizer
diff --git a/lerobot/src/lerobot/optim/schedulers.py b/lerobot/src/lerobot/optim/schedulers.py
new file mode 100644
index 0000000000000000000000000000000000000000..19c3fd7bd681390a3f5b8a9af9548668380f7d19
--- /dev/null
+++ b/lerobot/src/lerobot/optim/schedulers.py
@@ -0,0 +1,143 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import abc
+import logging
+import math
+from dataclasses import asdict, dataclass
+from pathlib import Path
+
+import draccus
+from torch.optim import Optimizer
+from torch.optim.lr_scheduler import LambdaLR, LRScheduler
+
+from lerobot.datasets.io_utils import write_json
+from lerobot.utils.constants import SCHEDULER_STATE
+from lerobot.utils.io_utils import deserialize_json_into_object
+
+
+@dataclass
+class LRSchedulerConfig(draccus.ChoiceRegistry, abc.ABC):
+    num_warmup_steps: int | None
+
+    @property
+    def type(self) -> str:
+        return self.get_choice_name(self.__class__)
+
+    @abc.abstractmethod
+    def build(self, optimizer: Optimizer, num_training_steps: int) -> LRScheduler | None:
+        raise NotImplementedError
+
+
+@LRSchedulerConfig.register_subclass("diffuser")
+@dataclass
+class DiffuserSchedulerConfig(LRSchedulerConfig):
+    name: str = "cosine"
+    num_warmup_steps: int | None = None
+
+    def build(self, optimizer: Optimizer, num_training_steps: int) -> LambdaLR:
+        from diffusers.optimization import get_scheduler
+
+        kwargs = {**asdict(self), "num_training_steps": num_training_steps, "optimizer": optimizer}
+        return get_scheduler(**kwargs)
+
+
+@LRSchedulerConfig.register_subclass("vqbet")
+@dataclass
+class VQBeTSchedulerConfig(LRSchedulerConfig):
+    num_warmup_steps: int
+    num_vqvae_training_steps: int
+    num_cycles: float = 0.5
+
+    def build(self, optimizer: Optimizer, num_training_steps: int) -> LambdaLR:
+        def lr_lambda(current_step):
+            if current_step < self.num_vqvae_training_steps:
+                return float(1)
+            else:
+                adjusted_step = current_step - self.num_vqvae_training_steps
+                if adjusted_step < self.num_warmup_steps:
+                    return float(adjusted_step) / float(max(1, self.num_warmup_steps))
+                progress = float(adjusted_step - self.num_warmup_steps) / float(
+                    max(1, num_training_steps - self.num_warmup_steps)
+                )
+                return max(0.0, 0.5 * (1.0 + math.cos(math.pi * float(self.num_cycles) * 2.0 * progress)))
+
+        return LambdaLR(optimizer, lr_lambda, -1)
+
+
+@LRSchedulerConfig.register_subclass("cosine_decay_with_warmup")
+@dataclass
+class CosineDecayWithWarmupSchedulerConfig(LRSchedulerConfig):
+    """Used by Physical Intelligence to train Pi0.
+
+    Automatically scales warmup and decay steps if num_training_steps < num_decay_steps.
+    This ensures the learning rate schedule completes properly even with shorter training runs.
+    """
+
+    num_warmup_steps: int
+    num_decay_steps: int
+    peak_lr: float
+    decay_lr: float
+
+    def build(self, optimizer: Optimizer, num_training_steps: int) -> LambdaLR:
+        # Auto-scale scheduler parameters if training steps are shorter than configured decay steps
+        actual_warmup_steps = self.num_warmup_steps
+        actual_decay_steps = self.num_decay_steps
+
+        if num_training_steps < self.num_decay_steps:
+            # Calculate scaling factor to fit the schedule into the available training steps
+            scale_factor = num_training_steps / self.num_decay_steps
+            actual_warmup_steps = int(self.num_warmup_steps * scale_factor)
+            actual_decay_steps = num_training_steps
+
+            logging.info(
+                f"Auto-scaling LR scheduler: "
+                f"num_training_steps ({num_training_steps}) < num_decay_steps ({self.num_decay_steps}). "
+                f"Scaling warmup: {self.num_warmup_steps} → {actual_warmup_steps}, "
+                f"decay: {self.num_decay_steps} → {actual_decay_steps} "
+                f"(scale factor: {scale_factor:.3f})"
+            )
+
+        def lr_lambda(current_step):
+            def linear_warmup_schedule(current_step):
+                if current_step <= 0:
+                    return 1 / (actual_warmup_steps + 1)
+                frac = 1 - current_step / actual_warmup_steps
+                return (1 / (actual_warmup_steps + 1) - 1) * frac + 1
+
+            def cosine_decay_schedule(current_step):
+                step = min(current_step, actual_decay_steps)
+                cosine_decay = 0.5 * (1 + math.cos(math.pi * step / actual_decay_steps))
+                alpha = self.decay_lr / self.peak_lr
+                decayed = (1 - alpha) * cosine_decay + alpha
+                return decayed
+
+            if current_step < actual_warmup_steps:
+                return linear_warmup_schedule(current_step)
+
+            return cosine_decay_schedule(current_step)
+
+        return LambdaLR(optimizer, lr_lambda, -1)
+
+
+def save_scheduler_state(scheduler: LRScheduler, save_dir: Path) -> None:
+    state_dict = scheduler.state_dict()
+    write_json(state_dict, save_dir / SCHEDULER_STATE)
+
+
+def load_scheduler_state(scheduler: LRScheduler, save_dir: Path) -> LRScheduler:
+    state_dict = deserialize_json_into_object(save_dir / SCHEDULER_STATE, scheduler.state_dict())
+    scheduler.load_state_dict(state_dict)
+    return scheduler
diff --git a/lerobot/src/lerobot/policies/__init__.py b/lerobot/src/lerobot/policies/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..c7951f028eeeed4c299e7f74910c7d99465a5f6e
--- /dev/null
+++ b/lerobot/src/lerobot/policies/__init__.py
@@ -0,0 +1,41 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .act.configuration_act import ACTConfig as ACTConfig
+from .diffusion.configuration_diffusion import DiffusionConfig as DiffusionConfig
+from .groot.configuration_groot import GrootConfig as GrootConfig
+from .pi0.configuration_pi0 import PI0Config as PI0Config
+from .pi0_fast.configuration_pi0_fast import PI0FastConfig as PI0FastConfig
+from .pi05.configuration_pi05 import PI05Config as PI05Config
+from .smolvla.configuration_smolvla import SmolVLAConfig as SmolVLAConfig
+from .smolvla.processor_smolvla import SmolVLANewLineProcessor
+from .tdmpc.configuration_tdmpc import TDMPCConfig as TDMPCConfig
+from .vqbet.configuration_vqbet import VQBeTConfig as VQBeTConfig
+from .wall_x.configuration_wall_x import WallXConfig as WallXConfig
+from .xvla.configuration_xvla import XVLAConfig as XVLAConfig
+
+__all__ = [
+    "ACTConfig",
+    "DiffusionConfig",
+    "PI0Config",
+    "PI05Config",
+    "PI0FastConfig",
+    "SmolVLAConfig",
+    "SARMConfig",
+    "TDMPCConfig",
+    "VQBeTConfig",
+    "GrootConfig",
+    "XVLAConfig",
+    "WallXConfig",
+]
diff --git a/lerobot/src/lerobot/policies/act/README.md b/lerobot/src/lerobot/policies/act/README.md
new file mode 120000
index 0000000000000000000000000000000000000000..04602009852778a28be44647b6e7ba445dae3a95
--- /dev/null
+++ b/lerobot/src/lerobot/policies/act/README.md
@@ -0,0 +1 @@
+../../../../docs/source/policy_act_README.md
\ No newline at end of file
diff --git a/lerobot/src/lerobot/policies/act/configuration_act.py b/lerobot/src/lerobot/policies/act/configuration_act.py
new file mode 100644
index 0000000000000000000000000000000000000000..bd89185fdcd5b6c635bb6be82179b33d83bba56f
--- /dev/null
+++ b/lerobot/src/lerobot/policies/act/configuration_act.py
@@ -0,0 +1,177 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Tony Z. Zhao and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import NormalizationMode
+from lerobot.optim.optimizers import AdamWConfig
+
+
+@PreTrainedConfig.register_subclass("act")
+@dataclass
+class ACTConfig(PreTrainedConfig):
+    """Configuration class for the Action Chunking Transformers policy.
+
+    Defaults are configured for training on bimanual Aloha tasks like "insertion" or "transfer".
+
+    The parameters you will most likely need to change are the ones which depend on the environment / sensors.
+    Those are: `input_features` and `output_features`.
+
+    Notes on the inputs and outputs:
+        - Either:
+            - At least one key starting with "observation.image is required as an input.
+              AND/OR
+            - The key "observation.environment_state" is required as input.
+        - If there are multiple keys beginning with "observation.images." they are treated as multiple camera
+          views. Right now we only support all images having the same shape.
+        - May optionally work without an "observation.state" key for the proprioceptive robot state.
+        - "action" is required as an output key.
+
+    Args:
+        n_obs_steps: Number of environment steps worth of observations to pass to the policy (takes the
+            current step and additional steps going back).
+        chunk_size: The size of the action prediction "chunks" in units of environment steps.
+        n_action_steps: The number of action steps to run in the environment for one invocation of the policy.
+            This should be no greater than the chunk size. For example, if the chunk size size 100, you may
+            set this to 50. This would mean that the model predicts 100 steps worth of actions, runs 50 in the
+            environment, and throws the other 50 out.
+        input_features: A dictionary defining the PolicyFeature of the input data for the policy. The key represents
+            the input data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        output_features: A dictionary defining the PolicyFeature of the output data for the policy. The key represents
+            the output data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        normalization_mapping: A dictionary that maps from a str value of FeatureType (e.g., "STATE", "VISUAL") to
+            a corresponding NormalizationMode (e.g., NormalizationMode.MIN_MAX)
+        vision_backbone: Name of the torchvision resnet backbone to use for encoding images.
+        pretrained_backbone_weights: Pretrained weights from torchvision to initialize the backbone.
+            `None` means no pretrained weights.
+        replace_final_stride_with_dilation: Whether to replace the ResNet's final 2x2 stride with a dilated
+            convolution.
+        pre_norm: Whether to use "pre-norm" in the transformer blocks.
+        dim_model: The transformer blocks' main hidden dimension.
+        n_heads: The number of heads to use in the transformer blocks' multi-head attention.
+        dim_feedforward: The dimension to expand the transformer's hidden dimension to in the feed-forward
+            layers.
+        feedforward_activation: The activation to use in the transformer block's feed-forward layers.
+        n_encoder_layers: The number of transformer layers to use for the transformer encoder.
+        n_decoder_layers: The number of transformer layers to use for the transformer decoder.
+        use_vae: Whether to use a variational objective during training. This introduces another transformer
+            which is used as the VAE's encoder (not to be confused with the transformer encoder - see
+            documentation in the policy class).
+        latent_dim: The VAE's latent dimension.
+        n_vae_encoder_layers: The number of transformer layers to use for the VAE's encoder.
+        temporal_ensemble_coeff: Coefficient for the exponential weighting scheme to apply for temporal
+            ensembling. Defaults to None which means temporal ensembling is not used. `n_action_steps` must be
+            1 when using this feature, as inference needs to happen at every step to form an ensemble. For
+            more information on how ensembling works, please see `ACTTemporalEnsembler`.
+        dropout: Dropout to use in the transformer layers (see code for details).
+        kl_weight: The weight to use for the KL-divergence component of the loss if the variational objective
+            is enabled. Loss is then calculated as: `reconstruction_loss + kl_weight * kld_loss`.
+    """
+
+    # Input / output structure.
+    n_obs_steps: int = 1
+    chunk_size: int = 100
+    n_action_steps: int = 100
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.MEAN_STD,
+            "STATE": NormalizationMode.MEAN_STD,
+            "ACTION": NormalizationMode.MEAN_STD,
+        }
+    )
+
+    # Architecture.
+    # Vision backbone.
+    vision_backbone: str = "resnet18"
+    pretrained_backbone_weights: str | None = "ResNet18_Weights.IMAGENET1K_V1"
+    replace_final_stride_with_dilation: int = False
+    # Transformer layers.
+    pre_norm: bool = False
+    dim_model: int = 512
+    n_heads: int = 8
+    dim_feedforward: int = 3200
+    feedforward_activation: str = "relu"
+    n_encoder_layers: int = 4
+    # Note: Although the original ACT implementation has 7 for `n_decoder_layers`, there is a bug in the code
+    # that means only the first layer is used. Here we match the original implementation by setting this to 1.
+    # See this issue https://github.com/tonyzhaozh/act/issues/25#issue-2258740521.
+    n_decoder_layers: int = 1
+    # VAE.
+    use_vae: bool = True
+    latent_dim: int = 32
+    n_vae_encoder_layers: int = 4
+
+    # Inference.
+    # Note: the value used in ACT when temporal ensembling is enabled is 0.01.
+    temporal_ensemble_coeff: float | None = None
+
+    # Training and loss computation.
+    dropout: float = 0.1
+    kl_weight: float = 10.0
+
+    # Training preset
+    optimizer_lr: float = 1e-5
+    optimizer_weight_decay: float = 1e-4
+    optimizer_lr_backbone: float = 1e-5
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        """Input validation (not exhaustive)."""
+        if not self.vision_backbone.startswith("resnet"):
+            raise ValueError(
+                f"`vision_backbone` must be one of the ResNet variants. Got {self.vision_backbone}."
+            )
+        if self.temporal_ensemble_coeff is not None and self.n_action_steps > 1:
+            raise NotImplementedError(
+                "`n_action_steps` must be 1 when using temporal ensembling. This is "
+                "because the policy needs to be queried every step to compute the ensembled action."
+            )
+        if self.n_action_steps > self.chunk_size:
+            raise ValueError(
+                f"The chunk size is the upper bound for the number of action steps per model invocation. Got "
+                f"{self.n_action_steps} for `n_action_steps` and {self.chunk_size} for `chunk_size`."
+            )
+        if self.n_obs_steps != 1:
+            raise ValueError(
+                f"Multiple observation steps not handled yet. Got `nobs_steps={self.n_obs_steps}`"
+            )
+
+    def get_optimizer_preset(self) -> AdamWConfig:
+        return AdamWConfig(
+            lr=self.optimizer_lr,
+            weight_decay=self.optimizer_weight_decay,
+        )
+
+    def get_scheduler_preset(self) -> None:
+        return None
+
+    def validate_features(self) -> None:
+        if not self.image_features and not self.env_state_feature:
+            raise ValueError("You must provide at least one image or the environment state among the inputs.")
+
+    @property
+    def observation_delta_indices(self) -> None:
+        return None
+
+    @property
+    def action_delta_indices(self) -> list:
+        return list(range(self.chunk_size))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/act/modeling_act.py b/lerobot/src/lerobot/policies/act/modeling_act.py
new file mode 100644
index 0000000000000000000000000000000000000000..a5c48eb3dff39e29ce37676cfe5881abee8e9bb7
--- /dev/null
+++ b/lerobot/src/lerobot/policies/act/modeling_act.py
@@ -0,0 +1,746 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Tony Z. Zhao and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Action Chunking Transformer Policy
+
+As per Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware (https://huggingface.co/papers/2304.13705).
+The majority of changes here involve removing unused code, unifying naming, and adding helpful comments.
+"""
+
+import math
+from collections import deque
+from collections.abc import Callable
+from itertools import chain
+
+import einops
+import numpy as np
+import torch
+import torch.nn.functional as F  # noqa: N812
+import torchvision
+from torch import Tensor, nn
+from torchvision.models._utils import IntermediateLayerGetter
+from torchvision.ops.misc import FrozenBatchNorm2d
+
+from lerobot.policies.act.configuration_act import ACTConfig
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.utils.constants import ACTION, OBS_ENV_STATE, OBS_IMAGES, OBS_STATE
+
+
+class ACTPolicy(PreTrainedPolicy):
+    """
+    Action Chunking Transformer Policy as per Learning Fine-Grained Bimanual Manipulation with Low-Cost
+    Hardware (paper: https://huggingface.co/papers/2304.13705, code: https://github.com/tonyzhaozh/act)
+    """
+
+    config_class = ACTConfig
+    name = "act"
+
+    def __init__(
+        self,
+        config: ACTConfig,
+        **kwargs,
+    ):
+        """
+        Args:
+            config: Policy configuration class instance or None, in which case the default instantiation of
+                    the configuration class is used.
+        """
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        self.model = ACT(config)
+
+        if config.temporal_ensemble_coeff is not None:
+            self.temporal_ensembler = ACTTemporalEnsembler(config.temporal_ensemble_coeff, config.chunk_size)
+
+        self.reset()
+
+    def get_optim_params(self) -> dict:
+        # TODO(aliberts, rcadene): As of now, lr_backbone == lr
+        # Should we remove this and just `return self.parameters()`?
+        return [
+            {
+                "params": [
+                    p
+                    for n, p in self.named_parameters()
+                    if not n.startswith("model.backbone") and p.requires_grad
+                ]
+            },
+            {
+                "params": [
+                    p
+                    for n, p in self.named_parameters()
+                    if n.startswith("model.backbone") and p.requires_grad
+                ],
+                "lr": self.config.optimizer_lr_backbone,
+            },
+        ]
+
+    def reset(self):
+        """This should be called whenever the environment is reset."""
+        if self.config.temporal_ensemble_coeff is not None:
+            self.temporal_ensembler.reset()
+        else:
+            self._action_queue = deque([], maxlen=self.config.n_action_steps)
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select a single action given environment observations.
+
+        This method wraps `select_actions` in order to return one action at a time for execution in the
+        environment. It works by managing the actions in a queue and only calling `select_actions` when the
+        queue is empty.
+        """
+        self.eval()  # keeping the policy in eval mode as it could be set to train mode while queue is consumed
+
+        if self.config.temporal_ensemble_coeff is not None:
+            actions = self.predict_action_chunk(batch)
+            action = self.temporal_ensembler.update(actions)
+            return action
+
+        # Action queue logic for n_action_steps > 1. When the action_queue is depleted, populate it by
+        # querying the policy.
+        if len(self._action_queue) == 0:
+            actions = self.predict_action_chunk(batch)[:, : self.config.n_action_steps]
+
+            # `self.model.forward` returns a (batch_size, n_action_steps, action_dim) tensor, but the queue
+            # effectively has shape (n_action_steps, batch_size, *), hence the transpose.
+            self._action_queue.extend(actions.transpose(0, 1))
+        return self._action_queue.popleft()
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor]) -> Tensor:
+        """Predict a chunk of actions given environment observations."""
+        self.eval()
+
+        if self.config.image_features:
+            batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
+            batch[OBS_IMAGES] = [batch[key] for key in self.config.image_features]
+
+        actions = self.model(batch)[0]
+        return actions
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, dict]:
+        """Run the batch through the model and compute the loss for training or validation."""
+        if self.config.image_features:
+            batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
+            batch[OBS_IMAGES] = [batch[key] for key in self.config.image_features]
+
+        actions_hat, (mu_hat, log_sigma_x2_hat) = self.model(batch)
+
+        l1_loss = (
+            F.l1_loss(batch[ACTION], actions_hat, reduction="none") * ~batch["action_is_pad"].unsqueeze(-1)
+        ).mean()
+
+        loss_dict = {"l1_loss": l1_loss.item()}
+        if self.config.use_vae:
+            # Calculate Dₖₗ(latent_pdf || standard_normal). Note: After computing the KL-divergence for
+            # each dimension independently, we sum over the latent dimension to get the total
+            # KL-divergence per batch element, then take the mean over the batch.
+            # (See App. B of https://huggingface.co/papers/1312.6114 for more details).
+            mean_kld = (
+                (-0.5 * (1 + log_sigma_x2_hat - mu_hat.pow(2) - (log_sigma_x2_hat).exp())).sum(-1).mean()
+            )
+            loss_dict["kld_loss"] = mean_kld.item()
+            loss = l1_loss + mean_kld * self.config.kl_weight
+        else:
+            loss = l1_loss
+
+        return loss, loss_dict
+
+
+class ACTTemporalEnsembler:
+    def __init__(self, temporal_ensemble_coeff: float, chunk_size: int) -> None:
+        """Temporal ensembling as described in Algorithm 2 of https://huggingface.co/papers/2304.13705.
+
+        The weights are calculated as wᵢ = exp(-temporal_ensemble_coeff * i) where w₀ is the oldest action.
+        They are then normalized to sum to 1 by dividing by Σwᵢ. Here's some intuition around how the
+        coefficient works:
+            - Setting it to 0 uniformly weighs all actions.
+            - Setting it positive gives more weight to older actions.
+            - Setting it negative gives more weight to newer actions.
+        NOTE: The default value for `temporal_ensemble_coeff` used by the original ACT work is 0.01. This
+        results in older actions being weighed more highly than newer actions (the experiments documented in
+        https://github.com/huggingface/lerobot/pull/319 hint at why highly weighing new actions might be
+        detrimental: doing so aggressively may diminish the benefits of action chunking).
+
+        Here we use an online method for computing the average rather than caching a history of actions in
+        order to compute the average offline. For a simple 1D sequence it looks something like:
+
+        ```
+        import torch
+
+        seq = torch.linspace(8, 8.5, 100)
+        print(seq)
+
+        m = 0.01
+        exp_weights = torch.exp(-m * torch.arange(len(seq)))
+        print(exp_weights)
+
+        # Calculate offline
+        avg = (exp_weights * seq).sum() / exp_weights.sum()
+        print("offline", avg)
+
+        # Calculate online
+        for i, item in enumerate(seq):
+            if i == 0:
+                avg = item
+                continue
+            avg *= exp_weights[:i].sum()
+            avg += item * exp_weights[i]
+            avg /= exp_weights[: i + 1].sum()
+        print("online", avg)
+        ```
+        """
+        self.chunk_size = chunk_size
+        self.ensemble_weights = torch.exp(-temporal_ensemble_coeff * torch.arange(chunk_size))
+        self.ensemble_weights_cumsum = torch.cumsum(self.ensemble_weights, dim=0)
+        self.reset()
+
+    def reset(self):
+        """Resets the online computation variables."""
+        self.ensembled_actions = None
+        # (chunk_size,) count of how many actions are in the ensemble for each time step in the sequence.
+        self.ensembled_actions_count = None
+
+    def update(self, actions: Tensor) -> Tensor:
+        """
+        Takes a (batch, chunk_size, action_dim) sequence of actions, update the temporal ensemble for all
+        time steps, and pop/return the next batch of actions in the sequence.
+        """
+        self.ensemble_weights = self.ensemble_weights.to(device=actions.device)
+        self.ensemble_weights_cumsum = self.ensemble_weights_cumsum.to(device=actions.device)
+        if self.ensembled_actions is None:
+            # Initializes `self._ensembled_action` to the sequence of actions predicted during the first
+            # time step of the episode.
+            self.ensembled_actions = actions.clone()
+            # Note: The last dimension is unsqueeze to make sure we can broadcast properly for tensor
+            # operations later.
+            self.ensembled_actions_count = torch.ones(
+                (self.chunk_size, 1), dtype=torch.long, device=self.ensembled_actions.device
+            )
+        else:
+            # self.ensembled_actions will have shape (batch_size, chunk_size - 1, action_dim). Compute
+            # the online update for those entries.
+            self.ensembled_actions *= self.ensemble_weights_cumsum[self.ensembled_actions_count - 1]
+            self.ensembled_actions += actions[:, :-1] * self.ensemble_weights[self.ensembled_actions_count]
+            self.ensembled_actions /= self.ensemble_weights_cumsum[self.ensembled_actions_count]
+            self.ensembled_actions_count = torch.clamp(self.ensembled_actions_count + 1, max=self.chunk_size)
+            # The last action, which has no prior online average, needs to get concatenated onto the end.
+            self.ensembled_actions = torch.cat([self.ensembled_actions, actions[:, -1:]], dim=1)
+            self.ensembled_actions_count = torch.cat(
+                [self.ensembled_actions_count, torch.ones_like(self.ensembled_actions_count[-1:])]
+            )
+        # "Consume" the first action.
+        action, self.ensembled_actions, self.ensembled_actions_count = (
+            self.ensembled_actions[:, 0],
+            self.ensembled_actions[:, 1:],
+            self.ensembled_actions_count[1:],
+        )
+        return action
+
+
+class ACT(nn.Module):
+    """Action Chunking Transformer: The underlying neural network for ACTPolicy.
+
+    Note: In this code we use the terms `vae_encoder`, 'encoder', `decoder`. The meanings are as follows.
+        - The `vae_encoder` is, as per the literature around variational auto-encoders (VAE), the part of the
+          model that encodes the target data (a sequence of actions), and the condition (the robot
+          joint-space).
+        - A transformer with an `encoder` (not the VAE encoder) and `decoder` (not the VAE decoder) with
+          cross-attention is used as the VAE decoder. For these terms, we drop the `vae_` prefix because we
+          have an option to train this model without the variational objective (in which case we drop the
+          `vae_encoder` altogether, and nothing about this model has anything to do with a VAE).
+
+                                 Transformer
+                                 Used alone for inference
+                                 (acts as VAE decoder
+                                  during training)
+                                ┌───────────────────────┐
+                                │             Outputs   │
+                                │                ▲      │
+                                │     ┌─────►┌───────┐  │
+                   ┌──────┐     │     │      │Transf.│  │
+                   │      │     │     ├─────►│decoder│  │
+              ┌────┴────┐ │     │     │      │       │  │
+              │         │ │     │ ┌───┴───┬─►│       │  │
+              │ VAE     │ │     │ │       │  └───────┘  │
+              │ encoder │ │     │ │Transf.│             │
+              │         │ │     │ │encoder│             │
+              └───▲─────┘ │     │ │       │             │
+                  │       │     │ └▲──▲─▲─┘             │
+                  │       │     │  │  │ │               │
+                inputs    └─────┼──┘  │ image emb.      │
+                                │    state emb.         │
+                                └───────────────────────┘
+    """
+
+    def __init__(self, config: ACTConfig):
+        # BERT style VAE encoder with input tokens [cls, robot_state, *action_sequence].
+        # The cls token forms parameters of the latent's distribution (like this [*means, *log_variances]).
+        super().__init__()
+        self.config = config
+
+        if self.config.use_vae:
+            self.vae_encoder = ACTEncoder(config, is_vae_encoder=True)
+            self.vae_encoder_cls_embed = nn.Embedding(1, config.dim_model)
+            # Projection layer for joint-space configuration to hidden dimension.
+            if self.config.robot_state_feature:
+                self.vae_encoder_robot_state_input_proj = nn.Linear(
+                    self.config.robot_state_feature.shape[0], config.dim_model
+                )
+            # Projection layer for action (joint-space target) to hidden dimension.
+            self.vae_encoder_action_input_proj = nn.Linear(
+                self.config.action_feature.shape[0],
+                config.dim_model,
+            )
+            # Projection layer from the VAE encoder's output to the latent distribution's parameter space.
+            self.vae_encoder_latent_output_proj = nn.Linear(config.dim_model, config.latent_dim * 2)
+            # Fixed sinusoidal positional embedding for the input to the VAE encoder. Unsqueeze for batch
+            # dimension.
+            num_input_token_encoder = 1 + config.chunk_size
+            if self.config.robot_state_feature:
+                num_input_token_encoder += 1
+            self.register_buffer(
+                "vae_encoder_pos_enc",
+                create_sinusoidal_pos_embedding(num_input_token_encoder, config.dim_model).unsqueeze(0),
+            )
+
+        # Backbone for image feature extraction.
+        if self.config.image_features:
+            backbone_model = getattr(torchvision.models, config.vision_backbone)(
+                replace_stride_with_dilation=[False, False, config.replace_final_stride_with_dilation],
+                weights=config.pretrained_backbone_weights,
+                norm_layer=FrozenBatchNorm2d,
+            )
+            # Note: The assumption here is that we are using a ResNet model (and hence layer4 is the final
+            # feature map).
+            # Note: The forward method of this returns a dict: {"feature_map": output}.
+            self.backbone = IntermediateLayerGetter(backbone_model, return_layers={"layer4": "feature_map"})
+
+        # Transformer (acts as VAE decoder when training with the variational objective).
+        self.encoder = ACTEncoder(config)
+        self.decoder = ACTDecoder(config)
+
+        # Transformer encoder input projections. The tokens will be structured like
+        # [latent, (robot_state), (env_state), (image_feature_map_pixels)].
+        if self.config.robot_state_feature:
+            self.encoder_robot_state_input_proj = nn.Linear(
+                self.config.robot_state_feature.shape[0], config.dim_model
+            )
+        if self.config.env_state_feature:
+            self.encoder_env_state_input_proj = nn.Linear(
+                self.config.env_state_feature.shape[0], config.dim_model
+            )
+        self.encoder_latent_input_proj = nn.Linear(config.latent_dim, config.dim_model)
+        if self.config.image_features:
+            self.encoder_img_feat_input_proj = nn.Conv2d(
+                backbone_model.fc.in_features, config.dim_model, kernel_size=1
+            )
+        # Transformer encoder positional embeddings.
+        n_1d_tokens = 1  # for the latent
+        if self.config.robot_state_feature:
+            n_1d_tokens += 1
+        if self.config.env_state_feature:
+            n_1d_tokens += 1
+        self.encoder_1d_feature_pos_embed = nn.Embedding(n_1d_tokens, config.dim_model)
+        if self.config.image_features:
+            self.encoder_cam_feat_pos_embed = ACTSinusoidalPositionEmbedding2d(config.dim_model // 2)
+
+        # Transformer decoder.
+        # Learnable positional embedding for the transformer's decoder (in the style of DETR object queries).
+        self.decoder_pos_embed = nn.Embedding(config.chunk_size, config.dim_model)
+
+        # Final action regression head on the output of the transformer's decoder.
+        self.action_head = nn.Linear(config.dim_model, self.config.action_feature.shape[0])
+
+        self._reset_parameters()
+
+    def _reset_parameters(self):
+        """Xavier-uniform initialization of the transformer parameters as in the original code."""
+        for p in chain(self.encoder.parameters(), self.decoder.parameters()):
+            if p.dim() > 1:
+                nn.init.xavier_uniform_(p)
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, tuple[Tensor, Tensor] | tuple[None, None]]:
+        """A forward pass through the Action Chunking Transformer (with optional VAE encoder).
+
+        `batch` should have the following structure:
+        {
+            [robot_state_feature] (optional): (B, state_dim) batch of robot states.
+
+            [image_features]: (B, n_cameras, C, H, W) batch of images.
+                AND/OR
+            [env_state_feature]: (B, env_dim) batch of environment states.
+
+            [action_feature] (optional, only if training with VAE): (B, chunk_size, action dim) batch of actions.
+        }
+
+        Returns:
+            (B, chunk_size, action_dim) batch of action sequences
+            Tuple containing the latent PDF's parameters (mean, log(σ²)) both as (B, L) tensors where L is the
+            latent dimension.
+        """
+        if self.config.use_vae and self.training:
+            assert ACTION in batch, (
+                "actions must be provided when using the variational objective in training mode."
+            )
+
+        batch_size = batch[OBS_IMAGES][0].shape[0] if OBS_IMAGES in batch else batch[OBS_ENV_STATE].shape[0]
+
+        # Prepare the latent for input to the transformer encoder.
+        if self.config.use_vae and ACTION in batch and self.training:
+            # Prepare the input to the VAE encoder: [cls, *joint_space_configuration, *action_sequence].
+            cls_embed = einops.repeat(
+                self.vae_encoder_cls_embed.weight, "1 d -> b 1 d", b=batch_size
+            )  # (B, 1, D)
+            if self.config.robot_state_feature:
+                robot_state_embed = self.vae_encoder_robot_state_input_proj(batch[OBS_STATE])
+                robot_state_embed = robot_state_embed.unsqueeze(1)  # (B, 1, D)
+            action_embed = self.vae_encoder_action_input_proj(batch[ACTION])  # (B, S, D)
+
+            if self.config.robot_state_feature:
+                vae_encoder_input = [cls_embed, robot_state_embed, action_embed]  # (B, S+2, D)
+            else:
+                vae_encoder_input = [cls_embed, action_embed]
+            vae_encoder_input = torch.cat(vae_encoder_input, axis=1)
+
+            # Prepare fixed positional embedding.
+            # Note: detach() shouldn't be necessary but leaving it the same as the original code just in case.
+            pos_embed = self.vae_encoder_pos_enc.clone().detach()  # (1, S+2, D)
+
+            # Prepare key padding mask for the transformer encoder. We have 1 or 2 extra tokens at the start of the
+            # sequence depending whether we use the input states or not (cls and robot state)
+            # False means not a padding token.
+            cls_joint_is_pad = torch.full(
+                (batch_size, 2 if self.config.robot_state_feature else 1),
+                False,
+                device=batch[OBS_STATE].device,
+            )
+            key_padding_mask = torch.cat(
+                [cls_joint_is_pad, batch["action_is_pad"]], axis=1
+            )  # (bs, seq+1 or 2)
+
+            # Forward pass through VAE encoder to get the latent PDF parameters.
+            cls_token_out = self.vae_encoder(
+                vae_encoder_input.permute(1, 0, 2),
+                pos_embed=pos_embed.permute(1, 0, 2),
+                key_padding_mask=key_padding_mask,
+            )[0]  # select the class token, with shape (B, D)
+            latent_pdf_params = self.vae_encoder_latent_output_proj(cls_token_out)
+            mu = latent_pdf_params[:, : self.config.latent_dim]
+            # This is 2log(sigma). Done this way to match the original implementation.
+            log_sigma_x2 = latent_pdf_params[:, self.config.latent_dim :]
+
+            # Sample the latent with the reparameterization trick.
+            latent_sample = mu + log_sigma_x2.div(2).exp() * torch.randn_like(mu)
+        else:
+            # When not using the VAE encoder, we set the latent to be all zeros.
+            mu = log_sigma_x2 = None
+            # TODO(rcadene, alexander-soare): remove call to `.to` to speedup forward ; precompute and use buffer
+            latent_sample = torch.zeros([batch_size, self.config.latent_dim], dtype=torch.float32).to(
+                batch[OBS_STATE].device
+            )
+
+        # Prepare transformer encoder inputs.
+        encoder_in_tokens = [self.encoder_latent_input_proj(latent_sample)]
+        encoder_in_pos_embed = list(self.encoder_1d_feature_pos_embed.weight.unsqueeze(1))
+        # Robot state token.
+        if self.config.robot_state_feature:
+            encoder_in_tokens.append(self.encoder_robot_state_input_proj(batch[OBS_STATE]))
+        # Environment state token.
+        if self.config.env_state_feature:
+            encoder_in_tokens.append(self.encoder_env_state_input_proj(batch[OBS_ENV_STATE]))
+
+        if self.config.image_features:
+            # For a list of images, the H and W may vary but H*W is constant.
+            # NOTE: If modifying this section, verify on MPS devices that
+            # gradients remain stable (no explosions or NaNs).
+            for img in batch[OBS_IMAGES]:
+                cam_features = self.backbone(img)["feature_map"]
+                cam_pos_embed = self.encoder_cam_feat_pos_embed(cam_features).to(dtype=cam_features.dtype)
+                cam_features = self.encoder_img_feat_input_proj(cam_features)
+
+                # Rearrange features to (sequence, batch, dim).
+                cam_features = einops.rearrange(cam_features, "b c h w -> (h w) b c")
+                cam_pos_embed = einops.rearrange(cam_pos_embed, "b c h w -> (h w) b c")
+
+                # Extend immediately instead of accumulating and concatenating
+                # Convert to list to extend properly
+                encoder_in_tokens.extend(list(cam_features))
+                encoder_in_pos_embed.extend(list(cam_pos_embed))
+
+        # Stack all tokens along the sequence dimension.
+        encoder_in_tokens = torch.stack(encoder_in_tokens, axis=0)
+        encoder_in_pos_embed = torch.stack(encoder_in_pos_embed, axis=0)
+
+        # Forward pass through the transformer modules.
+        encoder_out = self.encoder(encoder_in_tokens, pos_embed=encoder_in_pos_embed)
+        # TODO(rcadene, alexander-soare): remove call to `device` ; precompute and use buffer
+        decoder_in = torch.zeros(
+            (self.config.chunk_size, batch_size, self.config.dim_model),
+            dtype=encoder_in_pos_embed.dtype,
+            device=encoder_in_pos_embed.device,
+        )
+        decoder_out = self.decoder(
+            decoder_in,
+            encoder_out,
+            encoder_pos_embed=encoder_in_pos_embed,
+            decoder_pos_embed=self.decoder_pos_embed.weight.unsqueeze(1),
+        )
+
+        # Move back to (B, S, C).
+        decoder_out = decoder_out.transpose(0, 1)
+
+        actions = self.action_head(decoder_out)
+
+        return actions, (mu, log_sigma_x2)
+
+
+class ACTEncoder(nn.Module):
+    """Convenience module for running multiple encoder layers, maybe followed by normalization."""
+
+    def __init__(self, config: ACTConfig, is_vae_encoder: bool = False):
+        super().__init__()
+        self.is_vae_encoder = is_vae_encoder
+        num_layers = config.n_vae_encoder_layers if self.is_vae_encoder else config.n_encoder_layers
+        self.layers = nn.ModuleList([ACTEncoderLayer(config) for _ in range(num_layers)])
+        self.norm = nn.LayerNorm(config.dim_model) if config.pre_norm else nn.Identity()
+
+    def forward(
+        self, x: Tensor, pos_embed: Tensor | None = None, key_padding_mask: Tensor | None = None
+    ) -> Tensor:
+        for layer in self.layers:
+            x = layer(x, pos_embed=pos_embed, key_padding_mask=key_padding_mask)
+        x = self.norm(x)
+        return x
+
+
+class ACTEncoderLayer(nn.Module):
+    def __init__(self, config: ACTConfig):
+        super().__init__()
+        self.self_attn = nn.MultiheadAttention(config.dim_model, config.n_heads, dropout=config.dropout)
+
+        # Feed forward layers.
+        self.linear1 = nn.Linear(config.dim_model, config.dim_feedforward)
+        self.dropout = nn.Dropout(config.dropout)
+        self.linear2 = nn.Linear(config.dim_feedforward, config.dim_model)
+
+        self.norm1 = nn.LayerNorm(config.dim_model)
+        self.norm2 = nn.LayerNorm(config.dim_model)
+        self.dropout1 = nn.Dropout(config.dropout)
+        self.dropout2 = nn.Dropout(config.dropout)
+
+        self.activation = get_activation_fn(config.feedforward_activation)
+        self.pre_norm = config.pre_norm
+
+    def forward(self, x, pos_embed: Tensor | None = None, key_padding_mask: Tensor | None = None) -> Tensor:
+        skip = x
+        if self.pre_norm:
+            x = self.norm1(x)
+        q = k = x if pos_embed is None else x + pos_embed
+        x = self.self_attn(q, k, value=x, key_padding_mask=key_padding_mask)
+        x = x[0]  # note: [0] to select just the output, not the attention weights
+        x = skip + self.dropout1(x)
+        if self.pre_norm:
+            skip = x
+            x = self.norm2(x)
+        else:
+            x = self.norm1(x)
+            skip = x
+        x = self.linear2(self.dropout(self.activation(self.linear1(x))))
+        x = skip + self.dropout2(x)
+        if not self.pre_norm:
+            x = self.norm2(x)
+        return x
+
+
+class ACTDecoder(nn.Module):
+    def __init__(self, config: ACTConfig):
+        """Convenience module for running multiple decoder layers followed by normalization."""
+        super().__init__()
+        self.layers = nn.ModuleList([ACTDecoderLayer(config) for _ in range(config.n_decoder_layers)])
+        self.norm = nn.LayerNorm(config.dim_model)
+
+    def forward(
+        self,
+        x: Tensor,
+        encoder_out: Tensor,
+        decoder_pos_embed: Tensor | None = None,
+        encoder_pos_embed: Tensor | None = None,
+    ) -> Tensor:
+        for layer in self.layers:
+            x = layer(
+                x, encoder_out, decoder_pos_embed=decoder_pos_embed, encoder_pos_embed=encoder_pos_embed
+            )
+        if self.norm is not None:
+            x = self.norm(x)
+        return x
+
+
+class ACTDecoderLayer(nn.Module):
+    def __init__(self, config: ACTConfig):
+        super().__init__()
+        self.self_attn = nn.MultiheadAttention(config.dim_model, config.n_heads, dropout=config.dropout)
+        self.multihead_attn = nn.MultiheadAttention(config.dim_model, config.n_heads, dropout=config.dropout)
+
+        # Feed forward layers.
+        self.linear1 = nn.Linear(config.dim_model, config.dim_feedforward)
+        self.dropout = nn.Dropout(config.dropout)
+        self.linear2 = nn.Linear(config.dim_feedforward, config.dim_model)
+
+        self.norm1 = nn.LayerNorm(config.dim_model)
+        self.norm2 = nn.LayerNorm(config.dim_model)
+        self.norm3 = nn.LayerNorm(config.dim_model)
+        self.dropout1 = nn.Dropout(config.dropout)
+        self.dropout2 = nn.Dropout(config.dropout)
+        self.dropout3 = nn.Dropout(config.dropout)
+
+        self.activation = get_activation_fn(config.feedforward_activation)
+        self.pre_norm = config.pre_norm
+
+    def maybe_add_pos_embed(self, tensor: Tensor, pos_embed: Tensor | None) -> Tensor:
+        return tensor if pos_embed is None else tensor + pos_embed
+
+    def forward(
+        self,
+        x: Tensor,
+        encoder_out: Tensor,
+        decoder_pos_embed: Tensor | None = None,
+        encoder_pos_embed: Tensor | None = None,
+    ) -> Tensor:
+        """
+        Args:
+            x: (Decoder Sequence, Batch, Channel) tensor of input tokens.
+            encoder_out: (Encoder Sequence, B, C) output features from the last layer of the encoder we are
+                cross-attending with.
+            encoder_pos_embed: (ES, 1, C) positional embedding for keys (from the encoder).
+            decoder_pos_embed: (DS, 1, C) positional embedding for the queries (from the decoder).
+        Returns:
+            (DS, B, C) tensor of decoder output features.
+        """
+        skip = x
+        if self.pre_norm:
+            x = self.norm1(x)
+        q = k = self.maybe_add_pos_embed(x, decoder_pos_embed)
+        x = self.self_attn(q, k, value=x)[0]  # select just the output, not the attention weights
+        x = skip + self.dropout1(x)
+        if self.pre_norm:
+            skip = x
+            x = self.norm2(x)
+        else:
+            x = self.norm1(x)
+            skip = x
+        x = self.multihead_attn(
+            query=self.maybe_add_pos_embed(x, decoder_pos_embed),
+            key=self.maybe_add_pos_embed(encoder_out, encoder_pos_embed),
+            value=encoder_out,
+        )[0]  # select just the output, not the attention weights
+        x = skip + self.dropout2(x)
+        if self.pre_norm:
+            skip = x
+            x = self.norm3(x)
+        else:
+            x = self.norm2(x)
+            skip = x
+        x = self.linear2(self.dropout(self.activation(self.linear1(x))))
+        x = skip + self.dropout3(x)
+        if not self.pre_norm:
+            x = self.norm3(x)
+        return x
+
+
+def create_sinusoidal_pos_embedding(num_positions: int, dimension: int) -> Tensor:
+    """1D sinusoidal positional embeddings as in Attention is All You Need.
+
+    Args:
+        num_positions: Number of token positions required.
+    Returns: (num_positions, dimension) position embeddings (the first dimension is the batch dimension).
+
+    """
+
+    def get_position_angle_vec(position):
+        return [position / np.power(10000, 2 * (hid_j // 2) / dimension) for hid_j in range(dimension)]
+
+    sinusoid_table = np.array([get_position_angle_vec(pos_i) for pos_i in range(num_positions)])
+    sinusoid_table[:, 0::2] = np.sin(sinusoid_table[:, 0::2])  # dim 2i
+    sinusoid_table[:, 1::2] = np.cos(sinusoid_table[:, 1::2])  # dim 2i+1
+    return torch.from_numpy(sinusoid_table).float()
+
+
+class ACTSinusoidalPositionEmbedding2d(nn.Module):
+    """2D sinusoidal positional embeddings similar to what's presented in Attention Is All You Need.
+
+    The variation is that the position indices are normalized in [0, 2π] (not quite: the lower bound is 1/H
+    for the vertical direction, and 1/W for the horizontal direction.
+    """
+
+    def __init__(self, dimension: int):
+        """
+        Args:
+            dimension: The desired dimension of the embeddings.
+        """
+        super().__init__()
+        self.dimension = dimension
+        self._two_pi = 2 * math.pi
+        self._eps = 1e-6
+        # Inverse "common ratio" for the geometric progression in sinusoid frequencies.
+        self._temperature = 10000
+
+    def forward(self, x: Tensor) -> Tensor:
+        """
+        Args:
+            x: A (B, C, H, W) batch of 2D feature map to generate the embeddings for.
+        Returns:
+            A (1, C, H, W) batch of corresponding sinusoidal positional embeddings.
+        """
+        not_mask = torch.ones_like(x[0, :1])  # (1, H, W)
+        # Note: These are like range(1, H+1) and range(1, W+1) respectively, but in most implementations
+        # they would be range(0, H) and range(0, W). Keeping it at as is to match the original code.
+        y_range = not_mask.cumsum(1, dtype=torch.float32)
+        x_range = not_mask.cumsum(2, dtype=torch.float32)
+
+        # "Normalize" the position index such that it ranges in [0, 2π].
+        # Note: Adding epsilon on the denominator should not be needed as all values of y_embed and x_range
+        # are non-zero by construction. This is an artifact of the original code.
+        y_range = y_range / (y_range[:, -1:, :] + self._eps) * self._two_pi
+        x_range = x_range / (x_range[:, :, -1:] + self._eps) * self._two_pi
+
+        inverse_frequency = self._temperature ** (
+            2 * (torch.arange(self.dimension, dtype=torch.float32, device=x.device) // 2) / self.dimension
+        )
+
+        x_range = x_range.unsqueeze(-1) / inverse_frequency  # (1, H, W, 1)
+        y_range = y_range.unsqueeze(-1) / inverse_frequency  # (1, H, W, 1)
+
+        # Note: this stack then flatten operation results in interleaved sine and cosine terms.
+        # pos_embed_x and pos_embed_y are (1, H, W, C // 2).
+        pos_embed_x = torch.stack((x_range[..., 0::2].sin(), x_range[..., 1::2].cos()), dim=-1).flatten(3)
+        pos_embed_y = torch.stack((y_range[..., 0::2].sin(), y_range[..., 1::2].cos()), dim=-1).flatten(3)
+        pos_embed = torch.cat((pos_embed_y, pos_embed_x), dim=3).permute(0, 3, 1, 2)  # (1, C, H, W)
+
+        return pos_embed
+
+
+def get_activation_fn(activation: str) -> Callable:
+    """Return an activation function given a string."""
+    if activation == "relu":
+        return F.relu
+    if activation == "gelu":
+        return F.gelu
+    if activation == "glu":
+        return F.glu
+    raise RuntimeError(f"activation should be relu/gelu/glu, not {activation}.")
diff --git a/lerobot/src/lerobot/policies/act/processor_act.py b/lerobot/src/lerobot/policies/act/processor_act.py
new file mode 100644
index 0000000000000000000000000000000000000000..727b18cef5654d31039907cd46663eab37994939
--- /dev/null
+++ b/lerobot/src/lerobot/policies/act/processor_act.py
@@ -0,0 +1,85 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Tony Z. Zhao and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from typing import Any
+
+import torch
+
+from lerobot.policies.act.configuration_act import ACTConfig
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    RenameObservationsProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.utils.constants import POLICY_POSTPROCESSOR_DEFAULT_NAME, POLICY_PREPROCESSOR_DEFAULT_NAME
+
+
+def make_act_pre_post_processors(
+    config: ACTConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """Creates the pre- and post-processing pipelines for the ACT policy.
+
+    The pre-processing pipeline handles normalization, batching, and device placement for the model inputs.
+    The post-processing pipeline handles unnormalization and moves the model outputs back to the CPU.
+
+    Args:
+        config (ACTConfig): The ACT policy configuration object.
+        dataset_stats (dict[str, dict[str, torch.Tensor]] | None): A dictionary containing dataset
+            statistics (e.g., mean and std) used for normalization. Defaults to None.
+
+    Returns:
+        tuple[PolicyProcessorPipeline[dict[str, Any], dict[str, Any]], PolicyProcessorPipeline[PolicyAction, PolicyAction]]: A tuple containing the
+        pre-processor pipeline and the post-processor pipeline.
+    """
+
+    input_steps = [
+        RenameObservationsProcessorStep(rename_map={}),
+        AddBatchDimensionProcessorStep(),
+        DeviceProcessorStep(device=config.device),
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+            device=config.device,
+        ),
+    ]
+    output_steps = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/diffusion/README.md b/lerobot/src/lerobot/policies/diffusion/README.md
new file mode 120000
index 0000000000000000000000000000000000000000..d332d79c89ecbb7abca0ad9d2aefb3db409cbb3f
--- /dev/null
+++ b/lerobot/src/lerobot/policies/diffusion/README.md
@@ -0,0 +1 @@
+../../../../docs/source/policy_diffusion_README.md
\ No newline at end of file
diff --git a/lerobot/src/lerobot/policies/diffusion/configuration_diffusion.py b/lerobot/src/lerobot/policies/diffusion/configuration_diffusion.py
new file mode 100644
index 0000000000000000000000000000000000000000..91b3df21496cd3bd9c5100ba30481d699727a136
--- /dev/null
+++ b/lerobot/src/lerobot/policies/diffusion/configuration_diffusion.py
@@ -0,0 +1,259 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Columbia Artificial Intelligence, Robotics Lab,
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import NormalizationMode
+from lerobot.optim.optimizers import AdamConfig
+from lerobot.optim.schedulers import DiffuserSchedulerConfig
+
+
+@PreTrainedConfig.register_subclass("diffusion")
+@dataclass
+class DiffusionConfig(PreTrainedConfig):
+    """Configuration class for DiffusionPolicy.
+
+    Defaults are configured for training with PushT providing proprioceptive and single camera observations.
+
+    The parameters you will most likely need to change are the ones which depend on the environment / sensors.
+    Those are: `input_features` and `output_features`.
+
+    Notes on the inputs and outputs:
+        - "observation.state" is required as an input key.
+        - Either:
+            - At least one key starting with "observation.image is required as an input.
+              AND/OR
+            - The key "observation.environment_state" is required as input.
+        - If there are multiple keys beginning with "observation.image" they are treated as multiple camera
+          views. Right now we only support all images having the same shape.
+        - "action" is required as an output key.
+
+    Args:
+        n_obs_steps: Number of environment steps worth of observations to pass to the policy (takes the
+            current step and additional steps going back).
+        horizon: Diffusion model action prediction size as detailed in `DiffusionPolicy.select_action`.
+        n_action_steps: The number of action steps to run in the environment for one invocation of the policy.
+            See `DiffusionPolicy.select_action` for more details.
+        input_features: A dictionary defining the PolicyFeature of the input data for the policy. The key represents
+            the input data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        output_features: A dictionary defining the PolicyFeature of the output data for the policy. The key represents
+            the output data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        normalization_mapping: A dictionary that maps from a str value of FeatureType (e.g., "STATE", "VISUAL") to
+            a corresponding NormalizationMode (e.g., NormalizationMode.MIN_MAX)
+        vision_backbone: Name of the torchvision resnet backbone to use for encoding images.
+        resize_shape: (H, W) shape to resize images to as a preprocessing step for the vision
+            backbone. If None, no resizing is done and the original image resolution is used.
+        crop_ratio: Ratio in (0, 1] used to derive the crop size from resize_shape
+            (crop_h = int(resize_shape[0] * crop_ratio), likewise for width).
+            Set to 1.0 to disable cropping. Only takes effect when resize_shape is not None.
+        crop_shape: (H, W) shape to crop images to. When resize_shape is set and crop_ratio < 1.0,
+            this is computed automatically. Can also be set directly for legacy configs that use
+            crop-only (without resize). If None and no derivation applies, no cropping is done.
+        crop_is_random: Whether the crop should be random at training time (it's always a center
+            crop in eval mode).
+        pretrained_backbone_weights: Pretrained weights from torchvision to initialize the backbone.
+            `None` means no pretrained weights.
+        use_group_norm: Whether to replace batch normalization with group normalization in the backbone.
+            The group sizes are set to be about 16 (to be precise, feature_dim // 16).
+        spatial_softmax_num_keypoints: Number of keypoints for SpatialSoftmax.
+        use_separate_rgb_encoder_per_camera: Whether to use a separate RGB encoder for each camera view.
+        down_dims: Feature dimension for each stage of temporal downsampling in the diffusion modeling Unet.
+            You may provide a variable number of dimensions, therefore also controlling the degree of
+            downsampling.
+        kernel_size: The convolutional kernel size of the diffusion modeling Unet.
+        n_groups: Number of groups used in the group norm of the Unet's convolutional blocks.
+        diffusion_step_embed_dim: The Unet is conditioned on the diffusion timestep via a small non-linear
+            network. This is the output dimension of that network, i.e., the embedding dimension.
+        use_film_scale_modulation: FiLM (https://huggingface.co/papers/1709.07871) is used for the Unet conditioning.
+            Bias modulation is used be default, while this parameter indicates whether to also use scale
+            modulation.
+        noise_scheduler_type: Name of the noise scheduler to use. Supported options: ["DDPM", "DDIM"].
+        num_train_timesteps: Number of diffusion steps for the forward diffusion schedule.
+        beta_schedule: Name of the diffusion beta schedule as per DDPMScheduler from Hugging Face diffusers.
+        beta_start: Beta value for the first forward-diffusion step.
+        beta_end: Beta value for the last forward-diffusion step.
+        prediction_type: The type of prediction that the diffusion modeling Unet makes. Choose from "epsilon"
+            or "sample". These have equivalent outcomes from a latent variable modeling perspective, but
+            "epsilon" has been shown to work better in many deep neural network settings.
+        clip_sample: Whether to clip the sample to [-`clip_sample_range`, +`clip_sample_range`] for each
+            denoising step at inference time. WARNING: you will need to make sure your action-space is
+            normalized to fit within this range.
+        clip_sample_range: The magnitude of the clipping range as described above.
+        num_inference_steps: Number of reverse diffusion steps to use at inference time (steps are evenly
+            spaced). If not provided, this defaults to be the same as `num_train_timesteps`.
+        do_mask_loss_for_padding: Whether to mask the loss when there are copy-padded actions. See
+            `LeRobotDataset` and `load_previous_and_future_frames` for more information. Note, this defaults
+            to False as the original Diffusion Policy implementation does the same.
+    """
+
+    # Inputs / output structure.
+    n_obs_steps: int = 2
+    horizon: int = 16
+    n_action_steps: int = 8
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.MEAN_STD,
+            "STATE": NormalizationMode.MIN_MAX,
+            "ACTION": NormalizationMode.MIN_MAX,
+        }
+    )
+
+    # The original implementation doesn't sample frames for the last 7 steps,
+    # which avoids excessive padding and leads to improved training results.
+    drop_n_last_frames: int = 7  # horizon - n_action_steps - n_obs_steps + 1
+
+    # Architecture / modeling.
+    # Vision backbone.
+    vision_backbone: str = "resnet18"
+    resize_shape: tuple[int, int] | None = None
+    crop_ratio: float = 1.0
+    crop_shape: tuple[int, int] | None = None
+    crop_is_random: bool = True
+    pretrained_backbone_weights: str | None = None
+    use_group_norm: bool = True
+    spatial_softmax_num_keypoints: int = 32
+    use_separate_rgb_encoder_per_camera: bool = False
+    # Unet.
+    down_dims: tuple[int, ...] = (512, 1024, 2048)
+    kernel_size: int = 5
+    n_groups: int = 8
+    diffusion_step_embed_dim: int = 128
+    use_film_scale_modulation: bool = True
+    # Noise scheduler.
+    noise_scheduler_type: str = "DDPM"
+    num_train_timesteps: int = 100
+    beta_schedule: str = "squaredcos_cap_v2"
+    beta_start: float = 0.0001
+    beta_end: float = 0.02
+    prediction_type: str = "epsilon"
+    clip_sample: bool = True
+    clip_sample_range: float = 1.0
+
+    # Inference
+    num_inference_steps: int | None = None
+
+    # Optimization
+    compile_model: bool = False
+    compile_mode: str = "reduce-overhead"
+
+    # Loss computation
+    do_mask_loss_for_padding: bool = False
+
+    # Training presets
+    optimizer_lr: float = 1e-4
+    optimizer_betas: tuple = (0.95, 0.999)
+    optimizer_eps: float = 1e-8
+    optimizer_weight_decay: float = 1e-6
+    scheduler_name: str = "cosine"
+    scheduler_warmup_steps: int = 500
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        """Input validation (not exhaustive)."""
+        if not self.vision_backbone.startswith("resnet"):
+            raise ValueError(
+                f"`vision_backbone` must be one of the ResNet variants. Got {self.vision_backbone}."
+            )
+
+        supported_prediction_types = ["epsilon", "sample"]
+        if self.prediction_type not in supported_prediction_types:
+            raise ValueError(
+                f"`prediction_type` must be one of {supported_prediction_types}. Got {self.prediction_type}."
+            )
+        supported_noise_schedulers = ["DDPM", "DDIM"]
+        if self.noise_scheduler_type not in supported_noise_schedulers:
+            raise ValueError(
+                f"`noise_scheduler_type` must be one of {supported_noise_schedulers}. "
+                f"Got {self.noise_scheduler_type}."
+            )
+
+        if self.resize_shape is not None and (
+            len(self.resize_shape) != 2 or any(d <= 0 for d in self.resize_shape)
+        ):
+            raise ValueError(f"`resize_shape` must be a pair of positive integers. Got {self.resize_shape}.")
+        if not (0 < self.crop_ratio <= 1.0):
+            raise ValueError(f"`crop_ratio` must be in (0, 1]. Got {self.crop_ratio}.")
+
+        if self.resize_shape is not None:
+            if self.crop_ratio < 1.0:
+                self.crop_shape = (
+                    int(self.resize_shape[0] * self.crop_ratio),
+                    int(self.resize_shape[1] * self.crop_ratio),
+                )
+            else:
+                # Explicitly disable cropping for resize+ratio path when crop_ratio == 1.0.
+                self.crop_shape = None
+        if self.crop_shape is not None and (self.crop_shape[0] <= 0 or self.crop_shape[1] <= 0):
+            raise ValueError(f"`crop_shape` must have positive dimensions. Got {self.crop_shape}.")
+
+        # Check that the horizon size and U-Net downsampling is compatible.
+        # U-Net downsamples by 2 with each stage.
+        downsampling_factor = 2 ** len(self.down_dims)
+        if self.horizon % downsampling_factor != 0:
+            raise ValueError(
+                "The horizon should be an integer multiple of the downsampling factor (which is determined "
+                f"by `len(down_dims)`). Got {self.horizon=} and {self.down_dims=}"
+            )
+
+    def get_optimizer_preset(self) -> AdamConfig:
+        return AdamConfig(
+            lr=self.optimizer_lr,
+            betas=self.optimizer_betas,
+            eps=self.optimizer_eps,
+            weight_decay=self.optimizer_weight_decay,
+        )
+
+    def get_scheduler_preset(self) -> DiffuserSchedulerConfig:
+        return DiffuserSchedulerConfig(
+            name=self.scheduler_name,
+            num_warmup_steps=self.scheduler_warmup_steps,
+        )
+
+    def validate_features(self) -> None:
+        if len(self.image_features) == 0 and self.env_state_feature is None:
+            raise ValueError("You must provide at least one image or the environment state among the inputs.")
+
+        if self.resize_shape is None and self.crop_shape is not None:
+            for key, image_ft in self.image_features.items():
+                if self.crop_shape[0] > image_ft.shape[1] or self.crop_shape[1] > image_ft.shape[2]:
+                    raise ValueError(
+                        f"`crop_shape` should fit within the image shapes. Got {self.crop_shape} "
+                        f"for `crop_shape` and {image_ft.shape} for `{key}`."
+                    )
+
+        # Check that all input images have the same shape.
+        if len(self.image_features) > 0:
+            first_image_key, first_image_ft = next(iter(self.image_features.items()))
+            for key, image_ft in self.image_features.items():
+                if image_ft.shape != first_image_ft.shape:
+                    raise ValueError(
+                        f"`{key}` does not match `{first_image_key}`, but we expect all image shapes to match."
+                    )
+
+    @property
+    def observation_delta_indices(self) -> list:
+        return list(range(1 - self.n_obs_steps, 1))
+
+    @property
+    def action_delta_indices(self) -> list:
+        return list(range(1 - self.n_obs_steps, 1 - self.n_obs_steps + self.horizon))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/diffusion/modeling_diffusion.py b/lerobot/src/lerobot/policies/diffusion/modeling_diffusion.py
new file mode 100644
index 0000000000000000000000000000000000000000..aa8d5dd14ca6cf8a316f59bf228d985f52a1da97
--- /dev/null
+++ b/lerobot/src/lerobot/policies/diffusion/modeling_diffusion.py
@@ -0,0 +1,784 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Columbia Artificial Intelligence, Robotics Lab,
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Diffusion Policy as per "Diffusion Policy: Visuomotor Policy Learning via Action Diffusion"
+
+TODO(alexander-soare):
+  - Remove reliance on diffusers for DDPMScheduler and LR scheduler.
+"""
+
+import math
+from collections import deque
+from collections.abc import Callable
+
+import einops
+import numpy as np
+import torch
+import torch.nn.functional as F  # noqa: N812
+import torchvision
+from diffusers.schedulers.scheduling_ddim import DDIMScheduler
+from diffusers.schedulers.scheduling_ddpm import DDPMScheduler
+from torch import Tensor, nn
+
+from lerobot.policies.diffusion.configuration_diffusion import DiffusionConfig
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.utils import (
+    get_device_from_parameters,
+    get_dtype_from_parameters,
+    get_output_shape,
+    populate_queues,
+)
+from lerobot.utils.constants import ACTION, OBS_ENV_STATE, OBS_IMAGES, OBS_STATE
+
+
+class DiffusionPolicy(PreTrainedPolicy):
+    """
+    Diffusion Policy as per "Diffusion Policy: Visuomotor Policy Learning via Action Diffusion"
+    (paper: https://huggingface.co/papers/2303.04137, code: https://github.com/real-stanford/diffusion_policy).
+    """
+
+    config_class = DiffusionConfig
+    name = "diffusion"
+
+    def __init__(
+        self,
+        config: DiffusionConfig,
+        **kwargs,
+    ):
+        """
+        Args:
+            config: Policy configuration class instance or None, in which case the default instantiation of
+                the configuration class is used.
+            dataset_stats: Dataset statistics to be used for normalization. If not passed here, it is expected
+                that they will be passed with a call to `load_state_dict` before the policy is used.
+        """
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        # queues are populated during rollout of the policy, they contain the n latest observations and actions
+        self._queues = None
+
+        self.diffusion = DiffusionModel(config)
+
+        self.reset()
+
+    def get_optim_params(self) -> dict:
+        return self.diffusion.parameters()
+
+    def reset(self):
+        """Clear observation and action queues. Should be called on `env.reset()`"""
+        self._queues = {
+            OBS_STATE: deque(maxlen=self.config.n_obs_steps),
+            ACTION: deque(maxlen=self.config.n_action_steps),
+        }
+        if self.config.image_features:
+            self._queues[OBS_IMAGES] = deque(maxlen=self.config.n_obs_steps)
+        if self.config.env_state_feature:
+            self._queues[OBS_ENV_STATE] = deque(maxlen=self.config.n_obs_steps)
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor], noise: Tensor | None = None) -> Tensor:
+        """Predict a chunk of actions given environment observations."""
+        # stack n latest observations from the queue
+        batch = {k: torch.stack(list(self._queues[k]), dim=1) for k in batch if k in self._queues}
+        actions = self.diffusion.generate_actions(batch, noise=noise)
+
+        return actions
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor], noise: Tensor | None = None) -> Tensor:
+        """Select a single action given environment observations.
+
+        This method handles caching a history of observations and an action trajectory generated by the
+        underlying diffusion model. Here's how it works:
+          - `n_obs_steps` steps worth of observations are cached (for the first steps, the observation is
+            copied `n_obs_steps` times to fill the cache).
+          - The diffusion model generates `horizon` steps worth of actions.
+          - `n_action_steps` worth of actions are actually kept for execution, starting from the current step.
+        Schematically this looks like:
+            ----------------------------------------------------------------------------------------------
+            (legend: o = n_obs_steps, h = horizon, a = n_action_steps)
+            |timestep            | n-o+1 | n-o+2 | ..... | n     | ..... | n+a-1 | n+a   | ..... | n-o+h |
+            |observation is used | YES   | YES   | YES   | YES   | NO    | NO    | NO    | NO    | NO    |
+            |action is generated | YES   | YES   | YES   | YES   | YES   | YES   | YES   | YES   | YES   |
+            |action is used      | NO    | NO    | NO    | YES   | YES   | YES   | NO    | NO    | NO    |
+            ----------------------------------------------------------------------------------------------
+        Note that this means we require: `n_action_steps <= horizon - n_obs_steps + 1`. Also, note that
+        "horizon" may not the best name to describe what the variable actually means, because this period is
+        actually measured from the first observation which (if `n_obs_steps` > 1) happened in the past.
+        """
+        # NOTE: for offline evaluation, we have action in the batch, so we need to pop it out
+        if ACTION in batch:
+            batch.pop(ACTION)
+
+        if self.config.image_features:
+            batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
+            batch[OBS_IMAGES] = torch.stack([batch[key] for key in self.config.image_features], dim=-4)
+        # NOTE: It's important that this happens after stacking the images into a single key.
+        self._queues = populate_queues(self._queues, batch)
+
+        if len(self._queues[ACTION]) == 0:
+            actions = self.predict_action_chunk(batch, noise=noise)
+            self._queues[ACTION].extend(actions.transpose(0, 1))
+
+        action = self._queues[ACTION].popleft()
+        return action
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, None]:
+        """Run the batch through the model and compute the loss for training or validation."""
+        if self.config.image_features:
+            batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
+            for key in self.config.image_features:
+                if self.config.n_obs_steps == 1 and batch[key].ndim == 4:
+                    batch[key] = batch[key].unsqueeze(1)
+            batch[OBS_IMAGES] = torch.stack([batch[key] for key in self.config.image_features], dim=-4)
+        loss = self.diffusion.compute_loss(batch)
+        # no output_dict so returning None
+        return loss, None
+
+
+def _make_noise_scheduler(name: str, **kwargs: dict) -> DDPMScheduler | DDIMScheduler:
+    """
+    Factory for noise scheduler instances of the requested type. All kwargs are passed
+    to the scheduler.
+    """
+    if name == "DDPM":
+        return DDPMScheduler(**kwargs)
+    elif name == "DDIM":
+        return DDIMScheduler(**kwargs)
+    else:
+        raise ValueError(f"Unsupported noise scheduler type {name}")
+
+
+class DiffusionModel(nn.Module):
+    def __init__(self, config: DiffusionConfig):
+        super().__init__()
+        self.config = config
+
+        # Build observation encoders (depending on which observations are provided).
+        global_cond_dim = self.config.robot_state_feature.shape[0]
+        if self.config.image_features:
+            num_images = len(self.config.image_features)
+            if self.config.use_separate_rgb_encoder_per_camera:
+                encoders = [DiffusionRgbEncoder(config) for _ in range(num_images)]
+                self.rgb_encoder = nn.ModuleList(encoders)
+                global_cond_dim += encoders[0].feature_dim * num_images
+            else:
+                self.rgb_encoder = DiffusionRgbEncoder(config)
+                global_cond_dim += self.rgb_encoder.feature_dim * num_images
+        if self.config.env_state_feature:
+            global_cond_dim += self.config.env_state_feature.shape[0]
+
+        self.unet = DiffusionConditionalUnet1d(config, global_cond_dim=global_cond_dim * config.n_obs_steps)
+
+        if config.compile_model:
+            # Compile the U-Net. "reduce-overhead" is preferred for the small-batch repetitive loops
+            # common in diffusion inference.
+            self.unet = torch.compile(self.unet, mode=config.compile_mode)
+
+        self.noise_scheduler = _make_noise_scheduler(
+            config.noise_scheduler_type,
+            num_train_timesteps=config.num_train_timesteps,
+            beta_start=config.beta_start,
+            beta_end=config.beta_end,
+            beta_schedule=config.beta_schedule,
+            clip_sample=config.clip_sample,
+            clip_sample_range=config.clip_sample_range,
+            prediction_type=config.prediction_type,
+        )
+
+        if config.num_inference_steps is None:
+            self.num_inference_steps = self.noise_scheduler.config.num_train_timesteps
+        else:
+            self.num_inference_steps = config.num_inference_steps
+
+    # ========= inference  ============
+    def conditional_sample(
+        self,
+        batch_size: int,
+        global_cond: Tensor | None = None,
+        generator: torch.Generator | None = None,
+        noise: Tensor | None = None,
+    ) -> Tensor:
+        device = get_device_from_parameters(self)
+        dtype = get_dtype_from_parameters(self)
+
+        # Sample prior.
+        sample = (
+            noise
+            if noise is not None
+            else torch.randn(
+                size=(batch_size, self.config.horizon, self.config.action_feature.shape[0]),
+                dtype=dtype,
+                device=device,
+                generator=generator,
+            )
+        )
+
+        self.noise_scheduler.set_timesteps(self.num_inference_steps)
+
+        for t in self.noise_scheduler.timesteps:
+            # Predict model output.
+            model_output = self.unet(
+                sample,
+                torch.full(sample.shape[:1], t, dtype=torch.long, device=sample.device),
+                global_cond=global_cond,
+            )
+            # Compute previous image: x_t -> x_t-1
+            sample = self.noise_scheduler.step(model_output, t, sample, generator=generator).prev_sample
+
+        return sample
+
+    def _prepare_global_conditioning(self, batch: dict[str, Tensor]) -> Tensor:
+        """Encode image features and concatenate them all together along with the state vector."""
+        batch_size, n_obs_steps = batch[OBS_STATE].shape[:2]
+        global_cond_feats = [batch[OBS_STATE]]
+        # Extract image features.
+        if self.config.image_features:
+            if self.config.use_separate_rgb_encoder_per_camera:
+                # Combine batch and sequence dims while rearranging to make the camera index dimension first.
+                images_per_camera = einops.rearrange(batch[OBS_IMAGES], "b s n ... -> n (b s) ...")
+                img_features_list = torch.cat(
+                    [
+                        encoder(images)
+                        for encoder, images in zip(self.rgb_encoder, images_per_camera, strict=True)
+                    ]
+                )
+                # Separate batch and sequence dims back out. The camera index dim gets absorbed into the
+                # feature dim (effectively concatenating the camera features).
+                img_features = einops.rearrange(
+                    img_features_list, "(n b s) ... -> b s (n ...)", b=batch_size, s=n_obs_steps
+                )
+            else:
+                # Combine batch, sequence, and "which camera" dims before passing to shared encoder.
+                img_features = self.rgb_encoder(
+                    einops.rearrange(batch[OBS_IMAGES], "b s n ... -> (b s n) ...")
+                )
+                # Separate batch dim and sequence dim back out. The camera index dim gets absorbed into the
+                # feature dim (effectively concatenating the camera features).
+                img_features = einops.rearrange(
+                    img_features, "(b s n) ... -> b s (n ...)", b=batch_size, s=n_obs_steps
+                )
+            global_cond_feats.append(img_features)
+
+        if self.config.env_state_feature:
+            global_cond_feats.append(batch[OBS_ENV_STATE])
+
+        # Concatenate features then flatten to (B, global_cond_dim).
+        return torch.cat(global_cond_feats, dim=-1).flatten(start_dim=1)
+
+    def generate_actions(self, batch: dict[str, Tensor], noise: Tensor | None = None) -> Tensor:
+        """
+        This function expects `batch` to have:
+        {
+            "observation.state": (B, n_obs_steps, state_dim)
+
+            "observation.images": (B, n_obs_steps, num_cameras, C, H, W)
+                AND/OR
+            "observation.environment_state": (B, n_obs_steps, environment_dim)
+        }
+        """
+        batch_size, n_obs_steps = batch[OBS_STATE].shape[:2]
+        assert n_obs_steps == self.config.n_obs_steps
+
+        # Encode image features and concatenate them all together along with the state vector.
+        global_cond = self._prepare_global_conditioning(batch)  # (B, global_cond_dim)
+
+        # run sampling
+        actions = self.conditional_sample(batch_size, global_cond=global_cond, noise=noise)
+
+        # Extract `n_action_steps` steps worth of actions (from the current observation).
+        start = n_obs_steps - 1
+        end = start + self.config.n_action_steps
+        actions = actions[:, start:end]
+
+        return actions
+
+    def compute_loss(self, batch: dict[str, Tensor]) -> Tensor:
+        """
+        This function expects `batch` to have (at least):
+        {
+            "observation.state": (B, n_obs_steps, state_dim)
+
+            "observation.images": (B, n_obs_steps, num_cameras, C, H, W)
+                AND/OR
+            "observation.environment_state": (B, n_obs_steps, environment_dim)
+
+            "action": (B, horizon, action_dim)
+            "action_is_pad": (B, horizon)
+        }
+        """
+        # Input validation.
+        assert set(batch).issuperset({OBS_STATE, ACTION, "action_is_pad"})
+        assert OBS_IMAGES in batch or OBS_ENV_STATE in batch
+        n_obs_steps = batch[OBS_STATE].shape[1]
+        horizon = batch[ACTION].shape[1]
+        assert horizon == self.config.horizon
+        assert n_obs_steps == self.config.n_obs_steps
+
+        # Encode image features and concatenate them all together along with the state vector.
+        global_cond = self._prepare_global_conditioning(batch)  # (B, global_cond_dim)
+
+        # Forward diffusion.
+        trajectory = batch[ACTION]
+        # Sample noise to add to the trajectory.
+        eps = torch.randn(trajectory.shape, device=trajectory.device)
+        # Sample a random noising timestep for each item in the batch.
+        timesteps = torch.randint(
+            low=0,
+            high=self.noise_scheduler.config.num_train_timesteps,
+            size=(trajectory.shape[0],),
+            device=trajectory.device,
+        ).long()
+        # Add noise to the clean trajectories according to the noise magnitude at each timestep.
+        noisy_trajectory = self.noise_scheduler.add_noise(trajectory, eps, timesteps)
+
+        # Run the denoising network (that might denoise the trajectory, or attempt to predict the noise).
+        pred = self.unet(noisy_trajectory, timesteps, global_cond=global_cond)
+
+        # Compute the loss.
+        # The target is either the original trajectory, or the noise.
+        if self.config.prediction_type == "epsilon":
+            target = eps
+        elif self.config.prediction_type == "sample":
+            target = batch[ACTION]
+        else:
+            raise ValueError(f"Unsupported prediction type {self.config.prediction_type}")
+
+        loss = F.mse_loss(pred, target, reduction="none")
+
+        # Mask loss wherever the action is padded with copies (edges of the dataset trajectory).
+        if self.config.do_mask_loss_for_padding:
+            if "action_is_pad" not in batch:
+                raise ValueError(
+                    "You need to provide 'action_is_pad' in the batch when "
+                    f"{self.config.do_mask_loss_for_padding=}."
+                )
+            in_episode_bound = ~batch["action_is_pad"]
+            loss = loss * in_episode_bound.unsqueeze(-1)
+
+        return loss.mean()
+
+
+class SpatialSoftmax(nn.Module):
+    """
+    Spatial Soft Argmax operation described in "Deep Spatial Autoencoders for Visuomotor Learning" by Finn et al.
+    (https://huggingface.co/papers/1509.06113). A minimal port of the robomimic implementation.
+
+    At a high level, this takes 2D feature maps (from a convnet/ViT) and returns the "center of mass"
+    of activations of each channel, i.e., keypoints in the image space for the policy to focus on.
+
+    Example: take feature maps of size (512x10x12). We generate a grid of normalized coordinates (10x12x2):
+    -----------------------------------------------------
+    | (-1., -1.)   | (-0.82, -1.)   | ... | (1., -1.)   |
+    | (-1., -0.78) | (-0.82, -0.78) | ... | (1., -0.78) |
+    | ...          | ...            | ... | ...         |
+    | (-1., 1.)    | (-0.82, 1.)    | ... | (1., 1.)    |
+    -----------------------------------------------------
+    This is achieved by applying channel-wise softmax over the activations (512x120) and computing the dot
+    product with the coordinates (120x2) to get expected points of maximal activation (512x2).
+
+    The example above results in 512 keypoints (corresponding to the 512 input channels). We can optionally
+    provide num_kp != None to control the number of keypoints. This is achieved by a first applying a learnable
+    linear mapping (in_channels, H, W) -> (num_kp, H, W).
+    """
+
+    def __init__(self, input_shape, num_kp=None):
+        """
+        Args:
+            input_shape (list): (C, H, W) input feature map shape.
+            num_kp (int): number of keypoints in output. If None, output will have the same number of channels as input.
+        """
+        super().__init__()
+
+        assert len(input_shape) == 3
+        self._in_c, self._in_h, self._in_w = input_shape
+
+        if num_kp is not None:
+            self.nets = torch.nn.Conv2d(self._in_c, num_kp, kernel_size=1)
+            self._out_c = num_kp
+        else:
+            self.nets = None
+            self._out_c = self._in_c
+
+        # we could use torch.linspace directly but that seems to behave slightly differently than numpy
+        # and causes a small degradation in pc_success of pre-trained models.
+        pos_x, pos_y = np.meshgrid(np.linspace(-1.0, 1.0, self._in_w), np.linspace(-1.0, 1.0, self._in_h))
+        pos_x = torch.from_numpy(pos_x.reshape(self._in_h * self._in_w, 1)).float()
+        pos_y = torch.from_numpy(pos_y.reshape(self._in_h * self._in_w, 1)).float()
+        # register as buffer so it's moved to the correct device.
+        self.register_buffer("pos_grid", torch.cat([pos_x, pos_y], dim=1))
+
+    def forward(self, features: Tensor) -> Tensor:
+        """
+        Args:
+            features: (B, C, H, W) input feature maps.
+        Returns:
+            (B, K, 2) image-space coordinates of keypoints.
+        """
+        if self.nets is not None:
+            features = self.nets(features)
+
+        # [B, K, H, W] -> [B * K, H * W] where K is number of keypoints
+        features = features.reshape(-1, self._in_h * self._in_w)
+        # 2d softmax normalization
+        attention = F.softmax(features, dim=-1)
+        # [B * K, H * W] x [H * W, 2] -> [B * K, 2] for spatial coordinate mean in x and y dimensions
+        expected_xy = attention @ self.pos_grid
+        # reshape to [B, K, 2]
+        feature_keypoints = expected_xy.view(-1, self._out_c, 2)
+
+        return feature_keypoints
+
+
+class DiffusionRgbEncoder(nn.Module):
+    """Encodes an RGB image into a 1D feature vector.
+
+    Includes the ability to normalize and crop the image first.
+    """
+
+    def __init__(self, config: DiffusionConfig):
+        super().__init__()
+        # Set up optional preprocessing.
+        if config.resize_shape is not None:
+            self.resize = torchvision.transforms.Resize(config.resize_shape)
+        else:
+            self.resize = None
+
+        crop_shape = config.crop_shape
+        if crop_shape is not None:
+            self.do_crop = True
+            # Always use center crop for eval
+            self.center_crop = torchvision.transforms.CenterCrop(crop_shape)
+            if config.crop_is_random:
+                self.maybe_random_crop = torchvision.transforms.RandomCrop(crop_shape)
+            else:
+                self.maybe_random_crop = self.center_crop
+        else:
+            self.do_crop = False
+
+        # Set up backbone.
+        backbone_model = getattr(torchvision.models, config.vision_backbone)(
+            weights=config.pretrained_backbone_weights
+        )
+        # Note: This assumes that the layer4 feature map is children()[-3]
+        # TODO(alexander-soare): Use a safer alternative.
+        self.backbone = nn.Sequential(*(list(backbone_model.children())[:-2]))
+        if config.use_group_norm:
+            if config.pretrained_backbone_weights:
+                raise ValueError(
+                    "You can't replace BatchNorm in a pretrained model without ruining the weights!"
+                )
+            self.backbone = _replace_submodules(
+                root_module=self.backbone,
+                predicate=lambda x: isinstance(x, nn.BatchNorm2d),
+                func=lambda x: nn.GroupNorm(num_groups=x.num_features // 16, num_channels=x.num_features),
+            )
+
+        # Set up pooling and final layers.
+        # Use a dry run to get the feature map shape.
+        # The dummy shape mirrors the runtime preprocessing order: resize -> crop.
+
+        # Note: we have a check in the config class to make sure all images have the same shape.
+        images_shape = next(iter(config.image_features.values())).shape
+        if config.crop_shape is not None:
+            dummy_shape_h_w = config.crop_shape
+        elif config.resize_shape is not None:
+            dummy_shape_h_w = config.resize_shape
+        else:
+            dummy_shape_h_w = images_shape[1:]
+        dummy_shape = (1, images_shape[0], *dummy_shape_h_w)
+        feature_map_shape = get_output_shape(self.backbone, dummy_shape)[1:]
+
+        self.pool = SpatialSoftmax(feature_map_shape, num_kp=config.spatial_softmax_num_keypoints)
+        self.feature_dim = config.spatial_softmax_num_keypoints * 2
+        self.out = nn.Linear(config.spatial_softmax_num_keypoints * 2, self.feature_dim)
+        self.relu = nn.ReLU()
+
+    def forward(self, x: Tensor) -> Tensor:
+        """
+        Args:
+            x: (B, C, H, W) image tensor with pixel values in [0, 1].
+        Returns:
+            (B, D) image feature.
+        """
+        # Preprocess: resize if configured, then crop if configured.
+
+        if self.resize is not None:
+            x = self.resize(x)
+        if self.do_crop:
+            if self.training:  # noqa: SIM108
+                x = self.maybe_random_crop(x)
+            else:
+                # Always use center crop for eval.
+                x = self.center_crop(x)
+        # Extract backbone feature.
+        x = torch.flatten(self.pool(self.backbone(x)), start_dim=1)
+        # Final linear layer with non-linearity.
+        x = self.relu(self.out(x))
+        return x
+
+
+def _replace_submodules(
+    root_module: nn.Module, predicate: Callable[[nn.Module], bool], func: Callable[[nn.Module], nn.Module]
+) -> nn.Module:
+    """
+    Args:
+        root_module: The module for which the submodules need to be replaced
+        predicate: Takes a module as an argument and must return True if the that module is to be replaced.
+        func: Takes a module as an argument and returns a new module to replace it with.
+    Returns:
+        The root module with its submodules replaced.
+    """
+    if predicate(root_module):
+        return func(root_module)
+
+    replace_list = [k.split(".") for k, m in root_module.named_modules(remove_duplicate=True) if predicate(m)]
+    for *parents, k in replace_list:
+        parent_module = root_module
+        if len(parents) > 0:
+            parent_module = root_module.get_submodule(".".join(parents))
+        if isinstance(parent_module, nn.Sequential):
+            src_module = parent_module[int(k)]
+        else:
+            src_module = getattr(parent_module, k)
+        tgt_module = func(src_module)
+        if isinstance(parent_module, nn.Sequential):
+            parent_module[int(k)] = tgt_module
+        else:
+            setattr(parent_module, k, tgt_module)
+    # verify that all BN are replaced
+    assert not any(predicate(m) for _, m in root_module.named_modules(remove_duplicate=True))
+    return root_module
+
+
+class DiffusionSinusoidalPosEmb(nn.Module):
+    """1D sinusoidal positional embeddings as in Attention is All You Need."""
+
+    def __init__(self, dim: int):
+        super().__init__()
+        self.dim = dim
+
+    def forward(self, x: Tensor) -> Tensor:
+        device = x.device
+        half_dim = self.dim // 2
+        emb = math.log(10000) / (half_dim - 1)
+        emb = torch.exp(torch.arange(half_dim, device=device) * -emb)
+        emb = x.unsqueeze(-1) * emb.unsqueeze(0)
+        emb = torch.cat((emb.sin(), emb.cos()), dim=-1)
+        return emb
+
+
+class DiffusionConv1dBlock(nn.Module):
+    """Conv1d --> GroupNorm --> Mish"""
+
+    def __init__(self, inp_channels, out_channels, kernel_size, n_groups=8):
+        super().__init__()
+
+        self.block = nn.Sequential(
+            nn.Conv1d(inp_channels, out_channels, kernel_size, padding=kernel_size // 2),
+            nn.GroupNorm(n_groups, out_channels),
+            nn.Mish(),
+        )
+
+    def forward(self, x):
+        return self.block(x)
+
+
+class DiffusionConditionalUnet1d(nn.Module):
+    """A 1D convolutional UNet with FiLM modulation for conditioning.
+
+    Note: this removes local conditioning as compared to the original diffusion policy code.
+    """
+
+    def __init__(self, config: DiffusionConfig, global_cond_dim: int):
+        super().__init__()
+
+        self.config = config
+
+        # Encoder for the diffusion timestep.
+        self.diffusion_step_encoder = nn.Sequential(
+            DiffusionSinusoidalPosEmb(config.diffusion_step_embed_dim),
+            nn.Linear(config.diffusion_step_embed_dim, config.diffusion_step_embed_dim * 4),
+            nn.Mish(),
+            nn.Linear(config.diffusion_step_embed_dim * 4, config.diffusion_step_embed_dim),
+        )
+
+        # The FiLM conditioning dimension.
+        cond_dim = config.diffusion_step_embed_dim + global_cond_dim
+
+        # In channels / out channels for each downsampling block in the Unet's encoder. For the decoder, we
+        # just reverse these.
+        in_out = [(config.action_feature.shape[0], config.down_dims[0])] + list(
+            zip(config.down_dims[:-1], config.down_dims[1:], strict=True)
+        )
+
+        # Unet encoder.
+        common_res_block_kwargs = {
+            "cond_dim": cond_dim,
+            "kernel_size": config.kernel_size,
+            "n_groups": config.n_groups,
+            "use_film_scale_modulation": config.use_film_scale_modulation,
+        }
+        self.down_modules = nn.ModuleList([])
+        for ind, (dim_in, dim_out) in enumerate(in_out):
+            is_last = ind >= (len(in_out) - 1)
+            self.down_modules.append(
+                nn.ModuleList(
+                    [
+                        DiffusionConditionalResidualBlock1d(dim_in, dim_out, **common_res_block_kwargs),
+                        DiffusionConditionalResidualBlock1d(dim_out, dim_out, **common_res_block_kwargs),
+                        # Downsample as long as it is not the last block.
+                        nn.Conv1d(dim_out, dim_out, 3, 2, 1) if not is_last else nn.Identity(),
+                    ]
+                )
+            )
+
+        # Processing in the middle of the auto-encoder.
+        self.mid_modules = nn.ModuleList(
+            [
+                DiffusionConditionalResidualBlock1d(
+                    config.down_dims[-1], config.down_dims[-1], **common_res_block_kwargs
+                ),
+                DiffusionConditionalResidualBlock1d(
+                    config.down_dims[-1], config.down_dims[-1], **common_res_block_kwargs
+                ),
+            ]
+        )
+
+        # Unet decoder.
+        self.up_modules = nn.ModuleList([])
+        for ind, (dim_out, dim_in) in enumerate(reversed(in_out[1:])):
+            is_last = ind >= (len(in_out) - 1)
+            self.up_modules.append(
+                nn.ModuleList(
+                    [
+                        # dim_in * 2, because it takes the encoder's skip connection as well
+                        DiffusionConditionalResidualBlock1d(dim_in * 2, dim_out, **common_res_block_kwargs),
+                        DiffusionConditionalResidualBlock1d(dim_out, dim_out, **common_res_block_kwargs),
+                        # Upsample as long as it is not the last block.
+                        nn.ConvTranspose1d(dim_out, dim_out, 4, 2, 1) if not is_last else nn.Identity(),
+                    ]
+                )
+            )
+
+        self.final_conv = nn.Sequential(
+            DiffusionConv1dBlock(config.down_dims[0], config.down_dims[0], kernel_size=config.kernel_size),
+            nn.Conv1d(config.down_dims[0], config.action_feature.shape[0], 1),
+        )
+
+    def forward(self, x: Tensor, timestep: Tensor | int, global_cond=None) -> Tensor:
+        """
+        Args:
+            x: (B, T, input_dim) tensor for input to the Unet.
+            timestep: (B,) tensor of (timestep_we_are_denoising_from - 1).
+            global_cond: (B, global_cond_dim)
+            output: (B, T, input_dim)
+        Returns:
+            (B, T, input_dim) diffusion model prediction.
+        """
+        # For 1D convolutions we'll need feature dimension first.
+        x = einops.rearrange(x, "b t d -> b d t")
+
+        timesteps_embed = self.diffusion_step_encoder(timestep)
+
+        # If there is a global conditioning feature, concatenate it to the timestep embedding.
+        if global_cond is not None:
+            global_feature = torch.cat([timesteps_embed, global_cond], axis=-1)
+        else:
+            global_feature = timesteps_embed
+
+        # Run encoder, keeping track of skip features to pass to the decoder.
+        encoder_skip_features: list[Tensor] = []
+        for resnet, resnet2, downsample in self.down_modules:
+            x = resnet(x, global_feature)
+            x = resnet2(x, global_feature)
+            encoder_skip_features.append(x)
+            x = downsample(x)
+
+        for mid_module in self.mid_modules:
+            x = mid_module(x, global_feature)
+
+        # Run decoder, using the skip features from the encoder.
+        for resnet, resnet2, upsample in self.up_modules:
+            x = torch.cat((x, encoder_skip_features.pop()), dim=1)
+            x = resnet(x, global_feature)
+            x = resnet2(x, global_feature)
+            x = upsample(x)
+
+        x = self.final_conv(x)
+
+        x = einops.rearrange(x, "b d t -> b t d")
+        return x
+
+
+class DiffusionConditionalResidualBlock1d(nn.Module):
+    """ResNet style 1D convolutional block with FiLM modulation for conditioning."""
+
+    def __init__(
+        self,
+        in_channels: int,
+        out_channels: int,
+        cond_dim: int,
+        kernel_size: int = 3,
+        n_groups: int = 8,
+        # Set to True to do scale modulation with FiLM as well as bias modulation (defaults to False meaning
+        # FiLM just modulates bias).
+        use_film_scale_modulation: bool = False,
+    ):
+        super().__init__()
+
+        self.use_film_scale_modulation = use_film_scale_modulation
+        self.out_channels = out_channels
+
+        self.conv1 = DiffusionConv1dBlock(in_channels, out_channels, kernel_size, n_groups=n_groups)
+
+        # FiLM modulation (https://huggingface.co/papers/1709.07871) outputs per-channel bias and (maybe) scale.
+        cond_channels = out_channels * 2 if use_film_scale_modulation else out_channels
+        self.cond_encoder = nn.Sequential(nn.Mish(), nn.Linear(cond_dim, cond_channels))
+
+        self.conv2 = DiffusionConv1dBlock(out_channels, out_channels, kernel_size, n_groups=n_groups)
+
+        # A final convolution for dimension matching the residual (if needed).
+        self.residual_conv = (
+            nn.Conv1d(in_channels, out_channels, 1) if in_channels != out_channels else nn.Identity()
+        )
+
+    def forward(self, x: Tensor, cond: Tensor) -> Tensor:
+        """
+        Args:
+            x: (B, in_channels, T)
+            cond: (B, cond_dim)
+        Returns:
+            (B, out_channels, T)
+        """
+        out = self.conv1(x)
+
+        # Get condition embedding. Unsqueeze for broadcasting to `out`, resulting in (B, out_channels, 1).
+        cond_embed = self.cond_encoder(cond).unsqueeze(-1)
+        if self.use_film_scale_modulation:
+            # Treat the embedding as a list of scales and biases.
+            scale = cond_embed[:, : self.out_channels]
+            bias = cond_embed[:, self.out_channels :]
+            out = scale * out + bias
+        else:
+            # Treat the embedding as biases.
+            out = out + cond_embed
+
+        out = self.conv2(out)
+        out = out + self.residual_conv(x)
+        return out
diff --git a/lerobot/src/lerobot/policies/diffusion/processor_diffusion.py b/lerobot/src/lerobot/policies/diffusion/processor_diffusion.py
new file mode 100644
index 0000000000000000000000000000000000000000..a7799be64b621c507aaf17d1f784da3698e45df3
--- /dev/null
+++ b/lerobot/src/lerobot/policies/diffusion/processor_diffusion.py
@@ -0,0 +1,92 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Columbia Artificial Intelligence, Robotics Lab,
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from typing import Any
+
+import torch
+
+from lerobot.policies.diffusion.configuration_diffusion import DiffusionConfig
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    RenameObservationsProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.utils.constants import POLICY_POSTPROCESSOR_DEFAULT_NAME, POLICY_PREPROCESSOR_DEFAULT_NAME
+
+
+def make_diffusion_pre_post_processors(
+    config: DiffusionConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for a diffusion policy.
+
+    The pre-processing pipeline prepares the input data for the model by:
+    1. Renaming features.
+    2. Normalizing the input and output features based on dataset statistics.
+    3. Adding a batch dimension.
+    4. Moving the data to the specified device.
+
+    The post-processing pipeline handles the model's output by:
+    1. Moving the data to the CPU.
+    2. Unnormalizing the output features to their original scale.
+
+    Args:
+        config: The configuration object for the diffusion policy,
+            containing feature definitions, normalization mappings, and device information.
+        dataset_stats: A dictionary of statistics used for normalization.
+            Defaults to None.
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines.
+    """
+
+    input_steps = [
+        RenameObservationsProcessorStep(rename_map={}),
+        AddBatchDimensionProcessorStep(),
+        DeviceProcessorStep(device=config.device),
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+    ]
+    output_steps = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/factory.py b/lerobot/src/lerobot/policies/factory.py
new file mode 100644
index 0000000000000000000000000000000000000000..2320cd624d665dffee753b7b639608205c2788da
--- /dev/null
+++ b/lerobot/src/lerobot/policies/factory.py
@@ -0,0 +1,591 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from __future__ import annotations
+
+import importlib
+import logging
+from typing import Any, TypedDict, Unpack
+
+import torch
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType
+from lerobot.datasets.dataset_metadata import LeRobotDatasetMetadata
+from lerobot.datasets.feature_utils import dataset_to_policy_features
+from lerobot.envs.configs import EnvConfig
+from lerobot.envs.utils import env_to_policy_features
+from lerobot.policies.act.configuration_act import ACTConfig
+from lerobot.policies.diffusion.configuration_diffusion import DiffusionConfig
+from lerobot.policies.groot.configuration_groot import GrootConfig
+from lerobot.policies.pi0.configuration_pi0 import PI0Config
+from lerobot.policies.pi05.configuration_pi05 import PI05Config
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.sac.configuration_sac import SACConfig
+from lerobot.policies.sac.reward_model.configuration_classifier import RewardClassifierConfig
+from lerobot.policies.sarm.configuration_sarm import SARMConfig
+from lerobot.policies.smolvla.configuration_smolvla import SmolVLAConfig
+from lerobot.policies.tdmpc.configuration_tdmpc import TDMPCConfig
+from lerobot.policies.utils import validate_visual_features_consistency
+from lerobot.policies.vqbet.configuration_vqbet import VQBeTConfig
+from lerobot.policies.wall_x.configuration_wall_x import WallXConfig
+from lerobot.policies.xvla.configuration_xvla import XVLAConfig
+from lerobot.processor import PolicyProcessorPipeline
+from lerobot.processor.converters import (
+    batch_to_transition,
+    policy_action_to_transition,
+    transition_to_batch,
+    transition_to_policy_action,
+)
+from lerobot.types import PolicyAction
+from lerobot.utils.constants import (
+    ACTION,
+    POLICY_POSTPROCESSOR_DEFAULT_NAME,
+    POLICY_PREPROCESSOR_DEFAULT_NAME,
+)
+
+
+def get_policy_class(name: str) -> type[PreTrainedPolicy]:
+    """
+    Retrieves a policy class by its registered name.
+
+    This function uses dynamic imports to avoid loading all policy classes into memory
+    at once, improving startup time and reducing dependencies.
+
+    Args:
+        name: The name of the policy. Supported names are "tdmpc", "diffusion", "act",
+              "vqbet", "pi0", "pi05", "sac", "reward_classifier", "smolvla", "wall_x".
+
+    Returns:
+        The policy class corresponding to the given name.
+
+    Raises:
+        NotImplementedError: If the policy name is not recognized.
+    """
+    if name == "tdmpc":
+        from lerobot.policies.tdmpc.modeling_tdmpc import TDMPCPolicy
+
+        return TDMPCPolicy
+    elif name == "diffusion":
+        from lerobot.policies.diffusion.modeling_diffusion import DiffusionPolicy
+
+        return DiffusionPolicy
+    elif name == "act":
+        from lerobot.policies.act.modeling_act import ACTPolicy
+
+        return ACTPolicy
+    elif name == "vqbet":
+        from lerobot.policies.vqbet.modeling_vqbet import VQBeTPolicy
+
+        return VQBeTPolicy
+    elif name == "pi0":
+        from lerobot.policies.pi0.modeling_pi0 import PI0Policy
+
+        return PI0Policy
+    elif name == "pi0_fast":
+        from lerobot.policies.pi0_fast.modeling_pi0_fast import PI0FastPolicy
+
+        return PI0FastPolicy
+    elif name == "pi05":
+        from lerobot.policies.pi05.modeling_pi05 import PI05Policy
+
+        return PI05Policy
+    elif name == "sac":
+        from lerobot.policies.sac.modeling_sac import SACPolicy
+
+        return SACPolicy
+    elif name == "reward_classifier":
+        from lerobot.policies.sac.reward_model.modeling_classifier import Classifier
+
+        return Classifier
+    elif name == "smolvla":
+        from lerobot.policies.smolvla.modeling_smolvla import SmolVLAPolicy
+
+        return SmolVLAPolicy
+    elif name == "sarm":
+        from lerobot.policies.sarm.modeling_sarm import SARMRewardModel
+
+        return SARMRewardModel
+    elif name == "groot":
+        from lerobot.policies.groot.modeling_groot import GrootPolicy
+
+        return GrootPolicy
+    elif name == "xvla":
+        from lerobot.policies.xvla.modeling_xvla import XVLAPolicy
+
+        return XVLAPolicy
+    elif name == "wall_x":
+        from lerobot.policies.wall_x.modeling_wall_x import WallXPolicy
+
+        return WallXPolicy
+    else:
+        try:
+            return _get_policy_cls_from_policy_name(name=name)
+        except Exception as e:
+            raise ValueError(f"Policy type '{name}' is not available.") from e
+
+
+def make_policy_config(policy_type: str, **kwargs) -> PreTrainedConfig:
+    """
+    Instantiates a policy configuration object based on the policy type.
+
+    This factory function simplifies the creation of policy configuration objects by
+    mapping a string identifier to the corresponding config class.
+
+    Args:
+        policy_type: The type of the policy. Supported types include "tdmpc",
+                     "diffusion", "act", "vqbet", "pi0", "pi05", "sac", "smolvla",
+                     "reward_classifier", "wall_x".
+        **kwargs: Keyword arguments to be passed to the configuration class constructor.
+
+    Returns:
+        An instance of a `PreTrainedConfig` subclass.
+
+    Raises:
+        ValueError: If the `policy_type` is not recognized.
+    """
+    if policy_type == "tdmpc":
+        return TDMPCConfig(**kwargs)
+    elif policy_type == "diffusion":
+        return DiffusionConfig(**kwargs)
+    elif policy_type == "act":
+        return ACTConfig(**kwargs)
+    elif policy_type == "vqbet":
+        return VQBeTConfig(**kwargs)
+    elif policy_type == "pi0":
+        return PI0Config(**kwargs)
+    elif policy_type == "pi05":
+        return PI05Config(**kwargs)
+    elif policy_type == "sac":
+        return SACConfig(**kwargs)
+    elif policy_type == "smolvla":
+        return SmolVLAConfig(**kwargs)
+    elif policy_type == "reward_classifier":
+        return RewardClassifierConfig(**kwargs)
+    elif policy_type == "groot":
+        return GrootConfig(**kwargs)
+    elif policy_type == "xvla":
+        return XVLAConfig(**kwargs)
+    elif policy_type == "wall_x":
+        return WallXConfig(**kwargs)
+    else:
+        try:
+            config_cls = PreTrainedConfig.get_choice_class(policy_type)
+            return config_cls(**kwargs)
+        except Exception as e:
+            raise ValueError(f"Policy type '{policy_type}' is not available.") from e
+
+
+class ProcessorConfigKwargs(TypedDict, total=False):
+    """
+    A TypedDict defining the keyword arguments for processor configuration.
+
+    This provides type hints for the optional arguments passed to `make_pre_post_processors`,
+    improving code clarity and enabling static analysis.
+
+    Attributes:
+        preprocessor_config_filename: The filename for the preprocessor configuration.
+        postprocessor_config_filename: The filename for the postprocessor configuration.
+        preprocessor_overrides: A dictionary of overrides for the preprocessor configuration.
+        postprocessor_overrides: A dictionary of overrides for the postprocessor configuration.
+        dataset_stats: Dataset statistics for normalization.
+    """
+
+    preprocessor_config_filename: str | None
+    postprocessor_config_filename: str | None
+    preprocessor_overrides: dict[str, Any] | None
+    postprocessor_overrides: dict[str, Any] | None
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None
+
+
+def make_pre_post_processors(
+    policy_cfg: PreTrainedConfig,
+    pretrained_path: str | None = None,
+    **kwargs: Unpack[ProcessorConfigKwargs],
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Create or load pre- and post-processor pipelines for a given policy.
+
+    This function acts as a factory. It can either load existing processor pipelines
+    from a pretrained path or create new ones from scratch based on the policy
+    configuration. Each policy type has a dedicated factory function for its
+    processors (e.g., `make_tdmpc_pre_post_processors`).
+
+    Args:
+        policy_cfg: The configuration of the policy for which to create processors.
+        pretrained_path: An optional path to load pretrained processor pipelines from.
+            If provided, pipelines are loaded from this path.
+        **kwargs: Keyword arguments for processor configuration, as defined in
+            `ProcessorConfigKwargs`.
+
+    Returns:
+        A tuple containing the input (pre-processor) and output (post-processor) pipelines.
+
+    Raises:
+        NotImplementedError: If a processor factory is not implemented for the given
+            policy configuration type.
+    """
+    if pretrained_path:
+        # TODO(Steven): Temporary patch, implement correctly the processors for Gr00t
+        if isinstance(policy_cfg, GrootConfig):
+            # GROOT handles normalization in groot_pack_inputs_v3 step
+            # Need to override both stats AND normalize_min_max since saved config might be empty
+            preprocessor_overrides = {}
+            postprocessor_overrides = {}
+            preprocessor_overrides["groot_pack_inputs_v3"] = {
+                "stats": kwargs.get("dataset_stats"),
+                "normalize_min_max": True,
+            }
+
+            # Also ensure postprocessing slices to env action dim and unnormalizes with dataset stats
+            env_action_dim = policy_cfg.output_features[ACTION].shape[0]
+            postprocessor_overrides["groot_action_unpack_unnormalize_v1"] = {
+                "stats": kwargs.get("dataset_stats"),
+                "normalize_min_max": True,
+                "env_action_dim": env_action_dim,
+            }
+            kwargs["preprocessor_overrides"] = preprocessor_overrides
+            kwargs["postprocessor_overrides"] = postprocessor_overrides
+
+        return (
+            PolicyProcessorPipeline.from_pretrained(
+                pretrained_model_name_or_path=pretrained_path,
+                config_filename=kwargs.get(
+                    "preprocessor_config_filename", f"{POLICY_PREPROCESSOR_DEFAULT_NAME}.json"
+                ),
+                overrides=kwargs.get("preprocessor_overrides", {}),
+                to_transition=batch_to_transition,
+                to_output=transition_to_batch,
+            ),
+            PolicyProcessorPipeline.from_pretrained(
+                pretrained_model_name_or_path=pretrained_path,
+                config_filename=kwargs.get(
+                    "postprocessor_config_filename", f"{POLICY_POSTPROCESSOR_DEFAULT_NAME}.json"
+                ),
+                overrides=kwargs.get("postprocessor_overrides", {}),
+                to_transition=policy_action_to_transition,
+                to_output=transition_to_policy_action,
+            ),
+        )
+
+    # Create a new processor based on policy type
+    if isinstance(policy_cfg, TDMPCConfig):
+        from lerobot.policies.tdmpc.processor_tdmpc import make_tdmpc_pre_post_processors
+
+        processors = make_tdmpc_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, DiffusionConfig):
+        from lerobot.policies.diffusion.processor_diffusion import make_diffusion_pre_post_processors
+
+        processors = make_diffusion_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, ACTConfig):
+        from lerobot.policies.act.processor_act import make_act_pre_post_processors
+
+        processors = make_act_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, VQBeTConfig):
+        from lerobot.policies.vqbet.processor_vqbet import make_vqbet_pre_post_processors
+
+        processors = make_vqbet_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, PI0Config):
+        from lerobot.policies.pi0.processor_pi0 import make_pi0_pre_post_processors
+
+        processors = make_pi0_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, PI05Config):
+        from lerobot.policies.pi05.processor_pi05 import make_pi05_pre_post_processors
+
+        processors = make_pi05_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, SACConfig):
+        from lerobot.policies.sac.processor_sac import make_sac_pre_post_processors
+
+        processors = make_sac_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, RewardClassifierConfig):
+        from lerobot.policies.sac.reward_model.processor_classifier import make_classifier_processor
+
+        processors = make_classifier_processor(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, SmolVLAConfig):
+        from lerobot.policies.smolvla.processor_smolvla import make_smolvla_pre_post_processors
+
+        processors = make_smolvla_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, SARMConfig):
+        from lerobot.policies.sarm.processor_sarm import make_sarm_pre_post_processors
+
+        processors = make_sarm_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+            dataset_meta=kwargs.get("dataset_meta"),
+        )
+    elif isinstance(policy_cfg, GrootConfig):
+        from lerobot.policies.groot.processor_groot import make_groot_pre_post_processors
+
+        processors = make_groot_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, XVLAConfig):
+        from lerobot.policies.xvla.processor_xvla import (
+            make_xvla_pre_post_processors,
+        )
+
+        processors = make_xvla_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    elif isinstance(policy_cfg, WallXConfig):
+        from lerobot.policies.wall_x.processor_wall_x import make_wall_x_pre_post_processors
+
+        processors = make_wall_x_pre_post_processors(
+            config=policy_cfg,
+            dataset_stats=kwargs.get("dataset_stats"),
+        )
+
+    else:
+        try:
+            processors = _make_processors_from_policy_config(
+                config=policy_cfg,
+                dataset_stats=kwargs.get("dataset_stats"),
+            )
+        except Exception as e:
+            raise ValueError(f"Processor for policy type '{policy_cfg.type}' is not implemented.") from e
+
+    return processors
+
+
+def make_policy(
+    cfg: PreTrainedConfig,
+    ds_meta: LeRobotDatasetMetadata | None = None,
+    env_cfg: EnvConfig | None = None,
+    rename_map: dict[str, str] | None = None,
+) -> PreTrainedPolicy:
+    """
+    Instantiate a policy model.
+
+    This factory function handles the logic of creating a policy, which requires
+    determining the input and output feature shapes. These shapes can be derived
+    either from a `LeRobotDatasetMetadata` object or an `EnvConfig` object. The function
+    can either initialize a new policy from scratch or load a pretrained one.
+
+    Args:
+        cfg: The configuration for the policy to be created. If `cfg.pretrained_path` is
+             set, the policy will be loaded with weights from that path.
+        ds_meta: Dataset metadata used to infer feature shapes and types. Also provides
+                 statistics for normalization layers.
+        env_cfg: Environment configuration used to infer feature shapes and types.
+                 One of `ds_meta` or `env_cfg` must be provided.
+        rename_map: Optional mapping of dataset or environment feature keys to match
+                 expected policy feature names (e.g., `"left"` → `"camera1"`).
+
+    Returns:
+        An instantiated and device-placed policy model.
+
+    Raises:
+        ValueError: If both or neither of `ds_meta` and `env_cfg` are provided.
+        NotImplementedError: If attempting to use an unsupported policy-backend
+                             combination (e.g., VQBeT with 'mps').
+    """
+    if bool(ds_meta) == bool(env_cfg):
+        raise ValueError("Either one of a dataset metadata or a sim env must be provided.")
+
+    # NOTE: Currently, if you try to run vqbet with mps backend, you'll get this error.
+    # TODO(aliberts, rcadene): Implement a check_backend_compatibility in policies?
+    # NotImplementedError: The operator 'aten::unique_dim' is not currently implemented for the MPS device. If
+    # you want this op to be added in priority during the prototype phase of this feature, please comment on
+    # https://github.com/pytorch/pytorch/issues/77764. As a temporary fix, you can set the environment
+    # variable `PYTORCH_ENABLE_MPS_FALLBACK=1` to use the CPU as a fallback for this op. WARNING: this will be
+    # slower than running natively on MPS.
+    if cfg.type == "vqbet" and cfg.device == "mps":
+        raise NotImplementedError(
+            "Current implementation of VQBeT does not support `mps` backend. "
+            "Please use `cpu` or `cuda` backend."
+        )
+
+    policy_cls = get_policy_class(cfg.type)
+
+    kwargs = {}
+    if ds_meta is not None:
+        features = dataset_to_policy_features(ds_meta.features)
+    else:
+        if not cfg.pretrained_path:
+            logging.warning(
+                "You are instantiating a policy from scratch and its features are parsed from an environment "
+                "rather than a dataset. Normalization modules inside the policy will have infinite values "
+                "by default without stats from a dataset."
+            )
+        if env_cfg is None:
+            raise ValueError("env_cfg cannot be None when ds_meta is not provided")
+        features = env_to_policy_features(env_cfg)
+
+    cfg.output_features = {key: ft for key, ft in features.items() if ft.type is FeatureType.ACTION}
+    if not cfg.input_features:
+        cfg.input_features = {key: ft for key, ft in features.items() if key not in cfg.output_features}
+    kwargs["config"] = cfg
+
+    # Pass dataset_stats to the policy if available (needed for some policies like SARM)
+    if ds_meta is not None and hasattr(ds_meta, "stats"):
+        kwargs["dataset_stats"] = ds_meta.stats
+
+    if ds_meta is not None:
+        kwargs["dataset_meta"] = ds_meta
+
+    if not cfg.pretrained_path and cfg.use_peft:
+        raise ValueError(
+            "Instantiating a policy with `use_peft=True` without a checkpoint is not supported since that requires "
+            "the PEFT config parameters to be set. For training with PEFT, see `lerobot_train.py` on how to do that."
+        )
+
+    if cfg.pretrained_path and not cfg.use_peft:
+        # Load a pretrained policy and override the config if needed (for example, if there are inference-time
+        # hyperparameters that we want to vary).
+        kwargs["pretrained_name_or_path"] = cfg.pretrained_path
+        policy = policy_cls.from_pretrained(**kwargs)
+    elif cfg.pretrained_path and cfg.use_peft:
+        # Load a pretrained PEFT model on top of the policy. The pretrained path points to the folder/repo
+        # of the adapter and the adapter's config contains the path to the base policy. So we need the
+        # adapter config first, then load the correct policy and then apply PEFT.
+        from peft import PeftConfig, PeftModel
+
+        logging.info("Loading policy's PEFT adapter.")
+
+        peft_pretrained_path = cfg.pretrained_path
+        peft_config = PeftConfig.from_pretrained(peft_pretrained_path)
+
+        kwargs["pretrained_name_or_path"] = peft_config.base_model_name_or_path
+        if not kwargs["pretrained_name_or_path"]:
+            # This means that there's a bug or we trained a policy from scratch using PEFT.
+            # It is more likely that this is a bug so we'll raise an error.
+            raise ValueError(
+                "No pretrained model name found in adapter config. Can't instantiate the pre-trained policy on which "
+                "the adapter was trained."
+            )
+
+        policy = policy_cls.from_pretrained(**kwargs)
+        policy = PeftModel.from_pretrained(policy, peft_pretrained_path, config=peft_config)
+
+    else:
+        # Make a fresh policy.
+        policy = policy_cls(**kwargs)
+
+    policy.to(cfg.device)
+    assert isinstance(policy, torch.nn.Module)
+
+    # policy = torch.compile(policy, mode="reduce-overhead")
+
+    if not rename_map:
+        validate_visual_features_consistency(cfg, features)
+        # TODO: (jadechoghari) - add a check_state(cfg, features) and check_action(cfg, features)
+
+    return policy
+
+
+def _get_policy_cls_from_policy_name(name: str) -> type[PreTrainedConfig]:
+    """Get policy class from its registered name using dynamic imports.
+
+    This is used as a helper function to import policies from 3rd party lerobot plugins.
+
+    Args:
+        name: The name of the policy.
+    Returns:
+        The policy class corresponding to the given name.
+    """
+    if name not in PreTrainedConfig.get_known_choices():
+        raise ValueError(
+            f"Unknown policy name '{name}'. Available policies: {PreTrainedConfig.get_known_choices()}"
+        )
+
+    config_cls = PreTrainedConfig.get_choice_class(name)
+    config_cls_name = config_cls.__name__
+
+    model_name = config_cls_name.removesuffix("Config")  # e.g., DiffusionConfig -> Diffusion
+    if model_name == config_cls_name:
+        raise ValueError(
+            f"The config class name '{config_cls_name}' does not follow the expected naming convention."
+            f"Make sure it ends with 'Config'!"
+        )
+    cls_name = model_name + "Policy"  # e.g., DiffusionConfig -> DiffusionPolicy
+    module_path = config_cls.__module__.replace(
+        "configuration_", "modeling_"
+    )  # e.g., configuration_diffusion -> modeling_diffusion
+
+    module = importlib.import_module(module_path)
+    policy_cls = getattr(module, cls_name)
+    return policy_cls
+
+
+def _make_processors_from_policy_config(
+    config: PreTrainedConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[Any, Any]:
+    """Create pre- and post-processors from a policy configuration using dynamic imports.
+
+    This is used as a helper function to import processor factories from 3rd party lerobot plugins.
+
+    Args:
+        config: The policy configuration object.
+        dataset_stats: Dataset statistics for normalization.
+    Returns:
+        A tuple containing the input (pre-processor) and output (post-processor) pipelines.
+    """
+
+    policy_type = config.type
+    function_name = f"make_{policy_type}_pre_post_processors"
+    module_path = config.__class__.__module__.replace(
+        "configuration_", "processor_"
+    )  # e.g., configuration_diffusion -> processor_diffusion
+    logging.debug(
+        f"Instantiating pre/post processors using function '{function_name}' from module '{module_path}'"
+    )
+    module = importlib.import_module(module_path)
+    function = getattr(module, function_name)
+    return function(config, dataset_stats=dataset_stats)
diff --git a/lerobot/src/lerobot/policies/groot/README.md b/lerobot/src/lerobot/policies/groot/README.md
new file mode 120000
index 0000000000000000000000000000000000000000..ff4937f5c25888cda9942094ed36e684a72dd564
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/README.md
@@ -0,0 +1 @@
+../../../../docs/source/policy_groot_README.md
\ No newline at end of file
diff --git a/lerobot/src/lerobot/policies/groot/__init__.py b/lerobot/src/lerobot/policies/groot/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..c8933ff56929676384496c10f485bfd1c8af6ba1
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/__init__.py
@@ -0,0 +1,21 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Nvidia and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configuration_groot import GrootConfig
+from .modeling_groot import GrootPolicy
+from .processor_groot import make_groot_pre_post_processors
+
+__all__ = ["GrootConfig", "GrootPolicy", "make_groot_pre_post_processors"]
diff --git a/lerobot/src/lerobot/policies/groot/action_head/__init__.py b/lerobot/src/lerobot/policies/groot/action_head/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..3159bfe65645499015bd92609b99d476d69544e9
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/action_head/__init__.py
@@ -0,0 +1,14 @@
+# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
diff --git a/lerobot/src/lerobot/policies/groot/action_head/action_encoder.py b/lerobot/src/lerobot/policies/groot/action_head/action_encoder.py
new file mode 100644
index 0000000000000000000000000000000000000000..c6fa0a77966b41059ad395ba94d7e7141fc0a149
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/action_head/action_encoder.py
@@ -0,0 +1,54 @@
+# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import torch
+import torch.nn as nn
+
+
+def swish(x):
+    return x * torch.sigmoid(x)
+
+
+class SinusoidalPositionalEncoding(nn.Module):
+    """
+    Produces a sinusoidal encoding of shape (B, T, w)
+    given timesteps of shape (B, T).
+    """
+
+    def __init__(self, embedding_dim):
+        super().__init__()
+        self.embedding_dim = embedding_dim
+
+    def forward(self, timesteps):
+        # timesteps: shape (B, T)
+        # We'll compute sin/cos frequencies across dim T
+        timesteps = timesteps.float()  # ensure float
+
+        b, t = timesteps.shape
+        device = timesteps.device
+
+        half_dim = self.embedding_dim // 2
+        # typical log space frequencies for sinusoidal encoding
+        exponent = -torch.arange(half_dim, dtype=torch.float, device=device) * (
+            torch.log(torch.tensor(10000.0)) / half_dim
+        )
+        # Expand timesteps to (B, T, 1) then multiply
+        freqs = timesteps.unsqueeze(-1) * exponent.exp()  # (B, T, half_dim)
+
+        sin = torch.sin(freqs)
+        cos = torch.cos(freqs)
+        enc = torch.cat([sin, cos], dim=-1)  # (B, T, w)
+
+        return enc
diff --git a/lerobot/src/lerobot/policies/groot/action_head/cross_attention_dit.py b/lerobot/src/lerobot/policies/groot/action_head/cross_attention_dit.py
new file mode 100755
index 0000000000000000000000000000000000000000..40f7ba60330af8348dd0fff874eaca662b5b93db
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/action_head/cross_attention_dit.py
@@ -0,0 +1,370 @@
+# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+
+import torch
+import torch.nn.functional as F  # noqa: N812
+from diffusers import ConfigMixin, ModelMixin
+from diffusers.configuration_utils import register_to_config
+from diffusers.models.attention import Attention, FeedForward
+from diffusers.models.embeddings import (
+    SinusoidalPositionalEmbedding,
+    TimestepEmbedding,
+    Timesteps,
+)
+from torch import nn
+
+
+class TimestepEncoder(nn.Module):
+    def __init__(self, embedding_dim, compute_dtype=torch.float32):
+        super().__init__()
+        self.time_proj = Timesteps(num_channels=256, flip_sin_to_cos=True, downscale_freq_shift=1)
+        self.timestep_embedder = TimestepEmbedding(in_channels=256, time_embed_dim=embedding_dim)
+
+    def forward(self, timesteps):
+        dtype = next(self.parameters()).dtype
+        timesteps_proj = self.time_proj(timesteps).to(dtype)
+        timesteps_emb = self.timestep_embedder(timesteps_proj)  # (N, D)
+        return timesteps_emb
+
+
+class AdaLayerNorm(nn.Module):
+    def __init__(
+        self,
+        embedding_dim: int,
+        norm_elementwise_affine: bool = False,
+        norm_eps: float = 1e-5,
+        chunk_dim: int = 0,
+    ):
+        super().__init__()
+        self.chunk_dim = chunk_dim
+        output_dim = embedding_dim * 2
+        self.silu = nn.SiLU()
+        self.linear = nn.Linear(embedding_dim, output_dim)
+        self.norm = nn.LayerNorm(output_dim // 2, norm_eps, norm_elementwise_affine)
+
+    def forward(
+        self,
+        x: torch.Tensor,
+        temb: torch.Tensor | None = None,
+    ) -> torch.Tensor:
+        temb = self.linear(self.silu(temb))
+        scale, shift = temb.chunk(2, dim=1)
+        x = self.norm(x) * (1 + scale[:, None]) + shift[:, None]
+        return x
+
+
+class BasicTransformerBlock(nn.Module):
+    def __init__(
+        self,
+        dim: int,
+        num_attention_heads: int,
+        attention_head_dim: int,
+        dropout=0.0,
+        cross_attention_dim: int | None = None,
+        activation_fn: str = "geglu",
+        attention_bias: bool = False,
+        upcast_attention: bool = False,
+        norm_elementwise_affine: bool = True,
+        norm_type: str = "layer_norm",  # 'layer_norm', 'ada_norm', 'ada_norm_zero', 'ada_norm_single', 'ada_norm_continuous', 'layer_norm_i2vgen'
+        norm_eps: float = 1e-5,
+        final_dropout: bool = False,
+        attention_type: str = "default",
+        positional_embeddings: str | None = None,
+        num_positional_embeddings: int | None = None,
+        ff_inner_dim: int | None = None,
+        ff_bias: bool = True,
+        attention_out_bias: bool = True,
+    ):
+        super().__init__()
+        self.dim = dim
+        self.num_attention_heads = num_attention_heads
+        self.attention_head_dim = attention_head_dim
+        self.dropout = dropout
+        self.cross_attention_dim = cross_attention_dim
+        self.activation_fn = activation_fn
+        self.attention_bias = attention_bias
+        self.norm_elementwise_affine = norm_elementwise_affine
+        self.positional_embeddings = positional_embeddings
+        self.num_positional_embeddings = num_positional_embeddings
+        self.norm_type = norm_type
+
+        if positional_embeddings and (num_positional_embeddings is None):
+            raise ValueError(
+                "If `positional_embeddings` type is defined, `num_positional_embeddings` must also be defined."
+            )
+
+        if positional_embeddings == "sinusoidal":
+            self.pos_embed = SinusoidalPositionalEmbedding(dim, max_seq_length=num_positional_embeddings)
+        else:
+            self.pos_embed = None
+
+        # Define 3 blocks. Each block has its own normalization layer.
+        # 1. Self-Attn
+        if norm_type == "ada_norm":
+            self.norm1 = AdaLayerNorm(dim)
+        else:
+            self.norm1 = nn.LayerNorm(dim, elementwise_affine=norm_elementwise_affine, eps=norm_eps)
+
+        self.attn1 = Attention(
+            query_dim=dim,
+            heads=num_attention_heads,
+            dim_head=attention_head_dim,
+            dropout=dropout,
+            bias=attention_bias,
+            cross_attention_dim=cross_attention_dim,
+            upcast_attention=upcast_attention,
+            out_bias=attention_out_bias,
+        )
+
+        # 3. Feed-forward
+        self.norm3 = nn.LayerNorm(dim, norm_eps, norm_elementwise_affine)
+        self.ff = FeedForward(
+            dim,
+            dropout=dropout,
+            activation_fn=activation_fn,
+            final_dropout=final_dropout,
+            inner_dim=ff_inner_dim,
+            bias=ff_bias,
+        )
+        if final_dropout:
+            self.final_dropout = nn.Dropout(dropout)
+        else:
+            self.final_dropout = None
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: torch.Tensor | None = None,
+        encoder_hidden_states: torch.Tensor | None = None,
+        encoder_attention_mask: torch.Tensor | None = None,
+        temb: torch.LongTensor | None = None,
+    ) -> torch.Tensor:
+        # 0. Self-Attention
+        if self.norm_type == "ada_norm":
+            norm_hidden_states = self.norm1(hidden_states, temb)
+        else:
+            norm_hidden_states = self.norm1(hidden_states)
+
+        if self.pos_embed is not None:
+            norm_hidden_states = self.pos_embed(norm_hidden_states)
+
+        attn_output = self.attn1(
+            norm_hidden_states,
+            encoder_hidden_states=encoder_hidden_states,
+            attention_mask=attention_mask,
+            # encoder_attention_mask=encoder_attention_mask,
+        )
+        if self.final_dropout:
+            attn_output = self.final_dropout(attn_output)
+
+        hidden_states = attn_output + hidden_states
+        if hidden_states.ndim == 4:
+            hidden_states = hidden_states.squeeze(1)
+
+        # 4. Feed-forward
+        norm_hidden_states = self.norm3(hidden_states)
+        ff_output = self.ff(norm_hidden_states)
+
+        hidden_states = ff_output + hidden_states
+        if hidden_states.ndim == 4:
+            hidden_states = hidden_states.squeeze(1)
+        return hidden_states
+
+
+class DiT(ModelMixin, ConfigMixin):
+    _supports_gradient_checkpointing = True
+
+    @register_to_config
+    def __init__(
+        self,
+        num_attention_heads: int = 8,
+        attention_head_dim: int = 64,
+        output_dim: int = 26,
+        num_layers: int = 12,
+        dropout: float = 0.1,
+        attention_bias: bool = True,
+        activation_fn: str = "gelu-approximate",
+        num_embeds_ada_norm: int | None = 1000,
+        upcast_attention: bool = False,
+        norm_type: str = "ada_norm",
+        norm_elementwise_affine: bool = False,
+        norm_eps: float = 1e-5,
+        max_num_positional_embeddings: int = 512,
+        compute_dtype=torch.float32,
+        final_dropout: bool = True,
+        positional_embeddings: str | None = "sinusoidal",
+        interleave_self_attention=False,
+        cross_attention_dim: int | None = None,
+    ):
+        super().__init__()
+
+        self.attention_head_dim = attention_head_dim
+        self.inner_dim = self.config.num_attention_heads * self.config.attention_head_dim
+        self.gradient_checkpointing = False
+
+        # Timestep encoder
+        self.timestep_encoder = TimestepEncoder(
+            embedding_dim=self.inner_dim, compute_dtype=self.config.compute_dtype
+        )
+
+        all_blocks = []
+        for idx in range(self.config.num_layers):
+            use_self_attn = idx % 2 == 1 and interleave_self_attention
+            curr_cross_attention_dim = cross_attention_dim if not use_self_attn else None
+
+            all_blocks += [
+                BasicTransformerBlock(
+                    self.inner_dim,
+                    self.config.num_attention_heads,
+                    self.config.attention_head_dim,
+                    dropout=self.config.dropout,
+                    activation_fn=self.config.activation_fn,
+                    attention_bias=self.config.attention_bias,
+                    upcast_attention=self.config.upcast_attention,
+                    norm_type=norm_type,
+                    norm_elementwise_affine=self.config.norm_elementwise_affine,
+                    norm_eps=self.config.norm_eps,
+                    positional_embeddings=positional_embeddings,
+                    num_positional_embeddings=self.config.max_num_positional_embeddings,
+                    final_dropout=final_dropout,
+                    cross_attention_dim=curr_cross_attention_dim,
+                )
+            ]
+        self.transformer_blocks = nn.ModuleList(all_blocks)
+
+        # Output blocks
+        self.norm_out = nn.LayerNorm(self.inner_dim, elementwise_affine=False, eps=1e-6)
+        self.proj_out_1 = nn.Linear(self.inner_dim, 2 * self.inner_dim)
+        self.proj_out_2 = nn.Linear(self.inner_dim, self.config.output_dim)
+        print(
+            "Total number of DiT parameters: ",
+            sum(p.numel() for p in self.parameters() if p.requires_grad),
+        )
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,  # Shape: (B, T, D)
+        encoder_hidden_states: torch.Tensor,  # Shape: (B, S, D)
+        timestep: torch.LongTensor | None = None,
+        encoder_attention_mask: torch.Tensor | None = None,
+        return_all_hidden_states: bool = False,
+    ):
+        # Encode timesteps
+        temb = self.timestep_encoder(timestep)
+
+        # Process through transformer blocks - single pass through the blocks
+        hidden_states = hidden_states.contiguous()
+        encoder_hidden_states = encoder_hidden_states.contiguous()
+
+        all_hidden_states = [hidden_states]
+
+        # Process through transformer blocks
+        for idx, block in enumerate(self.transformer_blocks):
+            if idx % 2 == 1 and self.config.interleave_self_attention:
+                hidden_states = block(
+                    hidden_states,
+                    attention_mask=None,
+                    encoder_hidden_states=None,
+                    encoder_attention_mask=None,
+                    temb=temb,
+                )
+            else:
+                hidden_states = block(
+                    hidden_states,
+                    attention_mask=None,
+                    encoder_hidden_states=encoder_hidden_states,
+                    encoder_attention_mask=None,
+                    temb=temb,
+                )
+            all_hidden_states.append(hidden_states)
+
+        # Output processing
+        conditioning = temb
+        shift, scale = self.proj_out_1(F.silu(conditioning)).chunk(2, dim=1)
+        hidden_states = self.norm_out(hidden_states) * (1 + scale[:, None]) + shift[:, None]
+        if return_all_hidden_states:
+            return self.proj_out_2(hidden_states), all_hidden_states
+        else:
+            return self.proj_out_2(hidden_states)
+
+
+class SelfAttentionTransformer(ModelMixin, ConfigMixin):
+    _supports_gradient_checkpointing = True
+
+    @register_to_config
+    def __init__(
+        self,
+        num_attention_heads: int = 8,
+        attention_head_dim: int = 64,
+        output_dim: int = 26,
+        num_layers: int = 12,
+        dropout: float = 0.1,
+        attention_bias: bool = True,
+        activation_fn: str = "gelu-approximate",
+        num_embeds_ada_norm: int | None = 1000,
+        upcast_attention: bool = False,
+        max_num_positional_embeddings: int = 512,
+        compute_dtype=torch.float32,
+        final_dropout: bool = True,
+        positional_embeddings: str | None = "sinusoidal",
+        interleave_self_attention=False,
+    ):
+        super().__init__()
+
+        self.attention_head_dim = attention_head_dim
+        self.inner_dim = self.config.num_attention_heads * self.config.attention_head_dim
+        self.gradient_checkpointing = False
+
+        self.transformer_blocks = nn.ModuleList(
+            [
+                BasicTransformerBlock(
+                    self.inner_dim,
+                    self.config.num_attention_heads,
+                    self.config.attention_head_dim,
+                    dropout=self.config.dropout,
+                    activation_fn=self.config.activation_fn,
+                    attention_bias=self.config.attention_bias,
+                    upcast_attention=self.config.upcast_attention,
+                    positional_embeddings=positional_embeddings,
+                    num_positional_embeddings=self.config.max_num_positional_embeddings,
+                    final_dropout=final_dropout,
+                )
+                for _ in range(self.config.num_layers)
+            ]
+        )
+        print(
+            "Total number of SelfAttentionTransformer parameters: ",
+            sum(p.numel() for p in self.parameters() if p.requires_grad),
+        )
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,  # Shape: (B, T, D)
+        return_all_hidden_states: bool = False,
+    ):
+        # Process through transformer blocks - single pass through the blocks
+        hidden_states = hidden_states.contiguous()
+        all_hidden_states = [hidden_states]
+
+        # Process through transformer blocks
+        for _idx, block in enumerate(self.transformer_blocks):
+            hidden_states = block(hidden_states)
+            all_hidden_states.append(hidden_states)
+
+        if return_all_hidden_states:
+            return hidden_states, all_hidden_states
+        else:
+            return hidden_states
diff --git a/lerobot/src/lerobot/policies/groot/action_head/flow_matching_action_head.py b/lerobot/src/lerobot/policies/groot/action_head/flow_matching_action_head.py
new file mode 100644
index 0000000000000000000000000000000000000000..bfc456ba0bf1912c51030588b3115b09635f129c
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/action_head/flow_matching_action_head.py
@@ -0,0 +1,406 @@
+# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+from typing import TYPE_CHECKING
+
+import torch
+import torch.nn.functional as F  # noqa: N812
+from torch import nn
+from torch.distributions import Beta
+
+from lerobot.utils.import_utils import _transformers_available
+
+# Conditional import for type checking and lazy loading
+if TYPE_CHECKING or _transformers_available:
+    from transformers import PretrainedConfig
+    from transformers.feature_extraction_utils import BatchFeature
+else:
+    PretrainedConfig = object
+    BatchFeature = None
+
+from lerobot.policies.groot.action_head.action_encoder import (
+    SinusoidalPositionalEncoding,
+    swish,
+)
+
+from .cross_attention_dit import DiT, SelfAttentionTransformer
+
+
+class CategorySpecificLinear(nn.Module):
+    def __init__(self, num_categories, input_dim, hidden_dim):
+        super().__init__()
+        self.num_categories = num_categories
+        # For each category, we have separate weights and biases.
+        self.W = nn.Parameter(0.02 * torch.randn(num_categories, input_dim, hidden_dim))
+        self.b = nn.Parameter(torch.zeros(num_categories, hidden_dim))
+
+    def forward(self, x, cat_ids):
+        selected_w = self.W[cat_ids]
+        selected_b = self.b[cat_ids]
+        return torch.bmm(x, selected_w) + selected_b.unsqueeze(1)
+
+
+class CategorySpecificMLP(nn.Module):
+    def __init__(self, num_categories, input_dim, hidden_dim, output_dim):
+        super().__init__()
+        self.num_categories = num_categories
+        self.layer1 = CategorySpecificLinear(num_categories, input_dim, hidden_dim)
+        self.layer2 = CategorySpecificLinear(num_categories, hidden_dim, output_dim)
+
+    def forward(self, x, cat_ids):
+        hidden = F.relu(self.layer1(x, cat_ids))
+        return self.layer2(hidden, cat_ids)
+
+
+class MultiEmbodimentActionEncoder(nn.Module):
+    def __init__(self, action_dim, hidden_size, num_embodiments):
+        super().__init__()
+        self.hidden_size = hidden_size
+        self.num_embodiments = num_embodiments
+
+        # W1: R^{w x d}, W2: R^{w x 2w}, W3: R^{w x w}
+        self.W1 = CategorySpecificLinear(num_embodiments, action_dim, hidden_size)  # (d -> w)
+        self.W2 = CategorySpecificLinear(num_embodiments, 2 * hidden_size, hidden_size)  # (2w -> w)
+        self.W3 = CategorySpecificLinear(num_embodiments, hidden_size, hidden_size)  # (w -> w)
+        self.pos_encoding = SinusoidalPositionalEncoding(hidden_size)
+
+    def forward(self, actions, timesteps, cat_ids):
+        """
+        actions:   shape (B, T, action_dim)
+        timesteps: shape (B,)  -- a single scalar per batch item
+        cat_ids:   shape (B,)
+        returns:   shape (B, T, hidden_size)
+        """
+        b, t, _ = actions.shape
+
+        # 1) Expand each batch's single scalar time 'tau' across all T steps
+        #    so that shape => (B, T)
+        #    e.g. if timesteps is (B,), replicate across T
+        if timesteps.dim() == 1 and timesteps.shape[0] == b:
+            # shape (B,) => (B,T)
+            timesteps = timesteps.unsqueeze(1).expand(-1, t)
+        else:
+            raise ValueError("Expected `timesteps` to have shape (B,) so we can replicate across T.")
+
+        # 2) Standard action MLP step for shape => (B, T, w)
+        a_emb = self.W1(actions, cat_ids)
+
+        # 3) Get the sinusoidal encoding (B, T, w)
+        tau_emb = self.pos_encoding(timesteps).to(dtype=a_emb.dtype)
+
+        # 4) Concat along last dim => (B, T, 2w), then W2 => (B, T, w), swish
+        x = torch.cat([a_emb, tau_emb], dim=-1)
+        x = swish(self.W2(x, cat_ids))
+
+        # 5) Finally W3 => (B, T, w)
+        x = self.W3(x, cat_ids)
+        return x
+
+
+@dataclass
+class FlowmatchingActionHeadConfig(PretrainedConfig):
+    """NOTE: N1.5 uses XEmbFlowmatchingPolicyHeadConfig as action head"""
+
+    add_pos_embed: bool = field(default=True, metadata={"help": "Whether to add positional embedding"})
+    model_dtype: str = field(default="float32", metadata={"help": "Model data type."})
+    diffusion_model_cfg: dict = field(default=None, metadata={"help": "Diffusion model configuration."})
+    input_embedding_dim: int = field(default=1536, metadata={"help": "Input embedding channel dimension."})
+    backbone_embedding_dim: int = field(
+        default=1536, metadata={"help": "Backbone embedding channel dimension."}
+    )
+
+    hidden_size: int = field(default=1024, metadata={"help": "Input embedding dimension."})
+    max_seq_len: int = field(default=1024, metadata={"help": "Maximum Sequence Length"})
+    action_dim: int = field(default=None, metadata={"help": "Action dimension."})
+    action_horizon: int = field(default=None, metadata={"help": "Action horizon."})
+    noise_beta_alpha: float = field(default=1.5, metadata={"help": ""})
+    noise_beta_beta: float = field(default=1.0, metadata={"help": ""})
+    noise_s: float = field(default=0.999, metadata={"help": "Flow matching noise Beta distribution s."})
+    num_timestep_buckets: int = field(
+        default=1000, metadata={"help": "Number of timestep discretization buckets."}
+    )
+    num_inference_timesteps: int = field(
+        default=None,
+        metadata={"help": "Number of inference steps for noise diffusion."},
+    )
+    max_num_embodiments: int = field(default=32, metadata={"help": "Number of embodiments."})
+    tune_projector: bool = field(default=True, metadata={"help": "Whether to tune the projector."})
+    tune_diffusion_model: bool = field(
+        default=True, metadata={"help": "Whether to tune the diffusion model."}
+    )
+    load_pretrained_det_decode_layer_path: str = field(
+        default=None, metadata={"help": "Path to pretrained detection model."}
+    )
+    detection_coeff: float = field(default=1.0, metadata={"help": "Detection coefficient."})
+
+    freeze_decode_layer: bool = field(default=False)
+    expand_batch: int = field(default=None)
+    use_vlln: bool = field(default=True)
+
+    vl_self_attention_cfg: dict = field(default=None)
+    num_target_vision_tokens: int = field(default=32, metadata={"help": "Number of target vision tokens."})
+
+    def __init__(self, **kwargs):
+        super().__init__(**kwargs)
+        for key, value in kwargs.items():
+            setattr(self, key, value)
+
+
+class FlowmatchingActionHead(nn.Module):
+    config_class = FlowmatchingActionHeadConfig
+    supports_gradient_checkpointing = True
+
+    def __init__(
+        self,
+        config: FlowmatchingActionHeadConfig,
+    ):
+        super().__init__()
+        self.hidden_size = config.hidden_size
+        self.input_embedding_dim = config.input_embedding_dim
+
+        self.model = DiT(**config.diffusion_model_cfg)
+        self.action_dim = config.action_dim
+        self.action_horizon = config.action_horizon
+        self.num_inference_timesteps = config.num_inference_timesteps
+
+        self.state_encoder = CategorySpecificMLP(
+            num_categories=config.max_num_embodiments,
+            input_dim=config.max_state_dim,
+            hidden_dim=self.hidden_size,
+            output_dim=self.input_embedding_dim,
+        )
+        self.action_encoder = MultiEmbodimentActionEncoder(
+            action_dim=config.action_dim,
+            hidden_size=self.input_embedding_dim,
+            num_embodiments=config.max_num_embodiments,
+        )
+        self.action_decoder = CategorySpecificMLP(
+            num_categories=config.max_num_embodiments,
+            input_dim=self.hidden_size,
+            hidden_dim=self.hidden_size,
+            output_dim=self.action_dim,
+        )
+        self.future_tokens = nn.Embedding(config.num_target_vision_tokens, self.input_embedding_dim)
+        nn.init.normal_(self.future_tokens.weight, mean=0.0, std=0.02)
+
+        self.vlln = nn.LayerNorm(config.backbone_embedding_dim) if config.use_vlln else nn.Identity()
+        self.vl_self_attention = (
+            SelfAttentionTransformer(**config.vl_self_attention_cfg) if config.use_vlln else nn.Identity()
+        )
+
+        if config.add_pos_embed:
+            self.position_embedding = nn.Embedding(config.max_seq_len, self.input_embedding_dim)
+            nn.init.normal_(self.position_embedding.weight, mean=0.0, std=0.02)
+
+        self.beta_dist = Beta(config.noise_beta_alpha, config.noise_beta_beta)
+        self.num_timestep_buckets = config.num_timestep_buckets
+        self.config = config
+        self.set_trainable_parameters(config.tune_projector, config.tune_diffusion_model)
+
+    def set_trainable_parameters(self, tune_projector: bool, tune_diffusion_model: bool):
+        self.tune_projector = tune_projector
+        self.tune_diffusion_model = tune_diffusion_model
+        for p in self.parameters():
+            p.requires_grad = True
+        if not tune_projector:
+            self.state_encoder.requires_grad_(False)
+            self.action_encoder.requires_grad_(False)
+            self.action_decoder.requires_grad_(False)
+            if self.config.add_pos_embed:
+                self.position_embedding.requires_grad_(False)
+        if not tune_diffusion_model:
+            self.model.requires_grad_(False)
+        print(f"Tune action head projector: {self.tune_projector}")
+        print(f"Tune action head diffusion model: {self.tune_diffusion_model}")
+        # Check if any parameters are still trainable. If not, print a warning.
+        if not tune_projector and not tune_diffusion_model:
+            for name, p in self.named_parameters():
+                if p.requires_grad:
+                    print(f"Action head trainable parameter: {name}")
+        if not any(p.requires_grad for p in self.parameters()):
+            print("Warning: No action head trainable parameters found.")
+
+    def set_frozen_modules_to_eval_mode(self):
+        """
+        Huggingface will call model.train() at each training_step. To ensure
+        the expected behaviors for modules like dropout, batchnorm, etc., we
+        need to call model.eval() for the frozen modules.
+        """
+        if self.training:
+            if not self.tune_projector:
+                self.state_encoder.eval()
+                self.action_encoder.eval()
+                self.action_decoder.eval()
+                if self.config.add_pos_embed:
+                    self.position_embedding.eval()
+            if not self.tune_diffusion_model:
+                self.model.eval()
+
+    def sample_time(self, batch_size, device, dtype):
+        sample = self.beta_dist.sample([batch_size]).to(device, dtype=dtype)
+        return (self.config.noise_s - sample) / self.config.noise_s
+
+    def prepare_input(self, batch: dict) -> BatchFeature:
+        return BatchFeature(data=batch)
+
+    def process_backbone_output(self, backbone_output: BatchFeature) -> BatchFeature:
+        backbone_features = backbone_output["backbone_features"]
+        backbone_features = self.vlln(backbone_features)
+        backbone_features = self.vl_self_attention(backbone_features)
+        backbone_output["backbone_features"] = backbone_features
+        return backbone_output
+
+    def forward(self, backbone_output: BatchFeature, action_input: BatchFeature) -> BatchFeature:
+        # Set frozen modules to eval
+        self.set_frozen_modules_to_eval_mode()
+
+        backbone_output = self.process_backbone_output(backbone_output)
+
+        if self.config.expand_batch is not None:
+            for k, v in backbone_output.items():
+                ndim = len(v.shape)
+                factors = [self.config.expand_batch]
+                while len(factors) < ndim:
+                    factors.append(1)
+                factors = tuple(factors)
+                expanded = v.repeat(*factors)
+                backbone_output[k] = expanded
+
+            for k, v in action_input.items():
+                ndim = len(v.shape)
+                factors = [self.config.expand_batch]
+                while len(factors) < ndim:
+                    factors.append(1)
+                factors = tuple(factors)
+                expanded = v.repeat(*factors)
+                action_input[k] = expanded
+
+        # Get vision and language embeddings.
+        vl_embs = backbone_output.backbone_features
+        device = vl_embs.device
+
+        # Get embodiment ID.
+        embodiment_id = action_input.embodiment_id
+
+        # Embed state.
+        state_features = self.state_encoder(action_input.state, embodiment_id)
+
+        # Embed noised action trajectory.
+        actions = action_input.action
+        noise = torch.randn(actions.shape, device=actions.device, dtype=actions.dtype)
+        t = self.sample_time(actions.shape[0], device=actions.device, dtype=actions.dtype)
+        t = t[:, None, None]  # shape (B,1,1) for broadcast
+
+        noisy_trajectory = (1 - t) * noise + t * actions
+        velocity = actions - noise
+
+        # Convert (continuous) t -> discrete if needed
+        t_discretized = (t[:, 0, 0] * self.num_timestep_buckets).long()
+        action_features = self.action_encoder(noisy_trajectory, t_discretized, embodiment_id)
+
+        # Maybe add position embedding.
+        if self.config.add_pos_embed:
+            pos_ids = torch.arange(action_features.shape[1], dtype=torch.long, device=device)
+            pos_embs = self.position_embedding(pos_ids).unsqueeze(0)
+            action_features = action_features + pos_embs
+
+        # Join vision, language, state and action embedding along sequence dimension.
+        future_tokens = self.future_tokens.weight.unsqueeze(0).expand(vl_embs.shape[0], -1, -1)
+        sa_embs = torch.cat((state_features, future_tokens, action_features), dim=1)
+
+        vl_attn_mask = backbone_output.backbone_attention_mask
+
+        model_output = self.model(
+            hidden_states=sa_embs,
+            encoder_hidden_states=vl_embs,
+            encoder_attention_mask=vl_attn_mask,
+            timestep=t_discretized,
+            return_all_hidden_states=False,  # NOTE (YL): not using flare now
+        )
+        pred = self.action_decoder(model_output, embodiment_id)
+        pred_actions = pred[:, -actions.shape[1] :]
+
+        # Slice out only the action portion of pred and target.
+        action_mask = action_input.action_mask
+        loss = F.mse_loss(pred_actions, velocity, reduction="none") * action_mask
+        loss = loss.sum() / action_mask.sum()
+        output_dict = {
+            "loss": loss,
+        }
+        return BatchFeature(data=output_dict)
+
+    @torch.no_grad()
+    def get_action(self, backbone_output: BatchFeature, action_input: BatchFeature) -> BatchFeature:
+        backbone_output = self.process_backbone_output(backbone_output)
+
+        # Get vision and language embeddings.
+        vl_embs = backbone_output.backbone_features
+        embodiment_id = action_input.embodiment_id
+
+        # Embed state.
+        state_features = self.state_encoder(action_input.state, embodiment_id)
+
+        # Set initial actions as the sampled noise.
+        batch_size = vl_embs.shape[0]
+        device = vl_embs.device
+        actions = torch.randn(
+            size=(batch_size, self.config.action_horizon, self.config.action_dim),
+            dtype=vl_embs.dtype,
+            device=device,
+        )
+
+        num_steps = self.num_inference_timesteps
+        dt = 1.0 / num_steps
+
+        # Run denoising steps.
+        for t in range(num_steps):
+            t_cont = t / float(num_steps)  # e.g. goes 0, 1/N, 2/N, ...
+            t_discretized = int(t_cont * self.num_timestep_buckets)
+
+            # Embed noised action trajectory.
+            timesteps_tensor = torch.full(size=(batch_size,), fill_value=t_discretized, device=device)
+            action_features = self.action_encoder(actions, timesteps_tensor, embodiment_id)
+            # Maybe add position embedding.
+            if self.config.add_pos_embed:
+                pos_ids = torch.arange(action_features.shape[1], dtype=torch.long, device=device)
+                pos_embs = self.position_embedding(pos_ids).unsqueeze(0)
+                action_features = action_features + pos_embs
+
+            # Join vision, language, state and action embedding along sequence dimension.
+            future_tokens = self.future_tokens.weight.unsqueeze(0).expand(vl_embs.shape[0], -1, -1)
+            sa_embs = torch.cat((state_features, future_tokens, action_features), dim=1)
+
+            # Run model forward.
+            model_output = self.model(
+                hidden_states=sa_embs,
+                encoder_hidden_states=vl_embs,
+                timestep=timesteps_tensor,
+            )
+            pred = self.action_decoder(model_output, embodiment_id)
+
+            pred_velocity = pred[:, -self.action_horizon :]
+
+            # Update actions using euler integration.
+            actions = actions + dt * pred_velocity
+        return BatchFeature(data={"action_pred": actions})
+
+    @property
+    def device(self):
+        return next(iter(self.parameters())).device
+
+    @property
+    def dtype(self):
+        return next(iter(self.parameters())).dtype
diff --git a/lerobot/src/lerobot/policies/groot/configuration_groot.py b/lerobot/src/lerobot/policies/groot/configuration_groot.py
new file mode 100644
index 0000000000000000000000000000000000000000..4f3d78222337eb1a77c79b5b294d52d9d286c801
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/configuration_groot.py
@@ -0,0 +1,202 @@
+#!/usr/bin/env python
+
+# Copyright 2024 NVIDIA Corporation and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.optim.optimizers import AdamWConfig
+from lerobot.optim.schedulers import CosineDecayWithWarmupSchedulerConfig
+from lerobot.utils.constants import ACTION, OBS_STATE
+
+
+@PreTrainedConfig.register_subclass("groot")
+@dataclass
+class GrootConfig(PreTrainedConfig):
+    """Configuration for Groot policy wrapper."""
+
+    # Basic policy settings
+    n_obs_steps: int = 1
+    chunk_size: int = 50
+    n_action_steps: int = 50
+
+    # Dimension settings (must match pretrained GR00T model expectations)
+    # Maximum state dimension. Shorter states will be zero-padded.
+    max_state_dim: int = 64
+
+    # Maximum action dimension. Shorter actions will be zero-padded.
+    max_action_dim: int = 32
+
+    # Normalization (start with identity, adjust as needed)
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.MEAN_STD,
+            "ACTION": NormalizationMode.MEAN_STD,
+        }
+    )
+
+    # Image preprocessing (adjust to match Groot's expected input)
+    image_size: tuple[int, int] = (224, 224)
+
+    # Groot-specific model parameters (from groot_finetune_script.py)
+
+    # Path or HuggingFace model ID for the base Groot model
+    base_model_path: str = "nvidia/GR00T-N1.5-3B"
+
+    # HF repo ID (or local path) that hosts vocab.json and merges.txt for Eagle tokenizer.
+    tokenizer_assets_repo: str = "lerobot/eagle2hg-processor-groot-n1p5"
+
+    # Embodiment tag to use for training (e.g. 'new_embodiment', 'gr1')
+    embodiment_tag: str = "new_embodiment"
+
+    # Fine-tuning control arguments
+
+    # Whether to fine-tune the llm backbone
+    tune_llm: bool = False
+
+    # Whether to fine-tune the vision tower
+    tune_visual: bool = False
+
+    # Whether to fine-tune the projector
+    tune_projector: bool = True
+
+    # Whether to fine-tune the diffusion model
+    tune_diffusion_model: bool = True
+
+    # LoRA parameters (from groot_finetune_script.py)
+    # Rank for the LORA model. If 0, no LORA will be used.
+    lora_rank: int = 0
+
+    # Alpha value for the LORA model
+    lora_alpha: int = 16
+
+    # Dropout rate for the LORA model
+    lora_dropout: float = 0.1
+
+    # Whether to use the full model for LORA
+    lora_full_model: bool = False
+
+    # Training parameters (matching groot_finetune_script.py)
+    optimizer_lr: float = 1e-4
+    optimizer_betas: tuple[float, float] = (0.95, 0.999)
+    optimizer_eps: float = 1e-8
+    optimizer_weight_decay: float = 1e-5
+    warmup_ratio: float = 0.05
+    use_bf16: bool = True
+
+    # Dataset parameters
+    # Video backend to use for training ('decord' or 'torchvision_av')
+    video_backend: str = "decord"
+
+    # Whether to balance dataset weights in mixture datasets
+    balance_dataset_weights: bool = True
+
+    # Whether to sample trajectories weighted by their length
+    balance_trajectory_weights: bool = True
+
+    # Optional dataset paths for delegating training to Isaac-GR00T runner
+    dataset_paths: list[str] | None = None
+    output_dir: str = "./tmp/gr00t"
+    save_steps: int = 1000
+    max_steps: int = 10000
+    batch_size: int = 32
+    dataloader_num_workers: int = 8
+    report_to: str = "wandb"
+    resume: bool = False
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        if self.n_action_steps > self.chunk_size:
+            raise ValueError(
+                f"n_action_steps ({self.n_action_steps}) cannot exceed chunk_size ({self.chunk_size})"
+            )
+
+        # groot_repo_path is now optional since we ported the components
+        # No validation needed
+
+    def validate_features(self) -> None:
+        """Validate and set up input/output features for Groot."""
+        image_features = [key for key, feat in self.input_features.items() if feat.type == FeatureType.VISUAL]
+        if not image_features:
+            raise ValueError(
+                "Groot policy requires at least one visual input feature. "
+                "No features of type FeatureType.VISUAL found in input_features."
+            )
+
+        if OBS_STATE not in self.input_features:
+            state_feature = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(self.max_state_dim,),
+            )
+            self.input_features[OBS_STATE] = state_feature
+        else:
+            state_shape = self.input_features[OBS_STATE].shape
+            state_dim = state_shape[0] if state_shape else 0
+            if state_dim > self.max_state_dim:
+                raise ValueError(
+                    f"State dimension {state_dim} exceeds max_state_dim {self.max_state_dim}. "
+                    f"Either reduce state dimension or increase max_state_dim in config."
+                )
+
+        if ACTION not in self.output_features:
+            action_feature = PolicyFeature(
+                type=FeatureType.ACTION,
+                shape=(self.max_action_dim,),
+            )
+            self.output_features[ACTION] = action_feature
+        else:
+            action_shape = self.output_features[ACTION].shape
+            action_dim = action_shape[0] if action_shape else 0
+            if action_dim > self.max_action_dim:
+                raise ValueError(
+                    f"Action dimension {action_dim} exceeds max_action_dim {self.max_action_dim}. "
+                    f"Either reduce action dimension or increase max_action_dim in config."
+                )
+
+    def get_optimizer_preset(self) -> AdamWConfig:
+        """Return optimizer configuration."""
+        return AdamWConfig(
+            lr=self.optimizer_lr,
+            betas=self.optimizer_betas,
+            eps=self.optimizer_eps,
+            weight_decay=self.optimizer_weight_decay,
+        )
+
+    def get_scheduler_preset(self) -> CosineDecayWithWarmupSchedulerConfig:
+        """Return scheduler configuration."""
+        return CosineDecayWithWarmupSchedulerConfig(
+            num_warmup_steps=int(10000 * self.warmup_ratio),  # 5% warmup by default
+            num_decay_steps=10000,  # Adjust based on training steps
+            peak_lr=self.optimizer_lr,
+            decay_lr=self.optimizer_lr * 0.1,
+        )
+
+    @property
+    def observation_delta_indices(self) -> None:
+        """Return indices for delta observations (None for Groot)."""
+        return None
+
+    @property
+    def action_delta_indices(self) -> list[int]:
+        """Return indices for delta actions."""
+        return list(range(min(self.chunk_size, 16)))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        """Return indices for delta rewards (None for Groot)."""
+        return None
diff --git a/lerobot/src/lerobot/policies/groot/eagle2_hg_model/configuration_eagle2_5_vl.py b/lerobot/src/lerobot/policies/groot/eagle2_hg_model/configuration_eagle2_5_vl.py
new file mode 100755
index 0000000000000000000000000000000000000000..526b4f7a2d1aa47d44bf0d8048fdf9a7f886728a
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/eagle2_hg_model/configuration_eagle2_5_vl.py
@@ -0,0 +1,135 @@
+# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+
+import copy
+
+from transformers.configuration_utils import PretrainedConfig
+from transformers.models.llama.configuration_llama import LlamaConfig
+from transformers.models.qwen2.configuration_qwen2 import Qwen2Config
+from transformers.models.qwen3.configuration_qwen3 import Qwen3Config
+from transformers.models.siglip.configuration_siglip import SiglipVisionConfig
+from transformers.utils import logging
+
+logger = logging.get_logger(__name__)
+
+
+class Eagle25VLConfig(PretrainedConfig):
+    model_type = "eagle_2_5_vl"
+    is_composition = True
+    sub_configs = {"vision_config": SiglipVisionConfig, "text_config": Qwen2Config}
+
+    def __init__(
+        self,
+        vision_config=None,
+        text_config=None,
+        use_backbone_lora=0,
+        use_llm_lora=0,
+        pad2square=False,
+        select_layer=-4,
+        force_image_size=None,
+        downsample_ratio=0.5,
+        template=None,
+        dynamic_image_size=False,
+        use_thumbnail=False,
+        loss_version="v1",
+        min_dynamic_tiles=1,
+        max_dynamic_tiles=6,
+        mlp_checkpoint=False,
+        initializer_range=0.02,
+        _attn_implementation="flash_attention_2",
+        _attn_implementation_autoset=False,
+        llm_config=None,
+        image_token_index=None,
+        use_pixel_shuffle=True,
+        mlp_connector_layers=2,
+        **kwargs,
+    ):
+        super().__init__(**kwargs)
+
+        if vision_config is None:
+            vision_config = {"model_type": "siglip_vision_model"}
+            logger.info("vision_config is None. Initializing the InternVisionConfig with default values.")
+
+        if text_config is None:
+            text_config = {"architectures": ["Qwen2ForCausalLM"]}
+            logger.info(
+                "text_config is None. Initializing the LlamaConfig config with default values (`LlamaConfig`)."
+            )
+
+        if vision_config["model_type"] == "siglip_vision_model":
+            self.vision_config = SiglipVisionConfig(**vision_config)
+        else:
+            raise ValueError("Unsupported model_type: {}".format(vision_config["model_type"]))
+
+        if text_config["architectures"][0] == "LlamaForCausalLM":
+            self.text_config = LlamaConfig(**text_config)
+        elif text_config["architectures"][0] == "Qwen2ForCausalLM":
+            self.text_config = Qwen2Config(**text_config)
+        elif text_config["architectures"][0] == "Qwen3ForCausalLM":
+            self.text_config = Qwen3Config(**text_config)
+        else:
+            raise ValueError("Unsupported architecture: {}".format(text_config["architectures"][0]))
+        self.use_backbone_lora = use_backbone_lora
+        self.use_llm_lora = use_llm_lora
+        self.mlp_checkpoint = mlp_checkpoint
+        self.pad2square = pad2square
+        self.select_layer = select_layer
+        self.force_image_size = force_image_size
+        self.downsample_ratio = downsample_ratio
+        self.template = template
+        self.dynamic_image_size = dynamic_image_size
+        self.use_thumbnail = use_thumbnail
+        self.loss_version = loss_version
+        self.initializer_range = initializer_range
+        self.min_dynamic_tiles = min_dynamic_tiles
+        self.max_dynamic_tiles = max_dynamic_tiles
+        self.tie_word_embeddings = self.text_config.tie_word_embeddings
+        self._attn_implementation = _attn_implementation
+        self._attn_implementation_autoset = _attn_implementation_autoset
+        self.image_token_index = image_token_index
+        self.use_pixel_shuffle = use_pixel_shuffle
+        self.mlp_connector_layers = mlp_connector_layers
+        logger.info(f"min_dynamic_tiles: {self.min_dynamic_tiles}")
+        logger.info(f"max_dynamic_tiles: {self.max_dynamic_tiles}")
+
+    def to_dict(self):
+        """
+        Serializes this instance to a Python dictionary. Override the default [`~PretrainedConfig.to_dict`].
+
+        Returns:
+            `Dict[str, any]`: Dictionary of all the attributes that make up this configuration instance,
+        """
+        output = copy.deepcopy(self.__dict__)
+        output["vision_config"] = self.vision_config.to_dict()
+        output["text_config"] = self.text_config.to_dict()
+        output["model_type"] = self.__class__.model_type
+        output["use_backbone_lora"] = self.use_backbone_lora
+        output["use_llm_lora"] = self.use_llm_lora
+        output["pad2square"] = self.pad2square
+        output["select_layer"] = self.select_layer
+        output["force_image_size"] = self.force_image_size
+        output["downsample_ratio"] = self.downsample_ratio
+        output["template"] = self.template
+        output["dynamic_image_size"] = self.dynamic_image_size
+        output["use_thumbnail"] = self.use_thumbnail
+        output["min_dynamic_tiles"] = self.min_dynamic_tiles
+        output["max_dynamic_tiles"] = self.max_dynamic_tiles
+        output["tie_word_embeddings"] = self.tie_word_embeddings
+        output["_attn_implementation"] = self._attn_implementation
+        output["_attn_implementation_autoset"] = self._attn_implementation_autoset
+        output["use_pixel_shuffle"] = self.use_pixel_shuffle
+        output["mlp_connector_layers"] = self.mlp_connector_layers
+        return output
diff --git a/lerobot/src/lerobot/policies/groot/eagle2_hg_model/image_processing_eagle2_5_vl_fast.py b/lerobot/src/lerobot/policies/groot/eagle2_hg_model/image_processing_eagle2_5_vl_fast.py
new file mode 100644
index 0000000000000000000000000000000000000000..90e9dceccfcefb4685bcac2ff00f73709a9002fd
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/eagle2_hg_model/image_processing_eagle2_5_vl_fast.py
@@ -0,0 +1,503 @@
+# --------------------------------------------------------
+# NVIDIA
+# Copyright (c) 2025 NVIDIA
+# Licensed under The MIT License [see LICENSE for details]
+# --------------------------------------------------------
+
+from __future__ import annotations
+
+# copy from https://github.com/huggingface/transformers/blob/main/src/transformers/models/llava_onevision/image_processing_llava_onevision_fast.py
+from transformers.image_processing_utils import (
+    BatchFeature,
+    get_patch_output_size,
+)
+from transformers.image_processing_utils_fast import (
+    BaseImageProcessorFast,
+    ImagesKwargs,
+    group_images_by_shape,
+    reorder_images,
+)
+from transformers.image_utils import (
+    IMAGENET_STANDARD_MEAN,  # 0.5, 0.5, 0.5
+    IMAGENET_STANDARD_STD,  # 0.5, 0.5, 0.5
+    ChannelDimension,
+    ImageInput,
+    PILImageResampling,
+    SizeDict,
+    get_image_size,
+    make_flat_list_of_images,
+    validate_kwargs,
+)
+from transformers.processing_utils import Unpack
+from transformers.utils import (
+    TensorType,
+    add_start_docstrings,
+    is_torch_available,
+    is_torchvision_v2_available,
+)
+from transformers.video_utils import VideoInput
+
+if is_torch_available():
+    import torch
+if is_torchvision_v2_available():
+    from torchvision.transforms.v2 import functional as F  # noqa: N812
+    from transformers.image_utils import pil_torch_interpolation_mapping
+else:
+    from torchvision.transforms import functional as F  # noqa: N812
+
+
+def crop(img: torch.Tensor, left: int, top: int, right: int, bottom: int) -> torch.Tensor:
+    """Crop the given numpy array.
+
+    Args:
+        img (torch.Tensor): Image to be cropped. Format should be (C, H, W).
+        left (int): The left coordinate of the crop box.
+        top (int): The top coordinate of the crop box.
+        right (int): The right coordinate of the crop box.
+        bottom (int): The bottom coordinate of the crop box.
+
+    Returns:
+        torch.Tensor: Cropped image.
+    """
+    if not isinstance(img, torch.Tensor):
+        raise TypeError(f"img should be torch.Tensor. Got {type(img)}")
+
+    if img.ndim not in [2, 3]:
+        raise ValueError(f"Image should have 2 or 3 dimensions. Got {img.ndim}")
+
+    img_height = img.shape[1]
+    img_width = img.shape[2]
+    if top < 0 or left < 0 or bottom > img_height or right > img_width:
+        raise ValueError("Crop coordinates out of bounds")
+
+    if top >= bottom or left >= right:
+        raise ValueError("Invalid crop coordinates")
+
+    return img[:, top:bottom, left:right]
+
+
+class Eagle25VLFastImageProcessorKwargs(ImagesKwargs):
+    max_dynamic_tiles: int | None
+    min_dynamic_tiles: int | None
+    use_thumbnail: bool | None
+    pad_during_tiling: bool | None
+    do_pad: bool | None
+
+
+@add_start_docstrings(
+    "Constructs a fast ConvNeXT image processor. Based on [`SiglipImageProcessor`] with incorporation of processing each video frame.",
+    # BASE_IMAGE_PROCESSOR_FAST_DOCSTRING, TODO: this was depreciated from transformers remove!
+    """
+        image_grid_pinpoints (`List[List[int]]`, *optional*):
+            A list of possible resolutions to use for processing high resolution images. The best resolution is selected
+            based on the original size of the image. Can be overridden by `image_grid_pinpoints` in the `preprocess`
+            method. Not used for processing videos.
+        do_pad (`bool`, *optional*):
+            Whether to pad the image. If `True`, will pad the patch dimension of the images in the batch to the largest
+            number of patches in the batch. Padding will be applied to the bottom and right with zeros.
+    """,
+)
+class Eagle25VLImageProcessorFast(BaseImageProcessorFast):
+    resample = PILImageResampling.BICUBIC
+    image_mean = IMAGENET_STANDARD_MEAN
+    image_std = IMAGENET_STANDARD_STD
+    size = {"height": 448, "width": 448}
+    default_to_square = False
+    crop_size = None
+    do_resize = True
+    do_center_crop = None
+    do_rescale = True
+    do_normalize = True
+    do_convert_rgb = True
+    do_pad = True
+    max_dynamic_tiles = 12
+    min_dynamic_tiles = 1
+    use_thumbnail = True
+    pad_during_tiling = False
+    valid_kwargs = Eagle25VLFastImageProcessorKwargs
+    model_input_names = ["pixel_values_videos"]
+
+    def __init__(self, **kwargs: Unpack[Eagle25VLFastImageProcessorKwargs]):
+        super().__init__(**kwargs)
+
+    @add_start_docstrings(
+        # BASE_IMAGE_PROCESSOR_FAST_DOCSTRING_PREPROCESS, TODO: this was depreciated from transformers remove!
+        """
+            max_dynamic_tiles (`int`, *optional*):
+                The maximum number of dynamic tiles to use for processing high resolution images.
+            min_dynamic_tiles (`int`, *optional*):
+                The minimum number of dynamic tiles to use for processing high resolution images.
+            use_thumbnail (`bool`, *optional*):
+                Whether to use a thumbnail for processing high resolution images.
+            pad_during_tiling (`bool`, *optional*):
+                Whether to pad the image during tiling.
+            do_pad (`bool`, *optional*):
+                    Whether to pad the image. If `True`, will pad the patch dimension of the images in the batch to the largest
+                    number of patches in the batch. Padding will be applied to the bottom and right with zeros.
+        """,
+    )
+
+    # NOTE(YL): we will overload the preprocess method to add the image_flags
+    # def preprocess(
+    #     self, images: ImageInput, **kwargs: Unpack[Eagle25VLFastImageProcessorKwargs]
+    # ) -> BatchFeature:
+    #     return super().preprocess(images, **kwargs)
+
+    def _prepare_images_structure(
+        self,
+        images: ImageInput,
+        expected_ndims: int = 3,
+    ) -> ImageInput:
+        """
+        Prepare the images structure for processing.
+
+        Args:
+            images (`ImageInput`):
+                The input images to process.
+            expected_ndims (`int`, *optional*, defaults to 3):
+                Expected number of dimensions for the images (added for transformers >=4.53.0 compatibility).
+
+        Returns:
+            `ImageInput`: The images with a valid nesting.
+        """
+        return make_flat_list_of_images(images)
+
+    def _resize_for_patching(
+        self,
+        image: torch.Tensor,
+        target_resolution: tuple,
+        interpolation: F.InterpolationMode,
+        input_data_format: ChannelDimension,
+    ) -> torch.Tensor:
+        """
+        Resizes an image to a target resolution while maintaining aspect ratio.
+
+        Args:
+            image ("torch.Tensor"):
+                The input image.
+            target_resolution (tuple):
+                The target resolution (height, width) of the image.
+            interpolation (`InterpolationMode`):
+                Resampling filter to use if resizing the image.
+            input_data_format (`ChannelDimension` or `str`):
+                The channel dimension format of the input image.
+
+        Returns:
+            "torch.Tensor": The resized and padded image.
+        """
+        new_height, new_width = get_patch_output_size(image, target_resolution, input_data_format)
+
+        # Resize the image
+        resized_image = F.resize(image, (new_height, new_width), interpolation=interpolation)
+
+        return resized_image
+
+    def find_closest_aspect_ratio(self, aspect_ratio, target_ratios, width, height, image_size):
+        """
+        previous version mainly focus on ratio.
+        We also consider area ratio here.
+        """
+        best_factor = float("-inf")
+        best_ratio = (1, 1)
+        area = width * height
+        for ratio in target_ratios:
+            target_aspect_ratio = ratio[0] / ratio[1]
+            # ratio_diff = abs(aspect_ratio - target_aspect_ratio)
+            # area_ratio = (ratio[0] * ratio[1] * image_size * image_size) / area
+            """
+            new area > 60% of original image area is enough.
+            """
+            factor_based_on_area_n_ratio = min(
+                (ratio[0] * ratio[1] * image_size * image_size) / area, 0.6
+            ) * min(target_aspect_ratio / aspect_ratio, aspect_ratio / target_aspect_ratio)
+
+            if factor_based_on_area_n_ratio > best_factor:
+                best_factor = factor_based_on_area_n_ratio
+                best_ratio = ratio
+
+        return best_ratio
+
+    def _pad_for_patching(
+        self, image: torch.Tensor, target_resolution: tuple, input_data_format: ChannelDimension
+    ) -> torch.Tensor:
+        """
+        Pad an image to a target resolution while maintaining aspect ratio.
+        """
+        target_height, target_width = target_resolution
+        new_height, new_width = get_patch_output_size(image, target_resolution, input_data_format)
+
+        paste_x = (target_width - new_width) // 2
+        paste_y = (target_height - new_height) // 2
+
+        padded_image = F.pad(image, padding=[paste_x, paste_y, paste_x, paste_y])
+
+        return padded_image
+
+    def _get_image_patches(
+        self,
+        image: torch.Tensor,
+        min_num: int,
+        max_num: int,
+        size: tuple,
+        tile_size: int,
+        use_thumbnail: bool,
+        interpolation: F.InterpolationMode,
+        pad_during_tiling: bool,
+    ) -> list[torch.Tensor]:
+        image_size = get_image_size(image, channel_dim=ChannelDimension.FIRST)
+        orig_height, orig_width = image_size
+        aspect_ratio = orig_width / orig_height
+
+        # calculate the existing image aspect ratio
+        target_ratios = {
+            (i, j)
+            for n in range(min_num, max_num + 1)
+            for i in range(1, n + 1)
+            for j in range(1, n + 1)
+            if i * j <= max_num and i * j >= min_num
+        }
+        target_ratios = sorted(target_ratios, key=lambda x: x[0] * x[1])
+
+        # find the closest aspect ratio to the target
+        target_aspect_ratio = self.find_closest_aspect_ratio(
+            aspect_ratio, target_ratios, orig_width, orig_height, tile_size
+        )
+
+        # calculate the target width and height
+        target_width = tile_size * target_aspect_ratio[0]
+        target_height = tile_size * target_aspect_ratio[1]
+        blocks = target_aspect_ratio[0] * target_aspect_ratio[1]
+        if pad_during_tiling:
+            resized_image = self._resize_for_patching(
+                image,
+                (target_height, target_width),
+                interpolation=interpolation,
+                input_data_format=ChannelDimension.FIRST,
+            )
+            padded_image = self._pad_for_patching(
+                resized_image,
+                (target_height, target_width),
+                input_data_format=ChannelDimension.FIRST,
+            )
+            image_used_to_split = padded_image
+        else:
+            image_used_to_split = F.resize(image, (target_height, target_width), interpolation=interpolation)
+
+        processed_tiles = []
+        for i in range(blocks):
+            box = (
+                (i % (target_width // tile_size)) * tile_size,
+                (i // (target_width // tile_size)) * tile_size,
+                ((i % (target_width // tile_size)) + 1) * tile_size,
+                ((i // (target_width // tile_size)) + 1) * tile_size,
+            )
+            # split the image
+            split_img = crop(image_used_to_split, box[0], box[1], box[2], box[3])
+            processed_tiles.append(split_img)
+        assert len(processed_tiles) == blocks
+
+        if use_thumbnail and len(processed_tiles) != 1:
+            thumbnail_img = F.resize(image, (tile_size, tile_size), interpolation=interpolation)
+            processed_tiles.append(thumbnail_img)
+
+        return processed_tiles
+
+    def _pad_for_batching(
+        self,
+        pixel_values: list[torch.Tensor],
+    ) -> list[torch.Tensor]:
+        """
+        Pads images on the `num_of_patches` dimension with zeros to form a batch of same number of patches.
+
+        Args:
+            pixel_values (`List[torch.Tensor]`):
+                An array of pixel values of each images of shape (`batch_size`, `num_patches`, `image_in_3D`)
+
+        Returns:
+            List[`torch.Tensor`]: The padded images.
+        """
+        max_patch = max(len(x) for x in pixel_values)
+        pixel_values = [
+            torch.nn.functional.pad(image, pad=[0, 0, 0, 0, 0, 0, 0, max_patch - image.shape[0]])
+            for image in pixel_values
+        ]
+
+        return pixel_values
+
+    def _preprocess(
+        self,
+        images: list[torch.Tensor],
+        do_resize: bool,
+        size: SizeDict,
+        max_dynamic_tiles: int,
+        min_dynamic_tiles: int,
+        use_thumbnail: bool,
+        pad_during_tiling: bool,
+        interpolation: F.InterpolationMode | None,
+        do_center_crop: bool,
+        crop_size: SizeDict,
+        do_rescale: bool,
+        rescale_factor: float,
+        do_normalize: bool,
+        image_mean: float | list[float] | None,
+        image_std: float | list[float] | None,
+        do_pad: bool,
+        return_tensors: str | TensorType | None,
+        pad_size: SizeDict | None = None,  # Added for transformers >=4.53.0 compatibility
+        disable_grouping: bool | None = None,  # Added for transformers >=4.53.0 compatibility
+    ) -> BatchFeature:
+        processed_images = []
+        image_sizes = []
+        # Determine the size tuple
+        if size and size.height and size.width:
+            size_tuple = (size.height, size.width)
+        else:
+            size_tuple = (size.shortest_edge, size.shortest_edge)
+
+        # Determine the patch size
+        if crop_size and crop_size.height:
+            tile_size = crop_size.height
+        elif size and size.height:
+            tile_size = size.height
+        else:
+            tile_size = size.shortest_edge
+
+        for image in images:
+            image_patches = self._get_image_patches(
+                image,
+                min_num=min_dynamic_tiles,
+                max_num=max_dynamic_tiles,
+                size=size_tuple,
+                tile_size=tile_size,
+                use_thumbnail=use_thumbnail,
+                interpolation=interpolation,
+                pad_during_tiling=pad_during_tiling,
+            )
+
+            # Group images by size for batched processing
+            processed_image_patches_grouped = {}
+            # Added for transformers >=4.53.0 compatibility
+            grouped_image_patches, grouped_image_patches_index = group_images_by_shape(
+                image_patches,
+                disable_grouping=disable_grouping,
+            )
+
+            for shape, stacked_image_patches in grouped_image_patches.items():
+                if do_resize:
+                    stacked_image_patches = self.resize(
+                        image=stacked_image_patches,
+                        size=size,
+                        interpolation=interpolation,
+                    )
+                if do_center_crop:
+                    stacked_image_patches = self.center_crop(stacked_image_patches, crop_size)
+                # Fused rescale and normalize
+                stacked_image_patches = self.rescale_and_normalize(
+                    stacked_image_patches,
+                    do_rescale,
+                    rescale_factor,
+                    do_normalize,
+                    image_mean,
+                    image_std,
+                )
+                processed_image_patches_grouped[shape] = stacked_image_patches
+            processed_image_patches = reorder_images(
+                processed_image_patches_grouped, grouped_image_patches_index
+            )
+            processed_image_patches = (
+                torch.stack(processed_image_patches, dim=0) if return_tensors else processed_image_patches
+            )
+            processed_images.append(processed_image_patches)
+            image_sizes.append(get_image_size(image, ChannelDimension.FIRST))
+
+        if do_pad:
+            processed_images = self._pad_for_batching(processed_images)
+
+        # processed_images = torch.stack(processed_images, dim=0) if return_tensors else processed_images
+        processed_images = torch.cat(processed_images, dim=0) if return_tensors else processed_images
+        return BatchFeature(
+            data={"pixel_values": processed_images, "image_sizes": image_sizes},
+            tensor_type=return_tensors,
+        )
+
+    def preprocess(
+        self,
+        images: ImageInput,
+        videos: VideoInput = None,
+        **kwargs: Unpack[Eagle25VLFastImageProcessorKwargs],
+    ) -> BatchFeature:
+        validate_kwargs(
+            captured_kwargs=kwargs.keys(),
+            valid_processor_keys=self.valid_kwargs.__annotations__.keys(),
+        )
+        # Set default kwargs from self. This ensures that if a kwarg is not provided
+        # by the user, it gets its default value from the instance, or is set to None.
+        for kwarg_name in self.valid_kwargs.__annotations__:
+            kwargs.setdefault(kwarg_name, getattr(self, kwarg_name, None))
+
+        # Extract parameters that are only used for preparing the input images
+        do_convert_rgb = kwargs.pop("do_convert_rgb")
+        input_data_format = kwargs.pop("input_data_format")
+        device = kwargs.pop("device")
+        # Prepare input images
+        # transformers >= 4.53.0: uses _prepare_image_like_inputs instead of _prepare_input_images
+        if images is not None:
+            images = self._prepare_image_like_inputs(
+                images=images,
+                do_convert_rgb=do_convert_rgb,
+                input_data_format=input_data_format,
+                device=device,
+            )
+
+        if videos is not None:
+            videos = self._prepare_image_like_inputs(
+                images=videos,
+                do_convert_rgb=do_convert_rgb,
+                input_data_format=input_data_format,
+                device=device,
+            )
+
+        # Update kwargs that need further processing before being validated
+        kwargs = self._further_process_kwargs(**kwargs)
+
+        # Validate kwargs
+        self._validate_preprocess_kwargs(**kwargs)
+
+        # torch resize uses interpolation instead of resample
+        # Added for transformers >=4.53.0 compatibility
+        resample = kwargs.pop("resample", self.resample)
+        kwargs["interpolation"] = (
+            pil_torch_interpolation_mapping[resample]
+            if isinstance(resample, PILImageResampling | int)
+            else resample
+        )
+
+        # Filter kwargs to only include those accepted by _preprocess
+        valid_preprocess_kwargs = {
+            "do_resize",
+            "size",
+            "max_dynamic_tiles",
+            "min_dynamic_tiles",
+            "use_thumbnail",
+            "pad_during_tiling",
+            "interpolation",
+            "do_center_crop",
+            "crop_size",
+            "do_rescale",
+            "rescale_factor",
+            "do_normalize",
+            "image_mean",
+            "image_std",
+            "do_pad",
+            "return_tensors",
+            "pad_size",
+            "disable_grouping",
+        }
+        filtered_kwargs = {k: v for k, v in kwargs.items() if k in valid_preprocess_kwargs}
+        if images is not None:
+            return self._preprocess(images, **filtered_kwargs)
+        elif videos is not None:
+            return self._preprocess(videos, **filtered_kwargs)
+
+
+__all__ = ["Eagle25VLImageProcessorFast"]
diff --git a/lerobot/src/lerobot/policies/groot/eagle2_hg_model/modeling_eagle2_5_vl.py b/lerobot/src/lerobot/policies/groot/eagle2_hg_model/modeling_eagle2_5_vl.py
new file mode 100755
index 0000000000000000000000000000000000000000..5a66cfbcea327917e8ecc6e94ecedcab6ee7438c
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/eagle2_hg_model/modeling_eagle2_5_vl.py
@@ -0,0 +1,395 @@
+# --------------------------------------------------------
+# NVIDIA
+# Copyright (c) 2025 NVIDIA
+# Licensed under The MIT License [see LICENSE for details]
+# --------------------------------------------------------
+
+import inspect
+
+import torch
+import torch.utils.checkpoint as cp
+from peft import LoraConfig, get_peft_model
+from torch import nn
+from torch.nn import CrossEntropyLoss
+from transformers import GenerationConfig
+from transformers.generation import GenerationMixin
+from transformers.modeling_outputs import CausalLMOutputWithPast
+from transformers.modeling_utils import PreTrainedModel
+from transformers.models.llama.modeling_llama import LlamaForCausalLM
+from transformers.models.qwen2.modeling_qwen2 import Qwen2ForCausalLM
+from transformers.models.qwen3.modeling_qwen3 import Qwen3ForCausalLM
+from transformers.models.siglip.modeling_siglip import SiglipVisionModel
+from transformers.utils import add_start_docstrings, logging
+
+from .configuration_eagle2_5_vl import Eagle25VLConfig
+
+logger = logging.get_logger(__name__)
+
+
+# copy from https://github.com/huggingface/transformers/blob/main/src/transformers/models/llava_onevision/modeling_llava_onevision.py#L241C1-L280C1
+EAGLE2_5_VL_START_DOCSTRING = r"""
+    This model inherits from [`PreTrainedModel`]. Check the superclass documentation for the generic methods the
+    library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads
+    etc.)
+
+    This model is also a PyTorch [torch.nn.Module](https://pytorch.org/docs/stable/nn.html#torch.nn.Module) subclass.
+    Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage
+    and behavior.
+
+    Parameters:
+        config ([`Eagle25VLConfig`]):
+            Model configuration class with all the parameters of the model. Initializing with a config file does not
+            load the weights associated with the model, only the configuration. Check out the
+            [`~PreTrainedModel.from_pretrained`] method to load the model weights.
+"""
+
+
+@add_start_docstrings(
+    "The bare Eagle2_5_VL Model outputting raw hidden-states without any specific head on top.",
+    EAGLE2_5_VL_START_DOCSTRING,
+)
+class Eagle25VLPreTrainedModel(PreTrainedModel):
+    config_class = Eagle25VLConfig
+    base_model_prefix = "model"
+    main_input_name = "input_ids"
+    supports_gradient_checkpointing = True
+    _no_split_modules = [
+        "Qwen2DecoderLayer",
+        "LlamaDecoderLayer",
+        "Siglip2EncoderLayer",
+        "SiglipEncoderLayer",
+    ]
+    _skip_keys_device_placement = "past_key_values"
+    _supports_flash_attn_2 = True
+    _supports_cache_class = True
+    _supports_static_cache = True
+    _supports_quantized_cache = True
+    _supports_sdpa = True
+
+    def _init_weights(self, module):
+        std = self.config.initializer_range
+        if isinstance(module, nn.Linear | nn.Conv2d):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.bias is not None:
+                module.bias.data.zero_()
+        elif isinstance(module, nn.Embedding):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.padding_idx is not None:
+                module.weight.data[module.padding_idx].zero_()
+
+
+class Eagle25VLForConditionalGeneration(Eagle25VLPreTrainedModel, GenerationMixin):
+    config_class = Eagle25VLConfig
+
+    def __init__(self, config: Eagle25VLConfig, vision_model=None, language_model=None):
+        super().__init__(config)
+
+        image_size = config.force_image_size or config.vision_config.image_size
+        patch_size = config.vision_config.patch_size
+        self.patch_size = patch_size
+        if config.use_pixel_shuffle:
+            self.num_image_token = int((image_size // patch_size) ** 2 * (config.downsample_ratio**2))
+        else:
+            self.num_image_token = int((image_size // patch_size) ** 2)
+
+        self.select_layer = config.select_layer
+        self.downsample_ratio = config.downsample_ratio
+        self.loss_version = config.loss_version
+        self.mlp_checkpoint = config.mlp_checkpoint
+        self.use_pixel_shuffle = config.use_pixel_shuffle
+        self.mlp_connector_layers = config.mlp_connector_layers
+        logger.info(f"num_image_token: {self.num_image_token}")
+        logger.info(f"mlp_checkpoint: {self.mlp_checkpoint}")
+        if vision_model is not None:
+            self.vision_model = vision_model
+        else:
+            if config.vision_config.model_type == "siglip_vision_model":
+                config.vision_config._attn_implementation = "flash_attention_2"
+                self.vision_model = SiglipVisionModel(config.vision_config)
+            else:
+                raise NotImplementedError(f"{config.vision_config.model_type} is not implemented.")
+
+        if language_model is not None:
+            self.language_model = language_model
+        else:
+            if config.text_config.architectures[0] == "LlamaForCausalLM":
+                self.language_model = LlamaForCausalLM(config.text_config)
+            elif config.text_config.architectures[0] == "Phi3ForCausalLM":
+                raise NotImplementedError("Phi3 is not implemented.")
+                # self.language_model = Phi3ForCausalLM(config.text_config)
+            elif config.text_config.architectures[0] == "Qwen2ForCausalLM":
+                assert config.text_config._attn_implementation == "flash_attention_2", (
+                    f"Qwen2 must use flash_attention_2 but got {config.text_config._attn_implementation}"
+                )
+                self.language_model = Qwen2ForCausalLM(config.text_config)
+            elif config.text_config.architectures[0] == "Qwen3ForCausalLM":
+                self.language_model = Qwen3ForCausalLM(config.text_config)
+            else:
+                raise NotImplementedError(f"{config.text_config.architectures[0]} is not implemented.")
+
+        vit_hidden_size = config.vision_config.hidden_size
+        llm_hidden_size = config.text_config.hidden_size
+
+        if config.mlp_connector_layers == 2:
+            self.mlp1 = nn.Sequential(
+                nn.LayerNorm(vit_hidden_size * int(1 / self.downsample_ratio) ** 2),
+                nn.Linear(vit_hidden_size * int(1 / self.downsample_ratio) ** 2, llm_hidden_size),
+                nn.GELU(),
+                nn.Linear(llm_hidden_size, llm_hidden_size),
+            )
+        elif config.mlp_connector_layers == 1 and config.use_pixel_shuffle:
+            self.mlp1 = nn.Sequential(
+                nn.Linear(vit_hidden_size * int(1 / self.downsample_ratio) ** 2, llm_hidden_size),
+            )
+        elif config.mlp_connector_layers == 1 and not config.use_pixel_shuffle:
+            self.mlp1 = nn.Sequential(
+                nn.Linear(vit_hidden_size, llm_hidden_size),
+            )
+        else:
+            raise NotImplementedError(f"{config.mlp_connector_layers} is not implemented.")
+
+        self.image_token_index = config.image_token_index
+        self.neftune_alpha = None
+
+        if config.use_backbone_lora:
+            self.wrap_backbone_lora(r=config.use_backbone_lora, lora_alpha=2 * config.use_backbone_lora)
+
+        self.use_llm_lora = config.use_llm_lora
+        if config.use_llm_lora:
+            self.wrap_llm_lora(r=config.use_llm_lora, lora_alpha=2 * config.use_llm_lora)
+
+        self.check_forward_kwargs()
+
+    def check_forward_kwargs(self):
+        # We intentionally avoid using **kwargs in forward because Hugging Face Transformers
+        # has special handling for functions with **kwargs parameters that would affect
+        # how our model is processed during training and inference.
+        forward_params = inspect.signature(self.forward).parameters
+        assert not any(k.kind == inspect.Parameter.VAR_KEYWORD for k in forward_params.values())
+
+    def wrap_backbone_lora(self, r=128, lora_alpha=256, lora_dropout=0.05):
+        lora_config = LoraConfig(
+            r=r,
+            target_modules=[
+                "self_attn.q_proj",
+                "self_attn.k_proj",
+                "self_attn.v_proj",
+                "self_attn.out_proj",
+                "mlp.fc1",
+                "mlp.fc2",
+            ],
+            lora_alpha=lora_alpha,
+            lora_dropout=lora_dropout,
+        )
+        self.vision_model = get_peft_model(self.vision_model, lora_config)
+        self.vision_model.print_trainable_parameters()
+
+    def wrap_llm_lora(self, r=128, lora_alpha=256, lora_dropout=0.05):
+        lora_config = LoraConfig(
+            r=r,
+            target_modules=[
+                "self_attn.q_proj",
+                "self_attn.k_proj",
+                "self_attn.v_proj",
+                "self_attn.o_proj",
+                "mlp.gate_proj",
+                "mlp.down_proj",
+                "mlp.up_proj",
+            ],
+            lora_alpha=lora_alpha,
+            lora_dropout=lora_dropout,
+            task_type="CAUSAL_LM",
+        )
+        self.language_model = get_peft_model(self.language_model, lora_config)
+        self.language_model.enable_input_require_grads()
+        self.language_model.print_trainable_parameters()
+        self.use_llm_lora = True
+
+    def forward(
+        self,
+        pixel_values: torch.FloatTensor,
+        input_ids: torch.LongTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        image_flags: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        labels: torch.LongTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+        num_tiles_list: list[torch.Tensor] | None = None,
+    ) -> tuple | CausalLMOutputWithPast:
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        input_embeds = self.language_model.get_input_embeddings()(input_ids)
+
+        vit_embeds = self.extract_feature(pixel_values)
+
+        if image_flags is not None:
+            image_flags = image_flags.view(-1)
+            vit_embeds = vit_embeds[image_flags == 1]
+
+        b, n, c = input_embeds.shape
+        input_embeds = input_embeds.reshape(b * n, c)
+
+        input_ids = input_ids.reshape(b * n)
+        selected = input_ids == self.image_token_index
+        try:
+            input_embeds[selected] = input_embeds[selected] * 0.0 + vit_embeds.reshape(-1, c)
+        except Exception as e:
+            vit_embeds = vit_embeds.reshape(-1, c)
+            print(
+                f"warning: {e}, input_embeds[selected].shape={input_embeds[selected].shape}, "
+                f"vit_embeds.shape={vit_embeds.shape}"
+            )
+            n_token = selected.sum()
+            input_embeds[selected] = input_embeds[selected] * 0.0 + vit_embeds[:n_token]
+
+        input_embeds = input_embeds.reshape(b, n, c)
+
+        outputs = self.language_model(
+            inputs_embeds=input_embeds,
+            attention_mask=attention_mask,
+            position_ids=position_ids,
+            past_key_values=past_key_values,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+        )
+        logits = outputs.logits
+
+        loss = None
+        if labels is not None:
+            # Shift so that tokens < n predict n
+            shift_logits = logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            # Flatten the tokens
+            loss_fct = CrossEntropyLoss()
+            shift_logits = shift_logits.view(-1, self.language_model.config.vocab_size)
+            shift_labels = shift_labels.view(-1)
+            # Enable model parallelism
+            shift_labels = shift_labels.to(shift_logits.device)
+            loss = loss_fct(shift_logits, shift_labels)
+
+        if not return_dict:
+            output = (logits,) + outputs[1:]
+            return (loss,) + output if loss is not None else output
+
+        return CausalLMOutputWithPast(
+            loss=loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+        )
+
+    def pixel_shuffle(self, x, scale_factor=0.5):
+        n, w, h, c = x.size()
+        # N, W, H, C --> N, W, H * scale, C // scale
+        x = x.view(n, w, int(h * scale_factor), int(c / scale_factor))
+        # N, W, H * scale, C // scale --> N, H * scale, W, C // scale
+        x = x.permute(0, 2, 1, 3).contiguous()
+        # N, H * scale, W, C // scale --> N, H * scale, W * scale, C // (scale ** 2)
+        x = x.view(n, int(h * scale_factor), int(w * scale_factor), int(c / (scale_factor * scale_factor)))
+
+        x = x.permute(0, 2, 1, 3).contiguous()
+        return x
+
+    def extract_feature(self, pixel_values):
+        if self.select_layer == -1:
+            vit_embeds = self.vision_model(
+                pixel_values=pixel_values, output_hidden_states=False, return_dict=True
+            )
+            if hasattr(vit_embeds, "last_hidden_state"):
+                vit_embeds = vit_embeds.last_hidden_state
+
+        else:
+            vit_embeds = self.vision_model(
+                pixel_values=pixel_values, output_hidden_states=True, return_dict=True
+            ).hidden_states[self.select_layer]
+
+        if self.use_pixel_shuffle:
+            h = w = int(vit_embeds.shape[1] ** 0.5)
+            vit_embeds = vit_embeds.reshape(vit_embeds.shape[0], h, w, -1)
+            vit_embeds = self.pixel_shuffle(
+                vit_embeds, scale_factor=self.downsample_ratio
+            )  # torch.Size([B, 1024, 1024]) -> torch.Size([B, 16, 16, 4096])
+            vit_embeds = vit_embeds.reshape(
+                vit_embeds.shape[0], -1, vit_embeds.shape[-1]
+            )  # torch.Size([B, 16, 16, 4096]) -> torch.Size([B, 256, 4096])
+
+        if self.mlp_checkpoint and vit_embeds.requires_grad:
+            vit_embeds = cp.checkpoint(self.mlp1, vit_embeds)
+        else:
+            vit_embeds = self.mlp1(vit_embeds)
+
+        return vit_embeds
+
+    @torch.no_grad()
+    def generate(
+        self,
+        pixel_values: torch.FloatTensor | None = None,
+        input_ids: torch.FloatTensor | None = None,
+        attention_mask: torch.LongTensor | None = None,
+        visual_features: torch.FloatTensor | None = None,
+        generation_config: GenerationConfig | None = None,
+        output_hidden_states: bool | None = None,
+        image_sizes: list[tuple[int, int]] | None = None,
+        **generate_kwargs,
+    ) -> torch.LongTensor:
+        if pixel_values is not None:
+            if visual_features is not None:
+                vit_embeds = visual_features
+            else:
+                vit_embeds = self.extract_feature(pixel_values)
+
+            input_embeds = self.language_model.get_input_embeddings()(input_ids)
+            b, n, c = input_embeds.shape
+            input_embeds = input_embeds.reshape(b * n, c)
+
+            input_ids = input_ids.reshape(b * n)
+            selected = input_ids == self.config.image_token_index
+            assert selected.sum() != 0
+            input_embeds[selected] = vit_embeds.reshape(-1, c).to(input_embeds.device)
+
+            input_embeds = input_embeds.reshape(b, n, c)
+        else:
+            input_embeds = self.language_model.get_input_embeddings()(input_ids)
+
+        if "use_cache" not in generate_kwargs:
+            generate_kwargs["use_cache"] = True
+
+        outputs = self.language_model.generate(
+            inputs_embeds=input_embeds,
+            attention_mask=attention_mask,
+            generation_config=generation_config,
+            output_hidden_states=output_hidden_states,
+            **generate_kwargs,
+        )
+
+        return outputs
+
+    # Copied from transformers.models.llava_next.modeling_llava_next.LlavaNextForConditionalGeneration.get_input_embeddings
+    def get_input_embeddings(self):
+        return self.language_model.get_input_embeddings()
+
+    # Copied from transformers.models.llava_next.modeling_llava_next.LlavaNextForConditionalGeneration.set_input_embeddings
+    def set_input_embeddings(self, value):
+        self.language_model.set_input_embeddings(value)
+
+    # Copied from transformers.models.llava_next.modeling_llava_next.LlavaNextForConditionalGeneration.get_output_embeddings
+    def get_output_embeddings(self):
+        return self.language_model.get_output_embeddings()
+
+    # Copied from transformers.models.llava_next.modeling_llava_next.LlavaNextForConditionalGeneration.set_output_embeddings
+    def set_output_embeddings(self, new_embeddings):
+        self.language_model.set_output_embeddings(new_embeddings)
+
+    # Copied from transformers.models.llava_next.modeling_llava_next.LlavaNextForConditionalGeneration.set_decoder
+    def set_decoder(self, decoder):
+        self.language_model.set_decoder(decoder)
+
+    # Copied from transformers.models.llava_next.modeling_llava_next.LlavaNextForConditionalGeneration.get_decoder
+    def get_decoder(self):
+        return self.language_model.get_decoder()
diff --git a/lerobot/src/lerobot/policies/groot/eagle2_hg_model/processing_eagle2_5_vl.py b/lerobot/src/lerobot/policies/groot/eagle2_hg_model/processing_eagle2_5_vl.py
new file mode 100755
index 0000000000000000000000000000000000000000..27f9b33454177319379a826035a91f72ab967291
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/eagle2_hg_model/processing_eagle2_5_vl.py
@@ -0,0 +1,518 @@
+# Copyright 2024 The HuggingFace Inc. team.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""
+Processor class for Eagle25VL.
+copy from https://github.com/huggingface/transformers/blob/main/src/transformers/models/llava_onevision/processing_llava_onevision.py
+"""
+
+import base64
+import os
+import re
+from io import BytesIO
+
+import requests
+import torch
+from PIL import Image
+from transformers.feature_extraction_utils import BatchFeature
+from transformers.image_utils import ImageInput
+from transformers.processing_utils import ProcessingKwargs, ProcessorMixin, Unpack
+from transformers.tokenization_utils_base import PreTokenizedInput, TextInput
+from transformers.utils import logging
+from transformers.video_utils import VideoInput
+
+logger = logging.get_logger(__name__)
+
+
+FRAME_FACTOR = 2
+FPS = 2.0
+FPS_MIN_FRAMES = 4
+FPS_MAX_FRAMES = 256
+
+
+def to_rgb(pil_image: Image.Image) -> Image.Image:
+    if pil_image.mode == "RGBA":
+        white_background = Image.new("RGB", pil_image.size, (255, 255, 255))
+        white_background.paste(pil_image, mask=pil_image.split()[3])  # Use alpha channel as mask
+        return white_background
+    else:
+        return pil_image.convert("RGB")
+
+
+def fetch_image(ele: dict[str, str | Image.Image]) -> Image.Image:
+    image = ele["image"] if "image" in ele else ele["image_url"]
+    image_obj = None
+    if isinstance(image, Image.Image):
+        image_obj = image
+    elif image.startswith("http://") or image.startswith("https://"):
+        response = requests.get(image, stream=True, timeout=10)
+        image_obj = Image.open(BytesIO(response.content))
+    elif image.startswith("file://"):
+        image_obj = Image.open(image[7:])
+    elif image.startswith("data:image"):
+        if "base64," in image:
+            _, base64_data = image.split("base64,", 1)
+            data = base64.b64decode(base64_data)
+            image_obj = Image.open(BytesIO(data))
+    else:
+        image_obj = Image.open(image)
+    if image_obj is None:
+        raise ValueError(
+            f"Unrecognized image input, support local path, http url, base64 and PIL.Image, got {image}"
+        )
+    image = to_rgb(image_obj)
+    if "scale_factor" in ele:
+        scale_factor = ele["scale_factor"]
+        image = image.resize((image.width * scale_factor, image.height * scale_factor), Image.BILINEAR)
+    return image
+
+
+class Eagle25VLProcessorKwargs(ProcessingKwargs, total=False):
+    # see processing_utils.ProcessingKwargs documentation for usage.
+    _defaults = {
+        "text_kwargs": {
+            "padding": False,
+        },
+        "images_kwargs": {},
+        "videos_kwargs": {"max_dynamic_tiles": 1},
+    }
+
+
+class Eagle25VLProcessor(ProcessorMixin):
+    r"""
+    Constructs a Eagle25VL processor which wraps a Eagle25VL video processor, Eagle25VL image processor and a Eagle25VL tokenizer into a single processor.
+
+    [`Eagle25VLProcessor`] offers all the functionalities of [`Eagle25VLVideoProcessor`], [`Eagle25VLImageProcessor`] and [`Eagle25VLTokenizer`]. See the
+    [`~Eagle25VLVideoProcessor.__call__`], [`~Eagle25VLProcessor.__call__`] and [`~Eagle25VLProcessor.decode`] for more information.
+
+    Args:
+        image_processor ([`LlavaOnevisionImageProcessor`], *optional*):
+            The image processor is a required input.
+        tokenizer ([`LlamaTokenizerFast`], *optional*):
+            The tokenizer is a required input.
+        num_image_tokens (`int`, *optional*):
+            Number of image tokens for one imagethat will be returned by vision tower.
+        vision_feature_select_strategy (`str`, *optional*):
+            The feature selection strategy used to select the vision feature from the vision backbone.
+            Should be same as in model's config
+        chat_template (`str`, *optional*): A Jinja template which will be used to convert lists of messages
+            in a chat into a tokenizable string.
+        image_token (`str`, *optional*, defaults to `"<image>"`):
+            Special token used to denote image location.
+        video_token (`str`, *optional*, defaults to `"<video>"`):
+            Special token used to denote video location.
+    """
+
+    attributes = ["image_processor", "tokenizer"]
+    valid_kwargs = [
+        "chat_template",
+        "num_image_tokens",
+        "vision_feature_select_strategy",
+        "image_token",
+        "video_token",
+        "images_kwargs",
+        "videos_kwargs",
+        "text_kwargs",
+    ]
+    image_processor_class = "AutoImageProcessor"
+    tokenizer_class = "AutoTokenizer"
+
+    def __init__(
+        self,
+        image_processor=None,
+        tokenizer=None,
+        vision_feature_select_strategy=None,
+        chat_template=None,
+        image_token="<IMG_CONTEXT>",  # nosec: B107
+        video_token="<IMG_CONTEXT>",  # nosec: B107
+        tokens_per_tile=256,
+        image_placeholder="image",
+        video_placeholder="video",
+        image_start_token="<img>",
+        image_end_token="</img>",
+        **kwargs,
+    ):
+        self.vision_feature_select_strategy = vision_feature_select_strategy
+        self.image_token = tokenizer.image_token if hasattr(tokenizer, "image_token") else image_token
+        self.video_token = tokenizer.video_token if hasattr(tokenizer, "video_token") else video_token
+        self.image_token_id = (
+            tokenizer.image_token_id
+            if getattr(tokenizer, "image_token_id", None)
+            else tokenizer.convert_tokens_to_ids(self.image_token)
+        )
+        self.video_token_id = (
+            tokenizer.video_token_id
+            if getattr(tokenizer, "video_token_id", None)
+            else tokenizer.convert_tokens_to_ids(self.video_token)
+        )
+        self.image_placeholder = image_placeholder
+        self.video_placeholder = video_placeholder
+        self.tokens_per_tile = tokens_per_tile
+        self.image_start_token = image_start_token
+        self.image_end_token = image_end_token
+        if "auto_map" in kwargs:
+            self.auto_map = kwargs["auto_map"]
+        super().__init__(image_processor, tokenizer, chat_template=chat_template)
+
+    def replace_media_placeholder(
+        self, text, image_list, video_list, timestamps_list, fps_list, **output_kwargs
+    ):
+        num_of_images_in_this_sample = 0
+        num_of_videos_in_this_sample = 0
+        # Regular expression pattern to match formats like <image-1> or <video-2>
+        pattern = re.compile(rf"<({self.image_placeholder}|{self.video_placeholder})-(\d+)>")
+        unified_frame_list = []
+
+        # image_min_dynamic_tiles = output_kwargs["images_kwargs"].get(
+        #     "min_dynamic_tiles", self.image_processor.min_dynamic_tiles
+        # )
+        # image_max_dynamic_tiles = output_kwargs["images_kwargs"].get(
+        #     "max_dynamic_tiles", self.image_processor.max_dynamic_tiles
+        # )
+        # image_use_thumbnail = output_kwargs["images_kwargs"].get(
+        #     "use_thumbnail", self.image_processor.use_thumbnail
+        # )
+        video_min_dynamic_tiles = output_kwargs["videos_kwargs"].get(
+            "min_dynamic_tiles", self.image_processor.min_dynamic_tiles
+        )
+        video_max_dynamic_tiles = output_kwargs["videos_kwargs"].get(
+            "max_dynamic_tiles", self.image_processor.max_dynamic_tiles
+        )
+        video_use_thumbnail = output_kwargs["videos_kwargs"].get(
+            "use_thumbnail", self.image_processor.use_thumbnail
+        )
+
+        tile_size = self.image_processor.size.get("height", 448)
+
+        # Function to replace tags in a single text
+        def replace_in_text(text):
+            # repl callback function for each match replacement operation
+            def repl(match):
+                nonlocal unified_frame_list
+                nonlocal num_of_images_in_this_sample
+                nonlocal num_of_videos_in_this_sample
+                media_type = match.group(1)  # 'image' or 'video'
+                idx_in_list = int(match.group(2)) - 1  # Convert to list index (0-based)
+                # Select the corresponding path based on media type
+                idx_mapper = {
+                    0: "first",
+                    1: "second",
+                    2: "third",
+                    3: "fourth",
+                    4: "fifth",
+                    5: "sixth",
+                    6: "seventh",
+                    7: "eighth",
+                    8: "ninth",
+                    9: "tenth",
+                }
+                if media_type == "image":
+                    image_inputs = self.image_processor(
+                        images=[image_list[idx_in_list]],
+                        videos=None,
+                        **output_kwargs["images_kwargs"],
+                    )
+                    num_all_tiles = image_inputs["pixel_values"].shape[0]
+                    special_placeholder = f"<image {idx_in_list + 1}>{self.image_start_token}{self.image_token * num_all_tiles * self.tokens_per_tile}{self.image_end_token}"
+                    unified_frame_list.append(image_inputs)
+                    num_of_images_in_this_sample += 1
+
+                elif media_type == "video":
+                    video_inputs = self.image_processor(
+                        images=None,
+                        videos=[video_list[idx_in_list]],
+                        **output_kwargs["videos_kwargs"],
+                    )
+                    num_all_tiles = video_inputs["pixel_values"].shape[0]
+                    image_sizes = video_inputs["image_sizes"]
+                    if timestamps_list is not None and -1 not in timestamps_list:
+                        frame_timestamps = timestamps_list[idx_in_list]
+                    else:
+                        frame_timestamps = None
+                    sampled_fps = fps_list[idx_in_list] if fps_list is not None else None
+
+                    num_of_tiles_each_frame = [
+                        self.get_number_tiles_based_on_image_size(
+                            image_size,
+                            video_min_dynamic_tiles,
+                            video_max_dynamic_tiles,
+                            video_use_thumbnail,
+                            tile_size,
+                        )
+                        for image_size in image_sizes
+                    ]
+                    assert sum(num_of_tiles_each_frame) == num_all_tiles, (
+                        f"The number of tiles in each frame is not equal to the total number of tiles: {sum(num_of_tiles_each_frame)} != {num_all_tiles}"
+                    )
+
+                    if frame_timestamps is not None:
+                        assert len(frame_timestamps) == len(num_of_tiles_each_frame), (
+                            f"The number of timestamps is not equal to the number of frames: {len(frame_timestamps)} != {len(num_of_tiles_each_frame)}"
+                        )
+                        special_placeholder = [
+                            f"Frame {i + 1} sample at {frame_timestamps[i]:.2f}s: {self.image_start_token}{self.image_token * num_of_tiles * self.tokens_per_tile}{self.image_end_token}"
+                            for i, num_of_tiles in enumerate(num_of_tiles_each_frame)
+                        ]
+                    else:
+                        special_placeholder = [
+                            f"Frame {i + 1}: {self.image_start_token}{self.image_token * num_of_tiles * self.tokens_per_tile}{self.image_end_token}"
+                            for i, num_of_tiles in enumerate(num_of_tiles_each_frame)
+                        ]
+
+                    if sampled_fps is not None:
+                        special_placeholder = (
+                            f"The {idx_mapper[idx_in_list]} video sampled with {sampled_fps:.2f} fps: "
+                            + "".join(special_placeholder)
+                        )
+                    else:
+                        special_placeholder = f"The {idx_mapper[idx_in_list]} video: " + "".join(
+                            special_placeholder
+                        )
+                    unified_frame_list.append(video_inputs)
+                    num_of_videos_in_this_sample += 1
+                else:
+                    raise ValueError(f"Unknown media type: {media_type}")
+                return special_placeholder
+
+            return pattern.sub(repl, text)
+
+        text = replace_in_text(text)
+        if len(unified_frame_list) > 0:
+            pixel_values = torch.cat([frame["pixel_values"] for frame in unified_frame_list])
+            image_sizes = torch.cat([frame["image_sizes"] for frame in unified_frame_list])
+        else:
+            pixel_values = None
+            image_sizes = None
+        return (
+            text,
+            pixel_values,
+            image_sizes,
+            num_of_images_in_this_sample,
+            num_of_videos_in_this_sample,
+        )
+
+    def __call__(
+        self,
+        images: ImageInput = None,
+        text: TextInput | PreTokenizedInput | list[TextInput] | list[PreTokenizedInput] = None,
+        audio=None,
+        videos: VideoInput = None,
+        **kwargs: Unpack[Eagle25VLProcessorKwargs],
+    ) -> BatchFeature:
+        """
+        Main method to prepare for the model one or several sequences(s) and image(s). This method forwards the `text`
+        and `kwargs` arguments to LlamaTokenizerFast's [`~LlamaTokenizerFast.__call__`] if `text` is not `None` to encode
+        the text. To prepare the image(s), this method forwards the `images` and `kwrags` arguments to
+        LlavaNextImageProcessor's [`~LlavaNextImageProcessor.__call__`] if `images` is not `None`. Please refer to the docstring
+        of the above two methods for more information.
+
+        Args:
+            images (`PIL.Image.Image`, `np.ndarray`, `torch.Tensor`, `List[PIL.Image.Image]`, `List[np.ndarray]`, `List[torch.Tensor]`):
+                The image or batch of images to be prepared. Each image can be a PIL image, NumPy array or PyTorch
+                tensor. Both channels-first and channels-last formats are supported.
+            text (`str`, `List[str]`, `List[List[str]]`):
+                The sequence or batch of sequences to be encoded. Each sequence can be a string or a list of strings
+                (pretokenized string). If the sequences are provided as list of strings (pretokenized), you must set
+                `is_split_into_words=True` (to lift the ambiguity with a batch of sequences).
+            videos (`np.ndarray`, `torch.Tensor`, `List[np.ndarray]`, `List[torch.Tensor]`):
+                The image or batch of videos to be prepared. Each video can be a 4D NumPy array or PyTorch
+
+        Returns:
+            [`BatchFeature`]: A [`BatchFeature`] with the following fields:
+
+            - **input_ids** -- List of token ids to be fed to a model. Returned when `text` is not `None`.
+            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
+              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
+              `None`).
+            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
+            - **pixel_values_videos** -- Pixel values of a video input to be fed to a model. Returned when `videos` is not `None`.
+            - **image_sizes** -- Size of each image that will be used to unpad an image. Returned when `images` is not `None`.
+        """
+
+        output_kwargs = self._merge_kwargs(
+            Eagle25VLProcessorKwargs,
+            tokenizer_init_kwargs=self.tokenizer.init_kwargs,
+            **kwargs,
+        )
+
+        if isinstance(text, str):
+            text_list = [text]
+        elif not isinstance(text, list) and not isinstance(text[0], str):
+            raise ValueError("Invalid input text. Please provide a string, or a list of strings")
+        elif isinstance(text, list) and isinstance(text[0], str):
+            text_list = text
+
+        if images is None:
+            images = []
+        if videos is None:
+            videos = []
+
+        pixel_values_list = []
+        image_sizes_list = []
+        new_sample_list = []
+        image_start_idx = 0
+        video_start_idx = 0
+        timestamps_batch = output_kwargs["videos_kwargs"].pop("timestamps", None)
+        fps_batch = output_kwargs["videos_kwargs"].pop("fps", None)
+        for sample in text_list:
+            timestamps_list = timestamps_batch[video_start_idx:] if timestamps_batch is not None else None
+            fps_list = fps_batch[video_start_idx:] if fps_batch is not None else None
+            (
+                sample,
+                pixel_values,
+                image_sizes,
+                num_of_images_in_this_sample,
+                num_of_videos_in_this_sample,
+            ) = self.replace_media_placeholder(
+                sample,
+                images[image_start_idx:],
+                videos[video_start_idx:],
+                timestamps_list,
+                fps_list,
+                **output_kwargs,
+            )
+            new_sample_list.append(sample)
+            if pixel_values is not None:
+                pixel_values_list.append(pixel_values)
+                image_sizes_list.append(image_sizes)
+            image_start_idx += num_of_images_in_this_sample
+            video_start_idx += num_of_videos_in_this_sample
+
+        if len(pixel_values_list) > 0:
+            image_inputs = {
+                "pixel_values": torch.cat(pixel_values_list),
+                "image_sizes": torch.cat(image_sizes_list),
+            }
+        else:
+            image_inputs = {}
+        video_inputs = {}
+        text_inputs = self.tokenizer(new_sample_list, **output_kwargs["text_kwargs"])
+        return BatchFeature(data={**text_inputs, **image_inputs, **video_inputs})
+
+    def get_number_tiles_based_on_image_size(
+        self, image_size: tuple, min_num: int, max_num: int, use_thumbnail: bool, tile_size: int
+    ) -> int:
+        """
+        Get the number of tiles based on the image size.
+        """
+        orig_height, orig_width = image_size
+        aspect_ratio = orig_width / orig_height
+        # calculate the existing image aspect ratio
+        target_ratios = {
+            (i, j)
+            for n in range(min_num, max_num + 1)
+            for i in range(1, n + 1)
+            for j in range(1, n + 1)
+            if i * j <= max_num and i * j >= min_num
+        }
+        target_ratios = sorted(target_ratios, key=lambda x: x[0] * x[1])
+
+        # find the closest aspect ratio to the target
+        target_aspect_ratio = self.image_processor.find_closest_aspect_ratio(
+            aspect_ratio, target_ratios, orig_width, orig_height, tile_size
+        )
+        tiles_num = target_aspect_ratio[0] * target_aspect_ratio[1]
+        if use_thumbnail and tiles_num > 1:
+            tiles_num += 1
+        return tiles_num
+
+    # Copied from transformers.models.clip.processing_clip.CLIPProcessor.batch_decode with CLIP->Llama
+    def batch_decode(self, *args, **kwargs):
+        """
+        This method forwards all its arguments to LlamaTokenizerFast's [`~PreTrainedTokenizer.batch_decode`]. Please
+        refer to the docstring of this method for more information.
+        """
+        return self.tokenizer.batch_decode(*args, **kwargs)
+
+    # Copied from transformers.models.clip.processing_clip.CLIPProcessor.decode with CLIP->Llama
+    def decode(self, *args, **kwargs):
+        """
+        This method forwards all its arguments to LlamaTokenizerFast's [`~PreTrainedTokenizer.decode`]. Please refer to
+        the docstring of this method for more information.
+        """
+        return self.tokenizer.decode(*args, **kwargs)
+
+    @property
+    # Copied from transformers.models.clip.processing_clip.CLIPProcessor.model_input_names
+    def model_input_names(self):
+        tokenizer_input_names = self.tokenizer.model_input_names
+        image_processor_input_names = self.image_processor.model_input_names
+        return list(dict.fromkeys(tokenizer_input_names + image_processor_input_names))
+
+    # override to save video-config in a separate config file
+    def save_pretrained(self, save_directory, **kwargs):
+        if os.path.isfile(save_directory):
+            raise ValueError(f"Provided path ({save_directory}) should be a directory, not a file")
+        os.makedirs(save_directory, exist_ok=True)
+
+        outputs = super().save_pretrained(save_directory, **kwargs)
+        return outputs
+
+    # override to load video-config from a separate config file
+    @classmethod
+    def from_pretrained(cls, pretrained_model_name_or_path, **kwargs):
+        processor = super().from_pretrained(pretrained_model_name_or_path, **kwargs)
+
+        # if return_unused_kwargs a tuple is returned where the second element is 'unused_kwargs'
+        if isinstance(processor, tuple):
+            processor = processor[0]
+        return processor
+
+    # Copy from https://github.com/QwenLM/Qwen2.5-VL/blob/main/qwen-vl-utils/src/qwen_vl_utils/vision_process.py
+    def process_vision_info(
+        self,
+        conversations: list[dict] | list[list[dict]],
+        return_video_kwargs: bool = False,
+    ) -> tuple[list[Image.Image] | None, list[torch.Tensor | list[Image.Image]] | None, dict | None]:
+        vision_infos = self.extract_vision_info(conversations)
+        ## Read images or videos
+        image_inputs = []
+        video_inputs = []
+        video_sample_fps_list = []
+        video_timestamps_list = []
+        for vision_info in vision_infos:
+            if "image" in vision_info or "image_url" in vision_info:
+                image_inputs.append(fetch_image(vision_info))
+            else:
+                raise ValueError("image, image_url or video should in content.")
+        if len(image_inputs) == 0:
+            image_inputs = None
+        if len(video_inputs) == 0:
+            video_inputs = None
+        if return_video_kwargs:
+            return (
+                image_inputs,
+                video_inputs,
+                {"fps": video_sample_fps_list, "timestamps": video_timestamps_list},
+            )
+        return image_inputs, video_inputs
+
+    def extract_vision_info(self, conversations: list[dict] | list[list[dict]]) -> list[dict]:
+        vision_infos = []
+        if isinstance(conversations[0], dict):
+            conversations = [conversations]
+        for conversation in conversations:
+            for message in conversation:
+                if isinstance(message["content"], list):
+                    for ele in message["content"]:
+                        if (
+                            "image" in ele
+                            or "image_url" in ele
+                            or "video" in ele
+                            or ele["type"] in ("image", "image_url", "video")
+                        ):
+                            vision_infos.append(ele)
+        return vision_infos
+
+
+__all__ = ["Eagle25VLProcessor"]
diff --git a/lerobot/src/lerobot/policies/groot/groot_n1.py b/lerobot/src/lerobot/policies/groot/groot_n1.py
new file mode 100644
index 0000000000000000000000000000000000000000..06ff5a04d617b42fedfa7d7f802aa380358dbebd
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/groot_n1.py
@@ -0,0 +1,376 @@
+# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import TYPE_CHECKING
+
+import numpy as np
+import torch
+import torch.nn as nn
+from huggingface_hub import snapshot_download
+from huggingface_hub.errors import HFValidationError, RepositoryNotFoundError
+
+from lerobot.utils.import_utils import _transformers_available
+
+# Conditional import for type checking and lazy loading
+if TYPE_CHECKING or _transformers_available:
+    from transformers import AutoConfig, AutoModel, PretrainedConfig, PreTrainedModel
+    from transformers.feature_extraction_utils import BatchFeature
+else:
+    AutoConfig = None
+    AutoModel = None
+    PretrainedConfig = object
+    PreTrainedModel = object
+    BatchFeature = None
+
+try:
+    import tree
+except ImportError:
+    tree = None
+
+from lerobot.policies.groot.action_head.flow_matching_action_head import (
+    FlowmatchingActionHead,
+    FlowmatchingActionHeadConfig,
+)
+from lerobot.policies.groot.utils import ensure_eagle_cache_ready
+from lerobot.utils.constants import ACTION, HF_LEROBOT_HOME
+
+DEFAULT_VENDOR_EAGLE_PATH = str((Path(__file__).resolve().parent / "eagle2_hg_model").resolve())
+DEFAULT_TOKENIZER_ASSETS_REPO = "lerobot/eagle2hg-processor-groot-n1p5"
+
+
+class EagleBackbone(nn.Module):
+    def __init__(
+        self,
+        tune_llm: bool = False,
+        tune_visual: bool = False,
+        select_layer: int = -1,
+        reproject_vision: bool = False,
+        use_flash_attention: bool = False,
+        load_bf16: bool = False,
+        eagle_path: str = DEFAULT_VENDOR_EAGLE_PATH,
+        tokenizer_assets_repo: str = DEFAULT_TOKENIZER_ASSETS_REPO,
+        project_to_dim: int = 1536,
+    ):
+        """
+        Args:
+            tune_llm: whether to tune the LLM model (default: True)
+            tune_visual: whether to tune the visual model (default: False)
+        """
+        super().__init__()
+        assert not reproject_vision, "Reproject vision is not implemented here, set to False"
+
+        # Prefer loading Eagle model config from the cache directory where vendor files were copied.
+        vendor_dir = DEFAULT_VENDOR_EAGLE_PATH
+        cache_dir = HF_LEROBOT_HOME / tokenizer_assets_repo
+        try:
+            ensure_eagle_cache_ready(vendor_dir, cache_dir, tokenizer_assets_repo)
+        except Exception as exc:  # nosec: B110
+            print(f"[GROOT] Warning: failed to prepare Eagle cache for backbone: {exc}")
+
+        config = AutoConfig.from_pretrained(str(cache_dir), trust_remote_code=True)
+        self.eagle_model = AutoModel.from_config(config, trust_remote_code=True)
+
+        if project_to_dim is not None:
+            self.eagle_linear = torch.nn.Linear(2048, project_to_dim)
+        else:
+            self.eagle_linear = torch.nn.Identity()
+
+        # needed since we don't use these layers. Also saves compute
+        while len(self.eagle_model.language_model.model.layers) > select_layer:
+            self.eagle_model.language_model.model.layers.pop(-1)
+
+        self.select_layer = select_layer
+        self.set_trainable_parameters(tune_llm, tune_visual)
+
+    def set_trainable_parameters(self, tune_llm: bool, tune_visual: bool):
+        self.tune_llm = tune_llm
+        self.tune_visual = tune_visual
+        for p in self.parameters():
+            p.requires_grad = True
+        if not tune_llm:
+            self.eagle_model.language_model.requires_grad_(False)
+        if not tune_visual:
+            self.eagle_model.vision_model.requires_grad_(False)
+            self.eagle_model.mlp1.requires_grad_(False)
+        print(f"Tune backbone llm: {self.tune_llm}")
+        print(f"Tune backbone visual: {self.tune_visual}")
+        # Check if any parameters are still trainable. If not, print a warning.
+        if not tune_llm and not tune_visual:
+            for name, p in self.named_parameters():
+                if p.requires_grad:
+                    print(f"Backbone trainable parameter: {name}")
+        if not any(p.requires_grad for p in self.parameters()):
+            print("Warning: No backbone trainable parameters found.")
+
+    def set_frozen_modules_to_eval_mode(self):
+        """
+        Huggingface will call model.train() at each training_step. To ensure
+        the expected behaviors for modules like dropout, batchnorm, etc., we
+        need to call model.eval() for the frozen modules.
+        """
+        if self.training:
+            if self.eagle_model.language_model and not self.tune_llm:
+                self.eagle_model.language_model.eval()
+            if self.eagle_model.vision_model and not self.tune_visual:
+                self.eagle_model.vision_model.eval()
+
+    def prepare_input(self, batch: dict) -> BatchFeature:
+        return BatchFeature(data=batch)
+
+    def forward_eagle(self, vl_input: BatchFeature) -> BatchFeature:
+        eagle_prefix = "eagle_"
+        eagle_input = {
+            k.removeprefix(eagle_prefix): v for k, v in vl_input.items() if k.startswith(eagle_prefix)
+        }
+        del eagle_input["image_sizes"]
+
+        eagle_output = self.eagle_model(**eagle_input, output_hidden_states=True, return_dict=True)
+        eagle_features = eagle_output.hidden_states[self.select_layer]
+
+        eagle_features = self.eagle_linear(eagle_features)
+        return eagle_features, eagle_input["attention_mask"]
+
+    def forward(self, vl_input: BatchFeature) -> BatchFeature:
+        self.set_frozen_modules_to_eval_mode()
+
+        eagle_embeds, eagle_mask = self.forward_eagle(vl_input)
+
+        # YL (TODO HACK): to resolve DDP issue when tune_visual=True
+        # Ensure all trainable parameters in vision_model are used in the forward pass for DDP compatibility
+        if self.training and self.tune_visual:
+            dummy_term = torch.tensor(
+                0.0, device=eagle_embeds.device, dtype=eagle_embeds.dtype, requires_grad=True
+            )
+            for param in self.eagle_model.vision_model.parameters():
+                if param.requires_grad:
+                    dummy_term = dummy_term + 0.0 * param.sum()
+            eagle_embeds = eagle_embeds + dummy_term
+
+        return BatchFeature(
+            data={"backbone_features": eagle_embeds, "backbone_attention_mask": eagle_mask}
+        )  # [B, T2, hidden_size]
+
+
+BACKBONE_FEATURE_KEY = "backbone_features"
+ACTION_KEY = "action_pred"
+LOSS_KEY = "loss"
+ERROR_MSG = "Error: unexpected input/output"
+N_COLOR_CHANNELS = 3
+
+
+# config
+@dataclass
+class GR00TN15Config(PretrainedConfig):
+    model_type = "gr00t_n1_5"
+    backbone_cfg: dict = field(init=False, metadata={"help": "Backbone configuration."})
+
+    action_head_cfg: dict = field(init=False, metadata={"help": "Action head configuration."})
+
+    action_horizon: int = field(init=False, metadata={"help": "Action horizon."})
+
+    action_dim: int = field(init=False, metadata={"help": "Action dimension."})
+    compute_dtype: str = field(default="float32", metadata={"help": "Compute dtype."})
+
+    def __init__(self, **kwargs):
+        super().__init__(**kwargs)
+        for key, value in kwargs.items():
+            setattr(self, key, value)
+
+
+# real model
+class GR00TN15(PreTrainedModel):
+    supports_gradient_checkpointing = True
+    config_class = GR00TN15Config
+    """
+    we expect the backbone output to have a key 'backbone_features' with shape (batch_size, n, hidden_size)
+    here n is variable and can be e.g. time, 1 or user specified
+    we expect the action head output to have a key 'action_pred' with shape (batch_size, time, action_dim) during inference time
+    we expect these to have type BatchFeature, and they can of course have many other user specified keys too
+    """
+
+    def __init__(
+        self,
+        config: GR00TN15Config,
+        local_model_path: str,
+    ):
+        assert isinstance(config.backbone_cfg, dict)
+        assert isinstance(config.action_head_cfg, dict)
+
+        super().__init__(config)
+        self.local_model_path = local_model_path
+
+        self.backbone = EagleBackbone(**config.backbone_cfg)
+        action_head_cfg = FlowmatchingActionHeadConfig(**config.action_head_cfg)
+        self.action_head = FlowmatchingActionHead(action_head_cfg)
+
+        self.action_horizon = config.action_horizon
+        self.action_dim = config.action_dim
+        self.compute_dtype = config.compute_dtype
+
+    def validate_inputs(self, inputs):
+        # NOTE -- this should be handled internally by the model
+        # however, doing that will likely be breaking changes -- so we'll need to do it after the deadline
+
+        detected_error = False
+        error_msg = ERROR_MSG
+        if ACTION in inputs:
+            action = inputs[ACTION]
+            # In inference, action may be omitted or None; validate only when it's a tensor.
+            if action is None:
+                pass  # allow None during inference
+            elif isinstance(action, torch.Tensor):
+                shape_ok = (
+                    len(action.shape) == 3
+                    and action.shape[1] == self.action_horizon
+                    and action.shape[2] == self.action_dim
+                )
+                if not shape_ok:
+                    error_msg += f"\n{action.shape=}"
+                    detected_error = True
+            else:
+                # Unexpected non-tensor type provided for action
+                error_msg += f"\nInvalid type for action: {type(action)}"
+                detected_error = True
+
+        if "video" in inputs:
+            video = inputs["video"]
+            type_ok = isinstance(video, np.ndarray)
+            dtype_ok = video.dtype == np.uint8
+            shape_ok = len(video.shape) == 6 and video.shape[3] == N_COLOR_CHANNELS
+            if not type_ok:
+                error_msg += f"\n{type(video)=}"
+                detected_error = True
+            if not dtype_ok:
+                error_msg += f"\n{video.dtype=}"
+                detected_error = True
+            if not shape_ok:
+                error_msg += f"\n{video.shape=}"
+                detected_error = True
+
+        if detected_error:
+            raise ValueError(error_msg)
+
+    def validate_data(self, action_head_outputs, backbone_outputs, is_training):
+        fail_backbone = (
+            not isinstance(backbone_outputs, BatchFeature) or BACKBONE_FEATURE_KEY not in backbone_outputs
+        )
+
+        if fail_backbone:
+            error_msg = ERROR_MSG
+            error_msg += f"\n{isinstance(backbone_outputs, BatchFeature)=}"
+            error_msg += f"\n{BACKBONE_FEATURE_KEY in backbone_outputs=}"
+            error_msg += f"\n{backbone_outputs[BACKBONE_FEATURE_KEY].shape=}"
+            raise ValueError(error_msg)
+
+        fail_action_head = (not isinstance(action_head_outputs, BatchFeature)) or not (
+            (
+                LOSS_KEY in action_head_outputs and is_training
+            )  # there might not be an action prediction during training
+            or (
+                ACTION_KEY in action_head_outputs
+                and action_head_outputs[ACTION_KEY].shape[1] == self.action_horizon
+                and action_head_outputs[ACTION_KEY].shape[2] == self.action_dim
+            )
+        )
+
+        if fail_action_head:
+            error_msg = ERROR_MSG
+            error_msg += f"\n{isinstance(action_head_outputs, BatchFeature)=}"
+            error_msg += f"\n{LOSS_KEY in action_head_outputs=}"
+            error_msg += f"\n{action_head_outputs[ACTION_KEY].shape=}"
+            error_msg += f"\n{self.action_horizon=}"
+            error_msg += f"\n{self.action_dim=}"
+            raise ValueError(error_msg)
+
+    def forward(
+        self,
+        inputs: dict,
+    ) -> BatchFeature:
+        backbone_inputs, action_inputs = self.prepare_input(inputs)
+        backbone_outputs = self.backbone(backbone_inputs)
+        action_head_outputs = self.action_head(backbone_outputs, action_inputs)
+        self.validate_data(action_head_outputs, backbone_outputs, is_training=True)
+        return action_head_outputs
+
+    def get_action(
+        self,
+        inputs: dict,
+    ) -> BatchFeature:
+        backbone_inputs, action_inputs = self.prepare_input(inputs)
+        # Because the behavior of backbones remains the same for training and inference, we can use `forward` for backbones.
+        backbone_outputs = self.backbone(backbone_inputs)
+        action_head_outputs = self.action_head.get_action(backbone_outputs, action_inputs)
+        self.validate_data(action_head_outputs, backbone_outputs, is_training=False)
+        return action_head_outputs
+
+    def prepare_input(self, inputs) -> tuple[BatchFeature, BatchFeature]:
+        self.validate_inputs(inputs)
+        backbone_inputs = self.backbone.prepare_input(inputs)
+        action_inputs = self.action_head.prepare_input(inputs)
+
+        def to_device_with_maybe_dtype(x):
+            # Cast floating tensors to a memory-efficient compute dtype when requested.
+            # Rationale: Upcasting backbone activations to fp32 significantly increases VRAM.
+            # When compute_dtype is bfloat16, prefer bf16 for activations to match AMP behavior.
+            if not isinstance(x, torch.Tensor):
+                return x
+            if torch.is_floating_point(x):
+                if getattr(self, "compute_dtype", None) == "bfloat16":
+                    return x.to(self.device, dtype=torch.bfloat16)
+                # Fallback: preserve previous behavior if not using bf16 compute
+                return x.to(self.device, dtype=self.action_head.dtype)
+            # Non-floating tensors: move device only
+            return x.to(self.device)
+
+        backbone_inputs = tree.map_structure(to_device_with_maybe_dtype, backbone_inputs)
+        action_inputs = tree.map_structure(to_device_with_maybe_dtype, action_inputs)
+        return backbone_inputs, action_inputs
+
+    @classmethod
+    def from_pretrained(cls, pretrained_model_name_or_path: str, **kwargs):
+        tune_visual = kwargs.pop("tune_visual", True)
+        tune_llm = kwargs.pop("tune_llm", False)
+        tune_projector = kwargs.pop("tune_projector", True)
+        tune_diffusion_model = kwargs.pop("tune_diffusion_model", True)
+
+        print(f"Loading pretrained dual brain from {pretrained_model_name_or_path}")
+        print(f"Tune backbone vision tower: {tune_visual}")
+        print(f"Tune backbone LLM: {tune_llm}")
+        print(f"Tune action head projector: {tune_projector}")
+        print(f"Tune action head DiT: {tune_diffusion_model}")
+
+        # get the current model path being downloaded
+        try:
+            # NOTE(YL) This downloads the model to the local cache and returns the local path to the model
+            # saved in ~/.cache/huggingface/hub/
+            local_model_path = snapshot_download(pretrained_model_name_or_path, repo_type="model")
+            # HFValidationError, RepositoryNotFoundError
+        except (HFValidationError, RepositoryNotFoundError):
+            print(
+                f"Model not found or avail in the huggingface hub. Loading from local path: {pretrained_model_name_or_path}"
+            )
+            local_model_path = pretrained_model_name_or_path
+
+        pretrained_model = super().from_pretrained(
+            local_model_path, local_model_path=local_model_path, **kwargs
+        )
+
+        pretrained_model.backbone.set_trainable_parameters(tune_visual=tune_visual, tune_llm=tune_llm)
+        pretrained_model.action_head.set_trainable_parameters(
+            tune_projector=tune_projector, tune_diffusion_model=tune_diffusion_model
+        )
+        return pretrained_model
diff --git a/lerobot/src/lerobot/policies/groot/modeling_groot.py b/lerobot/src/lerobot/policies/groot/modeling_groot.py
new file mode 100644
index 0000000000000000000000000000000000000000..9a479b8f9fbc34d72920acbde5cf4f02d67b54c4
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/modeling_groot.py
@@ -0,0 +1,328 @@
+#!/usr/bin/env python
+
+# Copyright 2024 NVIDIA Corporation and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Groot Policy Wrapper for LeRobot Integration
+
+Minimal integration that delegates to Isaac-GR00T components where possible
+without porting their code. The intent is to:
+
+- Download and load the pretrained GR00T model via GR00TN15.from_pretrained
+- Optionally align action horizon similar to gr00t_finetune.py
+- Expose predict_action via GR00T model.get_action
+- Provide a training forward that can call the GR00T model forward if batch
+  structure matches.
+
+Notes:
+- Dataset loading and full training orchestration is handled by Isaac-GR00T
+  TrainRunner in their codebase. If you want to invoke that flow end-to-end
+  from LeRobot, see `GrootPolicy.finetune_with_groot_runner` below.
+"""
+
+import builtins
+import os
+from collections import deque
+from pathlib import Path
+from typing import TypeVar
+
+import torch
+from torch import Tensor
+
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.policies.groot.configuration_groot import GrootConfig
+from lerobot.policies.groot.groot_n1 import GR00TN15
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.utils.constants import ACTION, OBS_IMAGES
+
+T = TypeVar("T", bound="GrootPolicy")
+
+
+class GrootPolicy(PreTrainedPolicy):
+    """Wrapper around external Groot model for LeRobot integration."""
+
+    name = "groot"
+    config_class = GrootConfig
+
+    def __init__(self, config: GrootConfig, **kwargs):
+        """Initialize Groot policy wrapper."""
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        # Initialize GR00T model using ported components
+        self._groot_model = self._create_groot_model()
+
+        self.reset()
+
+    def _create_groot_model(self):
+        """Create and initialize the GR00T model using Isaac-GR00T API.
+
+        This is only called when creating a NEW policy (not when loading from checkpoint).
+
+        Steps (delegating to Isaac-GR00T):
+        1) Download and load pretrained model via GR00TN15.from_pretrained
+        2) Align action horizon with data_config if provided
+        """
+        # Handle Flash Attention compatibility issues
+        self._handle_flash_attention_compatibility()
+
+        model = GR00TN15.from_pretrained(
+            pretrained_model_name_or_path=self.config.base_model_path,
+            tune_llm=self.config.tune_llm,
+            tune_visual=self.config.tune_visual,
+            tune_projector=self.config.tune_projector,
+            tune_diffusion_model=self.config.tune_diffusion_model,
+        )
+
+        model.compute_dtype = "bfloat16" if self.config.use_bf16 else model.compute_dtype
+        model.config.compute_dtype = model.compute_dtype
+
+        return model
+
+    def reset(self):
+        """Reset policy state when environment resets."""
+        self._action_queue = deque([], maxlen=self.config.n_action_steps)
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type[T],
+        pretrained_name_or_path: str | Path,
+        *,
+        config: GrootConfig | None = None,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        strict: bool = True,
+        **kwargs,
+    ) -> T:
+        """Load Groot policy from pretrained model.
+
+        Handles two cases:
+        1. Base GR00T models (e.g., 'nvidia/GR00T-N1.5-3B') - loads the raw model
+        2. Fine-tuned LeRobot checkpoints - loads config and weights from safetensors
+
+        Args:
+            pretrained_name_or_path: Path to the GR00T model or fine-tuned checkpoint
+            config: Optional GrootConfig. If None, loads from checkpoint or creates default
+            force_download: Force download even if cached
+            resume_download: Resume interrupted download
+            proxies: Proxy settings
+            token: HuggingFace authentication token
+            cache_dir: Cache directory path
+            local_files_only: Only use local files
+            revision: Specific model revision
+            strict: Strict state dict loading
+            **kwargs: Additional arguments (passed to config)
+
+        Returns:
+            Initialized GrootPolicy instance with loaded model
+        """
+        from huggingface_hub import hf_hub_download
+        from huggingface_hub.constants import SAFETENSORS_SINGLE_FILE
+        from huggingface_hub.errors import HfHubHTTPError
+
+        print(
+            "The Groot policy is a wrapper around Nvidia's GR00T N1.5 model.\n"
+            f"Loading pretrained model from: {pretrained_name_or_path}"
+        )
+
+        model_id = str(pretrained_name_or_path)
+        is_finetuned_checkpoint = False
+
+        # Check if this is a fine-tuned LeRobot checkpoint (has model.safetensors)
+        try:
+            if os.path.isdir(model_id):
+                is_finetuned_checkpoint = os.path.exists(os.path.join(model_id, SAFETENSORS_SINGLE_FILE))
+            else:
+                # Try to download the safetensors file to check if it exists
+                try:
+                    hf_hub_download(
+                        repo_id=model_id,
+                        filename=SAFETENSORS_SINGLE_FILE,
+                        revision=revision,
+                        cache_dir=cache_dir,
+                        force_download=False,  # Just check, don't force download
+                        proxies=proxies,
+                        token=token,
+                        local_files_only=local_files_only,
+                    )
+                    is_finetuned_checkpoint = True
+                except HfHubHTTPError:
+                    is_finetuned_checkpoint = False
+        except Exception:
+            is_finetuned_checkpoint = False
+
+        if is_finetuned_checkpoint:
+            # This is a fine-tuned LeRobot checkpoint - use parent class loading
+            print("Detected fine-tuned LeRobot checkpoint, loading with state dict...")
+            return super().from_pretrained(
+                pretrained_name_or_path=pretrained_name_or_path,
+                config=config,
+                force_download=force_download,
+                resume_download=resume_download,
+                proxies=proxies,
+                token=token,
+                cache_dir=cache_dir,
+                local_files_only=local_files_only,
+                revision=revision,
+                strict=strict,
+                **kwargs,
+            )
+
+        # This is a base GR00T model - load it fresh
+        print("Detected base GR00T model, loading from HuggingFace...")
+
+        if config is None:
+            # Create default config with the pretrained path
+            config = GrootConfig(base_model_path=str(pretrained_name_or_path))
+
+            # Add minimal visual feature required for validation
+            # validate_features() will automatically add state and action features
+            # These are placeholders - actual robot features come from the preprocessor
+            if not config.input_features:
+                config.input_features = {
+                    f"{OBS_IMAGES}.camera": PolicyFeature(
+                        type=FeatureType.VISUAL,
+                        shape=(3, 224, 224),  # Default image size from config
+                    ),
+                }
+        else:
+            # Override the base_model_path with the provided path
+            config.base_model_path = str(pretrained_name_or_path)
+
+        # Pass through any additional config overrides from kwargs
+        for key, value in kwargs.items():
+            if hasattr(config, key):
+                setattr(config, key, value)
+
+        # Create a fresh policy instance - this will automatically load the GR00T model
+        # in __init__ via _create_groot_model()
+        policy = cls(config)
+
+        policy.eval()
+        return policy
+
+    def get_optim_params(self) -> dict:
+        return self.parameters()
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, dict]:
+        """Training forward pass.
+
+        Delegates to Isaac-GR00T model.forward when inputs are compatible.
+        """
+        # Build a clean input dict for GR00T: keep only tensors GR00T consumes
+        allowed_base = {"state", "state_mask", "action", "action_mask", "embodiment_id"}
+        groot_inputs = {
+            k: v
+            for k, v in batch.items()
+            if (k in allowed_base or k.startswith("eagle_")) and not (k.startswith("next.") or k == "info")
+        }
+
+        # Get device from model parameters
+        device = next(self.parameters()).device
+
+        # Run GR00T forward under bf16 autocast when enabled to reduce activation memory
+        # Rationale: Matches original GR00T finetuning (bf16 compute, fp32 params) and avoids fp32 upcasts.
+        with torch.autocast(device_type=device.type, dtype=torch.bfloat16, enabled=self.config.use_bf16):
+            outputs = self._groot_model.forward(groot_inputs)
+
+        # Isaac-GR00T returns a BatchFeature; loss key is typically 'loss'
+        loss = outputs.get("loss")
+
+        loss_dict = {"loss": loss.item()}
+
+        return loss, loss_dict
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor]) -> Tensor:
+        """Predict a chunk of actions for inference by delegating to Isaac-GR00T.
+
+        Returns a tensor of shape (B, n_action_steps, action_dim).
+        """
+        self.eval()
+
+        # Build a clean input dict for GR00T: keep only tensors GR00T consumes
+        # Preprocessing is handled by the processor pipeline, so we just filter the batch
+        # NOTE: During inference, we should NOT pass action/action_mask (that's what we're predicting)
+        allowed_base = {"state", "state_mask", "embodiment_id"}
+        groot_inputs = {
+            k: v
+            for k, v in batch.items()
+            if (k in allowed_base or k.startswith("eagle_")) and not (k.startswith("next.") or k == "info")
+        }
+
+        # Get device from model parameters
+        device = next(self.parameters()).device
+
+        # Use bf16 autocast for inference to keep memory low and match backbone dtype
+        with torch.autocast(device_type=device.type, dtype=torch.bfloat16, enabled=self.config.use_bf16):
+            outputs = self._groot_model.get_action(groot_inputs)
+
+        actions = outputs.get("action_pred")
+
+        original_action_dim = self.config.output_features[ACTION].shape[0]
+        actions = actions[:, :, :original_action_dim]
+
+        return actions
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select single action from action queue."""
+        self.eval()
+
+        if len(self._action_queue) == 0:
+            actions = self.predict_action_chunk(batch)
+            self._action_queue.extend(actions.transpose(0, 1))
+        return self._action_queue.popleft()
+
+    # -------------------------
+    # Internal helpers
+    # -------------------------
+    def _handle_flash_attention_compatibility(self) -> None:
+        """Handle Flash Attention compatibility issues by setting environment variables.
+
+        This addresses the common 'undefined symbol' error that occurs when Flash Attention
+        is compiled against a different PyTorch version than what's currently installed.
+        """
+
+        # Set environment variables to handle Flash Attention compatibility
+        # These help with symbol resolution issues
+        os.environ.setdefault("FLASH_ATTENTION_FORCE_BUILD", "0")
+        os.environ.setdefault("FLASH_ATTENTION_SKIP_CUDA_BUILD", "0")
+
+        # Try to import flash_attn and handle failures gracefully
+        try:
+            import flash_attn
+
+            print(f"[GROOT] Flash Attention version: {flash_attn.__version__}")
+        except ImportError as e:
+            print(f"[GROOT] Flash Attention not available: {e}")
+            print("[GROOT] Will use fallback attention mechanism")
+        except Exception as e:
+            if "undefined symbol" in str(e):
+                print(f"[GROOT] Flash Attention compatibility issue detected: {e}")
+                print("[GROOT] This is likely due to PyTorch/Flash Attention version mismatch")
+                print("[GROOT] Consider reinstalling Flash Attention with compatible version:")
+                print("  pip uninstall flash-attn")
+                print("  pip install --no-build-isolation flash-attn==2.6.3")
+                print("[GROOT] Continuing with fallback attention mechanism")
+            else:
+                print(f"[GROOT] Flash Attention error: {e}")
+                print("[GROOT] Continuing with fallback attention mechanism")
diff --git a/lerobot/src/lerobot/policies/groot/processor_groot.py b/lerobot/src/lerobot/policies/groot/processor_groot.py
new file mode 100644
index 0000000000000000000000000000000000000000..8bf9dabcacebd1e50945149d737c6a36c15dabae
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/processor_groot.py
@@ -0,0 +1,668 @@
+#!/usr/bin/env python
+
+# Copyright 2024 NVIDIA Corporation and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+from typing import TYPE_CHECKING, Any
+
+import numpy as np
+import torch
+from einops import rearrange
+from PIL import Image
+
+from lerobot.utils.import_utils import _transformers_available
+
+if TYPE_CHECKING or _transformers_available:
+    from transformers import AutoProcessor, ProcessorMixin
+else:
+    AutoProcessor = None
+    ProcessorMixin = object
+
+from lerobot.configs.types import (
+    FeatureType,
+    NormalizationMode,
+    PolicyFeature,
+)
+from lerobot.policies.groot.configuration_groot import GrootConfig
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    RenameObservationsProcessorStep,
+)
+from lerobot.processor.converters import (
+    policy_action_to_transition,
+    transition_to_policy_action,
+)
+from lerobot.types import EnvTransition, TransitionKey
+from lerobot.utils.constants import (
+    ACTION,
+    HF_LEROBOT_HOME,
+    OBS_IMAGE,
+    OBS_IMAGES,
+    OBS_STATE,
+    POLICY_POSTPROCESSOR_DEFAULT_NAME,
+    POLICY_PREPROCESSOR_DEFAULT_NAME,
+)
+
+# Defaults for Eagle processor locations
+DEFAULT_TOKENIZER_ASSETS_REPO = "lerobot/eagle2hg-processor-groot-n1p5"
+
+
+def make_groot_pre_post_processors(
+    config: GrootConfig, dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """Create preprocessor and postprocessor for Groot policy.
+
+    This creates a processing pipeline that transforms LeRobot data format into
+    the format expected by Isaac-GR00T models:
+
+    Preprocessing steps:
+    1. Optional key renaming (dataset-specific key mapping)
+    2. Add batch dimension to unbatched data
+    3. Pack video/state/action/language/embodiment and apply optional min-max normalization before padding
+    4. Encode video+language with Eagle VLM into intermediate eagle_content
+    5. Collate eagle_content into batched eagle_* tensors
+    6. Move tensors to device (GPU)
+
+    NOTE: We optionally apply min-max normalization to STATE and ACTION using
+    dataset-provided statistics prior to padding, mapping values to [-1, 1].
+    This mirrors SO100-style preprocessing and keeps scales consistent with GR00T.
+
+    Args:
+        config: Groot configuration containing data_config, embodiment_tag, etc.
+        dataset_stats: Optional per-key min/max statistics for normalization before padding.
+
+    Returns:
+        Tuple of (preprocessor, postprocessor) pipelines
+    """
+
+    # Get horizon/dimension parameters from config
+    # These should match the config used for the pretrained model
+    # Default values match most GR00T configs (state_horizon=1, action_horizon=16)
+    state_horizon = 1
+    # CRITICAL: Pretrained GR00T models use action_horizon=16 max!
+    # The model architecture hardcodes this limit
+    action_horizon = min(config.chunk_size, 16)
+    max_state_dim = config.max_state_dim
+    max_action_dim = config.max_action_dim
+
+    # Pass raw dataset_stats; normalization will occur inside pack step before padding
+    padded_stats = dataset_stats or {}
+
+    # Define feature specs for optional normalization steps
+    _features: dict[str, PolicyFeature] = {
+        # Observation features (only add those we may normalize)
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(state_horizon, max_state_dim)),
+        # Action feature
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(action_horizon, max_action_dim)),
+    }
+
+    # Normalize STATE and ACTION with min_max (SO100-like default)
+    _norm_map = {
+        FeatureType.ACTION: NormalizationMode.MIN_MAX,
+        FeatureType.STATE: NormalizationMode.MIN_MAX,
+    }
+
+    # Determine env action dimension from config (simple, object-like PolicyFeature)
+    try:
+        env_action_dim = int(config.output_features[ACTION].shape[0])
+    except Exception:
+        env_action_dim = 0
+
+    input_steps: list[ProcessorStep] = [
+        # 1. Rename keys if needed (e.g., dataset-specific camera names)
+        # Leave empty for now - add mappings if your dataset uses different key names
+        RenameObservationsProcessorStep(rename_map={}),
+        # 2. Add batch dimension for single samples
+        AddBatchDimensionProcessorStep(),
+        # 3. Pack video/state/action/language/embodiment; apply optional min-max normalization before padding
+        GrootPackInputsStep(
+            state_horizon=state_horizon,
+            action_horizon=action_horizon,
+            max_state_dim=max_state_dim,
+            max_action_dim=max_action_dim,
+            language_key="task",
+            formalize_language=False,
+            embodiment_tag=config.embodiment_tag,
+            normalize_min_max=True,
+            stats=padded_stats,
+        ),
+        # 4. Eagle encode (creates eagle_content)
+        GrootEagleEncodeStep(
+            tokenizer_assets_repo=config.tokenizer_assets_repo,
+        ),
+        # 5. Collate eagle_content -> eagle_* tensors
+        GrootEagleCollateStep(
+            tokenizer_assets_repo=config.tokenizer_assets_repo,
+        ),
+        # 6. Move to device
+        DeviceProcessorStep(device=config.device),
+    ]
+
+    # Postprocessing: slice to env action dim and unnormalize to env scale, then move to CPU
+    output_steps: list[ProcessorStep] = [
+        GrootActionUnpackUnnormalizeStep(
+            env_action_dim=env_action_dim,
+            stats=padded_stats,
+            normalize_min_max=True,
+        ),
+        # Finally, move to CPU for env interaction
+        DeviceProcessorStep(device="cpu"),
+    ]
+
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
+
+
+# GR00T specific processor steps
+
+
+def _to_uint8_np_bhwc(img_t: torch.Tensor) -> np.ndarray:
+    # img_t: (B, C, H, W) float in [0,1] or uint8
+    if img_t.dtype.is_floating_point:
+        img_t = (img_t.clamp(0, 1) * 255.0).to(torch.uint8)
+    return rearrange(img_t.cpu().numpy(), "b c h w -> b h w c")
+
+
+def _build_eagle_processor(tokenizer_assets_repo: str = DEFAULT_TOKENIZER_ASSETS_REPO) -> ProcessorMixin:
+    # Validate that the cache directory is ready. If not, instruct the user.
+    cache_dir = HF_LEROBOT_HOME / tokenizer_assets_repo
+    required = [
+        cache_dir / "processor_config.json",
+        cache_dir / "preprocessor_config.json",
+        cache_dir / "image_processing_eagle2_5_vl_fast.py",
+    ]
+    if not all(p.exists() for p in required):
+        raise FileNotFoundError(
+            f"[GROOT] Eagle processor cache at '{cache_dir}' is not populated. "
+            "Vendor files are copied during model creation. Create the policy/model first, "
+            "or call ensure_eagle_cache_ready() before building processors."
+        )
+    proc = AutoProcessor.from_pretrained(str(cache_dir), trust_remote_code=True, use_fast=True)
+    proc.tokenizer.padding_side = "left"
+    return proc
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="groot_pack_inputs_v3")
+class GrootPackInputsStep(ProcessorStep):
+    state_horizon: int = 1
+    action_horizon: int = 16
+    max_state_dim: int = 64
+    max_action_dim: int = 32
+    language_key: str = "task"
+    formalize_language: bool = False
+    embodiment_tag: str = "new_embodiment"
+    embodiment_mapping: dict[str, int] = field(
+        default_factory=lambda: {
+            "new_embodiment": 31,  # Match original GR00T EMBODIMENT_TAG_MAPPING
+            "oxe_droid": 17,
+            "agibot_genie1": 26,
+            "gr1": 24,
+            "so100": 2,
+            "unitree_g1": 3,
+        }
+    )
+    # Min-max normalization (SO100-like) applied BEFORE padding
+    normalize_min_max: bool = True
+    stats: dict[str, dict[str, Any]] | None = None
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        obs = transition.get(TransitionKey.OBSERVATION, {}) or {}
+        comp = transition.get(TransitionKey.COMPLEMENTARY_DATA, {}) or {}
+
+        def _align_vec(vec: Any, target_dim: int, *, default: float) -> torch.Tensor:
+            t = torch.as_tensor(vec)
+            t = t.flatten().to(
+                dtype=torch.float32,
+                device=next(
+                    (v.device for v in obs.values() if isinstance(v, torch.Tensor)), torch.device("cpu")
+                ),
+            )
+            d = int(t.shape[-1]) if t.numel() > 0 else 0
+            if d == target_dim:
+                return t
+            if d < target_dim:
+                pad = torch.full((target_dim - d,), default, dtype=t.dtype, device=t.device)
+                return torch.cat([t, pad], dim=0)
+            return t[:target_dim]
+
+        def _min_max_norm(x: torch.Tensor, key: str) -> torch.Tensor:
+            if not self.normalize_min_max:
+                return x
+            if self.stats is None or key not in self.stats:
+                return x
+            stats_k = self.stats[key]
+            last_dim = x.shape[-1]
+            min_v = _align_vec(stats_k.get("min", torch.zeros(last_dim)), last_dim, default=0.0)
+            max_v = _align_vec(stats_k.get("max", torch.ones(last_dim)), last_dim, default=1.0)
+            denom = max_v - min_v
+            mask = denom != 0
+            safe_denom = torch.where(mask, denom, torch.ones_like(denom))
+            mapped = 2 * (x - min_v) / safe_denom - 1
+            return torch.where(mask, mapped, torch.zeros_like(mapped))
+
+        # 1) Video (B, T=1, V, H, W, C) uint8
+        img_keys = sorted([k for k in obs if k.startswith(OBS_IMAGES)])
+        if not img_keys and OBS_IMAGE in obs:
+            img_keys = [OBS_IMAGE]
+        if img_keys:
+            cams = [_to_uint8_np_bhwc(obs[k]) for k in img_keys]
+            video = np.stack(cams, axis=1)  # (B, V, H, W, C)
+            video = np.expand_dims(video, axis=1)  # (B, 1, V, H, W, C)
+            # GR00T validates that video.shape[3] == 3 (channels), so reorder to (B, T, V, C, H, W)
+            video = np.transpose(video, (0, 1, 2, 5, 3, 4))  # (B, 1, V, C, H, W)
+            obs["video"] = video
+            # Drop raw images to avoid confusion downstream
+            for k in img_keys:
+                obs.pop(k, None)
+
+        # 2) Language (string)
+        lang = comp.get(self.language_key)
+        if isinstance(lang, list):
+            lang = lang[0] if len(lang) > 0 else None
+        if not lang:
+            lang = "Perform the task."
+        if self.formalize_language:
+            lang = (lang or "").lower()
+            lang = "".join(ch for ch in lang if ch.isalnum() or ch.isspace())
+        comp["language"] = lang
+
+        # 3) State/state_mask -> (B, 1, max_state_dim)
+        if OBS_STATE in obs:
+            state = obs[OBS_STATE]  # (B, D)
+            if state.dim() != 2:
+                raise ValueError(f"state must be (B, D), got {tuple(state.shape)}")
+            bsz, d = state.shape
+            # Normalize BEFORE padding
+            if self.normalize_min_max:
+                state = _min_max_norm(state, OBS_STATE)
+            state = state.unsqueeze(1)  # (B, 1, D)
+            if d > self.max_state_dim:
+                state = state[:, :, : self.max_state_dim]
+                d = self.max_state_dim
+            elif d < self.max_state_dim:
+                pad = torch.zeros(bsz, 1, self.max_state_dim - d, dtype=state.dtype, device=state.device)
+                state = torch.cat([state, pad], dim=2)
+            state_mask = torch.zeros(bsz, 1, self.max_state_dim, dtype=torch.bool, device=state.device)
+            state_mask[:, :, :d] = True
+            obs["state"] = state
+            obs["state_mask"] = state_mask
+
+        # 4) Action/action_mask -> (B, action_horizon, max_action_dim)
+        action = transition.get(TransitionKey.ACTION)
+        if isinstance(action, torch.Tensor):
+            # Normalize BEFORE temporal expansion/padding
+            if self.normalize_min_max:
+                if action.dim() == 2:
+                    action = _min_max_norm(action, ACTION)
+                elif action.dim() == 3:
+                    b, t, d = action.shape
+                    flat = action.reshape(b * t, d)
+                    flat = _min_max_norm(flat, ACTION)
+                    action = flat.view(b, t, d)
+            if action.dim() == 2:
+                action = action.unsqueeze(1).repeat(1, self.action_horizon, 1)
+            elif action.dim() == 3:
+                b, t, d = action.shape
+                if t < self.action_horizon:
+                    last = action[:, -1:, :]
+                    pad = last.repeat(1, self.action_horizon - t, 1)
+                    action = torch.cat([action, pad], dim=1)
+                elif t > self.action_horizon:
+                    action = action[:, : self.action_horizon, :]
+            else:
+                raise ValueError(f"action must be (B, D) or (B, T, D), got {tuple(action.shape)}")
+
+            b, t, d = action.shape
+            if d > self.max_action_dim:
+                action = action[:, :, : self.max_action_dim]
+                d = self.max_action_dim
+            elif d < self.max_action_dim:
+                pad = torch.zeros(b, t, self.max_action_dim - d, dtype=action.dtype, device=action.device)
+                action = torch.cat([action, pad], dim=2)
+            action_mask = torch.zeros(b, t, self.max_action_dim, dtype=torch.bool, device=action.device)
+            action_mask[:, :, :d] = True
+            transition[TransitionKey.ACTION] = action
+            comp["action_mask"] = action_mask
+
+        # 5) Embodiment id as LongTensor (B,)
+        emb_id = self.embodiment_mapping.get(self.embodiment_tag, 0)
+        # Infer batch size/device from any tensor in obs or action
+        bsz = None
+        device = torch.device("cpu")
+        for v in list(obs.values()) + [transition.get(TransitionKey.ACTION)]:
+            if isinstance(v, torch.Tensor):
+                bsz = v.shape[0]
+                device = v.device
+                break
+        if bsz is None and "video" in obs and isinstance(obs["video"], np.ndarray):
+            bsz = obs["video"].shape[0]
+        if bsz is None:
+            bsz = 1
+        comp["embodiment_id"] = torch.full((bsz,), emb_id, dtype=torch.long, device=device)
+
+        transition[TransitionKey.OBSERVATION] = obs
+        transition[TransitionKey.COMPLEMENTARY_DATA] = comp
+        return transition
+
+    # Pipeline API requirement: declare how features change (we keep it simple)
+    def transform_features(self, features):
+        return features
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns a serializable dictionary of the processor's configuration.
+
+        Excludes 'stats' since they are saved separately via state_dict().
+        """
+        return {
+            "state_horizon": self.state_horizon,
+            "action_horizon": self.action_horizon,
+            "max_state_dim": self.max_state_dim,
+            "max_action_dim": self.max_action_dim,
+            "language_key": self.language_key,
+            "formalize_language": self.formalize_language,
+            "embodiment_tag": self.embodiment_tag,
+            "embodiment_mapping": self.embodiment_mapping,
+            "normalize_min_max": self.normalize_min_max,
+        }
+
+    def state_dict(self) -> dict[str, torch.Tensor]:
+        """
+        Returns normalization statistics as a flat state dictionary.
+
+        This enables saving stats to safetensors files, similar to normalizer_processor.
+        """
+        if not self.stats:
+            return {}
+
+        flat: dict[str, torch.Tensor] = {}
+        for key, sub in self.stats.items():
+            for stat_name, value in sub.items():
+                tensor = torch.as_tensor(value).cpu()
+                flat[f"{key}.{stat_name}"] = tensor
+        return flat
+
+    def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+        """
+        Loads normalization statistics from a flat state dictionary.
+
+        This enables loading stats from safetensors files during from_pretrained.
+        """
+        if not state:
+            return
+
+        reconstructed: dict[str, dict[str, Any]] = {}
+        for flat_key, tensor in state.items():
+            if "." in flat_key:
+                key, stat_name = flat_key.rsplit(".", 1)
+                if key not in reconstructed:
+                    reconstructed[key] = {}
+                reconstructed[key][stat_name] = tensor
+
+        if reconstructed:
+            self.stats = reconstructed
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="groot_eagle_encode_v3")
+class GrootEagleEncodeStep(ProcessorStep):
+    tokenizer_assets_repo: str = DEFAULT_TOKENIZER_ASSETS_REPO
+    _proc: ProcessorMixin | None = field(default=None, init=False, repr=False)
+
+    @property
+    def proc(self) -> ProcessorMixin:
+        if self._proc is None:
+            self._proc = _build_eagle_processor(self.tokenizer_assets_repo)
+        return self._proc
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        obs = transition.get(TransitionKey.OBSERVATION, {}) or {}
+        comp = transition.get(TransitionKey.COMPLEMENTARY_DATA, {}) or {}
+
+        if "video" not in obs:
+            return transition
+
+        video = obs["video"]  # (B, T, V, H, W, C) uint8
+        lang = comp.get("language", "Perform the task.")
+        if isinstance(lang, list):
+            lang = lang[0] if len(lang) > 0 else "Perform the task."
+
+        bsz = video.shape[0]
+        eagle_contents: list[dict[str, Any]] = []
+        for b in range(bsz):
+            vt = video[b]  # (T, V, C, H, W) after reorder
+            if vt.ndim != 5:
+                # Fallback: assume (T, V, H, W, C)
+                t, v, h, w, c = vt.shape
+                flat = rearrange(vt, "t v h w c -> (t v) h w c")
+            else:
+                t, v, c, h, w = vt.shape
+                flat = rearrange(vt, "t v c h w -> (t v) h w c")
+            images = [Image.fromarray(flat[i]) for i in range(t * v)]
+            # Format language as string list representation to match Original GROOT
+            lang_formatted = str([lang])
+            text_content = [{"type": "text", "text": lang_formatted}]
+            image_content = [{"type": "image", "image": img} for img in images]
+            conv = [{"role": "user", "content": image_content + text_content}]
+            text_list = [self.proc.apply_chat_template(conv, tokenize=False, add_generation_prompt=True)]
+            img_inputs, vid_inputs = self.proc.process_vision_info(conv)
+            eagle_contents.append(
+                {
+                    "text_list": text_list,
+                    "image_inputs": img_inputs,
+                    "video_inputs": vid_inputs,
+                }
+            )
+
+        comp["eagle_content"] = eagle_contents
+        transition[TransitionKey.OBSERVATION] = obs
+        transition[TransitionKey.COMPLEMENTARY_DATA] = comp
+        return transition
+
+    # Pipeline API requirement: declare how features change (no schema change here)
+    def transform_features(self, features):
+        return features
+
+
+# Original GR00T-style collate: converts eagle_content -> eagle_* tensors
+def collate(features: list[dict[str, Any]], eagle_processor: ProcessorMixin) -> dict[str, Any]:
+    batch: dict[str, Any] = {}
+    keys = features[0].keys()
+
+    for key in keys:
+        values = [elem[key] for elem in features]
+
+        if key == "eagle_content":
+            text_list: list[str] = []
+            image_inputs: list[Any] = []
+            for v in values:
+                curr_text_list = v["text_list"]
+                curr_image_inputs = v["image_inputs"]
+                text_list += curr_text_list
+                image_inputs += curr_image_inputs
+            eagle_inputs = eagle_processor(
+                text=text_list,
+                images=image_inputs,
+                images_kwargs={"min_dynamic_tiles": 1, "max_dynamic_tiles": 1, "use_thumbnail": False},
+                return_tensors="pt",
+                padding=True,
+            )
+            for k, v in eagle_inputs.items():
+                k = "eagle_" + k
+                batch[k] = v
+        elif key in ("pixel_values", "image_grid_thw", "attention_mask", "input_ids"):
+            # Concat in existing batch dimension.
+            batch[key] = torch.cat(values)
+        else:
+            # state, state_mask, action and action_mask.
+            # Stack to form the batch dimension.
+            batch[key] = torch.from_numpy(np.stack(values))
+    return batch
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="groot_eagle_collate_v3")
+class GrootEagleCollateStep(ProcessorStep):
+    tokenizer_assets_repo: str = DEFAULT_TOKENIZER_ASSETS_REPO
+    _proc: ProcessorMixin | None = field(default=None, init=False, repr=False)
+
+    @property
+    def proc(self) -> ProcessorMixin:
+        if self._proc is None:
+            self._proc = _build_eagle_processor(self.tokenizer_assets_repo)
+        return self._proc
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        obs = transition.get(TransitionKey.OBSERVATION, {}) or {}
+        comp = transition.get(TransitionKey.COMPLEMENTARY_DATA, {}) or {}
+        contents = comp.get("eagle_content")
+        if not contents:
+            return transition
+
+        # Build features list as original API expects: one dict per batch item
+        features = [{"eagle_content": content} for content in contents]
+        batched = collate(features, self.proc)
+
+        # Inject eagle_* tensors and remove the temporary content and raw video to free memory
+        for k, v in batched.items():
+            comp[k] = v
+        comp.pop("eagle_content", None)
+        obs.pop(
+            "video", None
+        )  # The video has been fully encoded into eagle_* tensors, so we don't need the raw video anymore
+        transition[TransitionKey.OBSERVATION] = obs
+        transition[TransitionKey.COMPLEMENTARY_DATA] = comp
+        return transition
+
+    def transform_features(self, features):
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="groot_action_unpack_unnormalize_v1")
+class GrootActionUnpackUnnormalizeStep(ProcessorStep):
+    env_action_dim: int = 0
+    # Apply inverse of min-max normalization if it was used in preprocessor
+    normalize_min_max: bool = True
+    stats: dict[str, dict[str, Any]] | None = None
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        # Expect model outputs to be in TransitionKey.ACTION as (B, T, D_model)
+        action = transition.get(TransitionKey.ACTION)
+        if not isinstance(action, torch.Tensor):
+            return transition
+
+        # Select last timestep and slice to env dimension
+        if action.dim() == 3:
+            action = action[:, -1, :]
+        # Now action is (B, D_model)
+        if self.env_action_dim and action.shape[-1] >= self.env_action_dim:
+            action = action[..., : self.env_action_dim]
+
+        # Inverse min-max normalization mirroring _min_max_norm:
+        # forward: y = 2 * (x - min) / denom - 1, with y=0 when denom==0
+        # inverse: x = (y+1)/2 * denom + min, and when denom==0 -> x = min
+        if self.normalize_min_max and self.stats is not None:
+            stats_k = self.stats.get(ACTION, {})
+            d = action.shape[-1]
+            min_v = torch.as_tensor(
+                stats_k.get("min", torch.zeros(d)), dtype=action.dtype, device=action.device
+            )
+            max_v = torch.as_tensor(
+                stats_k.get("max", torch.ones(d)), dtype=action.dtype, device=action.device
+            )
+            if min_v.numel() != d:
+                min_v = torch.nn.functional.pad(min_v.flatten()[:d], (0, max(0, d - min_v.numel())))
+                min_v = min_v.to(action.device, dtype=action.dtype)
+            if max_v.numel() != d:
+                max_v = torch.nn.functional.pad(max_v.flatten()[:d], (0, max(0, d - max_v.numel())))
+                max_v = max_v.to(action.device, dtype=action.dtype)
+            denom = max_v - min_v
+            mask = denom != 0
+            safe_denom = torch.where(mask, denom, torch.ones_like(denom))
+            inv = (action + 1.0) * 0.5 * safe_denom + min_v
+            action = torch.where(mask, inv, min_v)
+
+        transition[TransitionKey.ACTION] = action
+        return transition
+
+    def transform_features(self, features):
+        return features
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns a serializable dictionary of the processor's configuration.
+
+        Excludes 'stats' since they are saved separately via state_dict().
+        """
+        return {
+            "env_action_dim": self.env_action_dim,
+            "normalize_min_max": self.normalize_min_max,
+        }
+
+    def state_dict(self) -> dict[str, torch.Tensor]:
+        """
+        Returns normalization statistics as a flat state dictionary.
+
+        This enables saving stats to safetensors files, similar to normalizer_processor.
+        """
+        if not self.stats:
+            return {}
+
+        flat: dict[str, torch.Tensor] = {}
+        for key, sub in self.stats.items():
+            for stat_name, value in sub.items():
+                tensor = torch.as_tensor(value).cpu()
+                flat[f"{key}.{stat_name}"] = tensor
+        return flat
+
+    def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+        """
+        Loads normalization statistics from a flat state dictionary.
+
+        This enables loading stats from safetensors files during from_pretrained.
+        """
+        if not state:
+            return
+
+        reconstructed: dict[str, dict[str, Any]] = {}
+        for flat_key, tensor in state.items():
+            if "." in flat_key:
+                key, stat_name = flat_key.rsplit(".", 1)
+                if key not in reconstructed:
+                    reconstructed[key] = {}
+                reconstructed[key][stat_name] = tensor
+
+        if reconstructed:
+            self.stats = reconstructed
diff --git a/lerobot/src/lerobot/policies/groot/utils.py b/lerobot/src/lerobot/policies/groot/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..b5a3d0e6681500505736e765694ddf5efd69b7b9
--- /dev/null
+++ b/lerobot/src/lerobot/policies/groot/utils.py
@@ -0,0 +1,47 @@
+from pathlib import Path
+from shutil import copytree
+
+from huggingface_hub import hf_hub_download
+
+
+def ensure_eagle_cache_ready(vendor_dir: Path, cache_dir: Path, assets_repo: str) -> None:
+    """Populate the Eagle processor directory in cache and ensure tokenizer assets exist.
+
+    - Copies the vendored Eagle files into cache_dir (overwriting when needed).
+    - Downloads vocab.json and merges.txt into the same cache_dir if missing.
+    """
+    cache_dir = Path(cache_dir)
+    vendor_dir = Path(vendor_dir)
+
+    try:
+        # Populate/refresh cache with vendor files to ensure a complete processor directory
+        print(f"[GROOT] Copying vendor Eagle files to cache: {vendor_dir} -> {cache_dir}")
+        copytree(vendor_dir, cache_dir, dirs_exist_ok=True)
+    except Exception as exc:  # nosec: B110
+        print(f"[GROOT] Warning: Failed to copy vendor Eagle files to cache: {exc}")
+
+    required_assets = [
+        "vocab.json",
+        "merges.txt",
+        "added_tokens.json",
+        "chat_template.json",
+        "special_tokens_map.json",
+        "config.json",
+        "generation_config.json",
+        "preprocessor_config.json",
+        "processor_config.json",
+        "tokenizer_config.json",
+    ]
+
+    print(f"[GROOT] Assets repo: {assets_repo} \n Cache dir: {cache_dir}")
+
+    for fname in required_assets:
+        dst = cache_dir / fname
+        if not dst.exists():
+            print(f"[GROOT] Fetching {fname}")
+            hf_hub_download(
+                repo_id=assets_repo,
+                filename=fname,
+                repo_type="model",
+                local_dir=str(cache_dir),
+            )
diff --git a/lerobot/src/lerobot/policies/pi0/README.md b/lerobot/src/lerobot/policies/pi0/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..65b331e51120728232c2e1552ca4b6019ea0916c
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi0/README.md
@@ -0,0 +1,49 @@
+# π₀ (pi0)
+
+This repository contains the Hugging Face port of **π₀**, adapted from [OpenPI](https://github.com/Physical-Intelligence/openpi) by the Physical Intelligence.
+It is designed as a **Vision-Language-Action model for general robot control**.
+
+---
+
+## Model Overview
+
+| Feature              | π₀                                                     | π₀.₅                                      |
+| -------------------- | ------------------------------------------------------ | ----------------------------------------- |
+| Time Conditioning    | Concatenates time with actions via `action_time_mlp_*` | Uses `time_mlp_*` for AdaRMS conditioning |
+| AdaRMS               | Not used                                               | Used in action expert                     |
+| Tokenizer Length     | 48 tokens                                              | 200 tokens                                |
+| Discrete State Input | False (Uses `state_proj` layer)                        | True                                      |
+| Parameter Count      | Higher (includes state embedding)                      | Lower (no state embedding)                |
+
+---
+
+## Citation
+
+If you use this work, please cite both **OpenPI** and the π₀ paper:
+
+```bibtex
+@misc{openpi2024,
+  author       = {Physical Intelligence Lab},
+  title        = {OpenPI: PyTorch Implementation of π0 and π0.5 Policies},
+  year         = {2024},
+  publisher    = {GitHub},
+  howpublished = {\url{https://github.com/Physical-Intelligence/openpi}},
+  license      = {Apache-2.0}
+}
+
+@misc{black2024pi0visionlanguageactionflowmodel,
+  title        = {π₀: A Vision-Language-Action Flow Model for General Robot Control},
+  author       = {Kevin Black and Noah Brown and Danny Driess and Adnan Esmail and Michael Equi and Chelsea Finn and Niccolo Fusai and Lachy Groom and Karol Hausman and Brian Ichter and Szymon Jakubczak and Tim Jones and Liyiming Ke and Sergey Levine and Adrian Li-Bell and Mohith Mothukuri and Suraj Nair and Karl Pertsch and Lucy Xiaoyang Shi and James Tanner and Quan Vuong and Anna Walling and Haohuan Wang and Ury Zhilinsky},
+  year         = {2024},
+  eprint       = {2410.24164},
+  archivePrefix= {arXiv},
+  primaryClass = {cs.LG},
+  url          = {https://arxiv.org/abs/2410.24164},
+}
+```
+
+---
+
+## License
+
+This port follows the **Apache 2.0 License**, consistent with the original [OpenPI repository](https://github.com/Physical-Intelligence/openpi).
diff --git a/lerobot/src/lerobot/policies/pi0/__init__.py b/lerobot/src/lerobot/policies/pi0/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..ea3095b4ec9d6d95d762ed47ad1f69047519a350
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi0/__init__.py
@@ -0,0 +1,21 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configuration_pi0 import PI0Config
+from .modeling_pi0 import PI0Policy
+from .processor_pi0 import make_pi0_pre_post_processors
+
+__all__ = ["PI0Config", "PI0Policy", "make_pi0_pre_post_processors"]
diff --git a/lerobot/src/lerobot/policies/pi0/configuration_pi0.py b/lerobot/src/lerobot/policies/pi0/configuration_pi0.py
new file mode 100644
index 0000000000000000000000000000000000000000..be9b4530f060833b88f4fd62aa8b7a694fdf992f
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi0/configuration_pi0.py
@@ -0,0 +1,168 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.optim.optimizers import AdamWConfig
+from lerobot.optim.schedulers import CosineDecayWithWarmupSchedulerConfig
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+from lerobot.utils.constants import ACTION, OBS_IMAGES, OBS_STATE
+
+DEFAULT_IMAGE_SIZE = 224
+
+
+@PreTrainedConfig.register_subclass("pi0")
+@dataclass
+class PI0Config(PreTrainedConfig):
+    paligemma_variant: str = "gemma_2b"
+    action_expert_variant: str = "gemma_300m"
+    dtype: str = "float32"  # Options: "bfloat16", "float32"
+
+    n_obs_steps: int = 1
+    chunk_size: int = 50  # Number of action steps to predict, in openpi called "action_horizon"
+    n_action_steps: int = 50  # Number of action steps to execute
+
+    # Shorter state and action vectors will be padded to these dimensions
+    max_state_dim: int = 32
+    max_action_dim: int = 32
+
+    # Flow matching parameters: see openpi `PI0Pytorch`
+    num_inference_steps: int = 10  # Number of denoising steps during inference
+    time_sampling_beta_alpha: float = 1.5
+    time_sampling_beta_beta: float = 1.0
+    time_sampling_scale: float = 0.999
+    time_sampling_offset: float = 0.001
+    min_period: float = 4e-3
+    max_period: float = 4.0
+
+    # Real-Time Chunking (RTC) configuration
+    rtc_config: RTCConfig | None = None
+
+    image_resolution: tuple[int, int] = (
+        DEFAULT_IMAGE_SIZE,
+        DEFAULT_IMAGE_SIZE,
+    )  # see openpi `preprocessing_pytorch.py`
+
+    # Add empty images. Used to add empty cameras when no image features are present.
+    empty_cameras: int = 0
+
+    # Normalization
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.MEAN_STD,
+            "ACTION": NormalizationMode.MEAN_STD,
+        }
+    )
+
+    # Training settings
+    gradient_checkpointing: bool = False  # Enable gradient checkpointing for memory optimization
+    compile_model: bool = False  # Whether to use torch.compile for model optimization
+    compile_mode: str = "max-autotune"  # Torch compile mode
+    device: str | None = None  # Device to use for the model (None = auto-detect)
+
+    # Finetuning settings
+    freeze_vision_encoder: bool = False  # Freeze only the vision encoder
+    train_expert_only: bool = False  # Freeze entire VLM, train only action expert and projections
+
+    # Optimizer settings: see openpi `AdamW``
+    optimizer_lr: float = 2.5e-5  # see openpi `CosineDecaySchedule: peak_lr`
+    optimizer_betas: tuple[float, float] = (0.9, 0.95)
+    optimizer_eps: float = 1e-8
+    optimizer_weight_decay: float = 0.01
+    optimizer_grad_clip_norm: float = 1.0
+
+    # Scheduler settings: see openpi `CosineDecaySchedule`
+    # Note: These will auto-scale if --steps < scheduler_decay_steps
+    # For example, --steps=3000 will scale warmup to 100 and decay to 3000
+    scheduler_warmup_steps: int = 1_000
+    scheduler_decay_steps: int = 30_000
+    scheduler_decay_lr: float = 2.5e-6
+
+    tokenizer_max_length: int = 48  # see openpi `__post_init__`
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        # Validate configuration
+        if self.n_action_steps > self.chunk_size:
+            raise ValueError(
+                f"n_action_steps ({self.n_action_steps}) cannot be greater than chunk_size ({self.chunk_size})"
+            )
+
+        if self.paligemma_variant not in ["gemma_300m", "gemma_2b"]:
+            raise ValueError(f"Invalid paligemma_variant: {self.paligemma_variant}")
+
+        if self.action_expert_variant not in ["gemma_300m", "gemma_2b"]:
+            raise ValueError(f"Invalid action_expert_variant: {self.action_expert_variant}")
+
+        if self.dtype not in ["bfloat16", "float32"]:
+            raise ValueError(f"Invalid dtype: {self.dtype}")
+
+    def validate_features(self) -> None:
+        """Validate and set up input/output features."""
+        for i in range(self.empty_cameras):
+            key = f"{OBS_IMAGES}.empty_camera_{i}"
+            empty_camera = PolicyFeature(
+                type=FeatureType.VISUAL,
+                shape=(3, *self.image_resolution),  # Use configured image resolution
+            )
+            self.input_features[key] = empty_camera
+
+        if OBS_STATE not in self.input_features:
+            state_feature = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(self.max_state_dim,),  # Padded to max_state_dim
+            )
+            self.input_features[OBS_STATE] = state_feature
+
+        if ACTION not in self.output_features:
+            action_feature = PolicyFeature(
+                type=FeatureType.ACTION,
+                shape=(self.max_action_dim,),  # Padded to max_action_dim
+            )
+            self.output_features[ACTION] = action_feature
+
+    def get_optimizer_preset(self) -> AdamWConfig:
+        return AdamWConfig(
+            lr=self.optimizer_lr,
+            betas=self.optimizer_betas,
+            eps=self.optimizer_eps,
+            weight_decay=self.optimizer_weight_decay,
+            grad_clip_norm=self.optimizer_grad_clip_norm,
+        )
+
+    def get_scheduler_preset(self):
+        return CosineDecayWithWarmupSchedulerConfig(
+            peak_lr=self.optimizer_lr,
+            decay_lr=self.scheduler_decay_lr,
+            num_warmup_steps=self.scheduler_warmup_steps,
+            num_decay_steps=self.scheduler_decay_steps,
+        )
+
+    @property
+    def observation_delta_indices(self) -> None:
+        return None
+
+    @property
+    def action_delta_indices(self) -> list:
+        return list(range(self.chunk_size))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/pi0/modeling_pi0.py b/lerobot/src/lerobot/policies/pi0/modeling_pi0.py
new file mode 100644
index 0000000000000000000000000000000000000000..aebf329645734d645c8cff14b337fca2fae6a536
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi0/modeling_pi0.py
@@ -0,0 +1,1324 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import builtins
+import copy
+import logging
+import math
+from collections import deque
+from pathlib import Path
+from typing import TYPE_CHECKING, Literal, TypedDict, Unpack
+
+import torch
+import torch.nn.functional as F  # noqa: N812
+from torch import Tensor, nn
+
+from lerobot.utils.import_utils import _transformers_available
+
+# Conditional import for type checking and lazy loading
+if TYPE_CHECKING or _transformers_available:
+    from transformers.models.auto import CONFIG_MAPPING
+    from transformers.models.gemma import modeling_gemma
+
+    from lerobot.policies.pi_gemma import (
+        PaliGemmaForConditionalGenerationWithPiGemma,
+        PiGemmaForCausalLM,
+        _gated_residual,
+        layernorm_forward,
+    )
+else:
+    CONFIG_MAPPING = None
+    modeling_gemma = None
+    PiGemmaForCausalLM = None
+    _gated_residual = None
+    layernorm_forward = None
+    PaliGemmaForConditionalGenerationWithPiGemma = None
+
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.policies.pi0.configuration_pi0 import DEFAULT_IMAGE_SIZE, PI0Config
+from lerobot.policies.pretrained import PreTrainedPolicy, T
+from lerobot.policies.rtc.modeling_rtc import RTCProcessor
+from lerobot.utils.constants import (
+    ACTION,
+    OBS_LANGUAGE_ATTENTION_MASK,
+    OBS_LANGUAGE_TOKENS,
+    OBS_STATE,
+    OPENPI_ATTENTION_MASK_VALUE,
+)
+
+
+class ActionSelectKwargs(TypedDict, total=False):
+    inference_delay: int | None
+    prev_chunk_left_over: Tensor | None
+    execution_horizon: int | None
+
+
+def get_safe_dtype(target_dtype, device_type):
+    """Get a safe dtype for the given device type."""
+    if device_type == "mps" and target_dtype == torch.float64:
+        return torch.float32
+    if device_type == "cpu":
+        # CPU doesn't support bfloat16, use float32 instead
+        if target_dtype == torch.bfloat16:
+            return torch.float32
+        if target_dtype == torch.float64:
+            return torch.float64
+    return target_dtype
+
+
+def create_sinusoidal_pos_embedding(  # see openpi `create_sinusoidal_pos_embedding` (exact copy)
+    time: torch.Tensor, dimension: int, min_period: float, max_period: float, device="cpu"
+) -> Tensor:
+    """Computes sine-cosine positional embedding vectors for scalar positions."""
+    if dimension % 2 != 0:
+        raise ValueError(f"dimension ({dimension}) must be divisible by 2")
+
+    if time.ndim != 1:
+        raise ValueError("The time tensor is expected to be of shape `(batch_size, )`.")
+
+    dtype = get_safe_dtype(torch.float64, device.type)
+    fraction = torch.linspace(0.0, 1.0, dimension // 2, dtype=dtype, device=device)
+    period = min_period * (max_period / min_period) ** fraction
+
+    # Compute the outer product
+    scaling_factor = 1.0 / period * 2 * math.pi
+    sin_input = scaling_factor[None, :] * time[:, None]
+    return torch.cat([torch.sin(sin_input), torch.cos(sin_input)], dim=1)
+
+
+def sample_beta(alpha, beta, bsize, device):  # see openpi `sample_beta` (exact copy)
+    # Beta sampling uses _sample_dirichlet which isn't implemented for MPS, so sample on CPU
+    alpha_t = torch.tensor(alpha, dtype=torch.float32)
+    beta_t = torch.tensor(beta, dtype=torch.float32)
+    dist = torch.distributions.Beta(alpha_t, beta_t)
+    return dist.sample((bsize,)).to(device)
+
+
+def make_att_2d_masks(pad_masks, att_masks):  # see openpi `make_att_2d_masks` (exact copy)
+    """Copied from big_vision.
+
+    Tokens can attend to valid inputs tokens which have a cumulative mask_ar
+    smaller or equal to theirs. This way `mask_ar` int[B, N] can be used to
+    setup several types of attention, for example:
+
+      [[1 1 1 1 1 1]]: pure causal attention.
+
+      [[0 0 0 1 1 1]]: prefix-lm attention. The first 3 tokens can attend between
+          themselves and the last 3 tokens have a causal attention. The first
+          entry could also be a 1 without changing behaviour.
+
+      [[1 0 1 0 1 0 0 1 0 0]]: causal attention between 4 blocks. Tokens of a
+          block can attend all previous blocks and all tokens on the same block.
+
+    Args:
+      input_mask: bool[B, N] true if its part of the input, false if padding.
+      mask_ar: int32[B, N] mask that's 1 where previous tokens cannot depend on
+        it and 0 where it shares the same attention mask as the previous token.
+    """
+    if att_masks.ndim != 2:
+        raise ValueError(att_masks.ndim)
+    if pad_masks.ndim != 2:
+        raise ValueError(pad_masks.ndim)
+
+    cumsum = torch.cumsum(att_masks, dim=1)
+    att_2d_masks = cumsum[:, None, :] <= cumsum[:, :, None]
+    pad_2d_masks = pad_masks[:, None, :] * pad_masks[:, :, None]
+    return att_2d_masks & pad_2d_masks
+
+
+def pad_vector(vector, new_dim):
+    """Pad the last dimension of a vector to new_dim with zeros.
+
+    Can be (batch_size x sequence_length x features_dimension)
+    or (batch_size x features_dimension)
+    """
+    if vector.shape[-1] >= new_dim:
+        return vector
+    return F.pad(vector, (0, new_dim - vector.shape[-1]))
+
+
+def resize_with_pad_torch(  # see openpi `resize_with_pad_torch` (exact copy)
+    images: torch.Tensor,
+    height: int,
+    width: int,
+    mode: str = "bilinear",
+) -> torch.Tensor:
+    """PyTorch version of resize_with_pad. Resizes an image to a target height and width without distortion
+    by padding with black. If the image is float32, it must be in the range [-1, 1].
+
+    Args:
+        images: Tensor of shape [*b, h, w, c] or [*b, c, h, w]
+        height: Target height
+        width: Target width
+        mode: Interpolation mode ('bilinear', 'nearest', etc.)
+
+    Returns:
+        Resized and padded tensor with same shape format as input
+    """
+    # Check if input is in channels-last format [*b, h, w, c] or channels-first [*b, c, h, w]
+    if images.shape[-1] <= 4:  # Assume channels-last format
+        channels_last = True
+        if images.dim() == 3:
+            images = images.unsqueeze(0)  # Add batch dimension
+        images = images.permute(0, 3, 1, 2)  # [b, h, w, c] -> [b, c, h, w]
+    else:
+        channels_last = False
+        if images.dim() == 3:
+            images = images.unsqueeze(0)  # Add batch dimension
+
+    batch_size, channels, cur_height, cur_width = images.shape
+
+    # Calculate resize ratio
+    ratio = max(cur_width / width, cur_height / height)
+    resized_height = int(cur_height / ratio)
+    resized_width = int(cur_width / ratio)
+
+    # Resize
+    resized_images = F.interpolate(
+        images,
+        size=(resized_height, resized_width),
+        mode=mode,
+        align_corners=False if mode == "bilinear" else None,
+    )
+
+    # Handle dtype-specific clipping
+    if images.dtype == torch.uint8:
+        resized_images = torch.round(resized_images).clamp(0, 255).to(torch.uint8)
+    elif images.dtype == torch.float32:
+        resized_images = resized_images.clamp(0.0, 1.0)
+    else:
+        raise ValueError(f"Unsupported image dtype: {images.dtype}")
+
+    # Calculate padding
+    pad_h0, remainder_h = divmod(height - resized_height, 2)
+    pad_h1 = pad_h0 + remainder_h
+    pad_w0, remainder_w = divmod(width - resized_width, 2)
+    pad_w1 = pad_w0 + remainder_w
+
+    # Pad
+    constant_value = 0 if images.dtype == torch.uint8 else 0.0
+    padded_images = F.pad(
+        resized_images,
+        (pad_w0, pad_w1, pad_h0, pad_h1),  # left, right, top, bottom
+        mode="constant",
+        value=constant_value,
+    )
+
+    # Convert back to original format if needed
+    if channels_last:
+        padded_images = padded_images.permute(0, 2, 3, 1)  # [b, c, h, w] -> [b, h, w, c]
+
+    return padded_images
+
+
+# Define the complete layer computation function for gradient checkpointing
+def compute_layer_complete(
+    layer_idx, inputs_embeds, attention_mask, position_ids, adarms_cond, paligemma, gemma_expert
+):
+    models = [paligemma.model.language_model, gemma_expert.model]
+    query_states = []
+    key_states = []
+    value_states = []
+    gates = []
+    for i, hidden_states in enumerate(inputs_embeds):
+        layer = models[i].layers[layer_idx]
+        hidden_states, gate = layernorm_forward(layer.input_layernorm, hidden_states, adarms_cond[i])
+        gates.append(gate)
+        input_shape = hidden_states.shape[:-1]
+        hidden_shape = (*input_shape, -1, layer.self_attn.head_dim)
+        query_state = layer.self_attn.q_proj(hidden_states).view(hidden_shape).transpose(1, 2)
+        key_state = layer.self_attn.k_proj(hidden_states).view(hidden_shape).transpose(1, 2)
+        value_state = layer.self_attn.v_proj(hidden_states).view(hidden_shape).transpose(1, 2)
+        query_states.append(query_state)
+        key_states.append(key_state)
+        value_states.append(value_state)
+    # Concatenate and process attention
+    query_states = torch.cat(query_states, dim=2)
+    key_states = torch.cat(key_states, dim=2)
+    value_states = torch.cat(value_states, dim=2)
+    dummy_tensor = torch.zeros(
+        query_states.shape[0],
+        query_states.shape[2],
+        query_states.shape[-1],
+        device=query_states.device,
+        dtype=query_states.dtype,
+    )
+    cos, sin = paligemma.model.language_model.rotary_emb(dummy_tensor, position_ids)
+    query_states, key_states = modeling_gemma.apply_rotary_pos_emb(
+        query_states, key_states, cos, sin, unsqueeze_dim=1
+    )
+    batch_size = query_states.shape[0]
+    scaling = paligemma.model.language_model.layers[layer_idx].self_attn.scaling
+    # Attention computation
+    att_output, _ = modeling_gemma.eager_attention_forward(
+        paligemma.model.language_model.layers[layer_idx].self_attn,
+        query_states,
+        key_states,
+        value_states,
+        attention_mask,
+        scaling,
+    )
+    # Get head_dim from the current layer, not from the model
+    head_dim = paligemma.model.language_model.layers[layer_idx].self_attn.head_dim
+    att_output = att_output.reshape(batch_size, -1, 1 * 8 * head_dim)
+    # Process layer outputs
+    outputs_embeds = []
+    start_pos = 0
+    for i, hidden_states in enumerate(inputs_embeds):
+        layer = models[i].layers[layer_idx]
+        end_pos = start_pos + hidden_states.shape[1]
+        if att_output.dtype != layer.self_attn.o_proj.weight.dtype:
+            att_output = att_output.to(layer.self_attn.o_proj.weight.dtype)
+        out_emb = layer.self_attn.o_proj(att_output[:, start_pos:end_pos])
+        # first residual
+        out_emb = _gated_residual(hidden_states, out_emb, gates[i])
+        after_first_residual = out_emb.clone()
+        out_emb, gate = layernorm_forward(layer.post_attention_layernorm, out_emb, adarms_cond[i])
+        # Convert to bfloat16 if the next layer (mlp) uses bfloat16
+        if layer.mlp.up_proj.weight.dtype == torch.bfloat16:
+            out_emb = out_emb.to(dtype=torch.bfloat16)
+        out_emb = layer.mlp(out_emb)
+        # second residual
+        out_emb = _gated_residual(after_first_residual, out_emb, gate)
+        outputs_embeds.append(out_emb)
+        start_pos = end_pos
+    return outputs_embeds
+
+
+class GemmaConfig:  # see openpi `gemma.py: Config`
+    """Configuration for Gemma model variants."""
+
+    def __init__(self, width, depth, mlp_dim, num_heads, num_kv_heads, head_dim):
+        self.width = width
+        self.depth = depth
+        self.mlp_dim = mlp_dim
+        self.num_heads = num_heads
+        self.num_kv_heads = num_kv_heads
+        self.head_dim = head_dim
+
+
+def get_gemma_config(variant: str) -> GemmaConfig:  # see openpi `gemma.py: get_config`
+    """Returns config for specified gemma variant."""
+    if variant == "gemma_300m":
+        return GemmaConfig(
+            width=1024,
+            depth=18,
+            mlp_dim=4096,
+            num_heads=8,
+            num_kv_heads=1,
+            head_dim=256,
+        )
+    elif variant == "gemma_2b":
+        return GemmaConfig(
+            width=2048,
+            depth=18,
+            mlp_dim=16_384,
+            num_heads=8,
+            num_kv_heads=1,
+            head_dim=256,
+        )
+    else:
+        raise ValueError(f"Unknown variant: {variant}")
+
+
+class PaliGemmaWithExpertModel(
+    nn.Module
+):  # see openpi `gemma_pytorch.py: PaliGemmaWithExpertModel` this class is almost a exact copy of PaliGemmaWithExpertModel in openpi
+    """PaliGemma model with action expert for PI0."""
+
+    def __init__(
+        self,
+        vlm_config,
+        action_expert_config,
+        use_adarms=None,
+        precision: Literal["bfloat16", "float32"] = "bfloat16",
+        image_size: int = DEFAULT_IMAGE_SIZE,
+        freeze_vision_encoder: bool = False,
+        train_expert_only: bool = False,
+    ):
+        if use_adarms is None:
+            use_adarms = [False, False]
+        super().__init__()
+        self.freeze_vision_encoder = freeze_vision_encoder
+        self.train_expert_only = train_expert_only
+
+        vlm_config_hf = CONFIG_MAPPING["paligemma"]()
+        vlm_config_hf._vocab_size = 257152  # noqa: SLF001
+        vlm_config_hf.image_token_index = 257152
+        vlm_config_hf.text_config.hidden_size = vlm_config.width
+        vlm_config_hf.text_config.intermediate_size = vlm_config.mlp_dim
+        vlm_config_hf.text_config.num_attention_heads = vlm_config.num_heads
+        vlm_config_hf.text_config.head_dim = vlm_config.head_dim
+        vlm_config_hf.text_config.num_hidden_layers = vlm_config.depth
+        vlm_config_hf.text_config.num_key_value_heads = vlm_config.num_kv_heads
+        vlm_config_hf.text_config.hidden_activation = "gelu_pytorch_tanh"
+        vlm_config_hf.text_config.dtype = "float32"
+        vlm_config_hf.text_config.vocab_size = 257152
+        vlm_config_hf.text_config.use_adarms = use_adarms[0]
+        vlm_config_hf.text_config.adarms_cond_dim = vlm_config.width if use_adarms[0] else None
+        vlm_config_hf.vision_config.image_size = image_size
+        vlm_config_hf.vision_config.intermediate_size = 4304
+        vlm_config_hf.vision_config.projection_dim = 2048
+        vlm_config_hf.vision_config.projector_hidden_act = "gelu_fast"
+        vlm_config_hf.vision_config.dtype = "float32"
+
+        action_expert_config_hf = CONFIG_MAPPING["gemma"](
+            head_dim=action_expert_config.head_dim,
+            hidden_size=action_expert_config.width,
+            intermediate_size=action_expert_config.mlp_dim,
+            num_attention_heads=action_expert_config.num_heads,
+            num_hidden_layers=action_expert_config.depth,
+            num_key_value_heads=action_expert_config.num_kv_heads,
+            vocab_size=257152,
+            hidden_activation="gelu_pytorch_tanh",
+            dtype="float32",
+            use_adarms=use_adarms[1],
+            adarms_cond_dim=action_expert_config.width if use_adarms[1] else None,
+        )
+
+        self.paligemma = PaliGemmaForConditionalGenerationWithPiGemma(config=vlm_config_hf)
+        self.gemma_expert = PiGemmaForCausalLM(config=action_expert_config_hf)
+        self.gemma_expert.model.embed_tokens = None
+
+        self.to_bfloat16_for_selected_params(precision)
+        self._set_requires_grad()
+
+    def to_bfloat16_for_selected_params(self, precision: Literal["bfloat16", "float32"] = "bfloat16"):
+        if precision == "bfloat16":
+            self.to(dtype=torch.bfloat16)
+        elif precision == "float32":
+            self.to(dtype=torch.float32)
+            return
+        else:
+            raise ValueError(f"Invalid precision: {precision}")
+
+        # Keep full vision path in float32 so we never toggle (toggle causes optimizer
+        # "same dtype" error). Align with PI05.
+        params_to_keep_float32 = [
+            "vision_tower",
+            "multi_modal_projector",
+            "input_layernorm",
+            "post_attention_layernorm",
+            "model.norm",
+        ]
+
+        for name, param in self.named_parameters():
+            if any(selector in name for selector in params_to_keep_float32):
+                param.data = param.data.to(dtype=torch.float32)
+
+    def _set_requires_grad(self):
+        if self.freeze_vision_encoder:
+            self.paligemma.model.vision_tower.eval()
+            for param in self.paligemma.model.vision_tower.parameters():
+                param.requires_grad = False
+        if self.train_expert_only:
+            self.paligemma.eval()
+            for param in self.paligemma.parameters():
+                param.requires_grad = False
+
+    def train(self, mode: bool = True):
+        super().train(mode)
+        if self.freeze_vision_encoder:
+            self.paligemma.model.vision_tower.eval()
+        if self.train_expert_only:
+            self.paligemma.eval()
+
+    def embed_image(self, image: torch.Tensor):
+        # Vision tower and multi_modal_projector are kept in float32 (params_to_keep_float32). Align with PI05.
+        out_dtype = image.dtype
+        if image.dtype != torch.float32:
+            image = image.to(torch.float32)
+        image_outputs = self.paligemma.model.get_image_features(image)
+        features = image_outputs.pooler_output * self.paligemma.config.text_config.hidden_size**0.5
+        if features.dtype != out_dtype:
+            features = features.to(out_dtype)
+        return features
+
+    def embed_language_tokens(self, tokens: torch.Tensor):
+        return self.paligemma.model.language_model.embed_tokens(tokens)
+
+    def forward(
+        self,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: list[torch.FloatTensor] | None = None,
+        use_cache: bool | None = None,
+        adarms_cond: list[torch.Tensor] | None = None,
+    ):
+        if adarms_cond is None:
+            adarms_cond = [None, None]
+        if inputs_embeds[1] is None:
+            prefix_output = self.paligemma.model.language_model.forward(
+                inputs_embeds=inputs_embeds[0],
+                attention_mask=attention_mask,
+                position_ids=position_ids,
+                past_key_values=past_key_values,
+                use_cache=use_cache,
+                adarms_cond=adarms_cond[0] if adarms_cond is not None else None,
+            )
+            prefix_past_key_values = prefix_output.past_key_values
+            prefix_output = prefix_output.last_hidden_state
+            suffix_output = None
+        elif inputs_embeds[0] is None:
+            suffix_output = self.gemma_expert.model.forward(
+                inputs_embeds=inputs_embeds[1],
+                attention_mask=attention_mask,
+                position_ids=position_ids,
+                past_key_values=past_key_values,
+                use_cache=use_cache,
+                adarms_cond=adarms_cond[1] if adarms_cond is not None else None,
+            )
+            suffix_output = suffix_output.last_hidden_state
+            prefix_output = None
+            prefix_past_key_values = None
+        else:
+            models = [self.paligemma.model.language_model, self.gemma_expert.model]
+            num_layers = self.paligemma.config.text_config.num_hidden_layers
+
+            # Check if gradient checkpointing is enabled for any of the models
+            use_gradient_checkpointing = (
+                hasattr(self.gemma_expert.model, "gradient_checkpointing")
+                and self.gemma_expert.model.gradient_checkpointing
+                and self.training
+            ) or (hasattr(self, "gradient_checkpointing") and self.gradient_checkpointing and self.training)
+
+            # Process all layers with gradient checkpointing if enabled
+            for layer_idx in range(num_layers):
+                if use_gradient_checkpointing:
+                    inputs_embeds = torch.utils.checkpoint.checkpoint(
+                        compute_layer_complete,
+                        layer_idx,
+                        inputs_embeds,
+                        attention_mask,
+                        position_ids,
+                        adarms_cond,
+                        use_reentrant=False,
+                        preserve_rng_state=False,
+                        paligemma=self.paligemma,
+                        gemma_expert=self.gemma_expert,
+                    )
+                else:
+                    inputs_embeds = compute_layer_complete(
+                        layer_idx,
+                        inputs_embeds,
+                        attention_mask,
+                        position_ids,
+                        adarms_cond,
+                        paligemma=self.paligemma,
+                        gemma_expert=self.gemma_expert,
+                    )
+
+            # final norm
+            def compute_final_norms(inputs_embeds, adarms_cond):
+                outputs_embeds = []
+                for i, hidden_states in enumerate(inputs_embeds):
+                    out_emb, _ = layernorm_forward(models[i].norm, hidden_states, adarms_cond[i])
+                    outputs_embeds.append(out_emb)
+                return outputs_embeds
+
+            # Apply gradient checkpointing to final norm if enabled
+            if use_gradient_checkpointing:
+                outputs_embeds = torch.utils.checkpoint.checkpoint(
+                    compute_final_norms,
+                    inputs_embeds,
+                    adarms_cond,
+                    use_reentrant=False,
+                    preserve_rng_state=False,
+                )
+            else:
+                outputs_embeds = compute_final_norms(inputs_embeds, adarms_cond)
+
+            prefix_output = outputs_embeds[0]
+            suffix_output = outputs_embeds[1]
+            prefix_past_key_values = None
+
+        return [prefix_output, suffix_output], prefix_past_key_values
+
+
+class PI0Pytorch(nn.Module):  # see openpi `PI0Pytorch`
+    """Core PI0 PyTorch model."""
+
+    def __init__(self, config: PI0Config, rtc_processor: RTCProcessor | None = None):
+        super().__init__()
+        self.config = config
+        self.rtc_processor = rtc_processor
+
+        paligemma_config = get_gemma_config(config.paligemma_variant)
+        action_expert_config = get_gemma_config(config.action_expert_variant)
+
+        if config.image_resolution[0] != config.image_resolution[1]:
+            raise ValueError(
+                f"PaliGemma expects square image resolution, invalid resolution: {config.image_resolution}"
+            )
+
+        self.paligemma_with_expert = PaliGemmaWithExpertModel(
+            paligemma_config,
+            action_expert_config,
+            use_adarms=[False, False],
+            precision=config.dtype,
+            image_size=config.image_resolution[0],
+            freeze_vision_encoder=config.freeze_vision_encoder,
+            train_expert_only=config.train_expert_only,
+        )
+
+        self.action_in_proj = nn.Linear(config.max_action_dim, action_expert_config.width)
+        self.action_out_proj = nn.Linear(action_expert_config.width, config.max_action_dim)
+
+        self.state_proj = nn.Linear(config.max_state_dim, action_expert_config.width)
+        self.action_time_mlp_in = nn.Linear(2 * action_expert_config.width, action_expert_config.width)
+        self.action_time_mlp_out = nn.Linear(action_expert_config.width, action_expert_config.width)
+
+        # Initialize gradient checkpointing flag
+        self.gradient_checkpointing_enabled = False
+
+        # Compile model if requested
+        if config.compile_model:
+            torch.set_float32_matmul_precision("high")
+            self.sample_actions = torch.compile(self.sample_actions, mode=config.compile_mode)
+            # Also compile the main forward pass used during training
+            self.forward = torch.compile(self.forward, mode=config.compile_mode)
+
+    def gradient_checkpointing_enable(self):
+        """Enable gradient checkpointing for memory optimization."""
+        self.gradient_checkpointing_enabled = True
+        self.paligemma_with_expert.paligemma.model.language_model.gradient_checkpointing = True
+        self.paligemma_with_expert.paligemma.model.vision_tower.gradient_checkpointing = True
+        self.paligemma_with_expert.gemma_expert.model.gradient_checkpointing = True
+        logging.info("Enabled gradient checkpointing for PI0Pytorch model")
+
+    def gradient_checkpointing_disable(self):
+        """Disable gradient checkpointing."""
+        self.gradient_checkpointing_enabled = False
+        self.paligemma_with_expert.paligemma.model.language_model.gradient_checkpointing = False
+        self.paligemma_with_expert.paligemma.model.vision_tower.gradient_checkpointing = False
+        self.paligemma_with_expert.gemma_expert.model.gradient_checkpointing = False
+        logging.info("Disabled gradient checkpointing for PI0Pytorch model")
+
+    def _rtc_enabled(self):
+        return self.config.rtc_config is not None and self.config.rtc_config.enabled
+
+    def _apply_checkpoint(self, func, *args, **kwargs):
+        """Helper method to apply gradient checkpointing if enabled."""
+        if self.gradient_checkpointing_enabled and self.training:
+            return torch.utils.checkpoint.checkpoint(
+                func, *args, use_reentrant=False, preserve_rng_state=False, **kwargs
+            )
+        return func(*args, **kwargs)
+
+    def _prepare_attention_masks_4d(self, att_2d_masks):
+        """Helper method to prepare 4D attention masks for transformer."""
+        att_2d_masks_4d = att_2d_masks[:, None, :, :]
+        return torch.where(att_2d_masks_4d, 0.0, OPENPI_ATTENTION_MASK_VALUE)
+
+    def sample_noise(self, shape, device):
+        return torch.normal(
+            mean=0.0,
+            std=1.0,
+            size=shape,
+            dtype=torch.float32,
+            device=device,
+        )
+
+    def sample_time(self, bsize, device):
+        time_beta = sample_beta(
+            self.config.time_sampling_beta_alpha, self.config.time_sampling_beta_beta, bsize, device
+        )
+        time = time_beta * self.config.time_sampling_scale + self.config.time_sampling_offset
+        return time.to(dtype=torch.float32, device=device)
+
+    def embed_prefix(
+        self, images, img_masks, lang_tokens, lang_masks
+    ) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
+        """Embed images with SigLIP and language tokens with embedding layer."""
+        embs = []
+        pad_masks = []
+        att_masks = []
+
+        # Process images
+        for img, img_mask in zip(images, img_masks, strict=True):
+
+            def image_embed_func(img):
+                return self.paligemma_with_expert.embed_image(img)
+
+            img_emb = self._apply_checkpoint(image_embed_func, img)
+            bsize, num_img_embs = img_emb.shape[:2]
+
+            embs.append(img_emb)
+            pad_masks.append(img_mask[:, None].expand(bsize, num_img_embs))
+            att_masks += [0] * num_img_embs
+
+        # Process language tokens
+        def lang_embed_func(lang_tokens):
+            lang_emb = self.paligemma_with_expert.embed_language_tokens(lang_tokens)
+            lang_emb_dim = lang_emb.shape[-1]
+            return lang_emb * math.sqrt(lang_emb_dim)
+
+        lang_emb = self._apply_checkpoint(lang_embed_func, lang_tokens)
+        embs.append(lang_emb)
+        pad_masks.append(lang_masks)
+
+        num_lang_embs = lang_emb.shape[1]
+        att_masks += [0] * num_lang_embs
+
+        embs = torch.cat(embs, dim=1)
+        pad_masks = torch.cat(pad_masks, dim=1)
+        att_masks = torch.tensor(att_masks, dtype=torch.bool, device=pad_masks.device)
+
+        bsize = pad_masks.shape[0]
+        att_masks = att_masks[None, :].expand(bsize, len(att_masks))
+
+        return embs, pad_masks, att_masks
+
+    def embed_suffix(self, state, noisy_actions, timestep):
+        """Embed state, noisy_actions, timestep to prepare for Expert Gemma processing."""
+        embs = []
+        pad_masks = []
+        att_masks = []
+
+        if self.state_proj.weight.dtype == torch.float32:
+            state = state.to(torch.float32)
+
+        def state_proj_func(state):
+            return self.state_proj(state)
+
+        state_emb = self._apply_checkpoint(state_proj_func, state)
+        embs.append(state_emb[:, None, :])
+        bsize = state_emb.shape[0]
+        device = state_emb.device
+
+        state_mask = torch.ones(bsize, 1, dtype=torch.bool, device=device)
+        pad_masks.append(state_mask)
+        att_masks += [1]
+
+        # Embed timestep using sine-cosine positional encoding
+        time_emb = create_sinusoidal_pos_embedding(
+            timestep,
+            self.action_in_proj.out_features,
+            min_period=self.config.min_period,
+            max_period=self.config.max_period,
+            device=timestep.device,
+        )
+        time_emb = time_emb.type(dtype=timestep.dtype)
+
+        # Fuse timestep + action information using an MLP
+        def action_proj_func(noisy_actions):
+            return self.action_in_proj(noisy_actions)
+
+        action_emb = self._apply_checkpoint(action_proj_func, noisy_actions)
+
+        time_emb = time_emb[:, None, :].expand_as(action_emb)
+        action_time_emb = torch.cat([action_emb, time_emb], dim=2)
+
+        def mlp_func(action_time_emb):
+            x = self.action_time_mlp_in(action_time_emb)
+            x = F.silu(x)
+            return self.action_time_mlp_out(x)
+
+        action_time_emb = self._apply_checkpoint(mlp_func, action_time_emb)
+        adarms_cond = None
+
+        embs.append(action_time_emb)
+        bsize, action_time_dim = action_time_emb.shape[:2]
+        action_time_mask = torch.ones(bsize, action_time_dim, dtype=torch.bool, device=timestep.device)
+        pad_masks.append(action_time_mask)
+
+        # Set attention masks so that image, language and state inputs do not attend to action tokens
+        att_masks += [1] + ([0] * (self.config.chunk_size - 1))
+
+        embs = torch.cat(embs, dim=1)
+        pad_masks = torch.cat(pad_masks, dim=1)
+        att_masks = torch.tensor(att_masks, dtype=embs.dtype, device=embs.device)
+        att_masks = att_masks[None, :].expand(bsize, len(att_masks))
+
+        return embs, pad_masks, att_masks, adarms_cond
+
+    def forward(
+        self, images, img_masks, lang_tokens, lang_masks, state, actions, noise=None, time=None
+    ) -> Tensor:
+        """Do a full training forward pass and compute the loss."""
+        if noise is None:
+            noise = self.sample_noise(actions.shape, actions.device)
+
+        if time is None:
+            time = self.sample_time(actions.shape[0], actions.device)
+
+        time_expanded = time[:, None, None]
+        x_t = time_expanded * noise + (1 - time_expanded) * actions
+        u_t = noise - actions
+
+        prefix_embs, prefix_pad_masks, prefix_att_masks = self.embed_prefix(
+            images, img_masks, lang_tokens, lang_masks
+        )
+        suffix_embs, suffix_pad_masks, suffix_att_masks, adarms_cond = self.embed_suffix(state, x_t, time)
+
+        if (
+            self.paligemma_with_expert.paligemma.model.language_model.layers[0].self_attn.q_proj.weight.dtype
+            == torch.bfloat16
+        ):
+            suffix_embs = suffix_embs.to(dtype=torch.bfloat16)
+            prefix_embs = prefix_embs.to(dtype=torch.bfloat16)
+
+        pad_masks = torch.cat([prefix_pad_masks, suffix_pad_masks], dim=1)
+        att_masks = torch.cat([prefix_att_masks, suffix_att_masks], dim=1)
+
+        att_2d_masks = make_att_2d_masks(pad_masks, att_masks)
+        position_ids = torch.cumsum(pad_masks, dim=1) - 1
+
+        att_2d_masks_4d = self._prepare_attention_masks_4d(att_2d_masks)
+
+        def forward_func(prefix_embs, suffix_embs, att_2d_masks_4d, position_ids, adarms_cond):
+            (_, suffix_out), _ = self.paligemma_with_expert.forward(
+                attention_mask=att_2d_masks_4d,
+                position_ids=position_ids,
+                past_key_values=None,
+                inputs_embeds=[prefix_embs, suffix_embs],
+                use_cache=False,
+                adarms_cond=[None, adarms_cond],
+            )
+            return suffix_out
+
+        suffix_out = self._apply_checkpoint(
+            forward_func, prefix_embs, suffix_embs, att_2d_masks_4d, position_ids, adarms_cond
+        )
+
+        suffix_out = suffix_out[:, -self.config.chunk_size :]
+        suffix_out = suffix_out.to(dtype=torch.float32)
+
+        def action_out_proj_func(suffix_out):
+            return self.action_out_proj(suffix_out)
+
+        v_t = self._apply_checkpoint(action_out_proj_func, suffix_out)
+
+        return F.mse_loss(u_t, v_t, reduction="none")
+
+    @torch.no_grad()  # see openpi `sample_actions` (slightly adapted)
+    def sample_actions(
+        self,
+        images,
+        img_masks,
+        lang_tokens,
+        lang_masks,
+        state,
+        noise=None,
+        num_steps=None,
+        **kwargs: Unpack[ActionSelectKwargs],
+    ) -> Tensor:
+        """Do a full inference forward and compute the action."""
+        if num_steps is None:
+            num_steps = self.config.num_inference_steps
+
+        bsize = state.shape[0]
+        device = state.device
+
+        if noise is None:
+            # Sample noise with padded dimension as expected by action_in_proj
+            actions_shape = (
+                bsize,
+                self.config.chunk_size,
+                self.config.max_action_dim,
+            )  # Use config max_action_dim for internal processing
+            noise = self.sample_noise(actions_shape, device)
+
+        prefix_embs, prefix_pad_masks, prefix_att_masks = self.embed_prefix(
+            images, img_masks, lang_tokens, lang_masks
+        )
+        prefix_att_2d_masks = make_att_2d_masks(prefix_pad_masks, prefix_att_masks)
+        prefix_position_ids = torch.cumsum(prefix_pad_masks, dim=1) - 1
+
+        prefix_att_2d_masks_4d = self._prepare_attention_masks_4d(prefix_att_2d_masks)
+        self.paligemma_with_expert.paligemma.model.language_model.config._attn_implementation = "eager"  # noqa: SLF001
+
+        _, past_key_values = self.paligemma_with_expert.forward(
+            attention_mask=prefix_att_2d_masks_4d,
+            position_ids=prefix_position_ids,
+            past_key_values=None,
+            inputs_embeds=[prefix_embs, None],
+            use_cache=True,
+        )
+
+        dt = -1.0 / num_steps
+
+        x_t = noise
+        for step in range(num_steps):
+            time = 1.0 + step * dt
+            time_tensor = torch.tensor(time, dtype=torch.float32, device=device).expand(bsize)
+
+            def denoise_step_partial_call(input_x_t, current_timestep=time_tensor):
+                return self.denoise_step(
+                    state=state,
+                    prefix_pad_masks=prefix_pad_masks,
+                    past_key_values=past_key_values,
+                    x_t=input_x_t,
+                    timestep=current_timestep,
+                )
+
+            if self._rtc_enabled():
+                inference_delay = kwargs.get("inference_delay")
+                prev_chunk_left_over = kwargs.get("prev_chunk_left_over")
+                execution_horizon = kwargs.get("execution_horizon")
+
+                v_t = self.rtc_processor.denoise_step(
+                    x_t=x_t,
+                    prev_chunk_left_over=prev_chunk_left_over,
+                    inference_delay=inference_delay,
+                    time=time,
+                    original_denoise_step_partial=denoise_step_partial_call,
+                    execution_horizon=execution_horizon,
+                )
+            else:
+                v_t = denoise_step_partial_call(x_t)
+
+            x_t = x_t + dt * v_t
+
+            if self.rtc_processor is not None and self.rtc_processor.is_debug_enabled():
+                self.rtc_processor.track(time=time, x_t=x_t, v_t=v_t)
+
+        return x_t
+
+    def denoise_step(
+        self,
+        state,
+        prefix_pad_masks,
+        past_key_values,
+        x_t,
+        timestep,
+    ):
+        """Apply one denoising step of the noise `x_t` at a given timestep."""
+        suffix_embs, suffix_pad_masks, suffix_att_masks, adarms_cond = self.embed_suffix(state, x_t, timestep)
+
+        suffix_len = suffix_pad_masks.shape[1]
+        batch_size = prefix_pad_masks.shape[0]
+        prefix_len = prefix_pad_masks.shape[1]
+
+        prefix_pad_2d_masks = prefix_pad_masks[:, None, :].expand(batch_size, suffix_len, prefix_len)
+        suffix_att_2d_masks = make_att_2d_masks(suffix_pad_masks, suffix_att_masks)
+        full_att_2d_masks = torch.cat([prefix_pad_2d_masks, suffix_att_2d_masks], dim=2)
+
+        prefix_offsets = torch.sum(prefix_pad_masks, dim=-1)[:, None]
+        position_ids = prefix_offsets + torch.cumsum(suffix_pad_masks, dim=1) - 1
+
+        full_att_2d_masks_4d = self._prepare_attention_masks_4d(full_att_2d_masks)
+        self.paligemma_with_expert.gemma_expert.model.config._attn_implementation = "eager"  # noqa: SLF001
+
+        past_key_values = copy.deepcopy(past_key_values)
+        outputs_embeds, _ = self.paligemma_with_expert.forward(
+            attention_mask=full_att_2d_masks_4d,
+            position_ids=position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=[None, suffix_embs],
+            use_cache=False,
+            adarms_cond=[None, adarms_cond],
+        )
+
+        suffix_out = outputs_embeds[1]
+        suffix_out = suffix_out[:, -self.config.chunk_size :]
+        suffix_out = suffix_out.to(dtype=torch.float32)
+        return self.action_out_proj(suffix_out)
+
+
+class PI0Policy(PreTrainedPolicy):
+    """PI0 OpenPI Policy for LeRobot."""
+
+    config_class = PI0Config
+    name = "pi0"
+
+    def __init__(
+        self,
+        config: PI0Config,
+        **kwargs,
+    ):
+        """
+        Args:
+            config: Policy configuration class instance.
+        """
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        # Initialize the core PI0 model
+        self.init_rtc_processor()
+        self.model = PI0Pytorch(config, rtc_processor=self.rtc_processor)
+
+        # Enable gradient checkpointing if requested
+        if config.gradient_checkpointing:
+            self.model.gradient_checkpointing_enable()
+
+        self.model.to(config.device)
+
+        self.reset()
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type[T],
+        pretrained_name_or_path: str | Path,
+        *,
+        config: PreTrainedConfig | None = None,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        strict: bool = True,
+        **kwargs,
+    ) -> T:
+        """Override the from_pretrained method to handle key remapping and display important disclaimer."""
+        print(
+            "The PI0 model is a direct port of the OpenPI implementation. \n"
+            "This implementation follows the original OpenPI structure for compatibility. \n"
+            "Original implementation: https://github.com/Physical-Intelligence/openpi"
+        )
+        if pretrained_name_or_path is None:
+            raise ValueError("pretrained_name_or_path is required")
+
+        # Use provided config if available, otherwise create default config
+        if config is None:
+            config = PreTrainedConfig.from_pretrained(
+                pretrained_name_or_path=pretrained_name_or_path,
+                force_download=force_download,
+                resume_download=resume_download,
+                proxies=proxies,
+                token=token,
+                cache_dir=cache_dir,
+                local_files_only=local_files_only,
+                revision=revision,
+                **kwargs,
+            )
+
+        # Initialize model without loading weights
+        # Check if dataset_stats were provided in kwargs
+        model = cls(config, **kwargs)
+
+        # Load state dict (expects keys with "model." prefix)
+        try:
+            print(f"Loading model from: {pretrained_name_or_path}")
+            try:
+                from transformers.utils import cached_file
+
+                resolved_file = cached_file(
+                    pretrained_name_or_path,
+                    "model.safetensors",
+                    cache_dir=kwargs.get("cache_dir"),
+                    force_download=kwargs.get("force_download", False),
+                    resume_download=kwargs.get("resume_download"),
+                    proxies=kwargs.get("proxies"),
+                    token=kwargs.get("token"),
+                    revision=kwargs.get("revision"),
+                    local_files_only=kwargs.get("local_files_only", False),
+                )
+                from safetensors.torch import load_file
+
+                original_state_dict = load_file(resolved_file)
+                print("✓ Loaded state dict from model.safetensors")
+            except Exception as e:
+                print(f"Could not load state dict from remote files: {e}")
+                print("Returning model without loading pretrained weights")
+                return model
+
+            # First, fix any key differences (see openpi model.py, _fix_pytorch_state_dict_keys)
+            fixed_state_dict = model._fix_pytorch_state_dict_keys(original_state_dict, model.config)
+
+            # Then add "model." prefix for all keys that don't already have it
+            remapped_state_dict = {}
+            remap_count = 0
+
+            for key, value in fixed_state_dict.items():
+                if not key.startswith("model."):
+                    new_key = f"model.{key}"
+                    remapped_state_dict[new_key] = value
+                    remap_count += 1
+                else:
+                    remapped_state_dict[key] = value
+
+            if remap_count > 0:
+                print(f"Remapped {remap_count} state dict keys")
+
+            # Load the remapped state dict into the model
+            missing_keys, unexpected_keys = model.load_state_dict(remapped_state_dict, strict=strict)
+
+            if missing_keys:
+                print(f"Missing keys when loading state dict: {len(missing_keys)} keys")
+                if len(missing_keys) <= 5:
+                    for key in missing_keys:
+                        print(f"  - {key}")
+                else:
+                    for key in missing_keys[:5]:
+                        print(f"  - {key}")
+                    print(f"  ... and {len(missing_keys) - 5} more")
+
+            if unexpected_keys:
+                print(f"Unexpected keys when loading state dict: {len(unexpected_keys)} keys")
+                if len(unexpected_keys) <= 5:
+                    for key in unexpected_keys:
+                        print(f"  - {key}")
+                else:
+                    for key in unexpected_keys[:5]:
+                        print(f"  - {key}")
+                    print(f"  ... and {len(unexpected_keys) - 5} more")
+
+            if not missing_keys and not unexpected_keys:
+                print("All keys loaded successfully!")
+
+        except Exception as e:
+            print(f"Warning: Could not load state dict: {e}")
+
+        return model
+
+    def _fix_pytorch_state_dict_keys(
+        self, state_dict, model_config
+    ):  # see openpi `BaseModelConfig, _fix_pytorch_state_dict_keys`
+        """Fix state dict keys to match current model architecture."""
+        import re
+
+        fixed_state_dict = {}
+
+        for key, value in state_dict.items():
+            new_key = key
+
+            # Handle layer norm structure changes: .weight -> .dense.weight + .dense.bias
+            # For gemma expert layers
+            if re.match(
+                r"paligemma_with_expert\.gemma_expert\.model\.layers\.\d+\.(input_layernorm|post_attention_layernorm)\.weight",
+                key,
+            ):
+                # Check if the model actually has adaRMS enabled for the expert
+                expert_uses_adarms = getattr(
+                    self.model.paligemma_with_expert.gemma_expert.config, "use_adarms", False
+                )
+                if expert_uses_adarms:
+                    logging.warning(f"Skipping layer norm key (adaRMS mismatch): {key}")
+                    continue
+
+            if re.match(r"paligemma_with_expert\.gemma_expert\.model\.norm\.weight", key):
+                # Check if the model actually has adaRMS enabled for the expert
+                expert_uses_adarms = getattr(
+                    self.model.paligemma_with_expert.gemma_expert.config, "use_adarms", False
+                )
+                if expert_uses_adarms:
+                    logging.warning(f"Skipping norm key (adaRMS mismatch): {key}")
+                    continue
+
+            # Handle MLP naming changes for pi0
+            # non-pi05 model expects action_time_mlp_*, but checkpoint might have time_mlp_*
+            if key.startswith("time_mlp_in."):
+                new_key = key.replace("time_mlp_in.", "action_time_mlp_in.")
+            elif key.startswith("time_mlp_out."):
+                new_key = key.replace("time_mlp_out.", "action_time_mlp_out.")
+
+            # Handle vision tower embedding layer potential differences
+            if "patch_embedding" in key:
+                # Some checkpoints might have this, but current model expects different structure
+                logging.warning(f"Vision embedding key might need handling: {key}")
+
+            if (
+                key == "model.paligemma_with_expert.paligemma.lm_head.weight"
+                or key == "paligemma_with_expert.paligemma.lm_head.weight"
+            ):
+                fixed_state_dict[
+                    "model.paligemma_with_expert.paligemma.model.language_model.embed_tokens.weight"
+                ] = value.clone()
+
+            fixed_state_dict[new_key] = value
+
+        return fixed_state_dict
+
+    def get_optim_params(self) -> dict:
+        return self.parameters()
+
+    def reset(self):
+        """Reset internal state - called when environment resets."""
+        self._action_queue = deque(maxlen=self.config.n_action_steps)
+        self._queues = {
+            ACTION: deque(maxlen=self.config.n_action_steps),
+        }
+
+    def init_rtc_processor(self):
+        """Initialize RTC processor if RTC is enabled in config."""
+        self.rtc_processor = None
+
+        # Create processor if config provided
+        # If RTC is not enabled - we can still track the denoising data
+        if self.config.rtc_config is not None:
+            self.rtc_processor = RTCProcessor(self.config.rtc_config)
+
+            model_value = getattr(self, "model", None)
+            if model_value is not None:
+                model_value.rtc_processor = self.rtc_processor
+
+    def _rtc_enabled(self) -> bool:
+        return self.config.rtc_config is not None and self.config.rtc_config.enabled
+
+    def _preprocess_images(self, batch: dict[str, Tensor]) -> tuple[list[Tensor], list[Tensor]]:
+        """Preprocess images for the model.
+
+        Images from LeRobot are typically in [B, C, H, W] format and normalized to [0, 1].
+        PaliGemma expects images in [B, C, H, W] format and normalized to [-1, 1].
+        """
+        images = []
+        img_masks = []
+
+        # Get device from model parameters
+        device = next(self.parameters()).device
+
+        present_img_keys = [key for key in self.config.image_features if key in batch]
+        missing_img_keys = [key for key in self.config.image_features if key not in batch]
+
+        if len(present_img_keys) == 0:
+            raise ValueError(
+                f"All image features are missing from the batch. At least one expected. "
+                f"(batch: {batch.keys()}) (image_features: {self.config.image_features})"
+            )
+
+        for key in present_img_keys:
+            img = batch[key]
+
+            # Ensure tensor is on the same device as the model
+            if img.device != device:
+                img = img.to(device)
+
+            # Ensure float32 dtype for consistency
+            if img.dtype != torch.float32:
+                img = img.to(torch.float32)
+
+            # from openpi preprocess_observation_pytorch: Handle both [B, C, H, W] and [B, H, W, C] formats
+            is_channels_first = img.shape[1] == 3  # Check if channels are in dimension 1
+
+            if is_channels_first:
+                # Convert [B, C, H, W] to [B, H, W, C] for processing
+                img = img.permute(0, 2, 3, 1)
+
+            # from openpi preprocess_observation_pytorch: Resize with padding if needed
+            if img.shape[1:3] != self.config.image_resolution:
+                img = resize_with_pad_torch(img, *self.config.image_resolution)
+
+            # Normalize from [0,1] to [-1,1] as expected by siglip
+            img = img * 2.0 - 1.0
+
+            # from openpi preprocess_observation_pytorch: Convert back to [B, C, H, W] format if it was originally channels-first
+            if is_channels_first:
+                img = img.permute(0, 3, 1, 2)  # [B, H, W, C] -> [B, C, H, W]
+
+            images.append(img)
+            # Create mask (all ones for real images)
+            bsize = img.shape[0]
+            mask = torch.ones(bsize, dtype=torch.bool, device=device)
+            img_masks.append(mask)
+
+        # Create image features not present in the batch as fully 0 padded images
+        for _num_empty_cameras in range(len(missing_img_keys)):
+            img = torch.ones_like(img) * -1  # padded with -1 for SigLIP
+            mask = torch.zeros_like(mask)  # mask is zero for empty cameras
+            images.append(img)
+            img_masks.append(mask)
+
+        return images, img_masks
+
+    def prepare_state(self, batch):
+        """Pad state"""
+        state = pad_vector(batch[OBS_STATE], self.config.max_state_dim)
+        return state
+
+    def prepare_action(self, batch):
+        """Pad action"""
+        actions = pad_vector(batch[ACTION], self.config.max_action_dim)
+        return actions
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select a single action given environment observations."""
+        assert not self._rtc_enabled(), (
+            "RTC is not supported for select_action, use it with predict_action_chunk"
+        )
+
+        self.eval()
+
+        # Action queue logic for n_action_steps > 1
+        if len(self._action_queue) == 0:
+            actions = self.predict_action_chunk(batch)[:, : self.config.n_action_steps]
+            # Transpose to get shape (n_action_steps, batch_size, action_dim)
+            self._action_queue.extend(actions.transpose(0, 1))
+
+        return self._action_queue.popleft()
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]) -> Tensor:
+        """Predict a chunk of actions given environment observations."""
+        self.eval()
+
+        # Prepare inputs
+        images, img_masks = self._preprocess_images(batch)
+        lang_tokens, lang_masks = batch[f"{OBS_LANGUAGE_TOKENS}"], batch[f"{OBS_LANGUAGE_ATTENTION_MASK}"]
+        state = self.prepare_state(batch)
+
+        # Sample actions using the model (pass through RTC kwargs)
+        actions = self.model.sample_actions(images, img_masks, lang_tokens, lang_masks, state, **kwargs)
+
+        # Unpad actions to actual action dimension
+        original_action_dim = self.config.output_features[ACTION].shape[0]
+        actions = actions[:, :, :original_action_dim]
+
+        return actions
+
+    def forward(self, batch: dict[str, Tensor], reduction: str = "mean") -> tuple[Tensor, dict]:
+        """Run the batch through the model and compute the loss for training.
+
+        Args:
+            batch: Training batch containing observations and actions.
+            reduction: How to reduce the loss. Options:
+                - "mean": Return scalar mean loss (default, backward compatible)
+                - "none": Return per-sample losses of shape (batch_size,) for RA-BC weighting
+        """
+        # Prepare inputs
+        images, img_masks = self._preprocess_images(batch)
+        lang_tokens, lang_masks = batch[f"{OBS_LANGUAGE_TOKENS}"], batch[f"{OBS_LANGUAGE_ATTENTION_MASK}"]
+        state = self.prepare_state(batch)
+        actions = self.prepare_action(batch)
+
+        # Compute loss
+        losses = self.model.forward(images, img_masks, lang_tokens, lang_masks, state, actions)
+
+        # Truncate losses to actual action dimensions
+        original_action_dim = self.config.output_features[ACTION].shape[0]
+        losses = losses[:, :, :original_action_dim]
+
+        loss_dict = {
+            "loss_per_dim": losses.mean(dim=[0, 1]).detach().cpu().numpy().tolist(),
+        }
+
+        if reduction == "none":
+            # Return per-sample losses (B,) by averaging over time and action dims
+            per_sample_loss = losses.mean(dim=(1, 2))
+            loss_dict["loss"] = per_sample_loss.mean().item()
+            return per_sample_loss, loss_dict
+        else:
+            # Default: return scalar mean loss
+            loss = losses.mean()
+            loss_dict["loss"] = loss.item()
+            return loss, loss_dict
+
+    def _get_default_peft_targets(self) -> dict[str, any]:
+        """Return default PEFT target modules for PI0 fine-tuning."""
+        common_projections = (
+            "state_proj|action_in_proj|action_out_proj|action_time_mlp_in|action_time_mlp_out"
+        )
+        target_modules = rf"(.*\.gemma_expert\..*\.self_attn\.(q|v)_proj|model\.({common_projections}))"
+        return {
+            "target_modules": target_modules,
+            "modules_to_save": [],
+        }
diff --git a/lerobot/src/lerobot/policies/pi0/processor_pi0.py b/lerobot/src/lerobot/policies/pi0/processor_pi0.py
new file mode 100644
index 0000000000000000000000000000000000000000..50f5dec83ae8e417a07b3a0181bafd1bf485c5ea
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi0/processor_pi0.py
@@ -0,0 +1,166 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from typing import Any
+
+import torch
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.policies.pi0.configuration_pi0 import PI0Config
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    ComplementaryDataProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    RenameObservationsProcessorStep,
+    TokenizerProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.utils.constants import POLICY_POSTPROCESSOR_DEFAULT_NAME, POLICY_PREPROCESSOR_DEFAULT_NAME
+
+
+@ProcessorStepRegistry.register(name="pi0_new_line_processor")
+class Pi0NewLineProcessor(ComplementaryDataProcessorStep):
+    """
+    Ensures that the task description string ends with a newline character.
+
+    This processing step is required for compatibility with the PaliGemma tokenizer,
+    which expects a newline at the end of the text prompt. It handles both single
+    strings and lists of strings for the 'task' key in complementary data.
+    """
+
+    def complementary_data(self, complementary_data):
+        """
+        Adds a newline to the 'task' field if it doesn't already have one.
+
+        Args:
+            complementary_data: A dictionary that may contain a 'task' key with a
+                                string or list of strings.
+
+        Returns:
+            A new dictionary with the modified 'task' field.
+        """
+        if "task" not in complementary_data:
+            return complementary_data
+
+        task = complementary_data["task"]
+        if task is None:
+            return complementary_data
+
+        new_complementary_data = dict(complementary_data)
+
+        # Handle both string and list of strings
+        if isinstance(task, str):
+            # Single string: add newline if not present
+            if not task.endswith("\n"):
+                new_complementary_data["task"] = f"{task}\n"
+        elif isinstance(task, list) and all(isinstance(t, str) for t in task):
+            # List of strings: add newline to each if not present
+            new_complementary_data["task"] = [t if t.endswith("\n") else f"{t}\n" for t in task]
+        # If task is neither string nor list of strings, leave unchanged
+
+        return new_complementary_data
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        This step does not alter the feature definitions.
+
+        Args:
+            features: The input feature dictionary.
+
+        Returns:
+            The unchanged feature dictionary.
+        """
+        return features
+
+
+def make_pi0_pre_post_processors(
+    config: PI0Config,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for the PI0 policy.
+
+    The pre-processing pipeline prepares input data for the model by:
+    1. Renaming features to match pretrained configurations.
+    2. Normalizing input and output features based on dataset statistics.
+    3. Adding a batch dimension.
+    4. Appending a newline character to the task description for tokenizer compatibility.
+    5. Tokenizing the text prompt using the PaliGemma tokenizer.
+    6. Moving all data to the specified device.
+
+    The post-processing pipeline handles the model's output by:
+    1. Moving data to the CPU.
+    2. Unnormalizing the output features to their original scale.
+
+    Args:
+        config: The configuration object for the PI0 policy.
+        dataset_stats: A dictionary of statistics for normalization.
+        preprocessor_kwargs: Additional arguments for the pre-processor pipeline.
+        postprocessor_kwargs: Additional arguments for the post-processor pipeline.
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines.
+    """
+
+    # Add remaining processors
+    input_steps: list[ProcessorStep] = [
+        RenameObservationsProcessorStep(rename_map={}),  # To mimic the same processor as pretrained one
+        AddBatchDimensionProcessorStep(),
+        Pi0NewLineProcessor(),  # Add newlines before tokenization for PaliGemma
+        TokenizerProcessorStep(
+            tokenizer_name="google/paligemma-3b-pt-224",
+            max_length=config.tokenizer_max_length,
+            padding_side="right",
+            padding="max_length",
+        ),
+        DeviceProcessorStep(device=config.device),
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+    ]
+
+    output_steps: list[ProcessorStep] = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/pi05/README.md b/lerobot/src/lerobot/policies/pi05/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..2ae69d978d37c1f8027f8ae1603d8013cab6cd95
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi05/README.md
@@ -0,0 +1,49 @@
+# π₀.₅ (pi05)
+
+This repository contains the Hugging Face port of **π₀.₅**, adapted from [OpenPI](https://github.com/Physical-Intelligence/openpi) by the Physical Intelligence.
+It is designed as a **Vision-Language-Action model with open-world generalization**.
+
+---
+
+## Model Overview
+
+| Feature              | π₀                                                     | π₀.₅                                      |
+| -------------------- | ------------------------------------------------------ | ----------------------------------------- |
+| Time Conditioning    | Concatenates time with actions via `action_time_mlp_*` | Uses `time_mlp_*` for AdaRMS conditioning |
+| AdaRMS               | Not used                                               | Used in action expert                     |
+| Tokenizer Length     | 48 tokens                                              | 200 tokens                                |
+| Discrete State Input | False (Uses `state_proj` layer)                        | True                                      |
+| Parameter Count      | Higher (includes state embedding)                      | Lower (no state embedding)                |
+
+---
+
+## Citation
+
+If you use this work, please cite both **OpenPI** and the π₀.₅ paper:
+
+```bibtex
+@misc{openpi2024,
+  author       = {Physical Intelligence Lab},
+  title        = {OpenPI: PyTorch Implementation of π0 and π0.5 Policies},
+  year         = {2024},
+  publisher    = {GitHub},
+  howpublished = {\url{https://github.com/Physical-Intelligence/openpi}},
+  license      = {Apache-2.0}
+}
+
+@misc{intelligence2025pi05visionlanguageactionmodelopenworld,
+  title        = {π₀.₅: a Vision-Language-Action Model with Open-World Generalization},
+  author       = {Physical Intelligence and Kevin Black and Noah Brown and James Darpinian and Karan Dhabalia and Danny Driess and Adnan Esmail and Michael Equi and Chelsea Finn and Niccolo Fusai and Manuel Y. Galliker and Dibya Ghosh and Lachy Groom and Karol Hausman and Brian Ichter and Szymon Jakubczak and Tim Jones and Liyiming Ke and Devin LeBlanc and Sergey Levine and Adrian Li-Bell and Mohith Mothukuri and Suraj Nair and Karl Pertsch and Allen Z. Ren and Lucy Xiaoyang Shi and Laura Smith and Jost Tobias Springenberg and Kyle Stachowicz and James Tanner and Quan Vuong and Homer Walke and Anna Walling and Haohuan Wang and Lili Yu and Ury Zhilinsky},
+  year         = {2025},
+  eprint       = {2504.16054},
+  archivePrefix= {arXiv},
+  primaryClass = {cs.LG},
+  url          = {https://arxiv.org/abs/2504.16054},
+}
+```
+
+---
+
+## License
+
+This port follows the **Apache 2.0 License**, consistent with the original [OpenPI repository](https://github.com/Physical-Intelligence/openpi).
diff --git a/lerobot/src/lerobot/policies/pi05/__init__.py b/lerobot/src/lerobot/policies/pi05/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..4f9a9de4af65c583cc0db4c1a089609b7a03d813
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi05/__init__.py
@@ -0,0 +1,21 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configuration_pi05 import PI05Config
+from .modeling_pi05 import PI05Policy
+from .processor_pi05 import make_pi05_pre_post_processors
+
+__all__ = ["PI05Config", "PI05Policy", "make_pi05_pre_post_processors"]
diff --git a/lerobot/src/lerobot/policies/pi05/configuration_pi05.py b/lerobot/src/lerobot/policies/pi05/configuration_pi05.py
new file mode 100644
index 0000000000000000000000000000000000000000..b96e6d196359c2e99a3952c543856587a663b2c9
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi05/configuration_pi05.py
@@ -0,0 +1,169 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.optim.optimizers import AdamWConfig
+from lerobot.optim.schedulers import CosineDecayWithWarmupSchedulerConfig
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+from lerobot.utils.constants import ACTION, OBS_IMAGES, OBS_STATE
+
+DEFAULT_IMAGE_SIZE = 224
+
+
+@PreTrainedConfig.register_subclass("pi05")
+@dataclass
+class PI05Config(PreTrainedConfig):
+    paligemma_variant: str = "gemma_2b"
+    action_expert_variant: str = "gemma_300m"
+    dtype: str = "float32"  # Options: "bfloat16", "float32"
+
+    n_obs_steps: int = 1
+    chunk_size: int = 50  # Number of action steps to predict, in openpi called "action_horizon"
+    n_action_steps: int = 50  # Number of action steps to execute
+
+    # Shorter state and action vectors will be padded to these dimensions
+    max_state_dim: int = 32
+    max_action_dim: int = 32
+
+    # Flow matching parameters: see openpi `PI0Pytorch`
+    num_inference_steps: int = 10
+    time_sampling_beta_alpha: float = 1.5
+    time_sampling_beta_beta: float = 1.0
+    time_sampling_scale: float = 0.999
+    time_sampling_offset: float = 0.001
+    min_period: float = 4e-3
+    max_period: float = 4.0
+
+    # Real-Time Chunking (RTC) configuration
+    rtc_config: RTCConfig | None = None
+
+    image_resolution: tuple[int, int] = (
+        DEFAULT_IMAGE_SIZE,
+        DEFAULT_IMAGE_SIZE,
+    )  # see openpi `preprocessing_pytorch.py`
+
+    # Add empty images. Used to add empty cameras when no image features are present.
+    empty_cameras: int = 0
+
+    tokenizer_max_length: int = 200  # see openpi `__post_init__`
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.QUANTILES,  # Pi0.5 uses quantiles for state
+            "ACTION": NormalizationMode.QUANTILES,  # Pi0.5 uses quantiles for action
+        }
+    )
+
+    # Training settings
+    gradient_checkpointing: bool = False  # Enable gradient checkpointing for memory optimization
+    compile_model: bool = False  # Whether to use torch.compile for model optimization
+    compile_mode: str = "max-autotune"  # Torch compile mode
+    device: str | None = None  # Device to use for the model (None = auto-detect)
+
+    # Finetuning settings
+    freeze_vision_encoder: bool = False  # Freeze only the vision encoder
+    train_expert_only: bool = False  # Freeze entire VLM, train only action expert and projections
+
+    # Optimizer settings: see openpi `AdamW`
+    optimizer_lr: float = 2.5e-5  # see openpi `CosineDecaySchedule: peak_lr`
+    optimizer_betas: tuple[float, float] = (0.9, 0.95)
+    optimizer_eps: float = 1e-8
+    optimizer_weight_decay: float = 0.01
+    optimizer_grad_clip_norm: float = 1.0
+
+    # Scheduler settings: see openpi `CosineDecaySchedule`
+    # Note: These will auto-scale if --steps < scheduler_decay_steps
+    # For example, --steps=3000 will scale warmup to 100 and decay to 3000
+    scheduler_warmup_steps: int = 1_000
+    scheduler_decay_steps: int = 30_000
+    scheduler_decay_lr: float = 2.5e-6
+
+    tokenizer_max_length: int = 200  # see openpi `__post_init__`
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        # Validate configuration
+        if self.n_action_steps > self.chunk_size:
+            raise ValueError(
+                f"n_action_steps ({self.n_action_steps}) cannot be greater than chunk_size ({self.chunk_size})"
+            )
+
+        if self.paligemma_variant not in ["gemma_300m", "gemma_2b"]:
+            raise ValueError(f"Invalid paligemma_variant: {self.paligemma_variant}")
+
+        if self.action_expert_variant not in ["gemma_300m", "gemma_2b"]:
+            raise ValueError(f"Invalid action_expert_variant: {self.action_expert_variant}")
+
+        if self.dtype not in ["bfloat16", "float32"]:
+            raise ValueError(f"Invalid dtype: {self.dtype}")
+
+    def validate_features(self) -> None:
+        """Validate and set up input/output features."""
+        for i in range(self.empty_cameras):
+            key = OBS_IMAGES + f".empty_camera_{i}"
+            empty_camera = PolicyFeature(
+                type=FeatureType.VISUAL,
+                shape=(3, *self.image_resolution),  # Use configured image resolution
+            )
+            self.input_features[key] = empty_camera
+
+        if OBS_STATE not in self.input_features:
+            state_feature = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(self.max_state_dim,),  # Padded to max_state_dim
+            )
+            self.input_features[OBS_STATE] = state_feature
+
+        if ACTION not in self.output_features:
+            action_feature = PolicyFeature(
+                type=FeatureType.ACTION,
+                shape=(self.max_action_dim,),  # Padded to max_action_dim
+            )
+            self.output_features[ACTION] = action_feature
+
+    def get_optimizer_preset(self) -> AdamWConfig:
+        return AdamWConfig(
+            lr=self.optimizer_lr,
+            betas=self.optimizer_betas,
+            eps=self.optimizer_eps,
+            weight_decay=self.optimizer_weight_decay,
+            grad_clip_norm=self.optimizer_grad_clip_norm,
+        )
+
+    def get_scheduler_preset(self):
+        return CosineDecayWithWarmupSchedulerConfig(
+            peak_lr=self.optimizer_lr,
+            decay_lr=self.scheduler_decay_lr,
+            num_warmup_steps=self.scheduler_warmup_steps,
+            num_decay_steps=self.scheduler_decay_steps,
+        )
+
+    @property
+    def observation_delta_indices(self) -> None:
+        return None
+
+    @property
+    def action_delta_indices(self) -> list:
+        return list(range(self.chunk_size))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/pi05/modeling_pi05.py b/lerobot/src/lerobot/policies/pi05/modeling_pi05.py
new file mode 100644
index 0000000000000000000000000000000000000000..96c4002f214a6bdecede36882d75d5d4905742ca
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi05/modeling_pi05.py
@@ -0,0 +1,1294 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import builtins
+import copy
+import logging
+import math
+from collections import deque
+from pathlib import Path
+from typing import TYPE_CHECKING, Literal, TypedDict, Unpack
+
+import torch
+import torch.nn.functional as F  # noqa: N812
+from torch import Tensor, nn
+
+from lerobot.utils.import_utils import _transformers_available
+
+# Conditional import for type checking and lazy loading
+if TYPE_CHECKING or _transformers_available:
+    from transformers.models.auto import CONFIG_MAPPING
+    from transformers.models.gemma import modeling_gemma
+
+    from lerobot.policies.pi_gemma import (
+        PaliGemmaForConditionalGenerationWithPiGemma,
+        PiGemmaForCausalLM,
+        _gated_residual,
+        layernorm_forward,
+    )
+else:
+    CONFIG_MAPPING = None
+    modeling_gemma = None
+    PiGemmaForCausalLM = None
+    _gated_residual = None
+    layernorm_forward = None
+    PaliGemmaForConditionalGenerationWithPiGemma = None
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.policies.pi05.configuration_pi05 import DEFAULT_IMAGE_SIZE, PI05Config
+from lerobot.policies.pretrained import PreTrainedPolicy, T
+from lerobot.policies.rtc.modeling_rtc import RTCProcessor
+from lerobot.utils.constants import (
+    ACTION,
+    OBS_LANGUAGE_ATTENTION_MASK,
+    OBS_LANGUAGE_TOKENS,
+    OPENPI_ATTENTION_MASK_VALUE,
+)
+
+
+class ActionSelectKwargs(TypedDict, total=False):
+    inference_delay: int | None
+    prev_chunk_left_over: Tensor | None
+    execution_horizon: int | None
+
+
+def get_safe_dtype(target_dtype, device_type):
+    """Get a safe dtype for the given device type."""
+    if device_type == "mps" and target_dtype == torch.float64:
+        return torch.float32
+    if device_type == "cpu":
+        # CPU doesn't support bfloat16, use float32 instead
+        if target_dtype == torch.bfloat16:
+            return torch.float32
+        if target_dtype == torch.float64:
+            return torch.float64
+    return target_dtype
+
+
+def create_sinusoidal_pos_embedding(  # see openpi `create_sinusoidal_pos_embedding` (exact copy)
+    time: torch.Tensor, dimension: int, min_period: float, max_period: float, device="cpu"
+) -> Tensor:
+    """Computes sine-cosine positional embedding vectors for scalar positions."""
+    if dimension % 2 != 0:
+        raise ValueError(f"dimension ({dimension}) must be divisible by 2")
+
+    if time.ndim != 1:
+        raise ValueError("The time tensor is expected to be of shape `(batch_size, )`.")
+
+    dtype = get_safe_dtype(torch.float64, device.type)
+    fraction = torch.linspace(0.0, 1.0, dimension // 2, dtype=dtype, device=device)
+    period = min_period * (max_period / min_period) ** fraction
+
+    # Compute the outer product
+    scaling_factor = 1.0 / period * 2 * math.pi
+    sin_input = scaling_factor[None, :] * time[:, None]
+    return torch.cat([torch.sin(sin_input), torch.cos(sin_input)], dim=1)
+
+
+def sample_beta(alpha, beta, bsize, device):  # see openpi `sample_beta` (exact copy)
+    # Beta sampling uses _sample_dirichlet which isn't implemented for MPS, so sample on CPU
+    alpha_t = torch.tensor(alpha, dtype=torch.float32)
+    beta_t = torch.tensor(beta, dtype=torch.float32)
+    dist = torch.distributions.Beta(alpha_t, beta_t)
+    return dist.sample((bsize,)).to(device)
+
+
+def make_att_2d_masks(pad_masks, att_masks):  # see openpi `make_att_2d_masks` (exact copy)
+    """Copied from big_vision.
+
+    Tokens can attend to valid inputs tokens which have a cumulative mask_ar
+    smaller or equal to theirs. This way `mask_ar` int[B, N] can be used to
+    setup several types of attention, for example:
+
+      [[1 1 1 1 1 1]]: pure causal attention.
+
+      [[0 0 0 1 1 1]]: prefix-lm attention. The first 3 tokens can attend between
+          themselves and the last 3 tokens have a causal attention. The first
+          entry could also be a 1 without changing behaviour.
+
+      [[1 0 1 0 1 0 0 1 0 0]]: causal attention between 4 blocks. Tokens of a
+          block can attend all previous blocks and all tokens on the same block.
+
+    Args:
+      input_mask: bool[B, N] true if its part of the input, false if padding.
+      mask_ar: int32[B, N] mask that's 1 where previous tokens cannot depend on
+        it and 0 where it shares the same attention mask as the previous token.
+    """
+    if att_masks.ndim != 2:
+        raise ValueError(att_masks.ndim)
+    if pad_masks.ndim != 2:
+        raise ValueError(pad_masks.ndim)
+
+    cumsum = torch.cumsum(att_masks, dim=1)
+    att_2d_masks = cumsum[:, None, :] <= cumsum[:, :, None]
+    pad_2d_masks = pad_masks[:, None, :] * pad_masks[:, :, None]
+    return att_2d_masks & pad_2d_masks
+
+
+def pad_vector(vector, new_dim):
+    """Pad the last dimension of a vector to new_dim with zeros.
+
+    Can be (batch_size x sequence_length x features_dimension)
+    or (batch_size x features_dimension)
+    """
+    if vector.shape[-1] >= new_dim:
+        return vector
+    return F.pad(vector, (0, new_dim - vector.shape[-1]))
+
+
+def resize_with_pad_torch(  # see openpi `resize_with_pad_torch` (exact copy)
+    images: torch.Tensor,
+    height: int,
+    width: int,
+    mode: str = "bilinear",
+) -> torch.Tensor:
+    """PyTorch version of resize_with_pad. Resizes an image to a target height and width without distortion
+    by padding with black. If the image is float32, it must be in the range [-1, 1].
+
+    Args:
+        images: Tensor of shape [*b, h, w, c] or [*b, c, h, w]
+        height: Target height
+        width: Target width
+        mode: Interpolation mode ('bilinear', 'nearest', etc.)
+
+    Returns:
+        Resized and padded tensor with same shape format as input
+    """
+    # Check if input is in channels-last format [*b, h, w, c] or channels-first [*b, c, h, w]
+    if images.shape[-1] <= 4:  # Assume channels-last format
+        channels_last = True
+        if images.dim() == 3:
+            images = images.unsqueeze(0)  # Add batch dimension
+        images = images.permute(0, 3, 1, 2)  # [b, h, w, c] -> [b, c, h, w]
+    else:
+        channels_last = False
+        if images.dim() == 3:
+            images = images.unsqueeze(0)  # Add batch dimension
+
+    batch_size, channels, cur_height, cur_width = images.shape
+
+    # Calculate resize ratio
+    ratio = max(cur_width / width, cur_height / height)
+    resized_height = int(cur_height / ratio)
+    resized_width = int(cur_width / ratio)
+
+    # Resize
+    resized_images = F.interpolate(
+        images,
+        size=(resized_height, resized_width),
+        mode=mode,
+        align_corners=False if mode == "bilinear" else None,
+    )
+
+    # Handle dtype-specific clipping
+    if images.dtype == torch.uint8:
+        resized_images = torch.round(resized_images).clamp(0, 255).to(torch.uint8)
+    elif images.dtype == torch.float32:
+        resized_images = resized_images.clamp(0.0, 1.0)
+    else:
+        raise ValueError(f"Unsupported image dtype: {images.dtype}")
+
+    # Calculate padding
+    pad_h0, remainder_h = divmod(height - resized_height, 2)
+    pad_h1 = pad_h0 + remainder_h
+    pad_w0, remainder_w = divmod(width - resized_width, 2)
+    pad_w1 = pad_w0 + remainder_w
+
+    # Pad
+    constant_value = 0 if images.dtype == torch.uint8 else 0.0
+    padded_images = F.pad(
+        resized_images,
+        (pad_w0, pad_w1, pad_h0, pad_h1),  # left, right, top, bottom
+        mode="constant",
+        value=constant_value,
+    )
+
+    # Convert back to original format if needed
+    if channels_last:
+        padded_images = padded_images.permute(0, 2, 3, 1)  # [b, c, h, w] -> [b, h, w, c]
+
+    return padded_images
+
+
+# Define the complete layer computation function for gradient checkpointing
+def compute_layer_complete(
+    layer_idx, inputs_embeds, attention_mask, position_ids, adarms_cond, paligemma, gemma_expert
+):
+    models = [paligemma.model.language_model, gemma_expert.model]
+    query_states = []
+    key_states = []
+    value_states = []
+    gates = []
+    for i, hidden_states in enumerate(inputs_embeds):
+        layer = models[i].layers[layer_idx]
+        hidden_states, gate = layernorm_forward(layer.input_layernorm, hidden_states, adarms_cond[i])
+        gates.append(gate)
+        input_shape = hidden_states.shape[:-1]
+        hidden_shape = (*input_shape, -1, layer.self_attn.head_dim)
+        query_state = layer.self_attn.q_proj(hidden_states).view(hidden_shape).transpose(1, 2)
+        key_state = layer.self_attn.k_proj(hidden_states).view(hidden_shape).transpose(1, 2)
+        value_state = layer.self_attn.v_proj(hidden_states).view(hidden_shape).transpose(1, 2)
+        query_states.append(query_state)
+        key_states.append(key_state)
+        value_states.append(value_state)
+    # Concatenate and process attention
+    query_states = torch.cat(query_states, dim=2)
+    key_states = torch.cat(key_states, dim=2)
+    value_states = torch.cat(value_states, dim=2)
+    dummy_tensor = torch.zeros(
+        query_states.shape[0],
+        query_states.shape[2],
+        query_states.shape[-1],
+        device=query_states.device,
+        dtype=query_states.dtype,
+    )
+    cos, sin = paligemma.model.language_model.rotary_emb(dummy_tensor, position_ids)
+    query_states, key_states = modeling_gemma.apply_rotary_pos_emb(
+        query_states, key_states, cos, sin, unsqueeze_dim=1
+    )
+    batch_size = query_states.shape[0]
+    scaling = paligemma.model.language_model.layers[layer_idx].self_attn.scaling
+    # Attention computation
+    att_output, _ = modeling_gemma.eager_attention_forward(
+        paligemma.model.language_model.layers[layer_idx].self_attn,
+        query_states,
+        key_states,
+        value_states,
+        attention_mask,
+        scaling,
+    )
+    # Get head_dim from the current layer, not from the model
+    head_dim = paligemma.model.language_model.layers[layer_idx].self_attn.head_dim
+    att_output = att_output.reshape(batch_size, -1, 1 * 8 * head_dim)
+    # Process layer outputs
+    outputs_embeds = []
+    start_pos = 0
+    for i, hidden_states in enumerate(inputs_embeds):
+        layer = models[i].layers[layer_idx]
+        end_pos = start_pos + hidden_states.shape[1]
+        if att_output.dtype != layer.self_attn.o_proj.weight.dtype:
+            att_output = att_output.to(layer.self_attn.o_proj.weight.dtype)
+        out_emb = layer.self_attn.o_proj(att_output[:, start_pos:end_pos])
+        # first residual
+        out_emb = _gated_residual(hidden_states, out_emb, gates[i])
+        after_first_residual = out_emb.clone()
+        out_emb, gate = layernorm_forward(layer.post_attention_layernorm, out_emb, adarms_cond[i])
+        # Convert to bfloat16 if the next layer (mlp) uses bfloat16
+        if layer.mlp.up_proj.weight.dtype == torch.bfloat16:
+            out_emb = out_emb.to(dtype=torch.bfloat16)
+        out_emb = layer.mlp(out_emb)
+        # second residual
+        out_emb = _gated_residual(after_first_residual, out_emb, gate)
+        outputs_embeds.append(out_emb)
+        start_pos = end_pos
+    return outputs_embeds
+
+
+class GemmaConfig:  # see openpi `gemma.py: Config`
+    """Configuration for Gemma model variants."""
+
+    def __init__(self, width, depth, mlp_dim, num_heads, num_kv_heads, head_dim):
+        self.width = width
+        self.depth = depth
+        self.mlp_dim = mlp_dim
+        self.num_heads = num_heads
+        self.num_kv_heads = num_kv_heads
+        self.head_dim = head_dim
+
+
+def get_gemma_config(variant: str) -> GemmaConfig:  # see openpi `gemma.py: get_config`
+    """Returns config for specified gemma variant."""
+    if variant == "gemma_300m":
+        return GemmaConfig(
+            width=1024,
+            depth=18,
+            mlp_dim=4096,
+            num_heads=8,
+            num_kv_heads=1,
+            head_dim=256,
+        )
+    elif variant == "gemma_2b":
+        return GemmaConfig(
+            width=2048,
+            depth=18,
+            mlp_dim=16_384,
+            num_heads=8,
+            num_kv_heads=1,
+            head_dim=256,
+        )
+    else:
+        raise ValueError(f"Unknown variant: {variant}")
+
+
+class PaliGemmaWithExpertModel(
+    nn.Module
+):  # see openpi `gemma_pytorch.py: PaliGemmaWithExpertModel` this class is almost a exact copy of PaliGemmaWithExpertModel in openpi
+    """PaliGemma model with action expert for PI05."""
+
+    def __init__(
+        self,
+        vlm_config,
+        action_expert_config,
+        use_adarms=None,
+        precision: Literal["bfloat16", "float32"] = "bfloat16",
+        image_size: int = DEFAULT_IMAGE_SIZE,
+        freeze_vision_encoder: bool = False,
+        train_expert_only: bool = False,
+    ):
+        if use_adarms is None:
+            use_adarms = [False, False]
+        super().__init__()
+        self.freeze_vision_encoder = freeze_vision_encoder
+        self.train_expert_only = train_expert_only
+
+        vlm_config_hf = CONFIG_MAPPING["paligemma"]()
+        vlm_config_hf._vocab_size = 257152  # noqa: SLF001
+        vlm_config_hf.image_token_index = 257152
+        vlm_config_hf.text_config.hidden_size = vlm_config.width
+        vlm_config_hf.text_config.intermediate_size = vlm_config.mlp_dim
+        vlm_config_hf.text_config.num_attention_heads = vlm_config.num_heads
+        vlm_config_hf.text_config.head_dim = vlm_config.head_dim
+        vlm_config_hf.text_config.num_hidden_layers = vlm_config.depth
+        vlm_config_hf.text_config.num_key_value_heads = vlm_config.num_kv_heads
+        vlm_config_hf.text_config.hidden_activation = "gelu_pytorch_tanh"
+        vlm_config_hf.text_config.dtype = "float32"
+        vlm_config_hf.text_config.vocab_size = 257152
+        vlm_config_hf.text_config.use_adarms = use_adarms[0]
+        vlm_config_hf.text_config.adarms_cond_dim = vlm_config.width if use_adarms[0] else None
+        vlm_config_hf.vision_config.image_size = image_size
+        vlm_config_hf.vision_config.intermediate_size = 4304
+        vlm_config_hf.vision_config.projection_dim = 2048
+        vlm_config_hf.vision_config.projector_hidden_act = "gelu_fast"
+        vlm_config_hf.vision_config.dtype = "float32"
+
+        action_expert_config_hf = CONFIG_MAPPING["gemma"](
+            head_dim=action_expert_config.head_dim,
+            hidden_size=action_expert_config.width,
+            intermediate_size=action_expert_config.mlp_dim,
+            num_attention_heads=action_expert_config.num_heads,
+            num_hidden_layers=action_expert_config.depth,
+            num_key_value_heads=action_expert_config.num_kv_heads,
+            vocab_size=257152,
+            hidden_activation="gelu_pytorch_tanh",
+            dtype="float32",
+            use_adarms=use_adarms[1],
+            adarms_cond_dim=action_expert_config.width if use_adarms[1] else None,
+        )
+
+        self.paligemma = PaliGemmaForConditionalGenerationWithPiGemma(config=vlm_config_hf)
+        self.gemma_expert = PiGemmaForCausalLM(config=action_expert_config_hf)
+        self.gemma_expert.model.embed_tokens = None
+
+        self.to_bfloat16_for_selected_params(precision)
+        self._set_requires_grad()
+
+    def to_bfloat16_for_selected_params(self, precision: Literal["bfloat16", "float32"] = "bfloat16"):
+        if precision == "bfloat16":
+            self.to(dtype=torch.bfloat16)
+        elif precision == "float32":
+            self.to(dtype=torch.float32)
+            return
+        else:
+            raise ValueError(f"Invalid precision: {precision}")
+
+        # Keep full vision path in float32 so we never toggle (toggle causes optimizer
+        # "same dtype" error). Saves memory vs full float32; more memory than only 3 params.
+        params_to_keep_float32 = [
+            "vision_tower",
+            "multi_modal_projector",
+            "input_layernorm",
+            "post_attention_layernorm",
+            "model.norm",
+        ]
+
+        for name, param in self.named_parameters():
+            if any(selector in name for selector in params_to_keep_float32):
+                param.data = param.data.to(dtype=torch.float32)
+
+    def _set_requires_grad(self):
+        if self.freeze_vision_encoder:
+            self.paligemma.model.vision_tower.eval()
+            for param in self.paligemma.model.vision_tower.parameters():
+                param.requires_grad = False
+        if self.train_expert_only:
+            self.paligemma.eval()
+            for param in self.paligemma.parameters():
+                param.requires_grad = False
+
+    def train(self, mode: bool = True):
+        super().train(mode)
+        if self.freeze_vision_encoder:
+            self.paligemma.model.vision_tower.eval()
+        if self.train_expert_only:
+            self.paligemma.eval()
+
+    def embed_image(self, image: torch.Tensor):
+        # Vision tower and multi_modal_projector are kept in float32 (params_to_keep_float32).
+        out_dtype = image.dtype
+        if image.dtype != torch.float32:
+            image = image.to(torch.float32)
+        image_outputs = self.paligemma.model.get_image_features(image)
+        features = image_outputs.pooler_output * self.paligemma.config.text_config.hidden_size**0.5
+        if features.dtype != out_dtype:
+            features = features.to(out_dtype)
+        return features
+
+    def embed_language_tokens(self, tokens: torch.Tensor):
+        return self.paligemma.model.language_model.embed_tokens(tokens)
+
+    def forward(
+        self,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: list[torch.FloatTensor] | None = None,
+        use_cache: bool | None = None,
+        adarms_cond: list[torch.Tensor] | None = None,
+    ):
+        if adarms_cond is None:
+            adarms_cond = [None, None]
+        if inputs_embeds[1] is None:
+            prefix_output = self.paligemma.model.language_model.forward(
+                inputs_embeds=inputs_embeds[0],
+                attention_mask=attention_mask,
+                position_ids=position_ids,
+                past_key_values=past_key_values,
+                use_cache=use_cache,
+                adarms_cond=adarms_cond[0] if adarms_cond is not None else None,
+            )
+            prefix_past_key_values = prefix_output.past_key_values
+            prefix_output = prefix_output.last_hidden_state
+            suffix_output = None
+        elif inputs_embeds[0] is None:
+            suffix_output = self.gemma_expert.model.forward(
+                inputs_embeds=inputs_embeds[1],
+                attention_mask=attention_mask,
+                position_ids=position_ids,
+                past_key_values=past_key_values,
+                use_cache=use_cache,
+                adarms_cond=adarms_cond[1] if adarms_cond is not None else None,
+            )
+            suffix_output = suffix_output.last_hidden_state
+            prefix_output = None
+            prefix_past_key_values = None
+        else:
+            models = [self.paligemma.model.language_model, self.gemma_expert.model]
+            num_layers = self.paligemma.config.text_config.num_hidden_layers
+
+            # Check if gradient checkpointing is enabled for any of the models
+            use_gradient_checkpointing = (
+                hasattr(self.gemma_expert.model, "gradient_checkpointing")
+                and self.gemma_expert.model.gradient_checkpointing
+                and self.training
+            ) or (hasattr(self, "gradient_checkpointing") and self.gradient_checkpointing and self.training)
+
+            # Process all layers with gradient checkpointing if enabled
+            for layer_idx in range(num_layers):
+                if use_gradient_checkpointing:
+                    inputs_embeds = torch.utils.checkpoint.checkpoint(
+                        compute_layer_complete,
+                        layer_idx,
+                        inputs_embeds,
+                        attention_mask,
+                        position_ids,
+                        adarms_cond,
+                        use_reentrant=False,
+                        preserve_rng_state=False,
+                        paligemma=self.paligemma,
+                        gemma_expert=self.gemma_expert,
+                    )
+                else:
+                    inputs_embeds = compute_layer_complete(
+                        layer_idx,
+                        inputs_embeds,
+                        attention_mask,
+                        position_ids,
+                        adarms_cond,
+                        paligemma=self.paligemma,
+                        gemma_expert=self.gemma_expert,
+                    )
+
+            # final norm
+            def compute_final_norms(inputs_embeds, adarms_cond):
+                outputs_embeds = []
+                for i, hidden_states in enumerate(inputs_embeds):
+                    out_emb, _ = layernorm_forward(models[i].norm, hidden_states, adarms_cond[i])
+                    outputs_embeds.append(out_emb)
+                return outputs_embeds
+
+            # Apply gradient checkpointing to final norm if enabled
+            if use_gradient_checkpointing:
+                outputs_embeds = torch.utils.checkpoint.checkpoint(
+                    compute_final_norms,
+                    inputs_embeds,
+                    adarms_cond,
+                    use_reentrant=False,
+                    preserve_rng_state=False,
+                )
+            else:
+                outputs_embeds = compute_final_norms(inputs_embeds, adarms_cond)
+
+            prefix_output = outputs_embeds[0]
+            suffix_output = outputs_embeds[1]
+            prefix_past_key_values = None
+
+        return [prefix_output, suffix_output], prefix_past_key_values
+
+
+class PI05Pytorch(nn.Module):  # see openpi `PI0Pytorch`
+    """Core PI05 PyTorch model."""
+
+    def __init__(self, config: PI05Config, rtc_processor: RTCProcessor | None = None):
+        super().__init__()
+        self.config = config
+        self.rtc_processor = rtc_processor
+
+        paligemma_config = get_gemma_config(config.paligemma_variant)
+        action_expert_config = get_gemma_config(config.action_expert_variant)
+
+        if config.image_resolution[0] != config.image_resolution[1]:
+            raise ValueError(
+                f"PaliGemma expects square image resolution, invalid resolution: {config.image_resolution}"
+            )
+
+        self.paligemma_with_expert = PaliGemmaWithExpertModel(
+            paligemma_config,
+            action_expert_config,
+            use_adarms=[False, True],
+            precision=config.dtype,
+            image_size=config.image_resolution[0],
+            freeze_vision_encoder=config.freeze_vision_encoder,
+            train_expert_only=config.train_expert_only,
+        )
+
+        self.action_in_proj = nn.Linear(config.max_action_dim, action_expert_config.width)
+        self.action_out_proj = nn.Linear(action_expert_config.width, config.max_action_dim)
+
+        self.time_mlp_in = nn.Linear(action_expert_config.width, action_expert_config.width)
+        self.time_mlp_out = nn.Linear(action_expert_config.width, action_expert_config.width)
+
+        # Initialize gradient checkpointing flag
+        self.gradient_checkpointing_enabled = False
+
+        # Compile model if requested
+        if config.compile_model:
+            torch.set_float32_matmul_precision("high")
+            self.sample_actions = torch.compile(self.sample_actions, mode=config.compile_mode)
+            # Also compile the main forward pass used during training
+            self.forward = torch.compile(self.forward, mode=config.compile_mode)
+
+    def gradient_checkpointing_enable(self):
+        """Enable gradient checkpointing for memory optimization."""
+        self.gradient_checkpointing_enabled = True
+        self.paligemma_with_expert.paligemma.model.language_model.gradient_checkpointing = True
+        self.paligemma_with_expert.paligemma.model.vision_tower.gradient_checkpointing = True
+        self.paligemma_with_expert.gemma_expert.model.gradient_checkpointing = True
+        logging.info("Enabled gradient checkpointing for PI05Pytorch model")
+
+    def gradient_checkpointing_disable(self):
+        """Disable gradient checkpointing."""
+        self.gradient_checkpointing_enabled = False
+        self.paligemma_with_expert.paligemma.model.language_model.gradient_checkpointing = False
+        self.paligemma_with_expert.paligemma.model.vision_tower.gradient_checkpointing = False
+        self.paligemma_with_expert.gemma_expert.model.gradient_checkpointing = False
+        logging.info("Disabled gradient checkpointing for PI05Pytorch model")
+
+    def _rtc_enabled(self):
+        return self.config.rtc_config is not None and self.config.rtc_config.enabled
+
+    def _apply_checkpoint(self, func, *args, **kwargs):
+        """Helper method to apply gradient checkpointing if enabled."""
+        if self.gradient_checkpointing_enabled and self.training:
+            return torch.utils.checkpoint.checkpoint(
+                func, *args, use_reentrant=False, preserve_rng_state=False, **kwargs
+            )
+        return func(*args, **kwargs)
+
+    def _prepare_attention_masks_4d(self, att_2d_masks):
+        """Helper method to prepare 4D attention masks for transformer."""
+        att_2d_masks_4d = att_2d_masks[:, None, :, :]
+        return torch.where(att_2d_masks_4d, 0.0, OPENPI_ATTENTION_MASK_VALUE)
+
+    def sample_noise(self, shape, device):
+        return torch.normal(
+            mean=0.0,
+            std=1.0,
+            size=shape,
+            dtype=torch.float32,
+            device=device,
+        )
+
+    def sample_time(self, bsize, device):
+        time_beta = sample_beta(
+            self.config.time_sampling_beta_alpha, self.config.time_sampling_beta_beta, bsize, device
+        )
+        time = time_beta * self.config.time_sampling_scale + self.config.time_sampling_offset
+        return time.to(dtype=torch.float32, device=device)
+
+    def embed_prefix(
+        self, images, img_masks, tokens, masks
+    ) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
+        """Embed images with SigLIP and language tokens with embedding layer."""
+        embs = []
+        pad_masks = []
+        att_masks = []
+
+        # Process images
+        for img, img_mask in zip(images, img_masks, strict=True):
+
+            def image_embed_func(img):
+                return self.paligemma_with_expert.embed_image(img)
+
+            img_emb = self._apply_checkpoint(image_embed_func, img)
+            bsize, num_img_embs = img_emb.shape[:2]
+
+            embs.append(img_emb)
+            pad_masks.append(img_mask[:, None].expand(bsize, num_img_embs))
+            att_masks += [0] * num_img_embs
+
+        # Process language tokens
+        def lang_embed_func(tokens):
+            lang_emb = self.paligemma_with_expert.embed_language_tokens(tokens)
+            lang_emb_dim = lang_emb.shape[-1]
+            return lang_emb * math.sqrt(lang_emb_dim)
+
+        lang_emb = self._apply_checkpoint(lang_embed_func, tokens)
+        embs.append(lang_emb)
+        pad_masks.append(masks)
+
+        num_lang_embs = lang_emb.shape[1]
+        att_masks += [0] * num_lang_embs
+
+        embs = torch.cat(embs, dim=1)
+        pad_masks = torch.cat(pad_masks, dim=1)
+        att_masks = torch.tensor(att_masks, dtype=torch.bool, device=pad_masks.device)
+
+        bsize = pad_masks.shape[0]
+        att_masks = att_masks[None, :].expand(bsize, len(att_masks))
+
+        return embs, pad_masks, att_masks
+
+    def embed_suffix(self, noisy_actions, timestep):
+        """Embed noisy_actions, timestep to prepare for Expert Gemma processing."""
+        embs = []
+        pad_masks = []
+        att_masks = []
+
+        # Embed timestep using sine-cosine positional encoding
+        time_emb = create_sinusoidal_pos_embedding(
+            timestep,
+            self.action_in_proj.out_features,
+            min_period=self.config.min_period,
+            max_period=self.config.max_period,
+            device=timestep.device,
+        )
+        time_emb = time_emb.type(dtype=timestep.dtype)
+
+        # Fuse timestep + action information using an MLP
+        def action_proj_func(noisy_actions):
+            return self.action_in_proj(noisy_actions)
+
+        action_emb = self._apply_checkpoint(action_proj_func, noisy_actions)
+
+        def time_mlp_func(time_emb):
+            x = self.time_mlp_in(time_emb)
+            x = F.silu(x)
+            x = self.time_mlp_out(x)
+            return F.silu(x)
+
+        time_emb = self._apply_checkpoint(time_mlp_func, time_emb)
+        action_time_emb = action_emb
+        adarms_cond = time_emb
+
+        embs.append(action_time_emb)
+        bsize, action_time_dim = action_time_emb.shape[:2]
+        action_time_mask = torch.ones(bsize, action_time_dim, dtype=torch.bool, device=timestep.device)
+        pad_masks.append(action_time_mask)
+
+        # Set attention masks so that image, language and state inputs do not attend to action tokens
+        att_masks += [1] + ([0] * (self.config.chunk_size - 1))
+
+        embs = torch.cat(embs, dim=1)
+        pad_masks = torch.cat(pad_masks, dim=1)
+        att_masks = torch.tensor(att_masks, dtype=embs.dtype, device=embs.device)
+        att_masks = att_masks[None, :].expand(bsize, len(att_masks))
+
+        return embs, pad_masks, att_masks, adarms_cond
+
+    def forward(self, images, img_masks, tokens, masks, actions, noise=None, time=None) -> Tensor:
+        """Do a full training forward pass and compute the loss."""
+        if noise is None:
+            noise = self.sample_noise(actions.shape, actions.device)
+
+        if time is None:
+            time = self.sample_time(actions.shape[0], actions.device)
+
+        time_expanded = time[:, None, None]
+        x_t = time_expanded * noise + (1 - time_expanded) * actions
+        u_t = noise - actions
+
+        prefix_embs, prefix_pad_masks, prefix_att_masks = self.embed_prefix(images, img_masks, tokens, masks)
+        suffix_embs, suffix_pad_masks, suffix_att_masks, adarms_cond = self.embed_suffix(x_t, time)
+
+        if (
+            self.paligemma_with_expert.paligemma.model.language_model.layers[0].self_attn.q_proj.weight.dtype
+            == torch.bfloat16
+        ):
+            suffix_embs = suffix_embs.to(dtype=torch.bfloat16)
+            prefix_embs = prefix_embs.to(dtype=torch.bfloat16)
+
+        pad_masks = torch.cat([prefix_pad_masks, suffix_pad_masks], dim=1)
+        att_masks = torch.cat([prefix_att_masks, suffix_att_masks], dim=1)
+
+        att_2d_masks = make_att_2d_masks(pad_masks, att_masks)
+        position_ids = torch.cumsum(pad_masks, dim=1) - 1
+
+        att_2d_masks_4d = self._prepare_attention_masks_4d(att_2d_masks)
+
+        def forward_func(prefix_embs, suffix_embs, att_2d_masks_4d, position_ids, adarms_cond):
+            (_, suffix_out), _ = self.paligemma_with_expert.forward(
+                attention_mask=att_2d_masks_4d,
+                position_ids=position_ids,
+                past_key_values=None,
+                inputs_embeds=[prefix_embs, suffix_embs],
+                use_cache=False,
+                adarms_cond=[None, adarms_cond],
+            )
+            return suffix_out
+
+        suffix_out = self._apply_checkpoint(
+            forward_func, prefix_embs, suffix_embs, att_2d_masks_4d, position_ids, adarms_cond
+        )
+
+        suffix_out = suffix_out[:, -self.config.chunk_size :]
+        suffix_out = suffix_out.to(dtype=torch.float32)
+
+        def action_out_proj_func(suffix_out):
+            return self.action_out_proj(suffix_out)
+
+        v_t = self._apply_checkpoint(action_out_proj_func, suffix_out)
+
+        return F.mse_loss(u_t, v_t, reduction="none")
+
+    @torch.no_grad()  # see openpi `sample_actions` (slightly adapted)
+    def sample_actions(
+        self,
+        images,
+        img_masks,
+        tokens,
+        masks,
+        noise=None,
+        num_steps=None,
+        **kwargs: Unpack[ActionSelectKwargs],
+    ) -> Tensor:
+        """Do a full inference forward and compute the action."""
+        if num_steps is None:
+            num_steps = self.config.num_inference_steps
+
+        bsize = tokens.shape[0]
+        device = tokens.device
+
+        if noise is None:
+            # Sample noise with padded dimension as expected by action_in_proj
+            actions_shape = (
+                bsize,
+                self.config.chunk_size,
+                self.config.max_action_dim,
+            )  # Use config max_action_dim for internal processing
+            noise = self.sample_noise(actions_shape, device)
+
+        prefix_embs, prefix_pad_masks, prefix_att_masks = self.embed_prefix(images, img_masks, tokens, masks)
+        prefix_att_2d_masks = make_att_2d_masks(prefix_pad_masks, prefix_att_masks)
+        prefix_position_ids = torch.cumsum(prefix_pad_masks, dim=1) - 1
+
+        prefix_att_2d_masks_4d = self._prepare_attention_masks_4d(prefix_att_2d_masks)
+        self.paligemma_with_expert.paligemma.model.language_model.config._attn_implementation = "eager"  # noqa: SLF001
+
+        _, past_key_values = self.paligemma_with_expert.forward(
+            attention_mask=prefix_att_2d_masks_4d,
+            position_ids=prefix_position_ids,
+            past_key_values=None,
+            inputs_embeds=[prefix_embs, None],
+            use_cache=True,
+        )
+
+        dt = -1.0 / num_steps
+
+        x_t = noise
+        for step in range(num_steps):
+            time = 1.0 + step * dt
+            time_tensor = torch.tensor(time, dtype=torch.float32, device=device).expand(bsize)
+
+            def denoise_step_partial_call(input_x_t, current_timestep=time_tensor):
+                return self.denoise_step(
+                    prefix_pad_masks=prefix_pad_masks,
+                    past_key_values=past_key_values,
+                    x_t=input_x_t,
+                    timestep=current_timestep,
+                )
+
+            if self._rtc_enabled():
+                inference_delay = kwargs.get("inference_delay")
+                prev_chunk_left_over = kwargs.get("prev_chunk_left_over")
+                execution_horizon = kwargs.get("execution_horizon")
+
+                v_t = self.rtc_processor.denoise_step(
+                    x_t=x_t,
+                    prev_chunk_left_over=prev_chunk_left_over,
+                    inference_delay=inference_delay,
+                    time=time,
+                    original_denoise_step_partial=denoise_step_partial_call,
+                    execution_horizon=execution_horizon,
+                )
+            else:
+                v_t = denoise_step_partial_call(x_t)
+
+            x_t = x_t + dt * v_t
+
+            if self.rtc_processor is not None and self.rtc_processor.is_debug_enabled():
+                self.rtc_processor.track(time=time, x_t=x_t, v_t=v_t)
+
+        return x_t
+
+    def denoise_step(
+        self,
+        prefix_pad_masks,
+        past_key_values,
+        x_t,
+        timestep,
+    ):
+        """Apply one denoising step of the noise `x_t` at a given timestep."""
+        suffix_embs, suffix_pad_masks, suffix_att_masks, adarms_cond = self.embed_suffix(x_t, timestep)
+
+        suffix_len = suffix_pad_masks.shape[1]
+        batch_size = prefix_pad_masks.shape[0]
+        prefix_len = prefix_pad_masks.shape[1]
+
+        prefix_pad_2d_masks = prefix_pad_masks[:, None, :].expand(batch_size, suffix_len, prefix_len)
+        suffix_att_2d_masks = make_att_2d_masks(suffix_pad_masks, suffix_att_masks)
+        full_att_2d_masks = torch.cat([prefix_pad_2d_masks, suffix_att_2d_masks], dim=2)
+
+        prefix_offsets = torch.sum(prefix_pad_masks, dim=-1)[:, None]
+        position_ids = prefix_offsets + torch.cumsum(suffix_pad_masks, dim=1) - 1
+
+        full_att_2d_masks_4d = self._prepare_attention_masks_4d(full_att_2d_masks)
+        self.paligemma_with_expert.gemma_expert.model.config._attn_implementation = "eager"  # noqa: SLF001
+
+        past_key_values = copy.deepcopy(past_key_values)
+        outputs_embeds, _ = self.paligemma_with_expert.forward(
+            attention_mask=full_att_2d_masks_4d,
+            position_ids=position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=[None, suffix_embs],
+            use_cache=False,
+            adarms_cond=[None, adarms_cond],
+        )
+
+        suffix_out = outputs_embeds[1]
+        suffix_out = suffix_out[:, -self.config.chunk_size :]
+        suffix_out = suffix_out.to(dtype=torch.float32)
+        return self.action_out_proj(suffix_out)
+
+
+class PI05Policy(PreTrainedPolicy):
+    """PI05 Policy for LeRobot."""
+
+    config_class = PI05Config
+    name = "pi05"
+
+    def __init__(
+        self,
+        config: PI05Config,
+        **kwargs,
+    ):
+        """
+        Args:
+            config: Policy configuration class instance.
+        """
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        # Initialize the core PI05 model
+        self.init_rtc_processor()
+        self.model = PI05Pytorch(config, rtc_processor=self.rtc_processor)
+
+        # Enable gradient checkpointing if requested
+        if config.gradient_checkpointing:
+            self.model.gradient_checkpointing_enable()
+
+        self.model.to(config.device)
+
+        self.reset()
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type[T],
+        pretrained_name_or_path: str | Path,
+        *,
+        config: PreTrainedConfig | None = None,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        strict: bool = True,
+        **kwargs,
+    ) -> T:
+        """Override the from_pretrained method to handle key remapping and display important disclaimer."""
+        print(
+            "The PI05 model is a direct port of the OpenPI implementation. \n"
+            "This implementation follows the original OpenPI structure for compatibility. \n"
+            "Original implementation: https://github.com/Physical-Intelligence/openpi"
+        )
+        if pretrained_name_or_path is None:
+            raise ValueError("pretrained_name_or_path is required")
+
+        # Use provided config if available, otherwise create default config
+        if config is None:
+            config = PreTrainedConfig.from_pretrained(
+                pretrained_name_or_path=pretrained_name_or_path,
+                force_download=force_download,
+                resume_download=resume_download,
+                proxies=proxies,
+                token=token,
+                cache_dir=cache_dir,
+                local_files_only=local_files_only,
+                revision=revision,
+                **kwargs,
+            )
+
+        # Initialize model without loading weights
+        # Check if dataset_stats were provided in kwargs
+        model = cls(config, **kwargs)
+
+        # Load state dict (expects keys with "model." prefix)
+        try:
+            print(f"Loading model from: {pretrained_name_or_path}")
+            try:
+                from transformers.utils import cached_file
+
+                resolved_file = cached_file(
+                    pretrained_name_or_path,
+                    "model.safetensors",
+                    cache_dir=kwargs.get("cache_dir"),
+                    force_download=kwargs.get("force_download", False),
+                    resume_download=kwargs.get("resume_download"),
+                    proxies=kwargs.get("proxies"),
+                    token=kwargs.get("token"),
+                    revision=kwargs.get("revision"),
+                    local_files_only=kwargs.get("local_files_only", False),
+                )
+                from safetensors.torch import load_file
+
+                original_state_dict = load_file(resolved_file)
+                print("✓ Loaded state dict from model.safetensors")
+            except Exception as e:
+                print(f"Could not load state dict from remote files: {e}")
+                print("Returning model without loading pretrained weights")
+                return model
+
+            # First, fix any key differences (see openpi model.py, _fix_pytorch_state_dict_keys)
+            fixed_state_dict = model._fix_pytorch_state_dict_keys(original_state_dict, model.config)
+
+            # Then add "model." prefix for all keys that don't already have it
+            remapped_state_dict = {}
+            remap_count = 0
+
+            for key, value in fixed_state_dict.items():
+                if not key.startswith("model."):
+                    new_key = f"model.{key}"
+                    remapped_state_dict[new_key] = value
+                    remap_count += 1
+                else:
+                    remapped_state_dict[key] = value
+
+            if remap_count > 0:
+                print(f"Remapped {remap_count} state dict keys")
+
+            # Load the remapped state dict into the model
+            missing_keys, unexpected_keys = model.load_state_dict(remapped_state_dict, strict=strict)
+
+            if missing_keys:
+                print(f"Missing keys when loading state dict: {len(missing_keys)} keys")
+                if len(missing_keys) <= 5:
+                    for key in missing_keys:
+                        print(f"  - {key}")
+                else:
+                    for key in missing_keys[:5]:
+                        print(f"  - {key}")
+                    print(f"  ... and {len(missing_keys) - 5} more")
+
+            if unexpected_keys:
+                print(f"Unexpected keys when loading state dict: {len(unexpected_keys)} keys")
+                if len(unexpected_keys) <= 5:
+                    for key in unexpected_keys:
+                        print(f"  - {key}")
+                else:
+                    for key in unexpected_keys[:5]:
+                        print(f"  - {key}")
+                    print(f"  ... and {len(unexpected_keys) - 5} more")
+
+            if not missing_keys and not unexpected_keys:
+                print("All keys loaded successfully!")
+
+        except Exception as e:
+            print(f"Warning: Could not load state dict: {e}")
+
+        return model
+
+    def _fix_pytorch_state_dict_keys(
+        self, state_dict, model_config
+    ):  # see openpi `BaseModelConfig, _fix_pytorch_state_dict_keys`
+        """Fix state dict keys to match current model architecture."""
+        import re
+
+        fixed_state_dict = {}
+
+        for key, value in state_dict.items():
+            new_key = key
+
+            # Handle layer norm structure changes: .weight -> .dense.weight + .dense.bias
+            # For gemma expert layers
+            if re.match(
+                r"paligemma_with_expert\.gemma_expert\.model\.layers\.\d+\.(input_layernorm|post_attention_layernorm)\.weight",
+                key,
+            ):
+                # Check if the model actually has adaRMS enabled for the expert
+                expert_uses_adarms = getattr(
+                    self.model.paligemma_with_expert.gemma_expert.config, "use_adarms", False
+                )
+                if expert_uses_adarms:
+                    logging.warning(f"Skipping layer norm key (adaRMS mismatch): {key}")
+                    continue
+
+            if re.match(r"paligemma_with_expert\.gemma_expert\.model\.norm\.weight", key):
+                # Check if the model actually has adaRMS enabled for the expert
+                expert_uses_adarms = getattr(
+                    self.model.paligemma_with_expert.gemma_expert.config, "use_adarms", False
+                )
+                if expert_uses_adarms:
+                    logging.warning(f"Skipping norm key (adaRMS mismatch): {key}")
+                    continue
+
+            # Handle MLP naming changes for pi05
+            # pi05 model expects time_mlp_*, but checkpoint might have action_time_mlp_*
+            if key.startswith("action_time_mlp_in."):
+                new_key = key.replace("action_time_mlp_in.", "time_mlp_in.")
+            elif key.startswith("action_time_mlp_out."):
+                new_key = key.replace("action_time_mlp_out.", "time_mlp_out.")
+            # Also handle state_proj which shouldn't exist in pi05
+            if key.startswith("state_proj."):
+                logging.warning(f"Skipping state_proj key in pi05 mode: {key}")
+                continue
+
+            # Handle vision tower embedding layer potential differences
+            if "patch_embedding" in key:
+                # Some checkpoints might have this, but current model expects different structure
+                logging.warning(f"Vision embedding key might need handling: {key}")
+
+            if (
+                key == "model.paligemma_with_expert.paligemma.lm_head.weight"
+                or key == "paligemma_with_expert.paligemma.lm_head.weight"
+            ):
+                fixed_state_dict[
+                    "model.paligemma_with_expert.paligemma.model.language_model.embed_tokens.weight"
+                ] = value.clone()
+
+            fixed_state_dict[new_key] = value
+
+        return fixed_state_dict
+
+    def get_optim_params(self) -> dict:
+        return self.parameters()
+
+    def reset(self):
+        """Reset internal state - called when environment resets."""
+        self._action_queue = deque(maxlen=self.config.n_action_steps)
+        self._queues = {
+            ACTION: deque(maxlen=self.config.n_action_steps),
+        }
+
+    def init_rtc_processor(self):
+        """Initialize RTC processor if RTC is enabled in config."""
+        self.rtc_processor = None
+
+        # Create processor if config provided
+        # If RTC is not enabled - we can still track the denoising data
+        if self.config.rtc_config is not None:
+            self.rtc_processor = RTCProcessor(self.config.rtc_config)
+
+            model_value = getattr(self, "model", None)
+            if model_value is not None:
+                model_value.rtc_processor = self.rtc_processor
+
+    def _rtc_enabled(self) -> bool:
+        return self.config.rtc_config is not None and self.config.rtc_config.enabled
+
+    def _preprocess_images(self, batch: dict[str, Tensor]) -> tuple[list[Tensor], list[Tensor]]:
+        """Preprocess images for the model.
+
+        Images from LeRobot are typically in [B, C, H, W] format and normalized to [0, 1].
+        PaliGemma expects images in [B, C, H, W] format and normalized to [-1, 1].
+        """
+        images = []
+        img_masks = []
+
+        # Get device from model parameters
+        device = next(self.parameters()).device
+
+        present_img_keys = [key for key in self.config.image_features if key in batch]
+        missing_img_keys = [key for key in self.config.image_features if key not in batch]
+
+        if len(present_img_keys) == 0:
+            raise ValueError(
+                f"All image features are missing from the batch. At least one expected. "
+                f"(batch: {batch.keys()}) (image_features: {self.config.image_features})"
+            )
+
+        # Preprocess image features present in the batch
+        for key in present_img_keys:
+            img = batch[key]
+
+            # Ensure tensor is on the same device as the model
+            if img.device != device:
+                img = img.to(device)
+
+            # Ensure float32 dtype for consistency
+            if img.dtype != torch.float32:
+                img = img.to(torch.float32)
+
+            # from openpi preprocess_observation_pytorch: Handle both [B, C, H, W] and [B, H, W, C] formats
+            is_channels_first = img.shape[1] == 3  # Check if channels are in dimension 1
+
+            if is_channels_first:
+                # Convert [B, C, H, W] to [B, H, W, C] for processing
+                img = img.permute(0, 2, 3, 1)
+
+            # from openpi preprocess_observation_pytorch: Resize with padding if needed
+            if img.shape[1:3] != self.config.image_resolution:
+                img = resize_with_pad_torch(img, *self.config.image_resolution)
+
+            # Normalize from [0,1] to [-1,1] as expected by siglip
+            img = img * 2.0 - 1.0
+
+            # from openpi preprocess_observation_pytorch: Convert back to [B, C, H, W] format if it was originally channels-first
+            if is_channels_first:
+                img = img.permute(0, 3, 1, 2)  # [B, H, W, C] -> [B, C, H, W]
+
+            images.append(img)
+            # Create mask (all ones for real images)
+            bsize = img.shape[0]
+            mask = torch.ones(bsize, dtype=torch.bool, device=device)
+            img_masks.append(mask)
+
+        # Create image features not present in the batch as fully 0 padded images
+        for _num_empty_cameras in range(len(missing_img_keys)):
+            img = torch.ones_like(img) * -1  # Padded with -1 for SigLIP
+            mask = torch.zeros_like(mask)  # Mask is zero for empty cameras
+            images.append(img)
+            img_masks.append(mask)
+
+        return images, img_masks
+
+    def prepare_action(self, batch):
+        """Pad action"""
+        actions = pad_vector(batch[ACTION], self.config.max_action_dim)
+        return actions
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select a single action given environment observations."""
+        assert not self._rtc_enabled(), (
+            "RTC is not supported for select_action, use it with predict_action_chunk"
+        )
+
+        self.eval()
+
+        # Action queue logic for n_action_steps > 1
+        if len(self._action_queue) == 0:
+            actions = self.predict_action_chunk(batch)[:, : self.config.n_action_steps]
+            # Transpose to get shape (n_action_steps, batch_size, action_dim)
+            self._action_queue.extend(actions.transpose(0, 1))
+
+        return self._action_queue.popleft()
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]) -> Tensor:
+        """Predict a chunk of actions given environment observations."""
+        self.eval()
+
+        # Prepare inputs
+        images, img_masks = self._preprocess_images(batch)
+        tokens, masks = batch[f"{OBS_LANGUAGE_TOKENS}"], batch[f"{OBS_LANGUAGE_ATTENTION_MASK}"]
+
+        # Sample actions using the model (pass through RTC kwargs, no separate state needed for PI05)
+        actions = self.model.sample_actions(images, img_masks, tokens, masks, **kwargs)
+
+        # Unpad actions to actual action dimension
+        original_action_dim = self.config.output_features[ACTION].shape[0]
+        actions = actions[:, :, :original_action_dim]
+
+        return actions
+
+    def forward(self, batch: dict[str, Tensor], reduction: str = "mean") -> tuple[Tensor, dict]:
+        """Run the batch through the model and compute the loss for training.
+
+        Args:
+            batch: Training batch containing observations and actions.
+            reduction: How to reduce the loss. Options:
+                - "mean": Return scalar mean loss (default, backward compatible)
+                - "none": Return per-sample losses of shape (batch_size,) for RA-BC weighting
+        """
+        # Prepare inputs
+        images, img_masks = self._preprocess_images(batch)
+        tokens, masks = batch[f"{OBS_LANGUAGE_TOKENS}"], batch[f"{OBS_LANGUAGE_ATTENTION_MASK}"]
+
+        actions = self.prepare_action(batch)
+
+        # Compute loss (no separate state needed for PI05)
+        losses = self.model.forward(images, img_masks, tokens, masks, actions)
+
+        # Truncate losses to actual action dimensions
+        original_action_dim = self.config.output_features[ACTION].shape[0]
+        losses = losses[:, :, :original_action_dim]
+
+        loss_dict = {
+            "loss_per_dim": losses.mean(dim=[0, 1]).detach().cpu().numpy().tolist(),
+        }
+
+        if reduction == "none":
+            # Return per-sample losses (B,) by averaging over time and action dims
+            per_sample_loss = losses.mean(dim=(1, 2))
+            loss_dict["loss"] = per_sample_loss.mean().item()
+            return per_sample_loss, loss_dict
+        else:
+            # Default: return scalar mean loss
+            loss = losses.mean()
+            loss_dict["loss"] = loss.item()
+            return loss, loss_dict
+
+    def _get_default_peft_targets(self) -> dict[str, any]:
+        """Return default PEFT target modules for PI0.5 fine-tuning."""
+        common_projections = (
+            "state_proj|action_in_proj|action_out_proj|action_time_mlp_in|action_time_mlp_out"
+        )
+        target_modules = rf"(.*\.gemma_expert\..*\.self_attn\.(q|v)_proj|model\.({common_projections}))"
+        return {
+            "target_modules": target_modules,
+            "modules_to_save": [],
+        }
diff --git a/lerobot/src/lerobot/policies/pi05/processor_pi05.py b/lerobot/src/lerobot/policies/pi05/processor_pi05.py
new file mode 100644
index 0000000000000000000000000000000000000000..425a85577eb01bc65efe76ad3e4bd1cb95b14987
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi05/processor_pi05.py
@@ -0,0 +1,167 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from copy import deepcopy
+from dataclasses import dataclass
+from typing import Any
+
+import numpy as np
+import torch
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.policies.pi05.configuration_pi05 import PI05Config
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    RenameObservationsProcessorStep,
+    TokenizerProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.types import EnvTransition, TransitionKey
+from lerobot.utils.constants import (
+    OBS_STATE,
+    POLICY_POSTPROCESSOR_DEFAULT_NAME,
+    POLICY_PREPROCESSOR_DEFAULT_NAME,
+)
+
+
+@ProcessorStepRegistry.register(name="pi05_prepare_state_tokenizer_processor_step")
+@dataclass
+class Pi05PrepareStateTokenizerProcessorStep(ProcessorStep):
+    """
+    Processor step to prepare the state and tokenize the language input.
+    """
+
+    max_state_dim: int = 32
+    task_key: str = "task"
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        transition = transition.copy()
+
+        state = transition.get(TransitionKey.OBSERVATION, {}).get(OBS_STATE)
+        if state is None:
+            raise ValueError("State is required for PI05")
+        tasks = transition.get(TransitionKey.COMPLEMENTARY_DATA, {}).get(self.task_key)
+        if tasks is None:
+            raise ValueError("No task found in complementary data")
+
+        # TODO: check if this necessary
+        state = deepcopy(state)
+
+        # State should already be normalized to [-1, 1] by the NormalizerProcessorStep that runs before this step
+        # Discretize into 256 bins (see openpi `PaligemmaTokenizer.tokenize()`)
+        state_np = state.cpu().numpy()
+        discretized_states = np.digitize(state_np, bins=np.linspace(-1, 1, 256 + 1)[:-1]) - 1
+
+        full_prompts = []
+        for i, task in enumerate(tasks):
+            cleaned_text = task.strip().replace("_", " ").replace("\n", " ")
+            state_str = " ".join(map(str, discretized_states[i]))
+            full_prompt = f"Task: {cleaned_text}, State: {state_str};\nAction: "
+            full_prompts.append(full_prompt)
+
+        transition[TransitionKey.COMPLEMENTARY_DATA][self.task_key] = full_prompts
+        # Normalize state to [-1, 1] range if needed (assuming it's already normalized by normalizer processor step!!)
+        # Discretize into 256 bins (see openpi `PaligemmaTokenizer.tokenize()`)
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        This step does not alter the feature definitions.
+        """
+        return features
+
+
+def make_pi05_pre_post_processors(
+    config: PI05Config,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for the PI0 policy.
+
+    The pre-processing pipeline prepares input data for the model by:
+    1. Renaming features to match pretrained configurations.
+    2. Normalizing input and output features based on dataset statistics.
+    3. Adding a batch dimension.
+    4. Appending a newline character to the task description for tokenizer compatibility.
+    5. Tokenizing the text prompt using the PaliGemma tokenizer.
+    6. Moving all data to the specified device.
+
+    The post-processing pipeline handles the model's output by:
+    1. Moving data to the CPU.
+    2. Unnormalizing the output features to their original scale.
+
+    Args:
+        config: The configuration object for the PI0 policy.
+        dataset_stats: A dictionary of statistics for normalization.
+        preprocessor_kwargs: Additional arguments for the pre-processor pipeline.
+        postprocessor_kwargs: Additional arguments for the post-processor pipeline.
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines.
+    """
+
+    # Add remaining processors
+    input_steps: list[ProcessorStep] = [
+        RenameObservationsProcessorStep(rename_map={}),  # To mimic the same processor as pretrained one
+        AddBatchDimensionProcessorStep(),
+        # NOTE: NormalizerProcessorStep MUST come before Pi05PrepareStateTokenizerProcessorStep
+        # because the tokenizer step expects normalized state in [-1, 1] range for discretization
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+        Pi05PrepareStateTokenizerProcessorStep(max_state_dim=config.max_state_dim),
+        TokenizerProcessorStep(
+            tokenizer_name="google/paligemma-3b-pt-224",
+            max_length=config.tokenizer_max_length,
+            padding_side="right",
+            padding="max_length",
+        ),
+        DeviceProcessorStep(device=config.device),
+    ]
+
+    output_steps: list[ProcessorStep] = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/pi0_fast/__init__.py b/lerobot/src/lerobot/policies/pi0_fast/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..a0277da0fc12dc964bf0c978b8f0935e88b1eb54
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi0_fast/__init__.py
@@ -0,0 +1,21 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configuration_pi0_fast import PI0FastConfig
+from .modeling_pi0_fast import PI0FastPolicy
+from .processor_pi0_fast import make_pi0_fast_pre_post_processors
+
+__all__ = ["PI0FastConfig", "PI0FastPolicy", "make_pi0_fast_pre_post_processors"]
diff --git a/lerobot/src/lerobot/policies/pi0_fast/configuration_pi0_fast.py b/lerobot/src/lerobot/policies/pi0_fast/configuration_pi0_fast.py
new file mode 100644
index 0000000000000000000000000000000000000000..e125228334af5c7f7011d4ba7c1bfdb64eae1da4
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi0_fast/configuration_pi0_fast.py
@@ -0,0 +1,162 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.optim.optimizers import AdamWConfig
+from lerobot.optim.schedulers import CosineDecayWithWarmupSchedulerConfig
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+from lerobot.utils.constants import ACTION, OBS_IMAGES, OBS_STATE
+
+DEFAULT_IMAGE_SIZE = 224
+
+
+@PreTrainedConfig.register_subclass("pi0_fast")
+@dataclass
+class PI0FastConfig(PreTrainedConfig):
+    paligemma_variant: str = "gemma_2b"
+    action_expert_variant: str = "gemma_300m"
+    dtype: str = "float32"  # Options: "bfloat16", "float32"
+
+    chunk_size: int = 50  # Number of action steps to predict, in openpi called "action_horizon"
+    n_action_steps: int = 50  # Number of action steps to execute
+
+    # Shorter state and action vectors will be padded to these dimensions
+    max_state_dim: int = 32
+    max_action_dim: int = 32
+    max_action_tokens: int = 256
+
+    # Real-Time Chunking (RTC) configuration
+    rtc_config: RTCConfig | None = None
+
+    image_resolution: tuple[int, int] = (
+        DEFAULT_IMAGE_SIZE,
+        DEFAULT_IMAGE_SIZE,
+    )  # see openpi `preprocessing_pytorch.py`
+
+    # Add empty images. Used to add empty cameras when no image features are present.
+    empty_cameras: int = 0
+
+    tokenizer_max_length: int = 200  # see openpi `__post_init__`
+    text_tokenizer_name: str = "google/paligemma-3b-pt-224"
+    action_tokenizer_name: str = "lerobot/fast-action-tokenizer"
+    temperature: float = 0.0
+    max_decoding_steps: int = 256
+    fast_skip_tokens: int = 128
+
+    # Whether to validate that decoded action tokens start with "Action: " prefix
+    validate_action_token_prefix: bool = True
+
+    # Whether to use KV cache for faster autoregressive decoding
+    use_kv_cache: bool = True
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.MEAN_STD,  # Pi0Fast uses quantiles for state
+            "ACTION": NormalizationMode.MEAN_STD,  # Pi0Fast uses quantiles for action
+        }
+    )
+
+    # Training settings
+    gradient_checkpointing: bool = False  # Enable gradient checkpointing for memory optimization
+    compile_model: bool = False  # Whether to use torch.compile for model optimization
+    compile_mode: str = "max-autotune"  # Torch compile mode
+    device: str | None = None  # Device to use for the model (None = auto-detect)
+
+    # Optimizer settings: see openpi `AdamW`
+    optimizer_lr: float = 2.5e-5  # see openpi `CosineDecaySchedule: peak_lr`
+    optimizer_betas: tuple[float, float] = (0.9, 0.95)
+    optimizer_eps: float = 1e-8
+    optimizer_weight_decay: float = 0.01
+    optimizer_grad_clip_norm: float = 1.0
+
+    # Scheduler settings: see openpi `CosineDecaySchedule`
+    # Note: These will auto-scale if --steps < scheduler_decay_steps
+    # For example, --steps=3000 will scale warmup to 100 and decay to 3000
+    scheduler_warmup_steps: int = 1_000
+    scheduler_decay_steps: int = 30_000
+    scheduler_decay_lr: float = 2.5e-6
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        # Validate configuration
+        if self.n_action_steps > self.chunk_size:
+            raise ValueError(
+                f"n_action_steps ({self.n_action_steps}) cannot be greater than chunk_size ({self.chunk_size})"
+            )
+
+        if self.paligemma_variant not in ["gemma_300m", "gemma_2b"]:
+            raise ValueError(f"Invalid paligemma_variant: {self.paligemma_variant}")
+
+        if self.dtype not in ["bfloat16", "float32"]:
+            raise ValueError(f"Invalid dtype: {self.dtype}")
+
+    def validate_features(self) -> None:
+        """Validate and set up input/output features."""
+        for i in range(self.empty_cameras):
+            key = OBS_IMAGES + f".empty_camera_{i}"
+            empty_camera = PolicyFeature(
+                type=FeatureType.VISUAL,
+                shape=(3, *self.image_resolution),  # Use configured image resolution
+            )
+            self.input_features[key] = empty_camera
+
+        if OBS_STATE not in self.input_features:
+            state_feature = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(self.max_state_dim,),  # Padded to max_state_dim
+            )
+            self.input_features[OBS_STATE] = state_feature
+
+        if ACTION not in self.output_features:
+            action_feature = PolicyFeature(
+                type=FeatureType.ACTION,
+                shape=(self.max_action_dim,),  # Padded to max_action_dim
+            )
+            self.output_features[ACTION] = action_feature
+
+    def get_optimizer_preset(self) -> AdamWConfig:
+        return AdamWConfig(
+            lr=self.optimizer_lr,
+            betas=self.optimizer_betas,
+            eps=self.optimizer_eps,
+            weight_decay=self.optimizer_weight_decay,
+            grad_clip_norm=self.optimizer_grad_clip_norm,
+        )
+
+    def get_scheduler_preset(self):
+        return CosineDecayWithWarmupSchedulerConfig(
+            peak_lr=self.optimizer_lr,
+            decay_lr=self.scheduler_decay_lr,
+            num_warmup_steps=self.scheduler_warmup_steps,
+            num_decay_steps=self.scheduler_decay_steps,
+        )
+
+    @property
+    def observation_delta_indices(self) -> None:
+        return None
+
+    @property
+    def action_delta_indices(self) -> list:
+        return list(range(self.chunk_size))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/pi0_fast/modeling_pi0_fast.py b/lerobot/src/lerobot/policies/pi0_fast/modeling_pi0_fast.py
new file mode 100644
index 0000000000000000000000000000000000000000..1bcf9794c1043b6983607e1e673844692f37e104
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi0_fast/modeling_pi0_fast.py
@@ -0,0 +1,1359 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import builtins
+import logging
+import math
+from collections import deque
+from pathlib import Path
+from typing import TYPE_CHECKING, Literal, TypedDict, Unpack
+
+import numpy as np
+import torch
+import torch.nn.functional as F  # noqa: N812
+from torch import Tensor, nn
+
+from lerobot.utils.import_utils import _scipy_available, _transformers_available
+
+# Conditional import for type checking and lazy loading
+if TYPE_CHECKING or _scipy_available:
+    from scipy.fftpack import idct
+else:
+    idct = None
+
+if TYPE_CHECKING or _transformers_available:
+    from transformers import AutoTokenizer
+    from transformers.models.auto import CONFIG_MAPPING
+
+    from lerobot.policies.pi_gemma import (
+        PaliGemmaForConditionalGenerationWithPiGemma,
+        PiGemmaModel,
+    )
+else:
+    CONFIG_MAPPING = None
+    AutoTokenizer = None
+    PiGemmaModel = None
+    PaliGemmaForConditionalGenerationWithPiGemma = None
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.policies.pi0_fast.configuration_pi0_fast import PI0FastConfig
+from lerobot.policies.pretrained import PreTrainedPolicy, T
+from lerobot.policies.rtc.modeling_rtc import RTCProcessor
+from lerobot.utils.constants import (
+    ACTION,
+    ACTION_TOKEN_MASK,
+    ACTION_TOKENS,
+    OBS_LANGUAGE_ATTENTION_MASK,
+    OBS_LANGUAGE_TOKENS,
+    OPENPI_ATTENTION_MASK_VALUE,
+)
+
+
+class ActionSelectKwargs(TypedDict, total=False):
+    temperature: float | None
+
+
+def pad_vector(vector, new_dim):
+    """Pad the last dimension of a vector to new_dim with zeros.
+
+    Can be (batch_size x sequence_length x features_dimension)
+    or (batch_size x features_dimension)
+    """
+    if vector.shape[-1] >= new_dim:
+        return vector
+    return F.pad(vector, (0, new_dim - vector.shape[-1]))
+
+
+def resize_with_pad_torch(  # see openpi `resize_with_pad_torch` (exact copy)
+    images: torch.Tensor,
+    height: int,
+    width: int,
+    mode: str = "bilinear",
+) -> torch.Tensor:
+    """PyTorch version of resize_with_pad. Resizes an image to a target height and width without distortion
+    by padding with black. If the image is float32, it must be in the range [-1, 1].
+
+    Args:
+        images: Tensor of shape [*b, h, w, c] or [*b, c, h, w]
+        height: Target height
+        width: Target width
+        mode: Interpolation mode ('bilinear', 'nearest', etc.)
+
+    Returns:
+        Resized and padded tensor with same shape format as input
+    """
+    # Check if input is in channels-last format [*b, h, w, c] or channels-first [*b, c, h, w]
+    if images.shape[-1] <= 4:  # Assume channels-last format
+        channels_last = True
+        if images.dim() == 3:
+            images = images.unsqueeze(0)  # Add batch dimension
+        images = images.permute(0, 3, 1, 2)  # [b, h, w, c] -> [b, c, h, w]
+    else:
+        channels_last = False
+        if images.dim() == 3:
+            images = images.unsqueeze(0)  # Add batch dimension
+
+    batch_size, channels, cur_height, cur_width = images.shape
+
+    # Calculate resize ratio
+    ratio = max(cur_width / width, cur_height / height)
+    resized_height = int(cur_height / ratio)
+    resized_width = int(cur_width / ratio)
+
+    # Resize
+    resized_images = F.interpolate(
+        images,
+        size=(resized_height, resized_width),
+        mode=mode,
+        align_corners=False if mode == "bilinear" else None,
+    )
+
+    # Handle dtype-specific clipping
+    if images.dtype == torch.uint8:
+        resized_images = torch.round(resized_images).clamp(0, 255).to(torch.uint8)
+    elif images.dtype == torch.float32:
+        resized_images = resized_images.clamp(0.0, 1.0)
+    else:
+        raise ValueError(f"Unsupported image dtype: {images.dtype}")
+
+    # Calculate padding
+    pad_h0, remainder_h = divmod(height - resized_height, 2)
+    pad_h1 = pad_h0 + remainder_h
+    pad_w0, remainder_w = divmod(width - resized_width, 2)
+    pad_w1 = pad_w0 + remainder_w
+
+    # Pad
+    constant_value = 0 if images.dtype == torch.uint8 else 0.0
+    padded_images = F.pad(
+        resized_images,
+        (pad_w0, pad_w1, pad_h0, pad_h1),  # left, right, top, bottom
+        mode="constant",
+        value=constant_value,
+    )
+
+    # Convert back to original format if needed
+    if channels_last:
+        padded_images = padded_images.permute(0, 2, 3, 1)  # [b, c, h, w] -> [b, h, w, c]
+
+    return padded_images
+
+
+class GemmaConfig:  # see openpi `gemma.py: Config`
+    """Configuration for Gemma model variants."""
+
+    def __init__(self, width, depth, mlp_dim, num_heads, num_kv_heads, head_dim):
+        self.width = width
+        self.depth = depth
+        self.mlp_dim = mlp_dim
+        self.num_heads = num_heads
+        self.num_kv_heads = num_kv_heads
+        self.head_dim = head_dim
+
+
+def get_gemma_config(variant: str) -> GemmaConfig:  # see openpi `gemma.py: get_config`
+    """Returns config for specified gemma variant."""
+    if variant == "gemma_300m":
+        return GemmaConfig(
+            width=1024,
+            depth=18,
+            mlp_dim=4096,
+            num_heads=8,
+            num_kv_heads=1,
+            head_dim=256,
+        )
+    elif variant == "gemma_2b":
+        return GemmaConfig(
+            width=2048,
+            depth=18,
+            mlp_dim=16_384,
+            num_heads=8,
+            num_kv_heads=1,
+            head_dim=256,
+        )
+    else:
+        raise ValueError(f"Unknown variant: {variant}")
+
+
+class PI0FastPaliGemma(nn.Module):
+    """PaliGemma model for PI0Fast"""
+
+    def __init__(
+        self,
+        vlm_config,
+        use_adarms=None,
+        precision: Literal["bfloat16", "float32"] = "bfloat16",
+    ):
+        if use_adarms is None:
+            use_adarms = [False, False]
+        super().__init__()
+
+        vlm_config_hf = CONFIG_MAPPING["paligemma"]()
+        vlm_config_hf._vocab_size = 257152  # noqa: SLF001
+        vlm_config_hf.image_token_index = 257152
+        vlm_config_hf.text_config.hidden_size = vlm_config.width
+        vlm_config_hf.text_config.intermediate_size = vlm_config.mlp_dim
+        vlm_config_hf.text_config.num_attention_heads = vlm_config.num_heads
+        vlm_config_hf.text_config.head_dim = vlm_config.head_dim
+        vlm_config_hf.text_config.num_hidden_layers = vlm_config.depth
+        vlm_config_hf.text_config.num_key_value_heads = vlm_config.num_kv_heads
+        vlm_config_hf.text_config.hidden_activation = "gelu_pytorch_tanh"
+        vlm_config_hf.text_config.dtype = "float32"
+        vlm_config_hf.text_config.vocab_size = 257152
+        vlm_config_hf.text_config.use_adarms = use_adarms[0]
+        vlm_config_hf.text_config.adarms_cond_dim = vlm_config.width if use_adarms[0] else None
+        vlm_config_hf.vision_config.intermediate_size = 4304
+        vlm_config_hf.vision_config.projection_dim = 2048
+        vlm_config_hf.vision_config.projector_hidden_act = "gelu_fast"
+        vlm_config_hf.vision_config.dtype = "float32"
+
+        self.paligemma = PaliGemmaForConditionalGenerationWithPiGemma(config=vlm_config_hf)
+
+        # Use PI Gemma (AdaRMS) as language model when use_adarms[0] is True so that
+        # forward(..., adarms_cond=...) is supported (same as pi0/pi05).
+        if use_adarms[0]:
+            text_config = self.paligemma.config.text_config
+            self.paligemma.model.language_model = PiGemmaModel(text_config)
+
+        self.to_bfloat16_for_selected_params(precision)
+
+    def to_bfloat16_for_selected_params(self, precision: Literal["bfloat16", "float32"] = "bfloat16"):
+        if precision == "bfloat16":
+            self.to(dtype=torch.bfloat16)
+        elif precision == "float32":
+            self.to(dtype=torch.float32)
+            return
+        else:
+            raise ValueError(f"Invalid precision: {precision}")
+
+        # Keep full vision path in float32 so we never toggle (toggle causes optimizer
+        # "same dtype" error). Align with PI05.
+        params_to_keep_float32 = [
+            "vision_tower",
+            "multi_modal_projector",
+            "input_layernorm",
+            "post_attention_layernorm",
+            "model.norm",
+        ]
+
+        for name, param in self.named_parameters():
+            if any(selector in name for selector in params_to_keep_float32):
+                param.data = param.data.to(dtype=torch.float32)
+
+    def embed_image(self, image: torch.Tensor):
+        # Vision tower and multi_modal_projector are kept in float32 (params_to_keep_float32). Align with PI05.
+        out_dtype = image.dtype
+        if image.dtype != torch.float32:
+            image = image.to(torch.float32)
+        image_outputs = self.paligemma.model.get_image_features(image)
+        features = image_outputs.pooler_output * self.paligemma.config.text_config.hidden_size**0.5
+        if features.dtype != out_dtype:
+            features = features.to(out_dtype)
+        return features
+
+    def embed_language_tokens(self, tokens: torch.Tensor):
+        return self.paligemma.model.language_model.embed_tokens(tokens)
+
+    def forward(
+        self,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: list[torch.FloatTensor] | None = None,
+        use_cache: bool | None = None,
+        adarms_cond: list[torch.Tensor] | None = None,
+    ):
+        if adarms_cond is None:
+            adarms_cond = [None, None]
+        if inputs_embeds[1] is None:
+            prefix_output = self.paligemma.model.language_model.forward(
+                inputs_embeds=inputs_embeds[0],
+                attention_mask=attention_mask,
+                position_ids=position_ids,
+                past_key_values=past_key_values,
+                use_cache=use_cache,
+                adarms_cond=adarms_cond[0] if adarms_cond is not None else None,
+            )
+            prefix_past_key_values = prefix_output.past_key_values
+            # prefix_output to be used for the language head
+            # shape: [batch_size, seq_len, hidden_size] with hidden_size = 2048
+            prefix_output = prefix_output.last_hidden_state
+            suffix_output = None
+        return [prefix_output, suffix_output], prefix_past_key_values
+
+
+class PI0FastPytorch(nn.Module):  # see openpi `PI0Pytorch`
+    """Core PI0Fast PyTorch model."""
+
+    def __init__(
+        self,
+        config: PI0FastConfig,
+        rtc_processor: RTCProcessor | None = None,
+        paligemma_tokenizer: "AutoTokenizer | None" = None,
+    ):
+        super().__init__()
+        self.config = config
+        self.rtc_processor = rtc_processor
+        self._paligemma_tokenizer = paligemma_tokenizer
+
+        paligemma_config = get_gemma_config(config.paligemma_variant)
+
+        self.paligemma_with_expert = PI0FastPaliGemma(
+            paligemma_config,
+            use_adarms=[False, True],
+            precision=config.dtype,
+        )
+
+        # Initialize gradient checkpointing flag
+        self.gradient_checkpointing_enabled = False
+
+        # Compile model if requested
+        if config.compile_model:
+            torch.set_float32_matmul_precision("high")
+            self.sample_actions_fast = torch.compile(self.sample_actions_fast, mode=config.compile_mode)
+            self.forward = torch.compile(self.forward, mode=config.compile_mode)
+
+    def gradient_checkpointing_enable(self):
+        """Enable gradient checkpointing for memory optimization."""
+        self.gradient_checkpointing_enabled = True
+        # Call the proper gradient_checkpointing_enable() method with use_reentrant=False for better memory efficiency
+        self.paligemma_with_expert.paligemma.model.language_model.gradient_checkpointing_enable(
+            gradient_checkpointing_kwargs={"use_reentrant": False}
+        )
+        self.paligemma_with_expert.paligemma.model.vision_tower.gradient_checkpointing_enable(
+            gradient_checkpointing_kwargs={"use_reentrant": False}
+        )
+        logging.info("Enabled gradient checkpointing for PI0FastPytorch model")
+
+    def gradient_checkpointing_disable(self):
+        """Disable gradient checkpointing."""
+        self.gradient_checkpointing_enabled = False
+        # Call the proper gradient_checkpointing_disable() method
+        self.paligemma_with_expert.paligemma.model.language_model.gradient_checkpointing_disable()
+        self.paligemma_with_expert.paligemma.model.vision_tower.gradient_checkpointing_disable()
+        logging.info("Disabled gradient checkpointing for PI0FastPytorch model")
+
+    def _apply_checkpoint(self, func, *args, **kwargs):
+        """Helper method to apply gradient checkpointing if enabled."""
+        if self.gradient_checkpointing_enabled and self.training:
+            return torch.utils.checkpoint.checkpoint(
+                func, *args, use_reentrant=False, preserve_rng_state=False, **kwargs
+            )
+        return func(*args, **kwargs)
+
+    def _prepare_attention_masks_4d(self, att_2d_masks, dtype=None):
+        """Helper method to prepare 4D attention masks for transformer."""
+        att_2d_masks_4d = att_2d_masks[:, None, :, :]
+        result = torch.where(att_2d_masks_4d, 0.0, OPENPI_ATTENTION_MASK_VALUE)
+        if dtype is not None:
+            result = result.to(dtype=dtype)
+        return result
+
+    def embed_prefix_fast(
+        self,
+        images,
+        img_masks,
+        tokens,
+        masks,
+        fast_action_tokens=None,
+        fast_action_masks=None,
+    ) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor, int, int]:
+        """Embed images, language tokens, and FAST action tokens.
+
+        Attention pattern:
+        - Images + Language: bidirectional among themselves
+        - FAST: attend to images + language, causal among themselves
+
+        Args:
+            images: List of image tensors
+            img_masks: List of image masks
+            tokens: Language instruction tokens
+            masks: Attention masks for tokens
+            fast_action_tokens: FAST action tokens (discrete token IDs)
+            fast_action_masks: Padding masks for FAST action tokens
+
+        Returns:
+            embs: Concatenated embeddings [images, tokens, fast_action_tokens]
+            pad_masks: Padding masks
+            att_masks: 2D attention mask
+            total_T_images: Total number of image tokens
+            num_fast_embs: Number of FAST action token embeddings
+        """
+        embs = []
+        pad_masks = []
+        att_mask_segments = []
+        total_t_images = 0
+        num_fast_embs = 0
+
+        # Process images
+        for img, img_mask in zip(images, img_masks, strict=True):
+
+            def image_embed_func(img):
+                return self.paligemma_with_expert.embed_image(img)
+
+            img_emb = self._apply_checkpoint(image_embed_func, img)
+            bsize, num_img_embs = img_emb.shape[:2]
+
+            embs.append(img_emb)
+            pad_masks.append(img_mask[:, None].expand(bsize, num_img_embs))
+            att_mask_segments.append(("image", num_img_embs))
+            total_t_images += num_img_embs
+
+        # Process language instruction tokens
+        def lang_embed_func(tokens):
+            lang_emb = self.paligemma_with_expert.embed_language_tokens(tokens)
+            lang_emb_dim = lang_emb.shape[-1]
+            return lang_emb * math.sqrt(lang_emb_dim)
+
+        lang_emb = self._apply_checkpoint(lang_embed_func, tokens)
+        embs.append(lang_emb)
+        pad_masks.append(masks)
+
+        num_lang_embs = lang_emb.shape[1]
+        att_mask_segments.append(("language", num_lang_embs))
+
+        # Process FAST action tokens (discrete token IDs)
+        if fast_action_tokens is not None:
+
+            def fast_action_embed_func(fast_action_tokens):
+                fast_emb = self.paligemma_with_expert.embed_language_tokens(fast_action_tokens)
+                fast_emb_dim = fast_emb.shape[-1]
+                return fast_emb * math.sqrt(fast_emb_dim)
+
+            fast_action_emb = self._apply_checkpoint(fast_action_embed_func, fast_action_tokens)
+            embs.append(fast_action_emb)
+
+            num_fast_embs = fast_action_tokens.shape[1]
+            pad_masks.append(fast_action_masks)
+            att_mask_segments.append(("fast", num_fast_embs))
+
+        embs = torch.cat(embs, dim=1)
+        pad_masks = torch.cat(pad_masks, dim=1)
+
+        # Create custom 2D attention mask:
+        # - Images + Language: bidirectional among themselves
+        # - FAST: attend to images + language, causal among themselves
+        att_masks = self._create_custom_attention_mask_fast(att_mask_segments, pad_masks, bsize)
+
+        return embs, pad_masks, att_masks, total_t_images, num_fast_embs
+
+    def _create_custom_attention_mask_fast(self, att_mask_segments, pad_masks, bsize):
+        """Create custom 2D attention mask.
+
+        Attention rules:
+        - Images + Language: bidirectional among themselves
+        - FAST: attend to images + language, causal among themselves
+        """
+        total_len = sum(length for _, length in att_mask_segments)
+        device = pad_masks.device
+
+        att_2d_masks = torch.zeros(bsize, total_len, total_len, dtype=torch.bool, device=device)
+
+        positions = []
+        current_pos = 0
+        for seg_type, seg_len in att_mask_segments:
+            positions.append((seg_type, current_pos, current_pos + seg_len))
+            current_pos += seg_len
+
+        for _i, (query_type, query_start, query_end) in enumerate(positions):
+            for _j, (key_type, key_start, key_end) in enumerate(positions):
+                # Images and Language can attend to each other bidirectionally
+                if (
+                    query_type in ["image", "language"]
+                    and key_type in ["image", "language"]
+                    or query_type == "fast"
+                    and key_type in ["image", "language"]
+                ):
+                    att_2d_masks[:, query_start:query_end, key_start:key_end] = True
+
+                # FAST tokens attend causally to themselves
+                elif query_type == "fast" and key_type == "fast":
+                    fast_len = query_end - query_start
+                    causal_mask = torch.tril(torch.ones(fast_len, fast_len, dtype=torch.bool, device=device))
+                    att_2d_masks[:, query_start:query_end, key_start:key_end] = causal_mask[None, :, :]
+
+        # Apply padding masks
+        pad_2d_masks = pad_masks[:, None, :] * pad_masks[:, :, None]
+        att_2d_masks = att_2d_masks & pad_2d_masks
+
+        return att_2d_masks
+
+    def forward(
+        self,
+        images,
+        img_masks,
+        tokens,
+        masks,
+        fast_action_tokens,
+        fast_action_masks,
+    ) -> dict:
+        """Forward pass for PI0Fast.
+
+        This implements the Pi0FAST training objective: predict next action token
+        using cross-entropy loss.
+
+        Args:
+            images: List of image tensors
+            img_masks: List of image masks
+            tokens: Language instruction tokens
+            masks: Attention masks for tokens
+            fast_action_tokens: Discrete action token IDs [B, max_action_tokens]
+            fast_action_masks: Padding masks for fast action tokens [B, max_action_tokens]
+
+        Returns:
+            Dictionary with 'fast_loss' and 'loss' keys
+        """
+        if fast_action_tokens is None or fast_action_masks is None:
+            raise ValueError("fast_action_tokens and fast_action_masks are required for FAST-only mode")
+
+        # Embed prefix with FAST tokens
+        prefix_embs, prefix_pad_masks, prefix_att_masks, total_t_images, num_fast_embs = (
+            self.embed_prefix_fast(
+                images,
+                img_masks,
+                tokens,
+                masks,
+                fast_action_tokens=fast_action_tokens,
+                fast_action_masks=fast_action_masks,
+            )
+        )
+
+        # Convert embeddings to bfloat16 if needed
+        if (
+            self.paligemma_with_expert.paligemma.model.language_model.layers[0].self_attn.q_proj.weight.dtype
+            == torch.bfloat16
+        ):
+            prefix_embs = prefix_embs.to(dtype=torch.bfloat16)
+
+        # for next-token prediction, input tokens [0:T-1] to predict tokens [1:T]
+        input_embs = prefix_embs
+        input_pad_masks = prefix_pad_masks
+        input_att_masks = prefix_att_masks
+
+        position_ids = torch.cumsum(input_pad_masks, dim=1) - 1
+        att_2d_4d = self._prepare_attention_masks_4d(input_att_masks, dtype=input_embs.dtype)
+
+        # forward pass through paligemma (language model)
+        (prefix_out, _), _ = self.paligemma_with_expert.forward(
+            attention_mask=att_2d_4d,
+            position_ids=position_ids,
+            past_key_values=None,
+            inputs_embeds=[input_embs, None],  # No suffix/action expert
+            use_cache=False,
+            adarms_cond=[None, None],
+        )
+
+        # Get logits for FAST action tokens using the FAST LM head
+        # only compute logits for the positions that predict FAST tokens
+        lm_head = self.paligemma_with_expert.paligemma.lm_head
+
+        # Targets are the FAST action tokens
+        fast_targets = fast_action_tokens  # (B, num_fast_embs)
+
+        # extract logits for FAST token prediction
+        fast_hidden = prefix_out[:, -fast_targets.shape[1] :, :]
+        fast_logits_for_pred = lm_head(fast_hidden)  # (B, num_fast_embs, gemma_vocab_size)
+
+        # Shift left for next-step prediction and shift target
+        # logits[:, i] predicts targets[:, i+1]
+        fast_logits_for_pred = fast_logits_for_pred[:, :-1, :]  # shift logits left
+        fast_targets = fast_targets[:, 1:]  # shift targets right
+        fast_action_masks = fast_action_masks[:, 1:]  # shift masks to match targets
+
+        # compute cross-entropy loss
+        loss_fct = torch.nn.CrossEntropyLoss(reduction="none")
+        fast_logits_flat = fast_logits_for_pred.reshape(-1, fast_logits_for_pred.size(-1))
+        fast_targets_flat = fast_targets.reshape(-1)
+
+        fast_loss_per_token = loss_fct(fast_logits_flat, fast_targets_flat)
+        fast_loss_per_token = fast_loss_per_token.reshape(fast_targets.shape)
+
+        # apply mask and compute mean loss
+        masked_fast_loss = fast_loss_per_token * fast_action_masks.float()
+        fast_loss = masked_fast_loss.sum() / fast_action_masks.sum().clamp(min=1)
+
+        return {
+            "ce_loss": fast_loss,
+            "loss": fast_loss,
+        }
+
+    @torch.no_grad()
+    def sample_actions_fast(
+        self,
+        images,
+        img_masks,
+        tokens,
+        masks,
+        max_decoding_steps=None,
+        temperature=0.0,
+    ) -> torch.Tensor:
+        """
+        Inefficient but safe autoregressive decoding for FAST tokens.
+        Matches the pattern of _generate_subtask_tokens.
+        TODO: jadechoghari, should we move this logic to PI0FastPolicy class?
+        """
+        if max_decoding_steps is None:
+            max_decoding_steps = self.config.max_action_tokens
+
+        bsize = tokens.shape[0]
+        device = tokens.device
+        lm_head = self.paligemma_with_expert.paligemma.lm_head
+
+        # add bos token after tokens
+        bos_token = torch.full(
+            (bsize, 1), self._paligemma_tokenizer.bos_token_id, dtype=torch.long, device=device
+        )
+        tokens = torch.cat([tokens, bos_token], dim=1)
+        masks = torch.cat([masks, torch.ones((bsize, 1), dtype=torch.bool, device=device)], dim=1)
+
+        # 1. Initial Embedding (matches training prefix)
+        # prefix_embs will include [Images, Language Prompt, BOS]
+        prefix_embs, prefix_pad_masks, prefix_att_masks, total_t_images, _ = self.embed_prefix_fast(
+            images, img_masks, tokens, masks, fast_action_tokens=None, fast_action_masks=None
+        )
+
+        if (
+            self.paligemma_with_expert.paligemma.model.language_model.layers[0].self_attn.q_proj.weight.dtype
+            == torch.bfloat16
+        ):
+            prefix_embs = prefix_embs.to(dtype=torch.bfloat16)
+
+        generated_action_tokens = torch.zeros((bsize, max_decoding_steps), dtype=torch.long, device=device)
+
+        # 2. Decoding Loop (each step re-computes full sequence)
+        for t in range(max_decoding_steps):
+            # always re-calculate position IDs from the current pad mask
+            position_ids = torch.cumsum(prefix_pad_masks, dim=1) - 1
+            att_4d = self._prepare_attention_masks_4d(prefix_att_masks, dtype=prefix_embs.dtype)
+
+            # full forward pass (no kv cache)
+            (prefix_out, _), _ = self.paligemma_with_expert.forward(
+                attention_mask=att_4d,
+                position_ids=position_ids,
+                past_key_values=None,
+                inputs_embeds=[prefix_embs, None],
+                use_cache=False,
+                adarms_cond=[None, None],
+            )
+
+            # predict next token from the very last sequence position
+            last_logits = lm_head(prefix_out[:, -1:, :])  # (B, 1, vocab_size)
+
+            if temperature > 0:
+                probs = torch.softmax(last_logits[:, -1] / temperature, dim=-1)
+                next_token = torch.multinomial(probs, num_samples=1)
+            else:
+                next_token = torch.argmax(last_logits[:, -1], dim=-1, keepdim=True)
+
+            generated_action_tokens[:, t] = next_token.squeeze(-1)
+
+            # 3. Update sequence for next iteration (unless it's the last step)
+            if t < max_decoding_steps - 1:
+                # embed the newly generated token
+                next_token_emb = self.paligemma_with_expert.embed_language_tokens(next_token)
+                next_token_emb = next_token_emb * math.sqrt(next_token_emb.shape[-1])
+                if prefix_embs.dtype == torch.bfloat16:
+                    next_token_emb = next_token_emb.to(dtype=torch.bfloat16)
+
+                # append to embeddings
+                prefix_embs = torch.cat([prefix_embs, next_token_emb], dim=1)
+
+                # update padding mask (new token is always valid/1)
+                prefix_pad_masks = torch.cat(
+                    [prefix_pad_masks, torch.ones((bsize, 1), dtype=torch.bool, device=device)], dim=1
+                )
+
+                # update 2d attention mask: grow the matrix
+                old_len = prefix_att_masks.shape[1]
+                new_len = old_len + 1
+                new_att_masks = torch.zeros((bsize, new_len, new_len), dtype=torch.bool, device=device)
+                new_att_masks[:, :old_len, :old_len] = prefix_att_masks
+                # new token attends to all non-padding tokens in the updated sequence
+                new_att_masks[:, -1, :] = prefix_pad_masks
+                prefix_att_masks = new_att_masks
+        return generated_action_tokens
+
+    @torch.no_grad()
+    def sample_actions_fast_kv_cache(
+        self,
+        images,
+        img_masks,
+        tokens,
+        masks,
+        max_decoding_steps=None,
+        temperature=0.0,
+    ) -> torch.Tensor:
+        """
+        Optimized autoregressive decoding for FAST tokens using KV Caching.
+        """
+        if max_decoding_steps is None:
+            max_decoding_steps = self.config.max_action_tokens
+
+        bsize = tokens.shape[0]
+        device = tokens.device
+        lm_head = self.paligemma_with_expert.paligemma.lm_head
+
+        # --- 1. PREFILL PHASE ---
+        # Process Images + Text Prompt + BOS token once to populate the KV cache.
+
+        # Add BOS token to the prompt
+        bos_token = torch.full(
+            (bsize, 1), self._paligemma_tokenizer.bos_token_id, dtype=torch.long, device=device
+        )
+        tokens_in = torch.cat([tokens, bos_token], dim=1)
+        masks_in = torch.cat([masks, torch.ones((bsize, 1), dtype=torch.bool, device=device)], dim=1)
+
+        # Embed prefix [Images, Language, BOS]
+        # fast_action_tokens=None means we are just embedding the condition (images+text)
+        prefix_embs, prefix_pad_masks, prefix_att_masks, total_t_images, _ = self.embed_prefix_fast(
+            images, img_masks, tokens_in, masks_in, fast_action_tokens=None, fast_action_masks=None
+        )
+
+        # Ensure correct precision (bfloat16/float32)
+        if (
+            self.paligemma_with_expert.paligemma.model.language_model.layers[0].self_attn.q_proj.weight.dtype
+            == torch.bfloat16
+        ):
+            prefix_embs = prefix_embs.to(dtype=torch.bfloat16)
+
+        # Create position IDs (cumsum of mask - 1)
+        position_ids = torch.cumsum(prefix_pad_masks, dim=1) - 1
+
+        # Create 4D mask for the prefix
+        att_4d = self._prepare_attention_masks_4d(prefix_att_masks, dtype=prefix_embs.dtype)
+
+        # Forward pass (Prefill) with use_cache=True
+        # We only pass [prefix_embs, None] because we aren't using the suffix (expert) model yet
+        (prefix_out, _), past_key_values = self.paligemma_with_expert.forward(
+            attention_mask=att_4d,
+            position_ids=position_ids,
+            past_key_values=None,
+            inputs_embeds=[prefix_embs, None],
+            use_cache=True,  # Enable caching
+            adarms_cond=[None, None],
+        )
+
+        # Sample the first action token from the last logit of the prefix
+        last_logits = lm_head(prefix_out[:, -1:, :])  # (B, 1, V)
+        if temperature > 0:
+            probs = torch.softmax(last_logits[:, -1] / temperature, dim=-1)
+            next_token = torch.multinomial(probs, num_samples=1)
+        else:
+            next_token = torch.argmax(last_logits[:, -1], dim=-1, keepdim=True)
+
+        # Initialize storage for generated tokens
+        generated_action_tokens = torch.zeros((bsize, max_decoding_steps), dtype=torch.long, device=device)
+        generated_action_tokens[:, 0] = next_token.squeeze(-1)
+
+        # Track valid tokens mask (0 for pad, 1 for valid)
+        # We need this to tell the new token what it can attend to (images + text + past actions)
+        current_pad_mask = prefix_pad_masks
+
+        # --- 2. DECODING PHASE ---
+        # Generate remaining tokens one by one using the cache.
+
+        for t in range(1, max_decoding_steps):
+            # Embed the single previous token
+            # We use embed_language_tokens directly to avoid overhead of full prefix embedding
+            next_token_emb = self.paligemma_with_expert.embed_language_tokens(next_token)
+            next_token_emb = next_token_emb * math.sqrt(next_token_emb.shape[-1])
+            if prefix_embs.dtype == torch.bfloat16:
+                next_token_emb = next_token_emb.to(dtype=torch.bfloat16)
+
+            # Update Pad Mask: append 1s for the new valid token
+            new_column = torch.ones((bsize, 1), dtype=torch.bool, device=device)
+            current_pad_mask = torch.cat([current_pad_mask, new_column], dim=1)
+
+            # Update Position IDs for the single new token
+            current_position_ids = (torch.sum(current_pad_mask, dim=1, keepdim=True) - 1).long()
+
+            # Create Attention Mask for the single new step
+            # The new token attends to all valid tokens in history (captured by current_pad_mask).
+            # Shape becomes (B, 1, 1, Total_Len) which works with HF's cache logic.
+            step_att_mask = self._prepare_attention_masks_4d(
+                current_pad_mask.unsqueeze(1), dtype=next_token_emb.dtype
+            )
+
+            # Forward pass (Decoding step)
+            # input_embeds is just the new token (B, 1, D)
+            (step_out, _), past_key_values = self.paligemma_with_expert.forward(
+                attention_mask=step_att_mask,
+                position_ids=current_position_ids,
+                past_key_values=past_key_values,  # Pass updated cache
+                inputs_embeds=[next_token_emb, None],
+                use_cache=True,
+                adarms_cond=[None, None],
+            )
+
+            # Sample next token
+            last_logits = lm_head(step_out[:, -1:, :])
+            if temperature > 0:
+                probs = torch.softmax(last_logits[:, -1] / temperature, dim=-1)
+                next_token = torch.multinomial(probs, num_samples=1)
+            else:
+                next_token = torch.argmax(last_logits[:, -1], dim=-1, keepdim=True)
+
+            generated_action_tokens[:, t] = next_token.squeeze(-1)
+
+        return generated_action_tokens
+
+
+class PI0FastPolicy(PreTrainedPolicy):
+    """PI0Fast Policy for LeRobot."""
+
+    config_class = PI0FastConfig
+    name = "pi0_fast"
+
+    def __init__(
+        self,
+        config: PI0FastConfig,
+        **kwargs,
+    ):
+        """
+        Args:
+            config: Policy configuration class instance.
+        """
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        # Load tokenizers first
+        try:
+            from transformers import AutoProcessor, AutoTokenizer
+
+            # Load FAST tokenizer
+            self.action_tokenizer = AutoProcessor.from_pretrained(
+                config.action_tokenizer_name, trust_remote_code=True
+            )
+
+            # Load PaliGemma tokenizer for token conversion
+            self._paligemma_tokenizer = AutoTokenizer.from_pretrained(
+                config.text_tokenizer_name, trust_remote_code=True, add_eos_token=True, add_bos_token=False
+            )
+
+            logging.info("Loaded FAST tokenizer for action detokenization")
+        except Exception as e:
+            logging.error(f"Failed to load FAST tokenizer for action detokenization: {e}")
+            logging.error("Tokenizer loading is required for proper policy initialization; aborting.")
+            raise RuntimeError("Failed to load required tokenizers for PI0FastPolicy initialization") from e
+
+        # Initialize the core PI0Fast model
+        self.init_rtc_processor()
+        self.model = PI0FastPytorch(
+            config, rtc_processor=self.rtc_processor, paligemma_tokenizer=self._paligemma_tokenizer
+        )
+
+        # Enable gradient checkpointing if requested
+        if config.gradient_checkpointing:
+            self.model.gradient_checkpointing_enable()
+
+        self.model.to(config.device)
+
+        self.reset()
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type[T],
+        pretrained_name_or_path: str | Path,
+        *,
+        config: PreTrainedConfig | None = None,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        strict: bool = True,
+        **kwargs,
+    ) -> T:
+        """Override the from_pretrained method to handle key remapping and display important disclaimer."""
+        print(
+            "The PI0Fast model is a direct port of the OpenPI implementation. \n"
+            "This implementation follows the original OpenPI structure for compatibility. \n"
+            "Original implementation: https://github.com/Physical-Intelligence/openpi"
+        )
+        if pretrained_name_or_path is None:
+            raise ValueError("pretrained_name_or_path is required")
+
+        # Use provided config if available, otherwise create default config
+        if config is None:
+            config = PreTrainedConfig.from_pretrained(
+                pretrained_name_or_path=pretrained_name_or_path,
+                force_download=force_download,
+                resume_download=resume_download,
+                proxies=proxies,
+                token=token,
+                cache_dir=cache_dir,
+                local_files_only=local_files_only,
+                revision=revision,
+                **kwargs,
+            )
+
+        # Initialize model without loading weights
+        # Check if dataset_stats were provided in kwargs
+        model = cls(config, **kwargs)
+
+        # Load state dict (expects keys with "model." prefix)
+        try:
+            print(f"Loading model from: {pretrained_name_or_path}")
+            try:
+                from transformers.utils import cached_file
+
+                resolved_file = cached_file(
+                    pretrained_name_or_path,
+                    "model.safetensors",
+                    cache_dir=kwargs.get("cache_dir"),
+                    force_download=kwargs.get("force_download", False),
+                    resume_download=kwargs.get("resume_download"),
+                    proxies=kwargs.get("proxies"),
+                    token=kwargs.get("token"),
+                    revision=kwargs.get("revision"),
+                    local_files_only=kwargs.get("local_files_only", False),
+                )
+                from safetensors.torch import load_file
+
+                original_state_dict = load_file(resolved_file)
+                print("✓ Loaded state dict from model.safetensors")
+            except Exception as e:
+                print(f"Could not load state dict from remote files: {e}")
+                print("Returning model without loading pretrained weights")
+                return model
+
+            # First, fix any key differences (see openpi model.py, _fix_pytorch_state_dict_keys)
+            fixed_state_dict = model._fix_pytorch_state_dict_keys(original_state_dict, model.config)
+
+            # Then add "model." prefix for all keys that don't already have it
+            remapped_state_dict = {}
+            remap_count = 0
+
+            for key, value in fixed_state_dict.items():
+                if not key.startswith("model."):
+                    new_key = f"model.{key}"
+                    remapped_state_dict[new_key] = value
+                    remap_count += 1
+                else:
+                    remapped_state_dict[key] = value
+
+            if remap_count > 0:
+                print(f"Remapped {remap_count} state dict keys")
+
+            # Load the remapped state dict into the model
+            missing_keys, unexpected_keys = model.load_state_dict(remapped_state_dict, strict=strict)
+
+            if missing_keys:
+                print(f"Missing keys when loading state dict: {len(missing_keys)} keys")
+                if len(missing_keys) <= 5:
+                    for key in missing_keys:
+                        print(f"  - {key}")
+                else:
+                    for key in missing_keys[:5]:
+                        print(f"  - {key}")
+                    print(f"  ... and {len(missing_keys) - 5} more")
+
+            if unexpected_keys:
+                print(f"Unexpected keys when loading state dict: {len(unexpected_keys)} keys")
+                if len(unexpected_keys) <= 5:
+                    for key in unexpected_keys:
+                        print(f"  - {key}")
+                else:
+                    for key in unexpected_keys[:5]:
+                        print(f"  - {key}")
+                    print(f"  ... and {len(unexpected_keys) - 5} more")
+
+            if not missing_keys and not unexpected_keys:
+                print("All keys loaded successfully!")
+
+        except Exception as e:
+            print(f"Warning: Could not load state dict: {e}")
+
+        return model
+
+    def _fix_pytorch_state_dict_keys(
+        self, state_dict, model_config
+    ):  # see openpi `BaseModelConfig, _fix_pytorch_state_dict_keys`
+        """Fix state dict keys to match current model architecture."""
+
+        fixed_state_dict = {}
+
+        for key, value in state_dict.items():
+            new_key = key
+
+            # Handle vision tower embedding layer potential differences
+            if "patch_embedding" in key:
+                # Some checkpoints might have this, but current model expects different structure
+                logging.warning(f"Vision embedding key might need handling: {key}")
+
+            if (
+                key == "model.paligemma_with_expert.paligemma.lm_head.weight"
+                or key == "paligemma_with_expert.paligemma.lm_head.weight"
+            ):
+                fixed_state_dict[
+                    "model.paligemma_with_expert.paligemma.model.language_model.embed_tokens.weight"
+                ] = value.clone()
+
+            fixed_state_dict[new_key] = value
+
+        return fixed_state_dict
+
+    def get_optim_params(self) -> dict:
+        return self.parameters()
+
+    def reset(self):
+        """Reset internal state - called when environment resets."""
+        self._action_queue = deque(maxlen=self.config.n_action_steps)
+        self._queues = {
+            ACTION: deque(maxlen=self.config.n_action_steps),
+        }
+
+    def init_rtc_processor(self):
+        """Initialize RTC processor if RTC is enabled in config."""
+        self.rtc_processor = None
+
+        # Create processor if config provided
+        # If RTC is not enabled - we can still track the denoising data
+        if self.config.rtc_config is not None:
+            self.rtc_processor = RTCProcessor(self.config.rtc_config)
+
+            model_value = getattr(self, "model", None)
+            if model_value is not None:
+                model_value.rtc_processor = self.rtc_processor
+
+    def _rtc_enabled(self) -> bool:
+        return self.config.rtc_config is not None and self.config.rtc_config.enabled
+
+    def _preprocess_images(self, batch: dict[str, Tensor]) -> tuple[list[Tensor], list[Tensor]]:
+        """Preprocess images for the model.
+
+        Images from LeRobot are typically in [B, C, H, W] format and normalized to [0, 1].
+        PaliGemma expects images in [B, C, H, W] format and normalized to [-1, 1].
+        """
+        images = []
+        img_masks = []
+
+        # Get device from model parameters
+        device = next(self.parameters()).device
+
+        present_img_keys = [key for key in self.config.image_features if key in batch]
+        missing_img_keys = [key for key in self.config.image_features if key not in batch]
+
+        if len(present_img_keys) == 0:
+            raise ValueError(
+                f"All image features are missing from the batch. At least one expected. "
+                f"(batch: {batch.keys()}) (image_features: {self.config.image_features})"
+            )
+
+        # Preprocess image features present in the batch
+        for key in present_img_keys:
+            img = batch[key]
+
+            # Ensure tensor is on the same device as the model
+            if img.device != device:
+                img = img.to(device)
+
+            # Ensure float32 dtype for consistency
+            if img.dtype != torch.float32:
+                img = img.to(torch.float32)
+
+            # from openpi preprocess_observation_pytorch: Handle both [B, C, H, W] and [B, H, W, C] formats
+            is_channels_first = img.shape[1] == 3  # Check if channels are in dimension 1
+
+            if is_channels_first:
+                # Convert [B, C, H, W] to [B, H, W, C] for processing
+                img = img.permute(0, 2, 3, 1)
+
+            # from openpi preprocess_observation_pytorch: Resize with padding if needed
+            if img.shape[1:3] != self.config.image_resolution:
+                img = resize_with_pad_torch(img, *self.config.image_resolution)
+
+            # Normalize from [0,1] to [-1,1] as expected by siglip
+            img = img * 2.0 - 1.0
+
+            # from openpi preprocess_observation_pytorch: Convert back to [B, C, H, W] format if it was originally channels-first
+            if is_channels_first:
+                img = img.permute(0, 3, 1, 2)  # [B, H, W, C] -> [B, C, H, W]
+
+            images.append(img)
+            # Create mask (all ones for real images)
+            bsize = img.shape[0]
+            mask = torch.ones(bsize, dtype=torch.bool, device=device)
+            img_masks.append(mask)
+
+        # Create image features not present in the batch as fully 0 padded images
+        for _num_empty_cameras in range(len(missing_img_keys)):
+            img = torch.ones_like(img) * -1  # Padded with -1 for SigLIP
+            mask = torch.zeros_like(mask)  # Mask is zero for empty cameras
+            images.append(img)
+            img_masks.append(mask)
+
+        return images, img_masks
+
+    def prepare_action(self, batch):
+        """Pad action"""
+        actions = pad_vector(batch[ACTION], self.config.max_action_dim)
+        return actions
+
+    def _paligemma_tokens_to_act_tokens(self, tokens: torch.Tensor) -> torch.Tensor:
+        """
+        Converts PaliGemma tokens back to action tokens (inverse of _act_tokens_to_paligemma_tokens).
+
+        Args:
+            tokens: PaliGemma token IDs
+
+        Returns:
+            Action token IDs
+        """
+        return self._paligemma_tokenizer.vocab_size - 1 - self.config.fast_skip_tokens - tokens
+
+    def decode_actions_with_fast(
+        self, token_ids: list[int], time_horizon: int, action_dim: int, relaxed_decoding: bool = True
+    ) -> np.ndarray:
+        """
+        Decodes action token IDs back to continuous action values using the FAST tokenizer.
+
+        Args:
+            token_ids: List of token IDs to decode.
+            time_horizon: The number of timesteps for actions.
+            action_dim: The dimensionality of each action.
+            relaxed_decoding: Whether to use relaxed decoding (allows partial sequences).
+
+        Returns:
+            A numpy array representing the decoded actions.
+        """
+        decoded_actions = []
+
+        for token in token_ids:
+            try:
+                decoded_tokens = self.action_tokenizer.bpe_tokenizer.decode(token)
+                decoded_dct_coeff = np.array(list(map(ord, decoded_tokens))) + self.action_tokenizer.min_token
+
+                if relaxed_decoding:
+                    # expected sequence length
+                    expected_seq_len = time_horizon * action_dim
+                    diff = expected_seq_len - decoded_dct_coeff.shape[0]
+
+                    # apply truncation if too long
+                    if diff < 0:
+                        decoded_dct_coeff = decoded_dct_coeff[:expected_seq_len]  # truncate on the right
+
+                    # apply padding if too short
+                    elif diff > 0:
+                        decoded_dct_coeff = np.pad(
+                            decoded_dct_coeff, (0, diff), mode="constant", constant_values=0
+                        )
+
+                decoded_dct_coeff = decoded_dct_coeff.reshape(-1, action_dim)
+                assert decoded_dct_coeff.shape == (
+                    time_horizon,
+                    action_dim,
+                ), (
+                    f"Decoded DCT coefficients have shape {decoded_dct_coeff.shape}, expected ({time_horizon}, {action_dim})"
+                )
+
+            except Exception as e:
+                logging.warning(f"Error decoding tokens: {e}")
+                logging.warning(f"Tokens: {token}")
+                decoded_dct_coeff = np.zeros((time_horizon, action_dim))
+
+            decoded_actions.append(
+                idct(decoded_dct_coeff / self.action_tokenizer.scale, axis=0, norm="ortho")
+            )
+
+        return np.stack(decoded_actions)
+
+    def detokenize_actions(self, tokens: torch.Tensor, action_horizon: int, action_dim: int) -> torch.Tensor:
+        """
+        Detokenizes action tokens back to continuous actions.
+
+        This method converts predicted action tokens from the model back to continuous action values
+        using the FAST tokenizer. It handles the conversion from PaliGemma token space to action token
+        space, then decodes the action tokens to continuous values using DCT decoding.
+
+        Args:
+            tokens: The input tensor of tokenized outputs. Shape: (B, seq_len) or (seq_len,)
+            action_horizon: The number of timesteps for actions.
+            action_dim: The dimensionality of each action.
+
+        Returns:
+            The continuous action tensor. Shape: (B, action_horizon, action_dim) or (action_horizon, action_dim)
+        """
+        if self.action_tokenizer is None or self._paligemma_tokenizer is None:
+            raise ValueError(
+                "Action tokenizer not initialized. Make sure fast_only=True in config and tokenizers loaded successfully."
+            )
+
+        # Handle single sample (add batch dimension)
+        single_sample = tokens.dim() == 1
+        if single_sample:
+            tokens = tokens.unsqueeze(0)
+
+        # Convert token IDs to token strings
+        decoded_tokens = [self._paligemma_tokenizer.convert_ids_to_tokens(seq.tolist()) for seq in tokens]
+        # Get the token sequence for "Action: " to remove it
+        action_prefix_ids = self._paligemma_tokenizer.encode("Action: ", add_special_tokens=False)
+        action_prefix_tokens = self._paligemma_tokenizer.convert_ids_to_tokens(action_prefix_ids)
+        action_prefix_len = len(action_prefix_tokens)
+
+        # Clean tokens by removing everything after the first "|" (end-of-action marker)
+        # and removing all occurrences of "Action: " token sequence
+        # assert that beginning contain "Action: "
+        if self.config.validate_action_token_prefix:
+            for token_seq in decoded_tokens:
+                assert len(token_seq) >= 2 and token_seq[0] == "Action" and token_seq[1] == ":", (
+                    f"Token sequence does not start with ['Action', ':']: {token_seq}"
+                )
+
+        cleaned_tokens = []
+        for token_seq in decoded_tokens:
+            # Remove everything after "|"
+            if "|" in token_seq:
+                token_seq = token_seq[: token_seq.index("|")]
+
+            # Remove all occurrences of "Action: " token sequence
+            i = 0
+            while i <= len(token_seq) - action_prefix_len:
+                if token_seq[i : i + action_prefix_len] == action_prefix_tokens:
+                    # Found a match, remove it
+                    token_seq = token_seq[:i] + token_seq[i + action_prefix_len :]
+                else:
+                    i += 1
+
+            cleaned_tokens.append(token_seq)
+
+        # Convert token strings back to IDs
+        raw_action_tokens = [
+            torch.tensor(
+                self._paligemma_tokenizer.convert_tokens_to_ids(token_seq),
+                dtype=torch.long,
+                device=tokens.device,
+            )
+            for token_seq in cleaned_tokens
+        ]
+
+        # Convert PaliGemma tokens to action tokens
+        action_tokens = [
+            self._paligemma_tokens_to_act_tokens(raw_action_token) for raw_action_token in raw_action_tokens
+        ]
+
+        # Decode action tokens to continuous actions
+        actions = self.decode_actions_with_fast(
+            action_tokens, time_horizon=action_horizon, action_dim=action_dim
+        )
+
+        # Convert to tensor and return
+        actions_tensor = torch.tensor(actions, dtype=torch.float32, device=tokens.device)
+
+        # Remove batch dimension if input was single sample
+        if single_sample:
+            actions_tensor = actions_tensor.squeeze(0)
+
+        return actions_tensor
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select a single action given environment observations."""
+        assert not self._rtc_enabled(), (
+            "RTC is not supported for select_action, use it with predict_action_chunk"
+        )
+
+        self.eval()
+
+        # Action queue logic for n_action_steps > 1
+        if len(self._action_queue) == 0:
+            actions = self.predict_action_chunk(batch)[:, : self.config.n_action_steps]
+            # Transpose to get shape (n_action_steps, batch_size, action_dim)
+            self._action_queue.extend(actions.transpose(0, 1))
+
+        return self._action_queue.popleft()
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]) -> Tensor:
+        """Predict a chunk of actions given environment observations."""
+        self.eval()
+        # Prepare inputs
+        images, img_masks = self._preprocess_images(batch)
+
+        # FAST-only mode: use autoregressive decoding
+        tokens = batch[f"{OBS_LANGUAGE_TOKENS}"]
+        masks = batch[f"{OBS_LANGUAGE_ATTENTION_MASK}"]
+
+        # Get decoding parameters
+        temperature = self.config.temperature
+        max_decoding_steps = self.config.max_decoding_steps
+
+        # Sample action tokens autoregressively
+        if self.config.use_kv_cache:
+            action_tokens = self.model.sample_actions_fast_kv_cache(
+                images,
+                img_masks,
+                tokens,
+                masks,
+                max_decoding_steps=max_decoding_steps,
+                temperature=temperature,
+            )
+        else:
+            action_tokens = self.model.sample_actions_fast(
+                images,
+                img_masks,
+                tokens,
+                masks,
+                max_decoding_steps=max_decoding_steps,
+                temperature=temperature,
+            )
+
+        # Detokenize action tokens to continuous actions
+        action_horizon = self.config.n_action_steps
+        action_dim = self.config.output_features[ACTION].shape[0]
+
+        continuous_actions = self.detokenize_actions(
+            action_tokens, action_horizon=action_horizon, action_dim=action_dim
+        )
+
+        return continuous_actions
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, dict]:
+        """Run the batch through the model and compute the loss for training."""
+
+        # Prepare inputs
+        images, img_masks = self._preprocess_images(batch)
+
+        # Get FAST action tokens from batch
+        fast_action_tokens = batch.get(ACTION_TOKENS)  # (B, max_action_tokens)
+        fast_action_masks = batch.get(ACTION_TOKEN_MASK)  # (B, max_action_tokens)
+
+        # Use full language tokens (no separation into high_level_task and subtask)
+        tokens = batch.get(OBS_LANGUAGE_TOKENS)
+        masks = batch.get(OBS_LANGUAGE_ATTENTION_MASK)
+
+        if fast_action_tokens is None or fast_action_masks is None:
+            raise ValueError(
+                f"PI0Fast requires {ACTION_TOKENS} and {ACTION_TOKEN_MASK} to be present in the batch"
+            )
+
+        loss_dict = self.model.forward(
+            images,
+            img_masks,
+            tokens,
+            masks,
+            fast_action_tokens,
+            fast_action_masks,
+        )
+
+        loss = loss_dict["loss"]
+        detailed_loss_dict = {
+            "loss": loss.item(),
+            "ce_loss": loss_dict["ce_loss"].item(),
+        }
+        return loss, detailed_loss_dict
diff --git a/lerobot/src/lerobot/policies/pi0_fast/processor_pi0_fast.py b/lerobot/src/lerobot/policies/pi0_fast/processor_pi0_fast.py
new file mode 100644
index 0000000000000000000000000000000000000000..46e54432a8dd026cb273eaae2c226c6a66f8dbad
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi0_fast/processor_pi0_fast.py
@@ -0,0 +1,173 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from copy import deepcopy
+from dataclasses import dataclass
+from typing import Any
+
+import numpy as np
+import torch
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.policies.pi0_fast.configuration_pi0_fast import PI0FastConfig
+from lerobot.processor import (
+    ActionTokenizerProcessorStep,
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    RenameObservationsProcessorStep,
+    TokenizerProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.types import EnvTransition, TransitionKey
+from lerobot.utils.constants import (
+    OBS_STATE,
+    POLICY_POSTPROCESSOR_DEFAULT_NAME,
+    POLICY_PREPROCESSOR_DEFAULT_NAME,
+)
+
+
+@ProcessorStepRegistry.register(name="pi0_fast_prepare_state_tokenizer_processor_step")
+@dataclass
+class Pi0FastPrepareStateAndLanguageTokenizerProcessorStep(ProcessorStep):
+    """
+    Processor step to prepare the state and tokenize the language input.
+    """
+
+    max_state_dim: int = 32
+    task_key: str = "task"
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        transition = transition.copy()
+
+        state = transition.get(TransitionKey.OBSERVATION, {}).get(OBS_STATE)
+        if state is None:
+            raise ValueError("State is required for PI0Fast")
+        tasks = transition.get(TransitionKey.COMPLEMENTARY_DATA, {}).get(self.task_key)
+        if tasks is None:
+            raise ValueError("No task found in complementary data")
+
+        # TODO: check if this necessary
+        state = deepcopy(state)
+
+        # State should already be normalized to [-1, 1] by the NormalizerProcessorStep that runs before this step
+        # Discretize into 256 bins (see openpi `PaligemmaTokenizer.tokenize()`)
+        state_np = state.cpu().numpy()
+        discretized_states = np.digitize(state_np, bins=np.linspace(-1, 1, 256 + 1)[:-1]) - 1
+
+        full_prompts = []
+        for i, task in enumerate(tasks):
+            cleaned_text = task.strip().replace("_", " ").replace("\n", " ")
+            state_str = " ".join(map(str, discretized_states[i]))
+            full_prompt = f"Task: {cleaned_text}, State: {state_str};\n"
+            full_prompts.append(full_prompt)
+
+        transition[TransitionKey.COMPLEMENTARY_DATA][self.task_key] = full_prompts
+        # Normalize state to [-1, 1] range if needed (assuming it's already normalized by normalizer processor step!!)
+        # Discretize into 256 bins (see openpi `PaligemmaTokenizer.tokenize()`)
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        This step does not alter the feature definitions.
+        """
+        return features
+
+
+def make_pi0_fast_pre_post_processors(
+    config: PI0FastConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for the PI0Fast policy.
+
+    The pre-processing pipeline prepares input data for the model by:
+    1. Renaming features to match pretrained configurations.
+    2. Normalizing input and output features based on dataset statistics.
+    3. Adding a batch dimension.
+    4. Appending a newline character to the task description for tokenizer compatibility.
+    5. Tokenizing the text prompt using the PaliGemma tokenizer.
+    6. Moving all data to the specified device.
+
+    The post-processing pipeline handles the model's output by:
+    1. Moving data to the CPU.
+    2. Unnormalizing the output features to their original scale.
+
+    Args:
+        config: The configuration object for the PI0Fast policy.
+        dataset_stats: A dictionary of statistics for normalization.
+        preprocessor_kwargs: Additional arguments for the pre-processor pipeline.
+        postprocessor_kwargs: Additional arguments for the post-processor pipeline.
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines.
+    """
+    # Add remaining processors
+    input_steps: list[ProcessorStep] = [
+        RenameObservationsProcessorStep(rename_map={}),  # To mimic the same processor as pretrained one
+        AddBatchDimensionProcessorStep(),
+        # NOTE: NormalizerProcessorStep MUST come before Pi0FastPrepareStateAndLanguageTokenizerProcessorStep
+        # because the tokenizer step expects normalized state in [-1, 1] range for discretization
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+        Pi0FastPrepareStateAndLanguageTokenizerProcessorStep(max_state_dim=config.max_state_dim),
+        TokenizerProcessorStep(
+            tokenizer_name=config.text_tokenizer_name,
+            max_length=config.tokenizer_max_length,
+            padding_side="right",
+            padding="max_length",
+        ),
+        ActionTokenizerProcessorStep(
+            action_tokenizer_name=config.action_tokenizer_name,
+            max_action_tokens=config.max_action_tokens,
+            fast_skip_tokens=config.fast_skip_tokens,
+            paligemma_tokenizer_name=config.text_tokenizer_name,
+        ),
+        DeviceProcessorStep(device=config.device),
+    ]
+
+    output_steps: list[ProcessorStep] = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/pi_gemma.py b/lerobot/src/lerobot/policies/pi_gemma.py
new file mode 100644
index 0000000000000000000000000000000000000000..05f031d085c1e79eec60f8ae7211c4f1df8edde4
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pi_gemma.py
@@ -0,0 +1,363 @@
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from __future__ import annotations
+
+from typing import TYPE_CHECKING
+
+import torch
+from torch import nn
+
+from lerobot.utils.import_utils import _transformers_available
+
+if TYPE_CHECKING or _transformers_available:
+    from transformers.cache_utils import DynamicCache
+    from transformers.masking_utils import create_causal_mask
+    from transformers.modeling_layers import GradientCheckpointingLayer
+    from transformers.modeling_outputs import BaseModelOutputWithPast
+    from transformers.models.gemma.modeling_gemma import (
+        GemmaAttention,
+        GemmaConfig,
+        GemmaForCausalLM,
+        GemmaMLP,
+        GemmaModel,
+    )
+    from transformers.models.paligemma.modeling_paligemma import (
+        PaliGemmaForConditionalGeneration,
+        PaliGemmaModel,
+    )
+else:
+    GemmaAttention = None
+    GemmaConfig = None
+    GemmaForCausalLM = None
+    GemmaMLP = None
+    GemmaModel = None
+    PaliGemmaModel = None
+    PaliGemmaForConditionalGeneration = None
+    DynamicCache = None
+    GradientCheckpointingLayer = None
+    BaseModelOutputWithPast = None
+    create_causal_mask = None
+
+
+def _gated_residual(
+    x: torch.Tensor | None,
+    y: torch.Tensor | None,
+    gate: torch.Tensor | None,
+) -> torch.Tensor | None:
+    """Gated residual: x + y when gate is None, else x + y * gate."""
+    if x is None and y is None:
+        return None
+    if x is None or y is None:
+        return x if x is not None else y
+    if gate is None:
+        return x + y
+    return x + y * gate
+
+
+def layernorm_forward(
+    layernorm: nn.Module,
+    x: torch.Tensor,
+    cond: torch.Tensor | None = None,
+):
+    """
+    call layernorm and return hidden states and gate
+    if cond is not None, use conditional norm
+    otherwise, use normal gemma norm
+    """
+    if cond is not None:
+        return layernorm(x, cond=cond)
+    else:
+        return layernorm(x)
+
+
+class PiGemmaRMSNorm(nn.Module):
+    """
+    Adaptive RMSNorm for PI Gemma (AdaRMS).
+    When cond_dim is set, uses cond to modulate scale/shift/gate; otherwise behaves like standard GemmaRMSNorm.
+    forward(x, cond=None) returns (output, gate) for use with _gated_residual.
+    """
+
+    def __init__(self, dim: int, eps: float = 1e-6, cond_dim: int | None = None):
+        super().__init__()
+        self.eps = eps
+        self.dim = dim
+        self.cond_dim = cond_dim
+        if cond_dim is not None:
+            self.dense = nn.Linear(cond_dim, dim * 3, bias=True)
+            nn.init.zeros_(self.dense.weight)
+        else:
+            self.weight = nn.Parameter(torch.zeros(dim))
+            self.dense = None
+
+    def _norm(self, x):
+        # Compute variance in float32 (like the source implementation)
+        var = torch.mean(torch.square(x.float()), dim=-1, keepdim=True)
+        # Compute normalization in float32
+        normed_inputs = x * torch.rsqrt(var + self.eps)
+        return normed_inputs
+
+    def forward(
+        self,
+        x: torch.Tensor,
+        cond: torch.Tensor | None = None,
+    ) -> tuple[torch.Tensor, torch.Tensor | None]:
+        dtype = x.dtype
+        normed = self._norm(x)
+        if cond is None or self.dense is None:
+            normed = normed * (1.0 + self.weight.float())
+            return normed.type_as(x), None
+        if cond.shape[-1] != self.cond_dim:
+            raise ValueError(f"Expected cond dim {self.cond_dim}, got {cond.shape[-1]}")
+        modulation = self.dense(cond)
+        if len(x.shape) == 3:
+            modulation = modulation.unsqueeze(1)
+        scale, shift, gate = modulation.chunk(3, dim=-1)
+        normed = normed * (1 + scale.float()) + shift.float()
+        return normed.to(dtype), gate.to(dtype)
+
+    def extra_repr(self) -> str:
+        if self.dense is not None:
+            return f"dim={self.dim}, eps={self.eps}, adaptive=True, cond_dim={self.cond_dim}"
+        return f"dim={self.dim}, eps={self.eps}"
+
+
+def _get_pi_gemma_decoder_layer_base():
+    """base for PiGemmaDecoderLayer"""
+
+    class _PiGemmaDecoderLayerBase(GradientCheckpointingLayer):
+        """Decoder layer that uses PiGemmaRMSNorm and _gated_residual, compatible with v5 Gemma."""
+
+        def __init__(self, config: GemmaConfig, layer_idx: int):
+            super().__init__()
+            self.hidden_size = config.hidden_size
+            self.self_attn = GemmaAttention(config=config, layer_idx=layer_idx)
+            self.mlp = GemmaMLP(config)
+            cond_dim = (
+                getattr(config, "adarms_cond_dim", None) if getattr(config, "use_adarms", False) else None
+            )
+            self.input_layernorm = PiGemmaRMSNorm(
+                config.hidden_size, eps=config.rms_norm_eps, cond_dim=cond_dim
+            )
+            self.post_attention_layernorm = PiGemmaRMSNorm(
+                config.hidden_size, eps=config.rms_norm_eps, cond_dim=cond_dim
+            )
+
+        def forward(
+            self,
+            hidden_states: torch.Tensor,
+            attention_mask: torch.Tensor | None = None,
+            position_ids: torch.LongTensor | None = None,
+            past_key_values=None,
+            use_cache: bool = False,
+            cache_position: torch.LongTensor | None = None,
+            position_embeddings: tuple[torch.Tensor, torch.Tensor] | None = None,
+            adarms_cond: torch.Tensor | None = None,
+            **kwargs,
+        ) -> torch.Tensor:
+            residual = hidden_states
+            hidden_states, gate = self.input_layernorm(hidden_states, cond=adarms_cond)
+            hidden_states, _ = self.self_attn(
+                hidden_states,
+                attention_mask=attention_mask,
+                position_ids=position_ids,
+                past_key_values=past_key_values,
+                use_cache=use_cache,
+                cache_position=cache_position,
+                position_embeddings=position_embeddings,
+                **kwargs,
+            )
+
+            hidden_states = _gated_residual(residual, hidden_states, gate)
+
+            residual = hidden_states
+            hidden_states, gate = self.post_attention_layernorm(hidden_states, cond=adarms_cond)
+            hidden_states = self.mlp(hidden_states)
+            hidden_states = _gated_residual(residual, hidden_states, gate)
+            return hidden_states
+
+    return _PiGemmaDecoderLayerBase
+
+
+class PiGemmaModel(GemmaModel):  # type: ignore[misc]
+    """
+    GemmaModel extended with AdaRMS (adaptive RMSNorm) and gated residuals when config.use_adarms is True.
+    """
+
+    def __init__(self, config: GemmaConfig, **kwargs):
+        super().__init__(config, **kwargs)
+        # if not getattr(config, "use_adarms", False):
+        #     return
+        cond_dim = getattr(config, "adarms_cond_dim", None)
+        pi_gemma_decoder_layer_base = _get_pi_gemma_decoder_layer_base()
+        self.layers = nn.ModuleList(
+            [pi_gemma_decoder_layer_base(config, layer_idx) for layer_idx in range(config.num_hidden_layers)]
+        )
+        self.norm = PiGemmaRMSNorm(config.hidden_size, eps=config.rms_norm_eps, cond_dim=cond_dim)
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor | None = None,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: DynamicCache | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        cache_position: torch.LongTensor | None = None,
+        adarms_cond: torch.Tensor | None = None,
+        **kwargs,
+    ) -> BaseModelOutputWithPast:
+        """
+        adarms_cond (`torch.Tensor` of shape `(batch_size, cond_dim)`, *optional*):
+            Condition for ADARMS.
+        """
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+
+        if (input_ids is None) ^ (inputs_embeds is not None):
+            raise ValueError("You must specify exactly one of input_ids or inputs_embeds")
+
+        if self.gradient_checkpointing and self.training and use_cache:
+            import logging
+
+            logging.warning(
+                "`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`."
+            )
+            use_cache = False
+
+        if inputs_embeds is None:
+            inputs_embeds = self.embed_tokens(input_ids)
+
+        if use_cache and past_key_values is None:
+            past_key_values = DynamicCache()
+
+        if cache_position is None:
+            past_seen_tokens = past_key_values.get_seq_length() if past_key_values is not None else 0
+            cache_position = torch.arange(
+                past_seen_tokens, past_seen_tokens + inputs_embeds.shape[1], device=inputs_embeds.device
+            )
+
+        if position_ids is None:
+            position_ids = cache_position.unsqueeze(0)
+
+        causal_mask = create_causal_mask(
+            config=self.config,
+            inputs_embeds=inputs_embeds,
+            attention_mask=attention_mask,
+            cache_position=cache_position,
+            past_key_values=past_key_values,
+            position_ids=position_ids,
+        )
+
+        # embed positions
+        hidden_states = inputs_embeds
+        # Convert to bfloat16 if the first layer uses bfloat16
+        if len(self.layers) > 0 and self.layers[0].self_attn.q_proj.weight.dtype == torch.bfloat16:
+            hidden_states = hidden_states.to(torch.bfloat16)
+
+        # create position embeddings to be shared across the decoder layers
+        position_embeddings = self.rotary_emb(hidden_states, position_ids)
+
+        # normalized
+        # Gemma downcasts the below to float16, causing sqrt(3072)=55.4256 to become 55.5
+        # See https://github.com/huggingface/transformers/pull/29402
+
+        # decoder layers
+        all_hidden_states = () if output_hidden_states else None
+        all_self_attns = () if output_attentions else None
+
+        for decoder_layer in self.layers[: self.config.num_hidden_layers]:
+            if output_hidden_states:
+                all_hidden_states += (hidden_states,)
+
+            layer_outputs = decoder_layer(
+                hidden_states,
+                attention_mask=causal_mask,
+                position_ids=position_ids,
+                past_key_values=past_key_values,
+                output_attentions=output_attentions,
+                use_cache=use_cache,
+                cache_position=cache_position,
+                position_embeddings=position_embeddings,
+                adarms_cond=adarms_cond,
+                **kwargs,
+            )
+
+            hidden_states = layer_outputs
+
+            if output_attentions:
+                all_self_attns += (layer_outputs[1],)
+
+        hidden_states, _ = self.norm(hidden_states, adarms_cond)
+
+        # add hidden states from the last decoder layer
+        if output_hidden_states:
+            all_hidden_states += (hidden_states,)
+
+        return BaseModelOutputWithPast(
+            last_hidden_state=hidden_states,
+            past_key_values=past_key_values if use_cache else None,
+            hidden_states=all_hidden_states,
+            attentions=all_self_attns,
+        )
+
+
+class PiGemmaForCausalLM(GemmaForCausalLM):  # type: ignore[misc]
+    """
+    Causal LM wrapper using PiGemmaModel as the backbone, for consistency with GemmaForCausalLM
+    and the language model used in pi0_fast. Use this for the action expert in pi0/pi05.
+    """
+
+    def __init__(self, config: GemmaConfig, **kwargs):
+        super().__init__(config, **kwargs)
+        self.model = PiGemmaModel(config)
+
+
+class PaliGemmaModelWithPiGemma(PaliGemmaModel):
+    """PaliGemmaModel whose language_model is PiGemmaModel (custom decoder with PiGemmaRMSNorm and gated residuals)."""
+
+    def __init__(self, config):
+        super().__init__(config)
+        self.language_model = PiGemmaModel(config.text_config)
+
+
+class PaliGemmaForConditionalGenerationWithPiGemma(PaliGemmaForConditionalGeneration):
+    """PaliGemmaForConditionalGeneration using PiGemma decoder for the language model."""
+
+    def __init__(self, config):
+        super().__init__(config)
+        self.model = PaliGemmaModelWithPiGemma(config)
+
+    # Make modules available through conditional class for BC
+    @property
+    def language_model(self):
+        return self.model.language_model
+
+
+__all__ = [
+    "PiGemmaModel",
+    "PiGemmaForCausalLM",
+    "PiGemmaRMSNorm",
+    "_gated_residual",
+    "layernorm_forward",
+    "PaliGemmaModelWithPiGemma",
+    "PaliGemmaForConditionalGenerationWithPiGemma",
+]
diff --git a/lerobot/src/lerobot/policies/pretrained.py b/lerobot/src/lerobot/policies/pretrained.py
new file mode 100644
index 0000000000000000000000000000000000000000..70efeba6f21b31611b90073710d0d372b3e0aae7
--- /dev/null
+++ b/lerobot/src/lerobot/policies/pretrained.py
@@ -0,0 +1,430 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import abc
+import builtins
+import dataclasses
+import logging
+import os
+from importlib.resources import files
+from pathlib import Path
+from tempfile import TemporaryDirectory
+from typing import TypedDict, TypeVar, Unpack
+
+import packaging
+import safetensors
+from huggingface_hub import HfApi, ModelCard, ModelCardData, hf_hub_download
+from huggingface_hub.constants import SAFETENSORS_SINGLE_FILE
+from huggingface_hub.errors import HfHubHTTPError
+from safetensors.torch import load_model as load_model_as_safetensor, save_model as save_model_as_safetensor
+from torch import Tensor, nn
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.policies.utils import log_model_loading_keys
+from lerobot.utils.hub import HubMixin
+
+T = TypeVar("T", bound="PreTrainedPolicy")
+
+
+class ActionSelectKwargs(TypedDict, total=False):
+    noise: Tensor | None
+
+
+class PreTrainedPolicy(nn.Module, HubMixin, abc.ABC):
+    """
+    Base class for policy models.
+    """
+
+    config_class: None
+    name: None
+
+    def __init__(self, config: PreTrainedConfig, *inputs, **kwargs):
+        super().__init__()
+        if not isinstance(config, PreTrainedConfig):
+            raise ValueError(
+                f"Parameter config in `{self.__class__.__name__}(config)` should be an instance of class "
+                "`PreTrainedConfig`. To create a model from a pretrained model use "
+                f"`model = {self.__class__.__name__}.from_pretrained(PRETRAINED_MODEL_NAME)`"
+            )
+        self.config = config
+
+    def __init_subclass__(cls, **kwargs):
+        super().__init_subclass__(**kwargs)
+        if not getattr(cls, "config_class", None):
+            raise TypeError(f"Class {cls.__name__} must define 'config_class'")
+        if not getattr(cls, "name", None):
+            raise TypeError(f"Class {cls.__name__} must define 'name'")
+
+    def _save_pretrained(self, save_directory: Path) -> None:
+        self.config._save_pretrained(save_directory)
+        model_to_save = self.module if hasattr(self, "module") else self
+        save_model_as_safetensor(model_to_save, str(save_directory / SAFETENSORS_SINGLE_FILE))
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type[T],
+        pretrained_name_or_path: str | Path,
+        *,
+        config: PreTrainedConfig | None = None,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        strict: bool = False,
+        **kwargs,
+    ) -> T:
+        """
+        The policy is set in evaluation mode by default using `policy.eval()` (dropout modules are
+        deactivated). To train it, you should first set it back in training mode with `policy.train()`.
+        """
+        if config is None:
+            config = PreTrainedConfig.from_pretrained(
+                pretrained_name_or_path=pretrained_name_or_path,
+                force_download=force_download,
+                resume_download=resume_download,
+                proxies=proxies,
+                token=token,
+                cache_dir=cache_dir,
+                local_files_only=local_files_only,
+                revision=revision,
+                **kwargs,
+            )
+        model_id = str(pretrained_name_or_path)
+        instance = cls(config, **kwargs)
+        if os.path.isdir(model_id):
+            print("Loading weights from local directory")
+            model_file = os.path.join(model_id, SAFETENSORS_SINGLE_FILE)
+            policy = cls._load_as_safetensor(instance, model_file, config.device, strict)
+        else:
+            try:
+                model_file = hf_hub_download(
+                    repo_id=model_id,
+                    filename=SAFETENSORS_SINGLE_FILE,
+                    revision=revision,
+                    cache_dir=cache_dir,
+                    force_download=force_download,
+                    proxies=proxies,
+                    resume_download=resume_download,
+                    token=token,
+                    local_files_only=local_files_only,
+                )
+                policy = cls._load_as_safetensor(instance, model_file, config.device, strict)
+            except HfHubHTTPError as e:
+                raise FileNotFoundError(
+                    f"{SAFETENSORS_SINGLE_FILE} not found on the HuggingFace Hub in {model_id}"
+                ) from e
+
+        policy.to(config.device)
+        policy.eval()
+        return policy
+
+    @classmethod
+    def _load_as_safetensor(cls, model: T, model_file: str, map_location: str, strict: bool) -> T:
+        # Create base kwargs
+        kwargs = {"strict": strict}
+
+        # Add device parameter for newer versions that support it
+        if packaging.version.parse(safetensors.__version__) >= packaging.version.parse("0.4.3"):
+            kwargs["device"] = map_location
+
+        # Load the model with appropriate kwargs
+        missing_keys, unexpected_keys = load_model_as_safetensor(model, model_file, **kwargs)
+        log_model_loading_keys(missing_keys, unexpected_keys)
+
+        # For older versions, manually move to device if needed
+        if "device" not in kwargs and map_location != "cpu":
+            logging.warning(
+                "Loading model weights on other devices than 'cpu' is not supported natively in your version of safetensors."
+                " This means that the model is loaded on 'cpu' first and then copied to the device."
+                " This leads to a slower loading time."
+                " Please update safetensors to version 0.4.3 or above for improved performance."
+            )
+            model.to(map_location)
+        return model
+
+    @abc.abstractmethod
+    def get_optim_params(self) -> dict:
+        """
+        Returns the policy-specific parameters dict to be passed on to the optimizer.
+        """
+        raise NotImplementedError
+
+    @abc.abstractmethod
+    def reset(self):
+        """To be called whenever the environment is reset.
+
+        Does things like clearing caches.
+        """
+        raise NotImplementedError
+
+    # TODO(aliberts, rcadene): split into 'forward' and 'compute_loss'?
+    @abc.abstractmethod
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, dict | None]:
+        """_summary_
+
+        Args:
+            batch (dict[str, Tensor]): _description_
+
+        Returns:
+            tuple[Tensor, dict | None]: The loss and potentially other information. Apart from the loss which
+                is a Tensor, all other items should be logging-friendly, native Python types.
+        """
+        raise NotImplementedError
+
+    @abc.abstractmethod
+    def predict_action_chunk(self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]) -> Tensor:
+        """Returns the action chunk (for action chunking policies) for a given observation, potentially in batch mode.
+
+        Child classes using action chunking should use this method within `select_action` to form the action chunk
+        cached for selection.
+        """
+        raise NotImplementedError
+
+    @abc.abstractmethod
+    def select_action(self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]) -> Tensor:
+        """Return one action to run in the environment (potentially in batch mode).
+
+        When the model uses a history of observations, or outputs a sequence of actions, this method deals
+        with caching.
+        """
+        raise NotImplementedError
+
+    def push_model_to_hub(
+        self,
+        cfg: TrainPipelineConfig,
+        peft_model=None,
+    ):
+        api = HfApi()
+        repo_id = api.create_repo(
+            repo_id=self.config.repo_id, private=self.config.private, exist_ok=True
+        ).repo_id
+
+        # Push the files to the repo in a single commit
+        with TemporaryDirectory(ignore_cleanup_errors=True) as tmp:
+            saved_path = Path(tmp) / repo_id
+
+            if peft_model is not None:
+                # Since PEFT just forwards calls to `push_model_to_hub`, `self` is not the PeftModel wrapper
+                # but the actual policy which is why we need the PEFT model passed to us to save the adapter.
+                # That also means that we need to store the policy config ourselves since PEFT can't.
+                peft_model.save_pretrained(saved_path)
+                self.config.save_pretrained(saved_path)
+            else:
+                self.save_pretrained(saved_path)  # Calls _save_pretrained and stores model tensors
+
+            card = self.generate_model_card(
+                cfg.dataset.repo_id, self.config.type, self.config.license, self.config.tags
+            )
+            card.save(str(saved_path / "README.md"))
+
+            cfg.save_pretrained(saved_path)  # Calls _save_pretrained and stores train config
+
+            commit_info = api.upload_folder(
+                repo_id=repo_id,
+                repo_type="model",
+                folder_path=saved_path,
+                commit_message="Upload policy weights, train config and readme",
+                allow_patterns=["*.safetensors", "*.json", "*.yaml", "*.md"],
+                ignore_patterns=["*.tmp", "*.log"],
+            )
+
+            logging.info(f"Model pushed to {commit_info.repo_url.url}")
+
+    def generate_model_card(
+        self, dataset_repo_id: str, model_type: str, license: str | None, tags: list[str] | None
+    ) -> ModelCard:
+        base_model = "lerobot/smolvla_base" if model_type == "smolvla" else None  # Set a base model
+
+        card_data = ModelCardData(
+            license=license or "apache-2.0",
+            library_name="lerobot",
+            pipeline_tag="robotics",
+            tags=list(set(tags or []).union({"robotics", "lerobot", model_type})),
+            model_name=model_type,
+            datasets=dataset_repo_id,
+            base_model=base_model,
+        )
+
+        template_card = (
+            files("lerobot.templates").joinpath("lerobot_modelcard_template.md").read_text(encoding="utf-8")
+        )
+        card = ModelCard.from_template(card_data, template_str=template_card)
+        card.validate()
+        return card
+
+    def wrap_with_peft(
+        self,
+        peft_config=None,
+        peft_cli_overrides: dict | None = None,
+    ) -> "PreTrainedPolicy":
+        """
+        Wrap this policy with PEFT adapters for parameter-efficient fine-tuning.
+
+        This method is the single entry point for PEFT integration. Subclasses should
+        override `_get_default_peft_targets()` to provide default target modules, and
+        `_validate_peft_config()` for policy-specific validation.
+
+        Args:
+            peft_config: Optional PEFT adapter configuration (e.g., LoraConfig).
+                If provided, used directly (with CLI overrides applied).
+            peft_cli_overrides: Optional dict of CLI overrides (method_type, target_modules, r, etc.)
+                These are merged with policy defaults to build the final config.
+        """
+        from peft import get_peft_model
+
+        # If user provided a complete config, use it directly (with overrides)
+        if peft_config is not None:
+            final_config = peft_config
+            if peft_cli_overrides:
+                final_config = self._apply_peft_cli_overrides(final_config, peft_cli_overrides)
+        else:
+            # Build config from defaults + CLI overrides
+            final_config = self._build_peft_config(peft_cli_overrides or {})
+
+        # Validate the configuration
+        self._validate_peft_config(final_config)
+
+        # Freeze base parameters, only adapter params will be trained
+        for p in self.parameters():
+            p.requires_grad_(False)
+
+        # Store pretrained path for PEFT's base_model_name_or_path
+        if self.config.pretrained_path:
+            self.name_or_path = str(self.config.pretrained_path)
+
+        # Wrap with PEFT
+        peft_model = get_peft_model(self, final_config)
+
+        # Mark config as using PEFT for proper loading later
+        peft_model.config.use_peft = True
+
+        logging.info(f"Wrapped {self.name} with PEFT ({type(final_config).__name__})")
+        return peft_model
+
+    def _get_default_peft_targets(self) -> dict[str, any] | None:
+        """
+        Return default PEFT target modules for this policy.
+
+        Override this in subclasses to provide policy-specific defaults. These defaults
+        are PEFT-method agnostic - they only specify which modules to target.
+
+        """
+        return None
+
+    def _validate_peft_config(self, peft_config) -> None:
+        """
+        Validate the PEFT configuration for this policy.
+
+        Override this in subclasses to add policy-specific validation or warnings.
+        The default implementation checks that a pretrained_path exists.
+
+        Args:
+            peft_config: The PEFT configuration to validate.
+
+        Raises:
+            ValueError: If the configuration is invalid.
+        """
+        if not self.config.pretrained_path:
+            raise ValueError(
+                "Training from scratch using PEFT is unlikely to yield good results. "
+                "Supply a `policy.pretrained_path` to fine-tune an existing model."
+            )
+
+    def _preprocess_peft_cli_overrides(self, cli_overrides: dict, peft_method_type) -> dict:
+        """
+        Preprocess CLI overrides: rename keys and handle method-specific init_type.
+
+        Args:
+            cli_overrides: Dict of CLI options (will be copied, not mutated).
+            peft_method_type: The PeftType enum value for the PEFT method.
+
+        Returns:
+            Preprocessed dict with renamed keys and init_type mapped to method-specific key.
+        """
+        from peft import PeftType
+
+        cli_overrides = cli_overrides.copy()
+
+        # Handle the full_training_modules -> modules_to_save rename
+        if "full_training_modules" in cli_overrides:
+            cli_overrides["modules_to_save"] = cli_overrides.pop("full_training_modules")
+
+        # Remove method_type as it's handled separately
+        cli_overrides.pop("method_type", None)
+
+        # Handle init_type specially based on PEFT method
+        init_type = cli_overrides.pop("init_type", None)
+        if init_type is not None:
+            if peft_method_type == PeftType.LORA:
+                cli_overrides["init_lora_weights"] = init_type
+            elif peft_method_type == PeftType.MISS:
+                cli_overrides["init_weights"] = init_type
+            else:
+                raise ValueError(f"Init type '{init_type}' unknown for PEFT method {peft_method_type}.")
+
+        return cli_overrides
+
+    def _build_peft_config(self, cli_overrides: dict):
+        """Build a PEFT config from policy defaults and CLI overrides."""
+        from peft import PEFT_TYPE_TO_CONFIG_MAPPING, PeftType
+
+        # Determine PEFT method type (default to LORA)
+        method_type_str = cli_overrides.get("method_type") or "lora"
+        peft_method_type = PeftType[method_type_str.upper()]
+        peft_config_cls = PEFT_TYPE_TO_CONFIG_MAPPING[peft_method_type]
+
+        # Preprocess CLI overrides
+        cli_overrides = self._preprocess_peft_cli_overrides(cli_overrides, peft_method_type)
+
+        # Start with policy defaults, apply CLI overrides
+        config_dict = dict(self._get_default_peft_targets() or {})
+        for key, value in cli_overrides.items():
+            if value is not None:
+                config_dict[key] = value
+
+        # Ensure we have target_modules
+        if not config_dict.get("target_modules"):
+            raise ValueError(
+                f"Policy '{self.name}' does not define default target_modules. "
+                "Please pass --peft.target_modules explicitly."
+            )
+
+        return peft_config_cls(**config_dict)
+
+    def _apply_peft_cli_overrides(self, peft_config, cli_overrides: dict):
+        """Apply CLI overrides to an existing PEFT config."""
+        from peft import PEFT_TYPE_TO_CONFIG_MAPPING, PeftType
+
+        # Get method type from existing config or CLI override
+        method_type_str = cli_overrides.get("method_type")
+        if method_type_str:
+            peft_method_type = PeftType[method_type_str.upper()]
+            peft_config_cls = PEFT_TYPE_TO_CONFIG_MAPPING[peft_method_type]
+        else:
+            peft_method_type = PeftType(peft_config.peft_type)
+            peft_config_cls = type(peft_config)
+
+        # Preprocess CLI overrides
+        cli_overrides = self._preprocess_peft_cli_overrides(cli_overrides, peft_method_type)
+
+        # Start with existing config, apply CLI overrides
+        config_dict = {k: v for k, v in dataclasses.asdict(peft_config).items() if not k.startswith("_")}
+        for key, value in cli_overrides.items():
+            if value is not None:
+                config_dict[key] = value
+
+        return peft_config_cls(**config_dict)
diff --git a/lerobot/src/lerobot/policies/rtc/README.md b/lerobot/src/lerobot/policies/rtc/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..926d4e8c469fbf249803330a062b78d791bf37c0
--- /dev/null
+++ b/lerobot/src/lerobot/policies/rtc/README.md
@@ -0,0 +1,38 @@
+# Real-Time Chunking (RTC)
+
+This module contains the LeRobot implementation of **Real-Time Chunking (RTC)**, an inference-time technique for flow-matching based policies.
+
+**Note**: RTC is not a policy itself, but rather an inference enhancement that works with flow-matching based policies including [π₀](../pi0/), [π₀.₅](../pi05/), and [SmolVLA](../smolvla/).
+
+---
+
+## Citation
+
+If you use Real-Time Chunking in your work, please cite:
+
+```bibtex
+@misc{openpi2024,
+  author       = {Physical Intelligence Lab},
+  title        = {OpenPI: PyTorch Implementation of π0 and π0.5 Policies},
+  year         = {2024},
+  publisher    = {GitHub},
+  howpublished = {\url{https://github.com/Physical-Intelligence/openpi}},
+  license      = {Apache-2.0}
+}
+
+@misc{black2025realtimeexecutionactionchunking,
+      title={Real-Time Execution of Action Chunking Flow Policies},
+      author={Kevin Black and Manuel Y. Galliker and Sergey Levine},
+      year={2025},
+      eprint={2506.07339},
+      archivePrefix={arXiv},
+      primaryClass={cs.RO},
+      url={https://arxiv.org/abs/2506.07339},
+}
+```
+
+---
+
+## License
+
+This implementation follows the **Apache 2.0 License**, consistent with the LeRobot project.
diff --git a/lerobot/src/lerobot/policies/rtc/action_queue.py b/lerobot/src/lerobot/policies/rtc/action_queue.py
new file mode 100644
index 0000000000000000000000000000000000000000..3f253ff8ca408f4a1d69e8327379cce133c67243
--- /dev/null
+++ b/lerobot/src/lerobot/policies/rtc/action_queue.py
@@ -0,0 +1,219 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Action queue management for Real-Time Chunking (RTC).
+
+This module provides ActionQueue, a thread-safe queue for managing action chunks
+in real-time control scenarios. It supports both RTC-enabled and non-RTC modes,
+handling action merging and leftover tracking.
+"""
+
+import logging
+from threading import Lock
+
+import torch
+from torch import Tensor
+
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+
+logger = logging.getLogger(__name__)
+
+
+class ActionQueue:
+    """Thread-safe queue for managing action chunks in real-time control.
+
+    This queue handles two types of action sequences:
+    - Original actions: Used for RTC to compute leftovers from previous chunks
+    - Processed actions: Post-processed actions ready for robot execution
+
+    The queue operates in two modes:
+    1. RTC-enabled: Replaces the entire queue with new actions, accounting for inference delay
+    2. RTC-disabled: Appends new actions to the queue, maintaining continuity
+
+    Args:
+        cfg (RTCConfig): Configuration for Real-Time Chunking behavior.
+
+    Attributes:
+        queue (Tensor | None): Processed actions for robot rollout (time_steps, action_dim).
+        original_queue (Tensor | None): Original actions for RTC computation (time_steps, action_dim).
+        last_index (int): Current consumption index in the queue.
+    """
+
+    def __init__(self, cfg: RTCConfig):
+        """Initialize the action queue.
+
+        Args:
+            cfg: RTC configuration controlling queue behavior.
+        """
+        self.queue = None  # Processed actions for robot rollout
+        self.original_queue = None  # Original actions for RTC
+        self.lock = Lock()
+        self.last_index = 0
+        self.cfg = cfg
+
+    def get(self) -> Tensor | None:
+        """Get the next action from the queue.
+
+        Returns:
+            Tensor | None: The next action (action_dim,) or None if queue is empty.
+                          Returns a clone to prevent external modifications.
+        """
+        with self.lock:
+            if self.queue is None or self.last_index >= len(self.queue):
+                return None
+
+            action = self.queue[self.last_index]
+            self.last_index += 1
+            return action.clone()
+
+    def qsize(self) -> int:
+        """Get the number of remaining actions in the queue.
+
+        Returns:
+            int: Number of unconsumed actions.
+        """
+        if self.queue is None:
+            return 0
+        length = len(self.queue)
+        return length - self.last_index
+
+    def empty(self) -> bool:
+        """Check if the queue is empty.
+
+        Returns:
+            bool: True if no actions remain, False otherwise.
+        """
+        if self.queue is None:
+            return True
+
+        length = len(self.queue)
+        return length - self.last_index <= 0
+
+    def get_action_index(self) -> int:
+        """Get the current action consumption index.
+
+        Returns:
+            int: Index of the next action to be consumed.
+        """
+        return self.last_index
+
+    def get_left_over(self) -> Tensor | None:
+        """Get leftover original actions for RTC prev_chunk_left_over.
+
+        These are the unconsumed actions from the current chunk, which will be
+        used by RTC to compute corrections for the next chunk.
+
+        Returns:
+            Tensor | None: Remaining original actions (remaining_steps, action_dim),
+                          or None if no original queue exists.
+        """
+        with self.lock:
+            if self.original_queue is None:
+                return None
+            return self.original_queue[self.last_index :]
+
+    def merge(
+        self,
+        original_actions: Tensor,
+        processed_actions: Tensor,
+        real_delay: int,
+        action_index_before_inference: int | None = 0,
+    ):
+        """Merge new actions into the queue.
+
+        This method operates differently based on RTC mode:
+        - RTC enabled: Replaces the queue, accounting for inference delay
+        - RTC disabled: Appends to the queue, maintaining continuity
+
+        Args:
+            original_actions: Unprocessed actions from policy (time_steps, action_dim).
+            processed_actions: Post-processed actions for robot (time_steps, action_dim).
+            real_delay: Number of time steps of inference delay.
+            action_index_before_inference: Index before inference started, for validation.
+        """
+        with self.lock:
+            self._check_delays(real_delay, action_index_before_inference)
+
+            if self.cfg.enabled:
+                self._replace_actions_queue(original_actions, processed_actions, real_delay)
+                return
+
+            self._append_actions_queue(original_actions, processed_actions)
+
+    def _replace_actions_queue(self, original_actions: Tensor, processed_actions: Tensor, real_delay: int):
+        """Replace the queue with new actions (RTC mode).
+
+        Discards the first `real_delay` actions since they correspond to the time
+        spent during inference, when the robot was executing previous actions.
+
+        Args:
+            original_actions: Unprocessed actions from policy.
+            processed_actions: Post-processed actions for robot.
+            real_delay: Number of time steps to skip due to inference delay.
+        """
+        self.original_queue = original_actions[real_delay:].clone()
+        self.queue = processed_actions[real_delay:].clone()
+
+        logger.debug(f"original_actions shape: {self.original_queue.shape}")
+        logger.debug(f"processed_actions shape: {self.queue.shape}")
+        logger.debug(f"real_delay: {real_delay}")
+
+        self.last_index = 0
+
+    def _append_actions_queue(self, original_actions: Tensor, processed_actions: Tensor):
+        """Append new actions to the queue (non-RTC mode).
+
+        Removes already-consumed actions and appends new ones, maintaining
+        queue continuity without replacement.
+
+        Args:
+            original_actions: Unprocessed actions from policy.
+            processed_actions: Post-processed actions for robot.
+        """
+        if self.queue is None:
+            self.original_queue = original_actions.clone()
+            self.queue = processed_actions.clone()
+            return
+
+        self.original_queue = torch.cat([self.original_queue, original_actions.clone()])
+        self.original_queue = self.original_queue[self.last_index :]
+
+        self.queue = torch.cat([self.queue, processed_actions.clone()])
+        self.queue = self.queue[self.last_index :]
+
+        self.last_index = 0
+
+    def _check_delays(self, real_delay: int, action_index_before_inference: int | None = None):
+        """Validate that computed delays match expectations.
+
+        Compares the delay computed from inference latency with the actual
+        number of actions consumed during inference.
+
+        Args:
+            real_delay: Delay computed from inference latency.
+            action_index_before_inference: Action index when inference started.
+        """
+        if action_index_before_inference is None:
+            return
+
+        indexes_diff = self.last_index - action_index_before_inference
+        if indexes_diff != real_delay:
+            # Let's check that action index difference (real delay calculated based on action queue)
+            # is the same as delay calculated based on inference latency
+            logger.warning(
+                f"[ACTION_QUEUE] Indexes diff is not equal to real delay. "
+                f"Indexes diff: {indexes_diff}, real delay: {real_delay}"
+            )
diff --git a/lerobot/src/lerobot/policies/rtc/configuration_rtc.py b/lerobot/src/lerobot/policies/rtc/configuration_rtc.py
new file mode 100644
index 0000000000000000000000000000000000000000..70a8dfb096a2ca4aacbe286f729d8ded3e4eda89
--- /dev/null
+++ b/lerobot/src/lerobot/policies/rtc/configuration_rtc.py
@@ -0,0 +1,55 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Real Time Chunking (RTC) and Bidirectional Decoding (BID) configuration classes.
+
+Based on:
+- Real Time Chunking: https://www.physicalintelligence.company/research/real_time_chunking
+"""
+
+from dataclasses import dataclass
+
+from lerobot.configs.types import RTCAttentionSchedule
+
+
+@dataclass
+class RTCConfig:
+    """Configuration for Real Time Chunking (RTC) inference.
+
+    RTC improves real-time inference by treating chunk generation as an inpainting problem,
+    strategically handling overlapping timesteps between action chunks using prefix attention.
+    """
+
+    # Infrastructure
+    enabled: bool = False
+
+    # Core RTC settings
+    # Todo change to exp
+    prefix_attention_schedule: RTCAttentionSchedule = RTCAttentionSchedule.LINEAR
+    max_guidance_weight: float = 10.0
+    execution_horizon: int = 10
+
+    # Debug settings
+    debug: bool = False
+    debug_maxlen: int = 100
+
+    def __post_init__(self):
+        """Validate RTC configuration parameters."""
+        if self.max_guidance_weight <= 0:
+            raise ValueError(f"max_guidance_weight must be positive, got {self.max_guidance_weight}")
+        if self.debug_maxlen <= 0:
+            raise ValueError(f"debug_maxlen must be positive, got {self.debug_maxlen}")
diff --git a/lerobot/src/lerobot/policies/rtc/debug_tracker.py b/lerobot/src/lerobot/policies/rtc/debug_tracker.py
new file mode 100644
index 0000000000000000000000000000000000000000..f143c223b3c94ba87b43ecee6782ff333bebc167
--- /dev/null
+++ b/lerobot/src/lerobot/policies/rtc/debug_tracker.py
@@ -0,0 +1,233 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Debug information handler for Real-Time Chunking (RTC)."""
+
+from dataclasses import dataclass, field
+from typing import Any
+
+import torch
+from torch import Tensor
+
+
+@dataclass
+class DebugStep:
+    """Container for debug information from a single denoising step.
+
+    Attributes:
+        step_idx (int): Step index/counter.
+        x_t (Tensor | None): Current latent/state tensor.
+        v_t (Tensor | None): Velocity from denoiser.
+        x1_t (Tensor | None): Denoised prediction (x_t - time * v_t).
+        correction (Tensor | None): Correction gradient tensor.
+        err (Tensor | None): Weighted error term.
+        weights (Tensor | None): Prefix attention weights.
+        guidance_weight (float | Tensor | None): Applied guidance weight.
+        time (float | Tensor | None): Time parameter.
+        inference_delay (int | None): Inference delay parameter.
+        execution_horizon (int | None): Execution horizon parameter.
+        metadata (dict[str, Any]): Additional metadata.
+    """
+
+    step_idx: int = 0
+    x_t: Tensor | None = None
+    v_t: Tensor | None = None
+    x1_t: Tensor | None = None
+    correction: Tensor | None = None
+    err: Tensor | None = None
+    weights: Tensor | None = None
+    guidance_weight: float | Tensor | None = None
+    time: float | Tensor | None = None
+    inference_delay: int | None = None
+    execution_horizon: int | None = None
+    metadata: dict[str, Any] = field(default_factory=dict)
+
+    def to_dict(self, include_tensors: bool = False) -> dict[str, Any]:
+        """Convert debug step to dictionary.
+
+        Args:
+            include_tensors (bool): If True, include tensor values. If False, only include
+                tensor statistics (shape, mean, std, min, max).
+
+        Returns:
+            Dictionary representation of the debug step.
+        """
+        result = {
+            "step_idx": self.step_idx,
+            "guidance_weight": (
+                self.guidance_weight.item()
+                if isinstance(self.guidance_weight, Tensor)
+                else self.guidance_weight
+            ),
+            "time": self.time.item() if isinstance(self.time, Tensor) else self.time,
+            "inference_delay": self.inference_delay,
+            "execution_horizon": self.execution_horizon,
+            "metadata": self.metadata.copy(),
+        }
+
+        # Add tensor information
+        tensor_fields = ["x_t", "v_t", "x1_t", "correction", "err", "weights"]
+        for field_name in tensor_fields:
+            tensor = getattr(self, field_name)
+            if tensor is not None:
+                if include_tensors:
+                    result[field_name] = tensor.detach().cpu()
+                else:
+                    result[f"{field_name}_stats"] = {
+                        "shape": tuple(tensor.shape),
+                        "mean": tensor.mean().item(),
+                        "std": tensor.std().item(),
+                        "min": tensor.min().item(),
+                        "max": tensor.max().item(),
+                    }
+
+        return result
+
+
+class Tracker:
+    """Collects and manages debug information for RTC processing.
+
+    This tracker stores debug information from recent denoising steps in a dictionary,
+    using time as the key for efficient lookups and updates.
+
+    Args:
+        enabled (bool): Whether debug collection is enabled.
+        maxlen (int | None): Optional sliding window size. If provided, only the
+            most recent ``maxlen`` debug steps are kept. If ``None``, keeps all.
+    """
+
+    def __init__(self, enabled: bool = False, maxlen: int = 100):
+        self.enabled = enabled
+        self._steps = {} if enabled else None  # Dictionary with time as key
+        self._maxlen = maxlen
+        self._step_counter = 0
+
+    def reset(self) -> None:
+        """Clear all recorded debug information."""
+        if self.enabled and self._steps is not None:
+            self._steps.clear()
+        self._step_counter = 0
+
+    @torch._dynamo.disable
+    def track(
+        self,
+        time: float | Tensor,
+        x_t: Tensor | None = None,
+        v_t: Tensor | None = None,
+        x1_t: Tensor | None = None,
+        correction: Tensor | None = None,
+        err: Tensor | None = None,
+        weights: Tensor | None = None,
+        guidance_weight: float | Tensor | None = None,
+        inference_delay: int | None = None,
+        execution_horizon: int | None = None,
+        **metadata,
+    ) -> None:
+        """Track debug information for a denoising step at a given time.
+
+        If a step with the given time already exists, it will be updated with the new data.
+        Otherwise, a new step will be created. Only non-None fields are updated/set.
+
+        Note: This method is excluded from torch.compile to avoid graph breaks from
+        operations like .item() which are incompatible with compiled graphs.
+
+        Args:
+            time (float | Tensor): Time parameter - used as the key to identify the step.
+            x_t (Tensor | None): Current latent/state tensor.
+            v_t (Tensor | None): Velocity from denoiser.
+            x1_t (Tensor | None): Denoised prediction.
+            correction (Tensor | None): Correction gradient tensor.
+            err (Tensor | None): Weighted error term.
+            weights (Tensor | None): Prefix attention weights.
+            guidance_weight (float | Tensor | None): Applied guidance weight.
+            inference_delay (int | None): Inference delay parameter.
+            execution_horizon (int | None): Execution horizon parameter.
+            **metadata: Additional metadata to store.
+        """
+        if not self.enabled:
+            return
+
+        # Convert time to float and round to avoid float precision issues
+        time_value = time.item() if isinstance(time, Tensor) else time
+        time_key = round(time_value, 6)  # Use rounded time as dictionary key
+
+        # Check if step with this time already exists
+        if time_key in self._steps:
+            # Update existing step with non-None fields
+            existing_step = self._steps[time_key]
+            if x_t is not None:
+                existing_step.x_t = x_t.detach().clone()
+            if v_t is not None:
+                existing_step.v_t = v_t.detach().clone()
+            if x1_t is not None:
+                existing_step.x1_t = x1_t.detach().clone()
+            if correction is not None:
+                existing_step.correction = correction.detach().clone()
+            if err is not None:
+                existing_step.err = err.detach().clone()
+            if weights is not None:
+                existing_step.weights = weights.detach().clone()
+            if guidance_weight is not None:
+                existing_step.guidance_weight = guidance_weight
+            if inference_delay is not None:
+                existing_step.inference_delay = inference_delay
+            if execution_horizon is not None:
+                existing_step.execution_horizon = execution_horizon
+            if metadata:
+                existing_step.metadata.update(metadata)
+        else:
+            # Create new step
+            step = DebugStep(
+                step_idx=self._step_counter,
+                x_t=x_t.detach().clone() if x_t is not None else None,
+                v_t=v_t.detach().clone() if v_t is not None else None,
+                x1_t=x1_t.detach().clone() if x1_t is not None else None,
+                correction=correction.detach().clone() if correction is not None else None,
+                err=err.detach().clone() if err is not None else None,
+                weights=weights.detach().clone() if weights is not None else None,
+                guidance_weight=guidance_weight,
+                time=time_value,
+                inference_delay=inference_delay,
+                execution_horizon=execution_horizon,
+                metadata=metadata,
+            )
+
+            # Add to dictionary
+            self._steps[time_key] = step
+            self._step_counter += 1
+
+            # Enforce maxlen if set
+            if self._maxlen is not None and len(self._steps) > self._maxlen:
+                # Remove oldest entry (first key in dict - Python 3.7+ preserves insertion order)
+                oldest_key = next(iter(self._steps))
+                del self._steps[oldest_key]
+
+    def get_all_steps(self) -> list[DebugStep]:
+        """Get all recorded debug steps.
+
+        Returns:
+            List of all DebugStep objects (may be empty if disabled).
+        """
+        if not self.enabled or self._steps is None:
+            return []
+
+        return list(self._steps.values())
+
+    def __len__(self) -> int:
+        """Return the number of recorded debug steps."""
+        if not self.enabled or self._steps is None:
+            return 0
+        return len(self._steps)
diff --git a/lerobot/src/lerobot/policies/rtc/debug_visualizer.py b/lerobot/src/lerobot/policies/rtc/debug_visualizer.py
new file mode 100644
index 0000000000000000000000000000000000000000..589c86c9595ddfd0daaba62b923f2d453aba00ef
--- /dev/null
+++ b/lerobot/src/lerobot/policies/rtc/debug_visualizer.py
@@ -0,0 +1,113 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Visualization utilities for RTC debug information."""
+
+import torch
+
+
+class RTCDebugVisualizer:
+    """Visualizer for RTC debug information.
+
+    This class provides methods to visualize debug information collected by the Tracker,
+    including corrections, errors, weights, and guidance weights over denoising steps.
+    """
+
+    @staticmethod
+    def plot_waypoints(
+        axes,
+        tensor,
+        start_from: int = 0,
+        color: str = "blue",
+        label: str = "",
+        alpha: float = 0.7,
+        linewidth: float = 2,
+        marker: str | None = None,
+        markersize: int = 4,
+    ):
+        """Plot trajectories across multiple dimensions.
+
+        This function plots a tensor's values across time for multiple dimensions,
+        with each dimension plotted on a separate axis.
+
+        Args:
+            axes: Array of matplotlib axes (one for each dimension).
+            tensor: The tensor to plot (can be torch.Tensor or numpy array).
+                   Shape should be (time_steps, num_dims) or (batch, time_steps, num_dims).
+            start_from: Starting index for the x-axis.
+            color: Color for the plot lines.
+            label: Label for the plot legend.
+            alpha: Transparency level for the plot.
+            linewidth: Width of the plot lines.
+            marker: Marker style for data points (e.g., 'o', 's', '^').
+            markersize: Size of the markers.
+        """
+        import numpy as np
+
+        # Handle None tensor
+        if tensor is None:
+            return
+
+        # Convert tensor to numpy if needed
+        tensor_np = tensor.detach().cpu().numpy() if isinstance(tensor, torch.Tensor) else tensor
+
+        # Handle different tensor shapes
+        if tensor_np.ndim == 3:
+            # If batch dimension present, take first batch
+            tensor_np = tensor_np[0]
+        elif tensor_np.ndim == 1:
+            # If 1D, reshape to (time_steps, 1)
+            tensor_np = tensor_np.reshape(-1, 1)
+
+        # Get dimensions
+        time_steps, num_dims = tensor_np.shape
+
+        # Create x-axis indices
+        x_indices = np.arange(start_from, start_from + time_steps)
+
+        # Plot each dimension on its corresponding axis
+        num_axes = len(axes) if hasattr(axes, "__len__") else 1
+        for dim_idx in range(min(num_dims, num_axes)):
+            ax = axes[dim_idx] if hasattr(axes, "__len__") else axes
+
+            # Plot the trajectory
+            if marker:
+                ax.plot(
+                    x_indices,
+                    tensor_np[:, dim_idx],
+                    color=color,
+                    label=label if dim_idx == 0 else "",  # Only show label once
+                    alpha=alpha,
+                    linewidth=linewidth,
+                    marker=marker,
+                    markersize=markersize,
+                )
+            else:
+                ax.plot(
+                    x_indices,
+                    tensor_np[:, dim_idx],
+                    color=color,
+                    label=label if dim_idx == 0 else "",  # Only show label once
+                    alpha=alpha,
+                    linewidth=linewidth,
+                )
+
+            # Add grid and labels if not already present
+            if not ax.xaxis.get_label().get_text():
+                ax.set_xlabel("Step", fontsize=10)
+            if not ax.yaxis.get_label().get_text():
+                ax.set_ylabel(f"Dim {dim_idx}", fontsize=10)
+            ax.grid(True, alpha=0.3)
diff --git a/lerobot/src/lerobot/policies/rtc/latency_tracker.py b/lerobot/src/lerobot/policies/rtc/latency_tracker.py
new file mode 100644
index 0000000000000000000000000000000000000000..e402cf1525c9a1853cc24cbd1f33ad3f92714e16
--- /dev/null
+++ b/lerobot/src/lerobot/policies/rtc/latency_tracker.py
@@ -0,0 +1,72 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Latency tracking utilities for Real-Time Chunking (RTC)."""
+
+from collections import deque
+
+import numpy as np
+
+
+class LatencyTracker:
+    """Tracks recent latencies and provides max/percentile queries.
+
+    Args:
+        maxlen (int | None): Optional sliding window size. If provided, only the
+            most recent ``maxlen`` latencies are kept. If ``None``, keeps all.
+    """
+
+    def __init__(self, maxlen: int = 100):
+        self._values = deque(maxlen=maxlen)
+        self.reset()
+
+    def reset(self) -> None:
+        """Clear all recorded latencies."""
+        self._values.clear()
+        self.max_latency = 0.0
+
+    def add(self, latency: float) -> None:
+        """Add a latency sample (seconds)."""
+        # Ensure numeric and non-negative
+        val = float(latency)
+
+        if val < 0:
+            return
+        self._values.append(val)
+        self.max_latency = max(self.max_latency, val)
+
+    def __len__(self) -> int:
+        return len(self._values)
+
+    def max(self) -> float | None:
+        """Return the maximum latency or None if empty."""
+        return self.max_latency
+
+    def percentile(self, q: float) -> float | None:
+        """Return the q-quantile (q in [0,1]) of recorded latencies or None if empty."""
+        if not self._values:
+            return 0.0
+        q = float(q)
+        if q <= 0.0:
+            return min(self._values)
+        if q >= 1.0:
+            return self.max_latency
+        vals = np.array(list(self._values), dtype=np.float32)
+        return float(np.quantile(vals, q))
+
+    def p95(self) -> float | None:
+        """Return the 95th percentile latency or None if empty."""
+        return self.percentile(0.95)
diff --git a/lerobot/src/lerobot/policies/rtc/modeling_rtc.py b/lerobot/src/lerobot/policies/rtc/modeling_rtc.py
new file mode 100644
index 0000000000000000000000000000000000000000..280905adf9fe5dd2493bb09827ddcf3d1a191f02
--- /dev/null
+++ b/lerobot/src/lerobot/policies/rtc/modeling_rtc.py
@@ -0,0 +1,297 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Real-Time Chunking (RTC) implementation for LeRobot.
+
+Based on Physical Intelligence's Kinetix implementation:
+https://github.com/Physical-Intelligence/real-time-chunking-kinetix/blob/main/src/model.py#L214
+"""
+
+import logging
+import math
+
+import torch
+from torch import Tensor
+
+from lerobot.configs.types import RTCAttentionSchedule
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+from lerobot.policies.rtc.debug_tracker import Tracker
+
+logger = logging.getLogger(__name__)
+
+
+class RTCProcessor:
+    """Real-Time Chunking processor for action chunking policies.
+
+    This class implements RTC techniques including velocity calculation,
+    prefix attention, and adaptive chunk processing.
+    """
+
+    def __init__(self, rtc_config: RTCConfig):
+        self.rtc_config = rtc_config
+
+        self.tracker = None
+
+        if rtc_config.debug:
+            self.tracker = Tracker(
+                enabled=rtc_config.debug,
+                maxlen=rtc_config.debug_maxlen,
+            )
+
+    # ====================== Tracker Proxy Methods ======================
+    def track(
+        self,
+        time: float | Tensor,
+        x_t: Tensor | None = None,
+        v_t: Tensor | None = None,
+        x1_t: Tensor | None = None,
+        correction: Tensor | None = None,
+        err: Tensor | None = None,
+        weights: Tensor | None = None,
+        guidance_weight: float | Tensor | None = None,
+        inference_delay: int | None = None,
+        execution_horizon: int | None = None,
+        **metadata,
+    ) -> None:
+        """Proxy method to track debug information.
+
+        If tracker is None or disabled, this method does nothing.
+        Otherwise, it forwards the call to tracker.track().
+        """
+        if self.tracker is not None:
+            self.tracker.track(
+                time=time,
+                x_t=x_t,
+                v_t=v_t,
+                x1_t=x1_t,
+                correction=correction,
+                err=err,
+                weights=weights,
+                guidance_weight=guidance_weight,
+                inference_delay=inference_delay,
+                execution_horizon=execution_horizon,
+                **metadata,
+            )
+
+    def get_all_debug_steps(self) -> list:
+        """Get all debug steps from tracker.
+
+        Returns empty list if tracker is disabled or None.
+        """
+        if self.tracker is not None:
+            return self.tracker.get_all_steps()
+        return []
+
+    def is_debug_enabled(self) -> bool:
+        """Check if debug tracking is enabled.
+
+        Returns True if tracker exists and is enabled.
+        """
+        return self.tracker is not None and self.tracker.enabled
+
+    def reset_tracker(self) -> None:
+        """Reset the tracker, clearing all recorded steps.
+
+        Does nothing if tracker is None.
+        """
+        if self.tracker is not None:
+            self.tracker.reset()
+
+    # ====================== End Tracker Proxy Methods ======================
+
+    def denoise_step(
+        self,
+        x_t,
+        prev_chunk_left_over,
+        inference_delay,
+        time,
+        original_denoise_step_partial,
+        execution_horizon=None,
+    ) -> Tensor:
+        """RTC guidance wrapper around an existing denoiser.
+
+        This method wraps an original denoising callable that only takes ``x_t`` and
+        returns a base denoised velocity ``v_t``. It then applies Real-Time Chunking
+        (RTC) prefix guidance using the leftover prefix from the previous chunk.
+
+        Args:
+            x_t (Tensor): Current latent/state to denoise. Shape ``(B, T, A)`` or ``(T, A)``.
+            prev_chunk_left_over (Tensor | None): Unexecuted prefix from the previous
+                chunk. Shape ``(B, T_prev, A)`` or ``(T_prev, A)``. If ``None``, no guidance
+                is applied and the method returns ``v_t`` from the original denoiser.
+            inference_delay (int): Number of timesteps from the prefix to use for guidance.
+            time (float | Tensor): Scalar in [0, 1] indicating normalized time. Must be
+                broadcastable with ``x_t``.
+            original_denoise_step_partial (Callable[[Tensor], Tensor]): Callable that
+                computes the base denoised velocity given only ``x_t``.
+            execution_horizon (int | None): Horizon used to build prefix weights. If
+                ``None``, defaults to ``self.rtc_config.execution_horizon``.
+
+        Returns:
+            Tensor: Guided velocity with the same shape as ``v_t``.
+
+        Notes:
+            - If inputs are 2D, a batch dimension is temporarily added and removed at the end.
+            - If ``prev_chunk_left_over`` is shorter than the current chunk length ``T``, it is
+              right-padded with zeros to match ``T``.
+            - Prefix weights are constructed via ``get_prefix_weights(inference_delay, execution_horizon, T)``
+              and broadcast to ``(B, T, A)``.
+            - Guidance correction is computed via autograd using ``x1_t = x_t + time * v_t`` and
+              ``error = (prev_chunk_left_over - x1_t) * weights``.
+            - The final guidance weight is clamped by ``max_guidance_weight`` from the config.
+
+        Reference:
+            https://www.physicalintelligence.company/download/real_time_chunking.pdf
+        """
+
+        # In the original implementation, the time goes from 0 to 1 and
+        # In our implementation, the time goes from 1 to 0
+        # So we need to invert the time
+        tau = 1 - time
+
+        if prev_chunk_left_over is None:
+            # First step, no guidance - return v_t
+            v_t = original_denoise_step_partial(x_t)
+            return v_t
+
+        x_t = x_t.clone().detach()
+
+        squeezed = False
+        if len(x_t.shape) < 3:
+            # Add batch dimension
+            x_t = x_t.unsqueeze(0)
+            squeezed = True
+
+        if len(prev_chunk_left_over.shape) < 3:
+            # Add batch dimension
+            prev_chunk_left_over = prev_chunk_left_over.unsqueeze(0)
+
+        if execution_horizon is None:
+            execution_horizon = self.rtc_config.execution_horizon
+
+        # If the previous action chunk is to short then it doesn't make sense to use long execution horizon
+        # because there is nothing to merge
+        if execution_horizon > prev_chunk_left_over.shape[1]:
+            execution_horizon = prev_chunk_left_over.shape[1]
+
+        batch_size = x_t.shape[0]
+        action_chunk_size = x_t.shape[1]
+        action_dim = x_t.shape[2]
+
+        if prev_chunk_left_over.shape[1] < action_chunk_size or prev_chunk_left_over.shape[2] < action_dim:
+            padded = torch.zeros(batch_size, action_chunk_size, action_dim).to(x_t.device)
+            padded[:, : prev_chunk_left_over.shape[1], : prev_chunk_left_over.shape[2]] = prev_chunk_left_over
+            prev_chunk_left_over = padded
+
+        assert prev_chunk_left_over.shape == x_t.shape, (
+            "The padded previous chunk must be the same size as the input tensor"
+        )
+
+        weights = (
+            self.get_prefix_weights(inference_delay, execution_horizon, action_chunk_size)
+            .to(x_t.device)
+            .unsqueeze(0)
+            .unsqueeze(-1)
+        )
+
+        with torch.enable_grad():
+            v_t = original_denoise_step_partial(x_t)
+            x_t.requires_grad_(True)
+
+            x1_t = x_t - time * v_t  # noqa: N806
+            err = (prev_chunk_left_over - x1_t) * weights
+            grad_outputs = err.clone().detach()
+            correction = torch.autograd.grad(x1_t, x_t, grad_outputs, retain_graph=False)[0]
+
+        max_guidance_weight = torch.as_tensor(self.rtc_config.max_guidance_weight)
+        tau_tensor = torch.as_tensor(tau)
+        squared_one_minus_tau = (1 - tau_tensor) ** 2
+        inv_r2 = (squared_one_minus_tau + tau_tensor**2) / (squared_one_minus_tau)
+        c = torch.nan_to_num((1 - tau_tensor) / tau_tensor, posinf=max_guidance_weight)
+        guidance_weight = torch.nan_to_num(c * inv_r2, posinf=max_guidance_weight)
+        guidance_weight = torch.minimum(guidance_weight, max_guidance_weight)
+
+        result = v_t - guidance_weight * correction
+
+        # Remove the batch dimension if it was added
+        if squeezed:
+            result = result.squeeze(0)
+            correction = correction.squeeze(0)
+            x1_t = x1_t.squeeze(0)
+            err = err.squeeze(0)
+
+        self.track(
+            time=time,
+            x1_t=x1_t,
+            correction=correction,
+            err=err,
+            weights=weights,
+            guidance_weight=guidance_weight,
+            inference_delay=inference_delay,
+            execution_horizon=execution_horizon,
+        )
+
+        return result
+
+    def get_prefix_weights(self, start, end, total):
+        start = min(start, end)
+
+        if self.rtc_config.prefix_attention_schedule == RTCAttentionSchedule.ZEROS:
+            weights = torch.zeros(total)
+            weights[:start] = 1.0
+        elif self.rtc_config.prefix_attention_schedule == RTCAttentionSchedule.ONES:
+            weights = torch.ones(total)
+            weights[end:] = 0.0
+        elif self.rtc_config.prefix_attention_schedule == RTCAttentionSchedule.LINEAR:
+            lin_weights = self._linweights(start, end, total)
+            weights = self._add_trailing_zeros(lin_weights, total, end)
+            weights = self._add_leading_ones(weights, start, total)
+        elif self.rtc_config.prefix_attention_schedule == RTCAttentionSchedule.EXP:
+            lin_weights = self._linweights(start, end, total)
+            lin_weights = lin_weights * torch.expm1(lin_weights).div(math.e - 1)
+            weights = self._add_trailing_zeros(lin_weights, total, end)
+            weights = self._add_leading_ones(weights, start, total)
+
+        return weights
+
+    def _linweights(self, start, end, total):
+        skip_steps_at_end = max(total - end, 0)
+
+        linspace_steps = total - skip_steps_at_end - start
+
+        if end <= start or linspace_steps <= 0:
+            return torch.tensor([])
+
+        return torch.linspace(1, 0, linspace_steps + 2)[1:-1]
+
+    def _add_trailing_zeros(self, weights, total, end):
+        zeros_len = total - end
+
+        if zeros_len <= 0:
+            return weights
+
+        zeros = torch.zeros(zeros_len)
+        return torch.cat([weights, zeros])
+
+    def _add_leading_ones(self, weights, start, total):
+        ones_len = min(start, total)
+
+        if ones_len <= 0:
+            return weights
+
+        ones = torch.ones(ones_len)
+        return torch.cat([ones, weights])
diff --git a/lerobot/src/lerobot/policies/sac/configuration_sac.py b/lerobot/src/lerobot/policies/sac/configuration_sac.py
new file mode 100644
index 0000000000000000000000000000000000000000..ada12330c94aa4249323ddcc699ba127744e150f
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sac/configuration_sac.py
@@ -0,0 +1,243 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team.
+# All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import NormalizationMode
+from lerobot.optim.optimizers import MultiAdamConfig
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_STATE
+
+
+def is_image_feature(key: str) -> bool:
+    """Check if a feature key represents an image feature.
+
+    Args:
+        key: The feature key to check
+
+    Returns:
+        True if the key represents an image feature, False otherwise
+    """
+    return key.startswith(OBS_IMAGE)
+
+
+@dataclass
+class ConcurrencyConfig:
+    """Configuration for the concurrency of the actor and learner.
+    Possible values are:
+    - "threads": Use threads for the actor and learner.
+    - "processes": Use processes for the actor and learner.
+    """
+
+    actor: str = "threads"
+    learner: str = "threads"
+
+
+@dataclass
+class ActorLearnerConfig:
+    learner_host: str = "127.0.0.1"
+    learner_port: int = 50051
+    policy_parameters_push_frequency: int = 4
+    queue_get_timeout: float = 2
+
+
+@dataclass
+class CriticNetworkConfig:
+    hidden_dims: list[int] = field(default_factory=lambda: [256, 256])
+    activate_final: bool = True
+    final_activation: str | None = None
+
+
+@dataclass
+class ActorNetworkConfig:
+    hidden_dims: list[int] = field(default_factory=lambda: [256, 256])
+    activate_final: bool = True
+
+
+@dataclass
+class PolicyConfig:
+    use_tanh_squash: bool = True
+    std_min: float = 1e-5
+    std_max: float = 10.0
+    init_final: float = 0.05
+
+
+@PreTrainedConfig.register_subclass("sac")
+@dataclass
+class SACConfig(PreTrainedConfig):
+    """Soft Actor-Critic (SAC) configuration.
+
+    SAC is an off-policy actor-critic deep RL algorithm based on the maximum entropy
+    reinforcement learning framework. It learns a policy and a Q-function simultaneously
+    using experience collected from the environment.
+
+    This configuration class contains all the parameters needed to define a SAC agent,
+    including network architectures, optimization settings, and algorithm-specific
+    hyperparameters.
+    """
+
+    # Mapping of feature types to normalization modes
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.MEAN_STD,
+            "STATE": NormalizationMode.MIN_MAX,
+            "ENV": NormalizationMode.MIN_MAX,
+            "ACTION": NormalizationMode.MIN_MAX,
+        }
+    )
+
+    # Statistics for normalizing different types of inputs
+    dataset_stats: dict[str, dict[str, list[float]]] | None = field(
+        default_factory=lambda: {
+            OBS_IMAGE: {
+                "mean": [0.485, 0.456, 0.406],
+                "std": [0.229, 0.224, 0.225],
+            },
+            OBS_STATE: {
+                "min": [0.0, 0.0],
+                "max": [1.0, 1.0],
+            },
+            ACTION: {
+                "min": [0.0, 0.0, 0.0],
+                "max": [1.0, 1.0, 1.0],
+            },
+        }
+    )
+
+    # Architecture specifics
+    # Device to run the model on (e.g., "cuda", "cpu")
+    device: str = "cpu"
+    # Device to store the model on
+    storage_device: str = "cpu"
+    # Name of the vision encoder model (Set to "helper2424/resnet10" for hil serl resnet10)
+    vision_encoder_name: str | None = None
+    # Whether to freeze the vision encoder during training
+    freeze_vision_encoder: bool = True
+    # Hidden dimension size for the image encoder
+    image_encoder_hidden_dim: int = 32
+    # Whether to use a shared encoder for actor and critic
+    shared_encoder: bool = True
+    # Number of discrete actions, eg for gripper actions
+    num_discrete_actions: int | None = None
+    # Dimension of the image embedding pooling
+    image_embedding_pooling_dim: int = 8
+
+    # Training parameter
+    # Number of steps for online training
+    online_steps: int = 1000000
+    # Capacity of the online replay buffer
+    online_buffer_capacity: int = 100000
+    # Capacity of the offline replay buffer
+    offline_buffer_capacity: int = 100000
+    # Whether to use asynchronous prefetching for the buffers
+    async_prefetch: bool = False
+    # Number of steps before learning starts
+    online_step_before_learning: int = 100
+    # Frequency of policy updates
+    policy_update_freq: int = 1
+
+    # SAC algorithm parameters
+    # Discount factor for the SAC algorithm
+    discount: float = 0.99
+    # Initial temperature value
+    temperature_init: float = 1.0
+    # Number of critics in the ensemble
+    num_critics: int = 2
+    # Number of subsampled critics for training
+    num_subsample_critics: int | None = None
+    # Learning rate for the critic network
+    critic_lr: float = 3e-4
+    # Learning rate for the actor network
+    actor_lr: float = 3e-4
+    # Learning rate for the temperature parameter
+    temperature_lr: float = 3e-4
+    # Weight for the critic target update
+    critic_target_update_weight: float = 0.005
+    # Update-to-data ratio for the UTD algorithm (If you want enable utd_ratio, you need to set it to >1)
+    utd_ratio: int = 1
+    # Hidden dimension size for the state encoder
+    state_encoder_hidden_dim: int = 256
+    # Dimension of the latent space
+    latent_dim: int = 256
+    # Target entropy for the SAC algorithm
+    target_entropy: float | None = None
+    # Whether to use backup entropy for the SAC algorithm
+    use_backup_entropy: bool = True
+    # Gradient clipping norm for the SAC algorithm
+    grad_clip_norm: float = 40.0
+
+    # Network configuration
+    # Configuration for the critic network architecture
+    critic_network_kwargs: CriticNetworkConfig = field(default_factory=CriticNetworkConfig)
+    # Configuration for the actor network architecture
+    actor_network_kwargs: ActorNetworkConfig = field(default_factory=ActorNetworkConfig)
+    # Configuration for the policy parameters
+    policy_kwargs: PolicyConfig = field(default_factory=PolicyConfig)
+    # Configuration for the discrete critic network
+    discrete_critic_network_kwargs: CriticNetworkConfig = field(default_factory=CriticNetworkConfig)
+    # Configuration for actor-learner architecture
+    actor_learner_config: ActorLearnerConfig = field(default_factory=ActorLearnerConfig)
+    # Configuration for concurrency settings (you can use threads or processes for the actor and learner)
+    concurrency: ConcurrencyConfig = field(default_factory=ConcurrencyConfig)
+
+    # Optimizations
+    use_torch_compile: bool = True
+
+    def __post_init__(self):
+        super().__post_init__()
+        # Any validation specific to SAC configuration
+
+    def get_optimizer_preset(self) -> MultiAdamConfig:
+        return MultiAdamConfig(
+            weight_decay=0.0,
+            optimizer_groups={
+                "actor": {"lr": self.actor_lr},
+                "critic": {"lr": self.critic_lr},
+                "temperature": {"lr": self.temperature_lr},
+            },
+        )
+
+    def get_scheduler_preset(self) -> None:
+        return None
+
+    def validate_features(self) -> None:
+        has_image = any(is_image_feature(key) for key in self.input_features)
+        has_state = OBS_STATE in self.input_features
+
+        if not (has_state or has_image):
+            raise ValueError(
+                "You must provide either 'observation.state' or an image observation (key starting with 'observation.image') in the input features"
+            )
+
+        if ACTION not in self.output_features:
+            raise ValueError("You must provide 'action' in the output features")
+
+    @property
+    def image_features(self) -> list[str]:
+        return [key for key in self.input_features if is_image_feature(key)]
+
+    @property
+    def observation_delta_indices(self) -> list:
+        return None
+
+    @property
+    def action_delta_indices(self) -> list:
+        return None  # SAC typically predicts one action at a time
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/sac/modeling_sac.py b/lerobot/src/lerobot/policies/sac/modeling_sac.py
new file mode 100644
index 0000000000000000000000000000000000000000..d5dd71a484182084e6a2d1b343000c545bd8b670
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sac/modeling_sac.py
@@ -0,0 +1,1064 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team.
+# All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import math
+from collections.abc import Callable
+from dataclasses import asdict
+from typing import Literal
+
+import einops
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.nn.functional as F  # noqa: N812
+from torch import Tensor
+from torch.distributions import MultivariateNormal, TanhTransform, Transform, TransformedDistribution
+
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.sac.configuration_sac import SACConfig, is_image_feature
+from lerobot.policies.utils import get_device_from_parameters
+from lerobot.utils.constants import ACTION, OBS_ENV_STATE, OBS_STATE
+
+DISCRETE_DIMENSION_INDEX = -1  # Gripper is always the last dimension
+
+
+class SACPolicy(
+    PreTrainedPolicy,
+):
+    config_class = SACConfig
+    name = "sac"
+
+    def __init__(
+        self,
+        config: SACConfig | None = None,
+    ):
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        # Determine action dimension and initialize all components
+        continuous_action_dim = config.output_features[ACTION].shape[0]
+        self._init_encoders()
+        self._init_critics(continuous_action_dim)
+        self._init_actor(continuous_action_dim)
+        self._init_temperature()
+
+    def get_optim_params(self) -> dict:
+        optim_params = {
+            "actor": [
+                p
+                for n, p in self.actor.named_parameters()
+                if not n.startswith("encoder") or not self.shared_encoder
+            ],
+            "critic": self.critic_ensemble.parameters(),
+            "temperature": self.log_alpha,
+        }
+        if self.config.num_discrete_actions is not None:
+            optim_params["discrete_critic"] = self.discrete_critic.parameters()
+        return optim_params
+
+    def reset(self):
+        """Reset the policy"""
+        pass
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor]) -> Tensor:
+        """Predict a chunk of actions given environment observations."""
+        raise NotImplementedError("SACPolicy does not support action chunking. It returns single actions!")
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select action for inference/evaluation"""
+
+        observations_features = None
+        if self.shared_encoder and self.actor.encoder.has_images:
+            observations_features = self.actor.encoder.get_cached_image_features(batch)
+
+        actions, _, _ = self.actor(batch, observations_features)
+
+        if self.config.num_discrete_actions is not None:
+            discrete_action_value = self.discrete_critic(batch, observations_features)
+            discrete_action = torch.argmax(discrete_action_value, dim=-1, keepdim=True)
+            actions = torch.cat([actions, discrete_action], dim=-1)
+
+        return actions
+
+    def critic_forward(
+        self,
+        observations: dict[str, Tensor],
+        actions: Tensor,
+        use_target: bool = False,
+        observation_features: Tensor | None = None,
+    ) -> Tensor:
+        """Forward pass through a critic network ensemble
+
+        Args:
+            observations: Dictionary of observations
+            actions: Action tensor
+            use_target: If True, use target critics, otherwise use ensemble critics
+
+        Returns:
+            Tensor of Q-values from all critics
+        """
+
+        critics = self.critic_target if use_target else self.critic_ensemble
+        q_values = critics(observations, actions, observation_features)
+        return q_values
+
+    def discrete_critic_forward(
+        self, observations, use_target=False, observation_features=None
+    ) -> torch.Tensor:
+        """Forward pass through a discrete critic network
+
+        Args:
+            observations: Dictionary of observations
+            use_target: If True, use target critics, otherwise use ensemble critics
+            observation_features: Optional pre-computed observation features to avoid recomputing encoder output
+
+        Returns:
+            Tensor of Q-values from the discrete critic network
+        """
+        discrete_critic = self.discrete_critic_target if use_target else self.discrete_critic
+        q_values = discrete_critic(observations, observation_features)
+        return q_values
+
+    def forward(
+        self,
+        batch: dict[str, Tensor | dict[str, Tensor]],
+        model: Literal["actor", "critic", "temperature", "discrete_critic"] = "critic",
+    ) -> dict[str, Tensor]:
+        """Compute the loss for the given model
+
+        Args:
+            batch: Dictionary containing:
+                - action: Action tensor
+                - reward: Reward tensor
+                - state: Observations tensor dict
+                - next_state: Next observations tensor dict
+                - done: Done mask tensor
+                - observation_feature: Optional pre-computed observation features
+                - next_observation_feature: Optional pre-computed next observation features
+            model: Which model to compute the loss for ("actor", "critic", "discrete_critic", or "temperature")
+
+        Returns:
+            The computed loss tensor
+        """
+        # Extract common components from batch
+        actions: Tensor = batch[ACTION]
+        observations: dict[str, Tensor] = batch["state"]
+        observation_features: Tensor = batch.get("observation_feature")
+
+        if model == "critic":
+            # Extract critic-specific components
+            rewards: Tensor = batch["reward"]
+            next_observations: dict[str, Tensor] = batch["next_state"]
+            done: Tensor = batch["done"]
+            next_observation_features: Tensor = batch.get("next_observation_feature")
+
+            loss_critic = self.compute_loss_critic(
+                observations=observations,
+                actions=actions,
+                rewards=rewards,
+                next_observations=next_observations,
+                done=done,
+                observation_features=observation_features,
+                next_observation_features=next_observation_features,
+            )
+
+            return {"loss_critic": loss_critic}
+
+        if model == "discrete_critic" and self.config.num_discrete_actions is not None:
+            # Extract critic-specific components
+            rewards: Tensor = batch["reward"]
+            next_observations: dict[str, Tensor] = batch["next_state"]
+            done: Tensor = batch["done"]
+            next_observation_features: Tensor = batch.get("next_observation_feature")
+            complementary_info = batch.get("complementary_info")
+            loss_discrete_critic = self.compute_loss_discrete_critic(
+                observations=observations,
+                actions=actions,
+                rewards=rewards,
+                next_observations=next_observations,
+                done=done,
+                observation_features=observation_features,
+                next_observation_features=next_observation_features,
+                complementary_info=complementary_info,
+            )
+            return {"loss_discrete_critic": loss_discrete_critic}
+        if model == "actor":
+            return {
+                "loss_actor": self.compute_loss_actor(
+                    observations=observations,
+                    observation_features=observation_features,
+                )
+            }
+
+        if model == "temperature":
+            return {
+                "loss_temperature": self.compute_loss_temperature(
+                    observations=observations,
+                    observation_features=observation_features,
+                )
+            }
+
+        raise ValueError(f"Unknown model type: {model}")
+
+    def update_target_networks(self):
+        """Update target networks with exponential moving average"""
+        for target_param, param in zip(
+            self.critic_target.parameters(),
+            self.critic_ensemble.parameters(),
+            strict=True,
+        ):
+            target_param.data.copy_(
+                param.data * self.config.critic_target_update_weight
+                + target_param.data * (1.0 - self.config.critic_target_update_weight)
+            )
+        if self.config.num_discrete_actions is not None:
+            for target_param, param in zip(
+                self.discrete_critic_target.parameters(),
+                self.discrete_critic.parameters(),
+                strict=True,
+            ):
+                target_param.data.copy_(
+                    param.data * self.config.critic_target_update_weight
+                    + target_param.data * (1.0 - self.config.critic_target_update_weight)
+                )
+
+    @property
+    def temperature(self) -> float:
+        """Return the current temperature value, always in sync with log_alpha."""
+        return self.log_alpha.exp().item()
+
+    def compute_loss_critic(
+        self,
+        observations,
+        actions,
+        rewards,
+        next_observations,
+        done,
+        observation_features: Tensor | None = None,
+        next_observation_features: Tensor | None = None,
+    ) -> Tensor:
+        with torch.no_grad():
+            next_action_preds, next_log_probs, _ = self.actor(next_observations, next_observation_features)
+
+            # 2- compute q targets
+            q_targets = self.critic_forward(
+                observations=next_observations,
+                actions=next_action_preds,
+                use_target=True,
+                observation_features=next_observation_features,
+            )
+
+            # subsample critics to prevent overfitting if use high UTD (update to date)
+            # TODO: Get indices before forward pass to avoid unnecessary computation
+            if self.config.num_subsample_critics is not None:
+                indices = torch.randperm(self.config.num_critics)
+                indices = indices[: self.config.num_subsample_critics]
+                q_targets = q_targets[indices]
+
+            # critics subsample size
+            min_q, _ = q_targets.min(dim=0)  # Get values from min operation
+            if self.config.use_backup_entropy:
+                min_q = min_q - (self.temperature * next_log_probs)
+
+            td_target = rewards + (1 - done) * self.config.discount * min_q
+
+        # 3- compute predicted qs
+        if self.config.num_discrete_actions is not None:
+            # NOTE: We only want to keep the continuous action part
+            # In the buffer we have the full action space (continuous + discrete)
+            # We need to split them before concatenating them in the critic forward
+            actions: Tensor = actions[:, :DISCRETE_DIMENSION_INDEX]
+        q_preds = self.critic_forward(
+            observations=observations,
+            actions=actions,
+            use_target=False,
+            observation_features=observation_features,
+        )
+
+        # 4- Calculate loss
+        # Compute state-action value loss (TD loss) for all of the Q functions in the ensemble.
+        td_target_duplicate = einops.repeat(td_target, "b -> e b", e=q_preds.shape[0])
+        # You compute the mean loss of the batch for each critic and then to compute the final loss you sum them up
+        critics_loss = (
+            F.mse_loss(
+                input=q_preds,
+                target=td_target_duplicate,
+                reduction="none",
+            ).mean(dim=1)
+        ).sum()
+        return critics_loss
+
+    def compute_loss_discrete_critic(
+        self,
+        observations,
+        actions,
+        rewards,
+        next_observations,
+        done,
+        observation_features=None,
+        next_observation_features=None,
+        complementary_info=None,
+    ):
+        # NOTE: We only want to keep the discrete action part
+        # In the buffer we have the full action space (continuous + discrete)
+        # We need to split them before concatenating them in the critic forward
+        actions_discrete: Tensor = actions[:, DISCRETE_DIMENSION_INDEX:].clone()
+        actions_discrete = torch.round(actions_discrete)
+        actions_discrete = actions_discrete.long()
+
+        discrete_penalties: Tensor | None = None
+        if complementary_info is not None:
+            discrete_penalties: Tensor | None = complementary_info.get("discrete_penalty")
+
+        with torch.no_grad():
+            # For DQN, select actions using online network, evaluate with target network
+            next_discrete_qs = self.discrete_critic_forward(
+                next_observations, use_target=False, observation_features=next_observation_features
+            )
+            best_next_discrete_action = torch.argmax(next_discrete_qs, dim=-1, keepdim=True)
+
+            # Get target Q-values from target network
+            target_next_discrete_qs = self.discrete_critic_forward(
+                observations=next_observations,
+                use_target=True,
+                observation_features=next_observation_features,
+            )
+
+            # Use gather to select Q-values for best actions
+            target_next_discrete_q = torch.gather(
+                target_next_discrete_qs, dim=1, index=best_next_discrete_action
+            ).squeeze(-1)
+
+            # Compute target Q-value with Bellman equation
+            rewards_discrete = rewards
+            if discrete_penalties is not None:
+                rewards_discrete = rewards + discrete_penalties
+            target_discrete_q = rewards_discrete + (1 - done) * self.config.discount * target_next_discrete_q
+
+        # Get predicted Q-values for current observations
+        predicted_discrete_qs = self.discrete_critic_forward(
+            observations=observations, use_target=False, observation_features=observation_features
+        )
+
+        # Use gather to select Q-values for taken actions
+        predicted_discrete_q = torch.gather(predicted_discrete_qs, dim=1, index=actions_discrete).squeeze(-1)
+
+        # Compute MSE loss between predicted and target Q-values
+        discrete_critic_loss = F.mse_loss(input=predicted_discrete_q, target=target_discrete_q)
+        return discrete_critic_loss
+
+    def compute_loss_temperature(self, observations, observation_features: Tensor | None = None) -> Tensor:
+        """Compute the temperature loss"""
+        # calculate temperature loss
+        with torch.no_grad():
+            _, log_probs, _ = self.actor(observations, observation_features)
+        temperature_loss = (-self.log_alpha.exp() * (log_probs + self.target_entropy)).mean()
+        return temperature_loss
+
+    def compute_loss_actor(
+        self,
+        observations,
+        observation_features: Tensor | None = None,
+    ) -> Tensor:
+        actions_pi, log_probs, _ = self.actor(observations, observation_features)
+
+        q_preds = self.critic_forward(
+            observations=observations,
+            actions=actions_pi,
+            use_target=False,
+            observation_features=observation_features,
+        )
+        min_q_preds = q_preds.min(dim=0)[0]
+
+        actor_loss = ((self.temperature * log_probs) - min_q_preds).mean()
+        return actor_loss
+
+    def _init_encoders(self):
+        """Initialize shared or separate encoders for actor and critic."""
+        self.shared_encoder = self.config.shared_encoder
+        self.encoder_critic = SACObservationEncoder(self.config)
+        self.encoder_actor = (
+            self.encoder_critic if self.shared_encoder else SACObservationEncoder(self.config)
+        )
+
+    def _init_critics(self, continuous_action_dim):
+        """Build critic ensemble, targets, and optional discrete critic."""
+        heads = [
+            CriticHead(
+                input_dim=self.encoder_critic.output_dim + continuous_action_dim,
+                **asdict(self.config.critic_network_kwargs),
+            )
+            for _ in range(self.config.num_critics)
+        ]
+        self.critic_ensemble = CriticEnsemble(encoder=self.encoder_critic, ensemble=heads)
+        target_heads = [
+            CriticHead(
+                input_dim=self.encoder_critic.output_dim + continuous_action_dim,
+                **asdict(self.config.critic_network_kwargs),
+            )
+            for _ in range(self.config.num_critics)
+        ]
+        self.critic_target = CriticEnsemble(encoder=self.encoder_critic, ensemble=target_heads)
+        self.critic_target.load_state_dict(self.critic_ensemble.state_dict())
+
+        if self.config.use_torch_compile:
+            self.critic_ensemble = torch.compile(self.critic_ensemble)
+            self.critic_target = torch.compile(self.critic_target)
+
+        if self.config.num_discrete_actions is not None:
+            self._init_discrete_critics()
+
+    def _init_discrete_critics(self):
+        """Build discrete discrete critic ensemble and target networks."""
+        self.discrete_critic = DiscreteCritic(
+            encoder=self.encoder_critic,
+            input_dim=self.encoder_critic.output_dim,
+            output_dim=self.config.num_discrete_actions,
+            **asdict(self.config.discrete_critic_network_kwargs),
+        )
+        self.discrete_critic_target = DiscreteCritic(
+            encoder=self.encoder_critic,
+            input_dim=self.encoder_critic.output_dim,
+            output_dim=self.config.num_discrete_actions,
+            **asdict(self.config.discrete_critic_network_kwargs),
+        )
+
+        # TODO: (maractingi, azouitine) Compile the discrete critic
+        self.discrete_critic_target.load_state_dict(self.discrete_critic.state_dict())
+
+    def _init_actor(self, continuous_action_dim):
+        """Initialize policy actor network and default target entropy."""
+        # NOTE: The actor select only the continuous action part
+        self.actor = Policy(
+            encoder=self.encoder_actor,
+            network=MLP(input_dim=self.encoder_actor.output_dim, **asdict(self.config.actor_network_kwargs)),
+            action_dim=continuous_action_dim,
+            encoder_is_shared=self.shared_encoder,
+            **asdict(self.config.policy_kwargs),
+        )
+
+        self.target_entropy = self.config.target_entropy
+        if self.target_entropy is None:
+            dim = continuous_action_dim + (1 if self.config.num_discrete_actions is not None else 0)
+            self.target_entropy = -np.prod(dim) / 2
+
+    def _init_temperature(self) -> None:
+        """Set up temperature parameter (log_alpha)."""
+        temp_init = self.config.temperature_init
+        self.log_alpha = nn.Parameter(torch.tensor([math.log(temp_init)]))
+
+
+class SACObservationEncoder(nn.Module):
+    """Encode image and/or state vector observations."""
+
+    def __init__(self, config: SACConfig) -> None:
+        super().__init__()
+        self.config = config
+        self._init_image_layers()
+        self._init_state_layers()
+        self._compute_output_dim()
+
+    def _init_image_layers(self) -> None:
+        self.image_keys = [k for k in self.config.input_features if is_image_feature(k)]
+        self.has_images = bool(self.image_keys)
+        if not self.has_images:
+            return
+
+        if self.config.vision_encoder_name is not None:
+            self.image_encoder = PretrainedImageEncoder(self.config)
+        else:
+            self.image_encoder = DefaultImageEncoder(self.config)
+
+        if self.config.freeze_vision_encoder:
+            freeze_image_encoder(self.image_encoder)
+
+        dummy = torch.zeros(1, *self.config.input_features[self.image_keys[0]].shape)
+        with torch.no_grad():
+            _, channels, height, width = self.image_encoder(dummy).shape
+
+        self.spatial_embeddings = nn.ModuleDict()
+        self.post_encoders = nn.ModuleDict()
+
+        for key in self.image_keys:
+            name = key.replace(".", "_")
+            self.spatial_embeddings[name] = SpatialLearnedEmbeddings(
+                height=height,
+                width=width,
+                channel=channels,
+                num_features=self.config.image_embedding_pooling_dim,
+            )
+            self.post_encoders[name] = nn.Sequential(
+                nn.Dropout(0.1),
+                nn.Linear(
+                    in_features=channels * self.config.image_embedding_pooling_dim,
+                    out_features=self.config.latent_dim,
+                ),
+                nn.LayerNorm(normalized_shape=self.config.latent_dim),
+                nn.Tanh(),
+            )
+
+    def _init_state_layers(self) -> None:
+        self.has_env = OBS_ENV_STATE in self.config.input_features
+        self.has_state = OBS_STATE in self.config.input_features
+        if self.has_env:
+            dim = self.config.input_features[OBS_ENV_STATE].shape[0]
+            self.env_encoder = nn.Sequential(
+                nn.Linear(dim, self.config.latent_dim),
+                nn.LayerNorm(self.config.latent_dim),
+                nn.Tanh(),
+            )
+        if self.has_state:
+            dim = self.config.input_features[OBS_STATE].shape[0]
+            self.state_encoder = nn.Sequential(
+                nn.Linear(dim, self.config.latent_dim),
+                nn.LayerNorm(self.config.latent_dim),
+                nn.Tanh(),
+            )
+
+    def _compute_output_dim(self) -> None:
+        out = 0
+        if self.has_images:
+            out += len(self.image_keys) * self.config.latent_dim
+        if self.has_env:
+            out += self.config.latent_dim
+        if self.has_state:
+            out += self.config.latent_dim
+        self._out_dim = out
+
+    def forward(
+        self, obs: dict[str, Tensor], cache: dict[str, Tensor] | None = None, detach: bool = False
+    ) -> Tensor:
+        parts = []
+        if self.has_images:
+            if cache is None:
+                cache = self.get_cached_image_features(obs)
+            parts.append(self._encode_images(cache, detach))
+        if self.has_env:
+            parts.append(self.env_encoder(obs[OBS_ENV_STATE]))
+        if self.has_state:
+            parts.append(self.state_encoder(obs[OBS_STATE]))
+        if parts:
+            return torch.cat(parts, dim=-1)
+
+        raise ValueError(
+            "No parts to concatenate, you should have at least one image or environment state or state"
+        )
+
+    def get_cached_image_features(self, obs: dict[str, Tensor]) -> dict[str, Tensor]:
+        """Extract and optionally cache image features from observations.
+
+        This function processes image observations through the vision encoder once and returns
+        the resulting features.
+        When the image encoder is shared between actor and critics AND frozen, these features can be safely cached and
+        reused across policy components (actor, critic, discrete_critic), avoiding redundant forward passes.
+
+        Performance impact:
+        - The vision encoder forward pass is typically the main computational bottleneck during training and inference
+        - Caching these features can provide 2-4x speedup in training and inference
+
+        Usage patterns:
+        - Called in select_action()
+        - Called in learner.py's get_observation_features() to pre-compute features for all policy components
+        - Called internally by forward()
+
+        Args:
+            obs: Dictionary of observation tensors containing image keys
+
+        Returns:
+            Dictionary mapping image keys to their corresponding encoded features
+        """
+        batched = torch.cat([obs[k] for k in self.image_keys], dim=0)
+        out = self.image_encoder(batched)
+        chunks = torch.chunk(out, len(self.image_keys), dim=0)
+        return dict(zip(self.image_keys, chunks, strict=False))
+
+    def _encode_images(self, cache: dict[str, Tensor], detach: bool) -> Tensor:
+        """Encode image features from cached observations.
+
+        This function takes pre-encoded image features from the cache and applies spatial embeddings and post-encoders.
+        It also supports detaching the encoded features if specified.
+
+        Args:
+            cache (dict[str, Tensor]): The cached image features.
+            detach (bool): Usually when the encoder is shared between actor and critics,
+            we want to detach the encoded features on the policy side to avoid backprop through the encoder.
+            More detail here `https://cdn.aaai.org/ojs/17276/17276-13-20770-1-2-20210518.pdf`
+
+        Returns:
+            Tensor: The encoded image features.
+        """
+        feats = []
+        for k, feat in cache.items():
+            safe_key = k.replace(".", "_")
+            x = self.spatial_embeddings[safe_key](feat)
+            x = self.post_encoders[safe_key](x)
+            if detach:
+                x = x.detach()
+            feats.append(x)
+        return torch.cat(feats, dim=-1)
+
+    @property
+    def output_dim(self) -> int:
+        return self._out_dim
+
+
+class MLP(nn.Module):
+    """Multi-layer perceptron builder.
+
+    Dynamically constructs a sequence of layers based on `hidden_dims`:
+      1) Linear (in_dim -> out_dim)
+      2) Optional Dropout if `dropout_rate` > 0 and (not final layer or `activate_final`)
+      3) LayerNorm on the output features
+      4) Activation (standard for intermediate layers, `final_activation` for last layer if `activate_final`)
+
+    Arguments:
+        input_dim (int): Size of input feature dimension.
+        hidden_dims (list[int]): Sizes for each hidden layer.
+        activations (Callable or str): Activation to apply between layers.
+        activate_final (bool): Whether to apply activation at the final layer.
+        dropout_rate (Optional[float]): Dropout probability applied before normalization and activation.
+        final_activation (Optional[Callable or str]): Activation for the final layer when `activate_final` is True.
+
+    For each layer, `in_dim` is updated to the previous `out_dim`. All constructed modules are
+    stored in `self.net` as an `nn.Sequential` container.
+    """
+
+    def __init__(
+        self,
+        input_dim: int,
+        hidden_dims: list[int],
+        activations: Callable[[torch.Tensor], torch.Tensor] | str = nn.SiLU(),
+        activate_final: bool = False,
+        dropout_rate: float | None = None,
+        final_activation: Callable[[torch.Tensor], torch.Tensor] | str | None = None,
+    ):
+        super().__init__()
+        layers: list[nn.Module] = []
+        in_dim = input_dim
+        total = len(hidden_dims)
+
+        for idx, out_dim in enumerate(hidden_dims):
+            # 1) linear transform
+            layers.append(nn.Linear(in_dim, out_dim))
+
+            is_last = idx == total - 1
+            # 2-4) optionally add dropout, normalization, and activation
+            if not is_last or activate_final:
+                if dropout_rate and dropout_rate > 0:
+                    layers.append(nn.Dropout(p=dropout_rate))
+                layers.append(nn.LayerNorm(out_dim))
+                act_cls = final_activation if is_last and final_activation else activations
+                act = act_cls if isinstance(act_cls, nn.Module) else getattr(nn, act_cls)()
+                layers.append(act)
+
+            in_dim = out_dim
+
+        self.net = nn.Sequential(*layers)
+
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        return self.net(x)
+
+
+class CriticHead(nn.Module):
+    def __init__(
+        self,
+        input_dim: int,
+        hidden_dims: list[int],
+        activations: Callable[[torch.Tensor], torch.Tensor] | str = nn.SiLU(),
+        activate_final: bool = False,
+        dropout_rate: float | None = None,
+        init_final: float | None = None,
+        final_activation: Callable[[torch.Tensor], torch.Tensor] | str | None = None,
+    ):
+        super().__init__()
+        self.net = MLP(
+            input_dim=input_dim,
+            hidden_dims=hidden_dims,
+            activations=activations,
+            activate_final=activate_final,
+            dropout_rate=dropout_rate,
+            final_activation=final_activation,
+        )
+        self.output_layer = nn.Linear(in_features=hidden_dims[-1], out_features=1)
+        if init_final is not None:
+            nn.init.uniform_(self.output_layer.weight, -init_final, init_final)
+            nn.init.uniform_(self.output_layer.bias, -init_final, init_final)
+        else:
+            orthogonal_init()(self.output_layer.weight)
+
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        return self.output_layer(self.net(x))
+
+
+class CriticEnsemble(nn.Module):
+    """
+    CriticEnsemble wraps multiple CriticHead modules into an ensemble.
+
+    Args:
+        encoder (SACObservationEncoder): encoder for observations.
+        ensemble (List[CriticHead]): list of critic heads.
+        init_final (float | None): optional initializer scale for final layers.
+
+    Forward returns a tensor of shape (num_critics, batch_size) containing Q-values.
+    """
+
+    def __init__(
+        self,
+        encoder: SACObservationEncoder,
+        ensemble: list[CriticHead],
+        init_final: float | None = None,
+    ):
+        super().__init__()
+        self.encoder = encoder
+        self.init_final = init_final
+        self.critics = nn.ModuleList(ensemble)
+
+    def forward(
+        self,
+        observations: dict[str, torch.Tensor],
+        actions: torch.Tensor,
+        observation_features: torch.Tensor | None = None,
+    ) -> torch.Tensor:
+        device = get_device_from_parameters(self)
+        # Move each tensor in observations to device
+        observations = {k: v.to(device) for k, v in observations.items()}
+
+        obs_enc = self.encoder(observations, cache=observation_features)
+
+        inputs = torch.cat([obs_enc, actions], dim=-1)
+
+        # Loop through critics and collect outputs
+        q_values = []
+        for critic in self.critics:
+            q_values.append(critic(inputs))
+
+        # Stack outputs to match expected shape [num_critics, batch_size]
+        q_values = torch.stack([q.squeeze(-1) for q in q_values], dim=0)
+        return q_values
+
+
+class DiscreteCritic(nn.Module):
+    def __init__(
+        self,
+        encoder: nn.Module,
+        input_dim: int,
+        hidden_dims: list[int],
+        output_dim: int = 3,
+        activations: Callable[[torch.Tensor], torch.Tensor] | str = nn.SiLU(),
+        activate_final: bool = False,
+        dropout_rate: float | None = None,
+        init_final: float | None = None,
+        final_activation: Callable[[torch.Tensor], torch.Tensor] | str | None = None,
+    ):
+        super().__init__()
+        self.encoder = encoder
+        self.output_dim = output_dim
+
+        self.net = MLP(
+            input_dim=input_dim,
+            hidden_dims=hidden_dims,
+            activations=activations,
+            activate_final=activate_final,
+            dropout_rate=dropout_rate,
+            final_activation=final_activation,
+        )
+
+        self.output_layer = nn.Linear(in_features=hidden_dims[-1], out_features=self.output_dim)
+        if init_final is not None:
+            nn.init.uniform_(self.output_layer.weight, -init_final, init_final)
+            nn.init.uniform_(self.output_layer.bias, -init_final, init_final)
+        else:
+            orthogonal_init()(self.output_layer.weight)
+
+    def forward(
+        self, observations: torch.Tensor, observation_features: torch.Tensor | None = None
+    ) -> torch.Tensor:
+        device = get_device_from_parameters(self)
+        observations = {k: v.to(device) for k, v in observations.items()}
+        obs_enc = self.encoder(observations, cache=observation_features)
+        return self.output_layer(self.net(obs_enc))
+
+
+class Policy(nn.Module):
+    def __init__(
+        self,
+        encoder: SACObservationEncoder,
+        network: nn.Module,
+        action_dim: int,
+        std_min: float = -5,
+        std_max: float = 2,
+        fixed_std: torch.Tensor | None = None,
+        init_final: float | None = None,
+        use_tanh_squash: bool = False,
+        encoder_is_shared: bool = False,
+    ):
+        super().__init__()
+        self.encoder: SACObservationEncoder = encoder
+        self.network = network
+        self.action_dim = action_dim
+        self.std_min = std_min
+        self.std_max = std_max
+        self.fixed_std = fixed_std
+        self.use_tanh_squash = use_tanh_squash
+        self.encoder_is_shared = encoder_is_shared
+
+        # Find the last Linear layer's output dimension
+        for layer in reversed(network.net):
+            if isinstance(layer, nn.Linear):
+                out_features = layer.out_features
+                break
+        # Mean layer
+        self.mean_layer = nn.Linear(out_features, action_dim)
+        if init_final is not None:
+            nn.init.uniform_(self.mean_layer.weight, -init_final, init_final)
+            nn.init.uniform_(self.mean_layer.bias, -init_final, init_final)
+        else:
+            orthogonal_init()(self.mean_layer.weight)
+
+        # Standard deviation layer or parameter
+        if fixed_std is None:
+            self.std_layer = nn.Linear(out_features, action_dim)
+            if init_final is not None:
+                nn.init.uniform_(self.std_layer.weight, -init_final, init_final)
+                nn.init.uniform_(self.std_layer.bias, -init_final, init_final)
+            else:
+                orthogonal_init()(self.std_layer.weight)
+
+    def forward(
+        self,
+        observations: torch.Tensor,
+        observation_features: torch.Tensor | None = None,
+    ) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
+        # We detach the encoder if it is shared to avoid backprop through it
+        # This is important to avoid the encoder to be updated through the policy
+        obs_enc = self.encoder(observations, cache=observation_features, detach=self.encoder_is_shared)
+
+        # Get network outputs
+        outputs = self.network(obs_enc)
+        means = self.mean_layer(outputs)
+
+        # Compute standard deviations
+        if self.fixed_std is None:
+            log_std = self.std_layer(outputs)
+            std = torch.exp(log_std)  # Match JAX "exp"
+            std = torch.clamp(std, self.std_min, self.std_max)  # Match JAX default clip
+        else:
+            std = self.fixed_std.expand_as(means)
+
+        # Build transformed distribution
+        dist = TanhMultivariateNormalDiag(loc=means, scale_diag=std)
+
+        # Sample actions (reparameterized)
+        actions = dist.rsample()
+
+        # Compute log_probs
+        log_probs = dist.log_prob(actions)
+
+        return actions, log_probs, means
+
+    def get_features(self, observations: torch.Tensor) -> torch.Tensor:
+        """Get encoded features from observations"""
+        device = get_device_from_parameters(self)
+        observations = observations.to(device)
+        if self.encoder is not None:
+            with torch.inference_mode():
+                return self.encoder(observations)
+        return observations
+
+
+class DefaultImageEncoder(nn.Module):
+    def __init__(self, config: SACConfig):
+        super().__init__()
+        image_key = next(key for key in config.input_features if is_image_feature(key))
+        self.image_enc_layers = nn.Sequential(
+            nn.Conv2d(
+                in_channels=config.input_features[image_key].shape[0],
+                out_channels=config.image_encoder_hidden_dim,
+                kernel_size=7,
+                stride=2,
+            ),
+            nn.ReLU(),
+            nn.Conv2d(
+                in_channels=config.image_encoder_hidden_dim,
+                out_channels=config.image_encoder_hidden_dim,
+                kernel_size=5,
+                stride=2,
+            ),
+            nn.ReLU(),
+            nn.Conv2d(
+                in_channels=config.image_encoder_hidden_dim,
+                out_channels=config.image_encoder_hidden_dim,
+                kernel_size=3,
+                stride=2,
+            ),
+            nn.ReLU(),
+            nn.Conv2d(
+                in_channels=config.image_encoder_hidden_dim,
+                out_channels=config.image_encoder_hidden_dim,
+                kernel_size=3,
+                stride=2,
+            ),
+            nn.ReLU(),
+        )
+
+    def forward(self, x):
+        x = self.image_enc_layers(x)
+        return x
+
+
+def freeze_image_encoder(image_encoder: nn.Module):
+    """Freeze all parameters in the encoder"""
+    for param in image_encoder.parameters():
+        param.requires_grad = False
+
+
+class PretrainedImageEncoder(nn.Module):
+    def __init__(self, config: SACConfig):
+        super().__init__()
+
+        self.image_enc_layers, self.image_enc_out_shape = self._load_pretrained_vision_encoder(config)
+
+    def _load_pretrained_vision_encoder(self, config: SACConfig):
+        """Set up CNN encoder"""
+        from transformers import AutoModel
+
+        self.image_enc_layers = AutoModel.from_pretrained(config.vision_encoder_name, trust_remote_code=True)
+
+        if hasattr(self.image_enc_layers.config, "hidden_sizes"):
+            self.image_enc_out_shape = self.image_enc_layers.config.hidden_sizes[-1]  # Last channel dimension
+        elif hasattr(self.image_enc_layers, "fc"):
+            self.image_enc_out_shape = self.image_enc_layers.fc.in_features
+        else:
+            raise ValueError("Unsupported vision encoder architecture, make sure you are using a CNN")
+        return self.image_enc_layers, self.image_enc_out_shape
+
+    def forward(self, x):
+        enc_feat = self.image_enc_layers(x).last_hidden_state
+        return enc_feat
+
+
+def orthogonal_init():
+    return lambda x: torch.nn.init.orthogonal_(x, gain=1.0)
+
+
+class SpatialLearnedEmbeddings(nn.Module):
+    def __init__(self, height, width, channel, num_features=8):
+        """
+        PyTorch implementation of learned spatial embeddings
+
+        Args:
+            height: Spatial height of input features
+            width: Spatial width of input features
+            channel: Number of input channels
+            num_features: Number of output embedding dimensions
+        """
+        super().__init__()
+        self.height = height
+        self.width = width
+        self.channel = channel
+        self.num_features = num_features
+
+        self.kernel = nn.Parameter(torch.empty(channel, height, width, num_features))
+
+        nn.init.kaiming_normal_(self.kernel, mode="fan_in", nonlinearity="linear")
+
+    def forward(self, features):
+        """
+        Forward pass for spatial embedding
+
+        Args:
+            features: Input tensor of shape [B, C, H, W] where B is batch size,
+                     C is number of channels, H is height, and W is width
+        Returns:
+            Output tensor of shape [B, C*F] where F is the number of features
+        """
+
+        features_expanded = features.unsqueeze(-1)  # [B, C, H, W, 1]
+        kernel_expanded = self.kernel.unsqueeze(0)  # [1, C, H, W, F]
+
+        # Element-wise multiplication and spatial reduction
+        output = (features_expanded * kernel_expanded).sum(dim=(2, 3))  # Sum over H,W dimensions
+
+        # Reshape to combine channel and feature dimensions
+        output = output.view(output.size(0), -1)  # [B, C*F]
+
+        return output
+
+
+class RescaleFromTanh(Transform):
+    def __init__(self, low: float = -1, high: float = 1):
+        super().__init__()
+
+        self.low = low
+
+        self.high = high
+
+    def _call(self, x):
+        # Rescale from (-1, 1) to (low, high)
+
+        return 0.5 * (x + 1.0) * (self.high - self.low) + self.low
+
+    def _inverse(self, y):
+        # Rescale from (low, high) back to (-1, 1)
+
+        return 2.0 * (y - self.low) / (self.high - self.low) - 1.0
+
+    def log_abs_det_jacobian(self, x, y):
+        # log|d(rescale)/dx| = sum(log(0.5 * (high - low)))
+
+        scale = 0.5 * (self.high - self.low)
+
+        return torch.sum(torch.log(scale), dim=-1)
+
+
+class TanhMultivariateNormalDiag(TransformedDistribution):
+    def __init__(self, loc, scale_diag, low=None, high=None):
+        base_dist = MultivariateNormal(loc, torch.diag_embed(scale_diag))
+
+        transforms = [TanhTransform(cache_size=1)]
+
+        if low is not None and high is not None:
+            low = torch.as_tensor(low)
+
+            high = torch.as_tensor(high)
+
+            transforms.insert(0, RescaleFromTanh(low, high))
+
+        super().__init__(base_dist, transforms)
+
+    def mode(self):
+        # Mode is mean of base distribution, passed through transforms
+
+        x = self.base_dist.mean
+
+        for transform in self.transforms:
+            x = transform(x)
+
+        return x
+
+    def stddev(self):
+        std = self.base_dist.stddev
+
+        x = std
+
+        for transform in self.transforms:
+            x = transform(x)
+
+        return x
diff --git a/lerobot/src/lerobot/policies/sac/processor_sac.py b/lerobot/src/lerobot/policies/sac/processor_sac.py
new file mode 100644
index 0000000000000000000000000000000000000000..cf90e3cb4a0ba1d34b412375cd726f6b71f06f20
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sac/processor_sac.py
@@ -0,0 +1,92 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team.
+# All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from typing import Any
+
+import torch
+
+from lerobot.policies.sac.configuration_sac import SACConfig
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    RenameObservationsProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.utils.constants import POLICY_POSTPROCESSOR_DEFAULT_NAME, POLICY_PREPROCESSOR_DEFAULT_NAME
+
+
+def make_sac_pre_post_processors(
+    config: SACConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for the SAC policy.
+
+    The pre-processing pipeline prepares input data for the model by:
+    1. Renaming features to match pretrained configurations.
+    2. Normalizing input and output features based on dataset statistics.
+    3. Adding a batch dimension.
+    4. Moving all data to the specified device.
+
+    The post-processing pipeline handles the model's output by:
+    1. Moving data to the CPU.
+    2. Unnormalizing the output features to their original scale.
+
+    Args:
+        config: The configuration object for the SAC policy.
+        dataset_stats: A dictionary of statistics for normalization.
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines.
+    """
+
+    # Add remaining processors
+    input_steps = [
+        RenameObservationsProcessorStep(rename_map={}),
+        AddBatchDimensionProcessorStep(),
+        DeviceProcessorStep(device=config.device),
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+    ]
+    output_steps = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/sac/reward_model/configuration_classifier.py b/lerobot/src/lerobot/policies/sac/reward_model/configuration_classifier.py
new file mode 100644
index 0000000000000000000000000000000000000000..879e3c1afa24e18dc7bd1e4632b942bf506c0de1
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sac/reward_model/configuration_classifier.py
@@ -0,0 +1,77 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import NormalizationMode
+from lerobot.optim.optimizers import AdamWConfig, OptimizerConfig
+from lerobot.optim.schedulers import LRSchedulerConfig
+from lerobot.utils.constants import OBS_IMAGE
+
+
+@PreTrainedConfig.register_subclass(name="reward_classifier")
+@dataclass
+class RewardClassifierConfig(PreTrainedConfig):
+    """Configuration for the Reward Classifier model."""
+
+    name: str = "reward_classifier"
+    num_classes: int = 2
+    hidden_dim: int = 256
+    latent_dim: int = 256
+    image_embedding_pooling_dim: int = 8
+    dropout_rate: float = 0.1
+    model_name: str = "helper2424/resnet10"  # TODO: This needs to be updated. The model on the Hub doesn't call self.post_init() in its __init__, which is required by transformers v5 to set all_tied_weights_keys. The from_pretrained call fails when it tries to access this attribute during _finalize_model_loading.
+    device: str = "cpu"
+    model_type: str = "cnn"  # "transformer" or "cnn"
+    num_cameras: int = 2
+    learning_rate: float = 1e-4
+    weight_decay: float = 0.01
+    grad_clip_norm: float = 1.0
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.MEAN_STD,
+        }
+    )
+
+    @property
+    def observation_delta_indices(self) -> list | None:
+        return None
+
+    @property
+    def action_delta_indices(self) -> list | None:
+        return None
+
+    @property
+    def reward_delta_indices(self) -> list | None:
+        return None
+
+    def get_optimizer_preset(self) -> OptimizerConfig:
+        return AdamWConfig(
+            lr=self.learning_rate,
+            weight_decay=self.weight_decay,
+            grad_clip_norm=self.grad_clip_norm,
+        )
+
+    def get_scheduler_preset(self) -> LRSchedulerConfig | None:
+        return None
+
+    def validate_features(self) -> None:
+        """Validate feature configurations."""
+        has_image = any(key.startswith(OBS_IMAGE) for key in self.input_features)
+        if not has_image:
+            raise ValueError(
+                "You must provide an image observation (key starting with 'observation.image') in the input features"
+            )
diff --git a/lerobot/src/lerobot/policies/sac/reward_model/modeling_classifier.py b/lerobot/src/lerobot/policies/sac/reward_model/modeling_classifier.py
new file mode 100644
index 0000000000000000000000000000000000000000..dba6a174b4bab6bb5f276128fb3e4a10c846c53a
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sac/reward_model/modeling_classifier.py
@@ -0,0 +1,308 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+
+import torch
+from torch import Tensor, nn
+
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.sac.reward_model.configuration_classifier import RewardClassifierConfig
+from lerobot.utils.constants import OBS_IMAGE, REWARD
+
+
+class ClassifierOutput:
+    """Wrapper for classifier outputs with additional metadata."""
+
+    def __init__(
+        self,
+        logits: Tensor,
+        probabilities: Tensor | None = None,
+        hidden_states: Tensor | None = None,
+    ):
+        self.logits = logits
+        self.probabilities = probabilities
+        self.hidden_states = hidden_states
+
+    def __repr__(self):
+        return (
+            f"ClassifierOutput(logits={self.logits}, "
+            f"probabilities={self.probabilities}, "
+            f"hidden_states={self.hidden_states})"
+        )
+
+
+class SpatialLearnedEmbeddings(nn.Module):
+    def __init__(self, height, width, channel, num_features=8):
+        """
+        PyTorch implementation of learned spatial embeddings
+
+        Args:
+            height: Spatial height of input features
+            width: Spatial width of input features
+            channel: Number of input channels
+            num_features: Number of output embedding dimensions
+        """
+        super().__init__()
+        self.height = height
+        self.width = width
+        self.channel = channel
+        self.num_features = num_features
+
+        self.kernel = nn.Parameter(torch.empty(channel, height, width, num_features))
+
+        nn.init.kaiming_normal_(self.kernel, mode="fan_in", nonlinearity="linear")
+
+    def forward(self, features):
+        """
+        Forward pass for spatial embedding
+
+        Args:
+            features: Input tensor of shape [B, H, W, C] or [H, W, C] if no batch
+        Returns:
+            Output tensor of shape [B, C*F] or [C*F] if no batch
+        """
+
+        features = features.last_hidden_state
+
+        original_shape = features.shape
+        if features.dim() == 3:
+            features = features.unsqueeze(0)  # Add batch dim
+
+        features_expanded = features.unsqueeze(-1)  # [B, H, W, C, 1]
+        kernel_expanded = self.kernel.unsqueeze(0)  # [1, H, W, C, F]
+
+        # Element-wise multiplication and spatial reduction
+        output = (features_expanded * kernel_expanded).sum(dim=(2, 3))  # Sum H,W
+
+        # Reshape to combine channel and feature dimensions
+        output = output.view(output.size(0), -1)  # [B, C*F]
+
+        # Remove batch dim
+        if len(original_shape) == 3:
+            output = output.squeeze(0)
+
+        return output
+
+
+class Classifier(PreTrainedPolicy):
+    """Image classifier built on top of a pre-trained encoder."""
+
+    name = "reward_classifier"
+    config_class = RewardClassifierConfig
+
+    def __init__(
+        self,
+        config: RewardClassifierConfig,
+    ):
+        from transformers import AutoModel
+
+        super().__init__(config)
+        self.config = config
+
+        # Set up encoder
+        encoder = AutoModel.from_pretrained(self.config.model_name, trust_remote_code=True)
+        # Extract vision model if we're given a multimodal model
+        if hasattr(encoder, "vision_model"):
+            logging.info("Multimodal model detected - using vision encoder only")
+            self.encoder = encoder.vision_model
+            self.vision_config = encoder.config.vision_config
+        else:
+            self.encoder = encoder
+            self.vision_config = getattr(encoder, "config", None)
+
+        # Model type from config
+        self.is_cnn = self.config.model_type == "cnn"
+
+        # For CNNs, initialize backbone
+        if self.is_cnn:
+            self._setup_cnn_backbone()
+
+        self._freeze_encoder()
+
+        # Extract image keys from input_features
+        self.image_keys = [
+            key.replace(".", "_") for key in config.input_features if key.startswith(OBS_IMAGE)
+        ]
+
+        if self.is_cnn:
+            self.encoders = nn.ModuleDict()
+            for image_key in self.image_keys:
+                encoder = self._create_single_encoder()
+                self.encoders[image_key] = encoder
+
+        self._build_classifier_head()
+
+    def _setup_cnn_backbone(self):
+        """Set up CNN encoder"""
+        if hasattr(self.encoder, "fc"):
+            self.feature_dim = self.encoder.fc.in_features
+            self.encoder = nn.Sequential(*list(self.encoder.children())[:-1])
+        elif hasattr(self.encoder.config, "hidden_sizes"):
+            self.feature_dim = self.encoder.config.hidden_sizes[-1]  # Last channel dimension
+        else:
+            raise ValueError("Unsupported CNN architecture")
+
+    def _freeze_encoder(self) -> None:
+        """Freeze the encoder parameters."""
+        for param in self.encoder.parameters():
+            param.requires_grad = False
+
+    def _create_single_encoder(self):
+        encoder = nn.Sequential(
+            self.encoder,
+            SpatialLearnedEmbeddings(
+                height=4,
+                width=4,
+                channel=self.feature_dim,
+                num_features=self.config.image_embedding_pooling_dim,
+            ),
+            nn.Dropout(self.config.dropout_rate),
+            nn.Linear(self.feature_dim * self.config.image_embedding_pooling_dim, self.config.latent_dim),
+            nn.LayerNorm(self.config.latent_dim),
+            nn.Tanh(),
+        )
+
+        return encoder
+
+    def _build_classifier_head(self) -> None:
+        """Initialize the classifier head architecture."""
+        # Get input dimension based on model type
+        if self.is_cnn:
+            input_dim = self.config.latent_dim
+        else:  # Transformer models
+            if hasattr(self.encoder.config, "hidden_size"):
+                input_dim = self.encoder.config.hidden_size
+            else:
+                raise ValueError("Unsupported transformer architecture since hidden_size is not found")
+
+        self.classifier_head = nn.Sequential(
+            nn.Linear(input_dim * self.config.num_cameras, self.config.hidden_dim),
+            nn.Dropout(self.config.dropout_rate),
+            nn.LayerNorm(self.config.hidden_dim),
+            nn.ReLU(),
+            nn.Linear(
+                self.config.hidden_dim,
+                1 if self.config.num_classes == 2 else self.config.num_classes,
+            ),
+        )
+
+    def _get_encoder_output(self, x: torch.Tensor, image_key: str) -> torch.Tensor:
+        """Extract the appropriate output from the encoder."""
+        with torch.no_grad():
+            if self.is_cnn:
+                # The HF ResNet applies pooling internally
+                outputs = self.encoders[image_key](x)
+                return outputs
+            else:  # Transformer models
+                outputs = self.encoder(x)
+                return outputs.last_hidden_state[:, 0, :]
+
+    def extract_images_and_labels(self, batch: dict[str, Tensor]) -> tuple[list, Tensor]:
+        """Extract image tensors and label tensors from batch."""
+        # Check for both OBS_IMAGE and OBS_IMAGES prefixes
+        images = [batch[key] for key in self.config.input_features if key.startswith(OBS_IMAGE)]
+        labels = batch[REWARD]
+
+        return images, labels
+
+    def predict(self, xs: list) -> ClassifierOutput:
+        """Forward pass of the classifier for inference."""
+        encoder_outputs = torch.hstack(
+            [self._get_encoder_output(x, img_key) for x, img_key in zip(xs, self.image_keys, strict=True)]
+        )
+        logits = self.classifier_head(encoder_outputs)
+
+        if self.config.num_classes == 2:
+            logits = logits.squeeze(-1)
+            probabilities = torch.sigmoid(logits)
+        else:
+            probabilities = torch.softmax(logits, dim=-1)
+
+        return ClassifierOutput(logits=logits, probabilities=probabilities, hidden_states=encoder_outputs)
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, dict[str, Tensor]]:
+        """Standard forward pass for training compatible with train.py."""
+        # Extract images and labels
+        images, labels = self.extract_images_and_labels(batch)
+
+        # Get predictions
+        outputs = self.predict(images)
+
+        # Calculate loss
+        if self.config.num_classes == 2:
+            # Binary classification
+            loss = nn.functional.binary_cross_entropy_with_logits(outputs.logits, labels)
+            predictions = (torch.sigmoid(outputs.logits) > 0.5).float()
+        else:
+            # Multi-class classification
+            loss = nn.functional.cross_entropy(outputs.logits, labels.long())
+            predictions = torch.argmax(outputs.logits, dim=1)
+
+        # Calculate accuracy for logging
+        correct = (predictions == labels).sum().item()
+        total = labels.size(0)
+        accuracy = 100 * correct / total
+
+        # Return loss and metrics for logging
+        output_dict = {
+            "accuracy": accuracy,
+            "correct": correct,
+            "total": total,
+        }
+
+        return loss, output_dict
+
+    def predict_reward(self, batch, threshold=0.5):
+        """Eval method. Returns predicted reward with the decision threshold as argument."""
+        # Check for both OBS_IMAGE and OBS_IMAGES prefixes
+        batch = self.normalize_inputs(batch)
+        batch = self.normalize_targets(batch)
+
+        # Extract images from batch dict
+        images = [batch[key] for key in self.config.input_features if key.startswith(OBS_IMAGE)]
+
+        if self.config.num_classes == 2:
+            probs = self.predict(images).probabilities
+            logging.debug(f"Predicted reward images: {probs}")
+            return (probs > threshold).float()
+        else:
+            return torch.argmax(self.predict(images).probabilities, dim=1)
+
+    def get_optim_params(self):
+        """Return optimizer parameters for the policy."""
+        return self.parameters()
+
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """
+        This method is required by PreTrainedPolicy but not used for reward classifiers.
+        The reward classifier is not an actor and does not select actions.
+        """
+        raise NotImplementedError("Reward classifiers do not select actions")
+
+    def predict_action_chunk(self, batch: dict[str, Tensor]) -> Tensor:
+        """
+        This method is required by PreTrainedPolicy but not used for reward classifiers.
+        The reward classifier is not an actor and does not produce action chunks.
+        """
+        raise NotImplementedError("Reward classifiers do not predict action chunks")
+
+    def reset(self):
+        """
+        This method is required by PreTrainedPolicy but not used for reward classifiers.
+        The reward classifier is not an actor and does not select actions.
+        """
+        pass
diff --git a/lerobot/src/lerobot/policies/sac/reward_model/processor_classifier.py b/lerobot/src/lerobot/policies/sac/reward_model/processor_classifier.py
new file mode 100644
index 0000000000000000000000000000000000000000..c2a34eab26a66711ef687286a9541233d5d0bea5
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sac/reward_model/processor_classifier.py
@@ -0,0 +1,82 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from typing import Any
+
+import torch
+
+from lerobot.policies.sac.reward_model.configuration_classifier import RewardClassifierConfig
+from lerobot.processor import (
+    DeviceProcessorStep,
+    IdentityProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+
+
+def make_classifier_processor(
+    config: RewardClassifierConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for the reward classifier.
+
+    The pre-processing pipeline prepares input data for the classifier by:
+    1. Normalizing both input and output features based on dataset statistics.
+    2. Moving the data to the specified device.
+
+    The post-processing pipeline handles the classifier's output by:
+    1. Moving the data to the CPU.
+    2. Applying an identity step, as no unnormalization is needed for the output logits.
+
+    Args:
+        config: The configuration object for the RewardClassifier.
+        dataset_stats: A dictionary of statistics for normalization.
+        preprocessor_kwargs: Additional arguments for the pre-processor pipeline.
+        postprocessor_kwargs: Additional arguments for the post-processor pipeline.
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines.
+    """
+
+    input_steps = [
+        NormalizerProcessorStep(
+            features=config.input_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        NormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device=config.device),
+    ]
+    output_steps = [DeviceProcessorStep(device="cpu"), IdentityProcessorStep()]
+
+    return (
+        PolicyProcessorPipeline(
+            steps=input_steps,
+            name="classifier_preprocessor",
+        ),
+        PolicyProcessorPipeline(
+            steps=output_steps,
+            name="classifier_postprocessor",
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/sarm/README.md b/lerobot/src/lerobot/policies/sarm/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..e0e49834bf47be5f5a06915e2e1c13066633cee0
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sarm/README.md
@@ -0,0 +1,14 @@
+## Paper
+
+https://arxiv.org/abs/2509.25358
+
+## Citation
+
+```bibtex
+@article{chen2025sarm,
+  title={SARM: Stage-Aware Reward Modeling for Long Horizon Robot Manipulation},
+  author={Chen, Qianzhong and Yu, Justin and Schwager, Mac and Abbeel, Pieter and Shentu, Yide and Wu, Philipp},
+  journal={arXiv preprint arXiv:2509.25358},
+  year={2025}
+}
+```
diff --git a/lerobot/src/lerobot/policies/sarm/compute_rabc_weights.py b/lerobot/src/lerobot/policies/sarm/compute_rabc_weights.py
new file mode 100644
index 0000000000000000000000000000000000000000..485c1096bc1312c6414f2a5f5faf5b1f86d3e57c
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sarm/compute_rabc_weights.py
@@ -0,0 +1,870 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Compute SARM progress values for RA-BC (Reward-Aware Behavior Cloning) weighting.
+
+This script processes all frames in a dataset with SARM to compute progress values [0, 1].
+The results are saved as a parquet file that can be loaded during training for RA-BC weighting.
+
+Uses multi-output extraction: each SARM query returns progress for 9 frames, so we only
+need ~num_frames/30 queries instead of one per frame (~30x speedup).
+
+Usage:
+    # Full RA-BC computation with visualizations
+    python src/lerobot/policies/sarm/compute_rabc_weights.py \\
+        --dataset-repo-id lerobot/aloha_sim_insertion_human \\
+        --reward-model-path <USER>/sarm_single_uni4
+
+    # Faster computation with stride (compute every 5 frames, interpolate the rest)
+    python src/lerobot/policies/sarm/compute_rabc_weights.py \\
+        --dataset-repo-id lerobot/aloha_sim_insertion_human \\
+        --reward-model-path <USER>/sarm_single_uni4 \\
+        --stride 5
+
+    # Visualize predictions only (no RA-BC computation)
+    python src/lerobot/policies/sarm/compute_rabc_weights.py \\
+        --dataset-repo-id lerobot/aloha_sim_insertion_human \\
+        --reward-model-path <USER>/sarm_single_uni4 \\
+        --visualize-only \\
+        --num-visualizations 5
+
+The output is saved to the dataset's local cache directory as 'sarm_progress.parquet'.
+"""
+
+import argparse
+import logging
+from pathlib import Path
+
+import matplotlib.gridspec as gridspec
+import matplotlib.pyplot as plt
+import numpy as np
+import pyarrow as pa
+import pyarrow.parquet as pq
+import torch
+from tqdm import tqdm
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.sarm.modeling_sarm import SARMRewardModel
+from lerobot.policies.sarm.processor_sarm import make_sarm_pre_post_processors
+from lerobot.policies.sarm.sarm_utils import normalize_stage_tau
+
+
+def get_reward_model_path_from_parquet(parquet_path: Path) -> str | None:
+    """Read reward_model_path from parquet metadata if available."""
+    if not parquet_path.exists():
+        return None
+    try:
+        metadata = pq.read_metadata(parquet_path).schema.to_arrow_schema().metadata
+        if metadata and b"reward_model_path" in metadata:
+            return metadata[b"reward_model_path"].decode()
+    except Exception:  # nosec B110
+        return None
+    return None
+
+
+def load_sarm_resources(
+    dataset_repo_id: str,
+    reward_model_path: str,
+    device: str = "cuda",
+) -> tuple[LeRobotDataset, SARMRewardModel, any]:
+    """
+    Load SARM model, dataset, and preprocessor.
+
+    Returns:
+        Tuple of (dataset, reward_model, preprocessor)
+    """
+    logging.info(f"Loading model: {reward_model_path}")
+    reward_model = SARMRewardModel.from_pretrained(reward_model_path)
+    reward_model.config.device = device
+    reward_model.to(device).eval()
+
+    image_key = reward_model.config.image_key
+    state_key = reward_model.config.state_key
+    delta_indices = reward_model.config.observation_delta_indices
+
+    logging.info(f"Loading dataset: {dataset_repo_id}")
+    temp_dataset = LeRobotDataset(dataset_repo_id, download_videos=True)
+    fps = temp_dataset.fps
+
+    delta_timestamps = {
+        image_key: [idx / fps for idx in delta_indices],
+        state_key: [idx / fps for idx in delta_indices],
+    }
+    dataset = LeRobotDataset(dataset_repo_id, delta_timestamps=delta_timestamps)
+    logging.info(f"Dataset: {dataset.num_episodes} episodes, {dataset.num_frames} frames")
+
+    preprocess, _ = make_sarm_pre_post_processors(
+        config=reward_model.config,
+        dataset_stats=dataset.meta.stats,
+        dataset_meta=dataset.meta,
+    )
+
+    return dataset, reward_model, preprocess
+
+
+def to_numpy_image(img) -> np.ndarray:
+    """Convert image tensor to numpy uint8 (H, W, C)."""
+    if isinstance(img, torch.Tensor):
+        img = img.cpu().numpy()
+    if img.ndim == 4:
+        # Take center frame for bidirectional sampling
+        img = img[img.shape[0] // 2]
+    if img.shape[0] in [1, 3]:
+        img = np.transpose(img, (1, 2, 0))
+    if img.dtype != np.uint8:
+        # Handle normalized images (may have negative values or values > 1)
+        img = img.astype(np.float32)
+        img = (img - img.min()) / (img.max() - img.min() + 1e-8)  # Normalize to [0, 1]
+        img = (img * 255).astype(np.uint8)
+    return img
+
+
+def visualize_episode(
+    frames, progress_preds, stage_preds, title, output_path, stage_labels, gt_progress=None, gt_stages=None
+):
+    """Create visualization with progress plot, stage probabilities, and sample frames.
+
+    Same as sarm_inference_visualization.py
+    """
+    num_stages = stage_preds.shape[1]
+    colors = plt.cm.tab10(np.linspace(0, 1, num_stages))
+    frame_indices = np.arange(len(progress_preds))
+
+    fig = plt.figure(figsize=(14, 12))
+    gs = gridspec.GridSpec(3, 1, height_ratios=[2, 1, 1], hspace=0.3)
+    ax_progress, ax_stages, ax_frames = fig.add_subplot(gs[0]), fig.add_subplot(gs[1]), fig.add_subplot(gs[2])
+
+    # Progress plot
+    ax_progress.plot(frame_indices, progress_preds, linewidth=2, color="#2E86AB", label="Predicted")
+    ax_progress.fill_between(frame_indices, 0, progress_preds, alpha=0.3, color="#2E86AB")
+    if gt_progress is not None:
+        ax_progress.plot(
+            frame_indices, gt_progress, linewidth=2, color="#28A745", linestyle="--", label="Ground Truth"
+        )
+    ax_progress.axhline(y=1.0, color="gray", linestyle="--", alpha=0.5)
+    ax_progress.set_ylabel("Progress")
+    ax_progress.set_title(f'Task: "{title}"', fontweight="bold")
+    ax_progress.set_ylim(-0.05, 1.1)
+    ax_progress.legend(loc="upper left")
+    ax_progress.grid(True, alpha=0.3)
+
+    # Stage predictions
+    ax_stages.stackplot(
+        frame_indices,
+        *[stage_preds[:, i] for i in range(num_stages)],
+        colors=colors,
+        alpha=0.8,
+        labels=stage_labels,
+    )
+    if gt_stages is not None:
+        for change_idx in np.where(np.diff(gt_stages) != 0)[0] + 1:
+            ax_stages.axvline(x=change_idx, color="black", linestyle="-", alpha=0.7, linewidth=1.5)
+    ax_stages.set_xlabel("Frame")
+    ax_stages.set_ylabel("Stage Probability")
+    ax_stages.set_ylim(0, 1)
+    ax_stages.legend(loc="upper left", ncol=min(num_stages, 5), fontsize=8)
+    ax_stages.grid(True, alpha=0.3)
+
+    # Sample frames
+    ax_frames.axis("off")
+    num_sample = 8
+    sample_indices = np.linspace(0, len(frames) - 1, num_sample, dtype=int)
+    h, w = frames[0].shape[:2]
+    combined = np.zeros((h, w * num_sample, 3), dtype=np.uint8)
+    for i, idx in enumerate(sample_indices):
+        frame = frames[idx]
+        if frame.shape[-1] == 1:
+            frame = np.repeat(frame, 3, axis=-1)
+        combined[:, i * w : (i + 1) * w] = frame
+        stage_name = stage_labels[np.argmax(stage_preds[idx])][:12]
+        ax_frames.text(
+            i * w + w / 2,
+            -10,
+            f"Frame {idx}\n{progress_preds[idx]:.2f}\n{stage_name}",
+            ha="center",
+            va="top",
+            fontsize=7,
+        )
+    ax_frames.imshow(combined)
+    ax_frames.set_title("Sample Frames", pad=20)
+
+    output_path.parent.mkdir(parents=True, exist_ok=True)
+    plt.savefig(output_path, dpi=150, bbox_inches="tight")
+    plt.close()
+    print(f"Saved: {output_path}")
+
+
+def visualize_sarm_predictions(
+    dataset: LeRobotDataset,
+    reward_model: SARMRewardModel,
+    preprocess,
+    episode_indices: list[int],
+    head_mode: str,
+    output_dir: Path,
+    num_display_frames: int = 5,
+    stride: int = 1,
+):
+    """
+    Visualize SARM predictions for multiple episodes.
+
+    Computes predictions for every frame by default. With stride > 1, computes predictions
+    every N frames and interpolates (progress + stage probabilities) for visualization.
+
+    Args:
+        dataset: LeRobotDataset with delta_timestamps configured
+        reward_model: Loaded SARM model
+        preprocess: Preprocessor from make_sarm_pre_post_processors
+        episode_indices: List of episode indices to visualize
+        head_mode: "sparse", "dense", or "both"
+        output_dir: Directory to save visualizations
+        num_display_frames: Number of frames to display in thumbnail strip (default: 5)
+        stride: Compute predictions every N frames, interpolate the rest (default: 1)
+    """
+    output_dir = Path(output_dir)
+    output_dir.mkdir(parents=True, exist_ok=True)
+
+    image_key = reward_model.config.image_key
+    state_key = reward_model.config.state_key
+    dual_mode = reward_model.config.uses_dual_heads
+    device = reward_model.device
+
+    # Center frame index for bidirectional sampling
+    target_idx = reward_model.config.n_obs_steps // 2
+
+    # Determine which heads to visualize
+    schemes_to_viz = []
+    if head_mode in ("sparse", "both") or not dual_mode:
+        schemes_to_viz.append("sparse")
+    if head_mode in ("dense", "both") and dual_mode:
+        schemes_to_viz.append("dense")
+
+    # Set preprocessor to eval mode to disable augmentations
+    if hasattr(preprocess, "eval"):
+        preprocess.eval()
+    for step in preprocess.steps:
+        if hasattr(step, "eval"):
+            step.eval()
+
+    for episode_idx in episode_indices:
+        ep = dataset.meta.episodes[episode_idx]
+        ep_start = ep["dataset_from_index"]
+        ep_end = ep["dataset_to_index"]
+        task = dataset[ep_start].get("task", "perform the task")
+        num_frames = ep_end - ep_start
+
+        # Select frames for display thumbnails (evenly sampled from begin to end)
+        display_indices = set(
+            [
+                ep_start + int(i * (num_frames - 1) / (num_display_frames - 1))
+                for i in range(num_display_frames)
+            ]
+            if num_frames >= num_display_frames
+            else list(range(ep_start, ep_end))
+        )
+        viz_frames = {}
+
+        # Load display frames up-front (stride mode might skip them otherwise).
+        for frame_idx in display_indices:
+            sample = dataset[frame_idx]
+            viz_frames[frame_idx] = to_numpy_image(sample[image_key])
+
+        # Initialize storage for each scheme
+        scheme_data = {}
+        for scheme in schemes_to_viz:
+            num_stages = getattr(reward_model.config, f"num_{scheme}_stages")
+            scheme_data[scheme] = {
+                "viz_progress": np.full(num_frames, np.nan),
+                "viz_stages": np.full((num_frames, num_stages), np.nan),
+                "viz_gt_progress": np.full(num_frames, np.nan),
+                "viz_gt_stages": np.full(num_frames, np.nan),
+                "target_key": f"{scheme}_targets",
+                "num_stages": num_stages,
+                "temporal_props": getattr(reward_model.config, f"{scheme}_temporal_proportions"),
+                "subtask_names": getattr(reward_model.config, f"{scheme}_subtask_names"),
+            }
+
+        if stride > 1:
+            logging.info(f"Visualization stride={stride}: inferring every {stride} frames and interpolating")
+
+        # Process frames one at a time to avoid memory buildup
+        frame_indices = list(range(ep_start, ep_end, stride))
+        if (ep_end - 1) not in frame_indices:
+            frame_indices.append(ep_end - 1)
+        frame_indices = sorted(set(frame_indices))
+
+        for frame_idx in tqdm(frame_indices, desc=f"Episode {episode_idx}", leave=False):
+            local_idx = frame_idx - ep_start
+            sample = dataset[frame_idx]
+
+            batch = {
+                image_key: sample[image_key],
+                "task": task,
+                "index": frame_idx,
+                "episode_index": episode_idx,
+            }
+            if state_key in sample:
+                batch[state_key] = sample[state_key]
+
+            with torch.no_grad():
+                processed = preprocess(batch)
+                video_features = processed["video_features"].to(device)
+                text_features = processed["text_features"].to(device)
+                state_features = processed.get("state_features")
+                if state_features is not None:
+                    state_features = state_features.to(device)
+                lengths = processed.get("lengths")
+
+                for scheme in schemes_to_viz:
+                    sd = scheme_data[scheme]
+
+                    # Ground truth
+                    # In stride visualization mode, ground-truth plots can be misleading
+                    # (only sparse points are available), so we skip GT.
+                    if stride == 1 and sd["target_key"] in processed:
+                        gt_target = processed[sd["target_key"]][0, target_idx].cpu().item()
+                        sd["viz_gt_stages"][local_idx] = int(gt_target)
+                        sd["viz_gt_progress"][local_idx] = normalize_stage_tau(
+                            gt_target,
+                            num_stages=sd["num_stages"],
+                            temporal_proportions=sd["temporal_props"],
+                            subtask_names=sd["subtask_names"],
+                        )
+
+                    # Predictions
+                    reward, stage_probs = reward_model.calculate_rewards(
+                        text_embeddings=text_features,
+                        video_embeddings=video_features,
+                        state_features=state_features,
+                        lengths=lengths,
+                        return_all_frames=True,
+                        return_stages=True,
+                        head_mode=scheme,
+                    )
+
+                    # Handle both tensor and numpy outputs
+                    if isinstance(reward, torch.Tensor):
+                        reward = reward.cpu().numpy()
+                        stage_probs = stage_probs.cpu().numpy()
+
+                    if reward.ndim == 2:
+                        sd["viz_progress"][local_idx] = reward[0, target_idx]
+                        sd["viz_stages"][local_idx] = stage_probs[0, target_idx, :]
+                    else:
+                        sd["viz_progress"][local_idx] = reward[target_idx]
+                        sd["viz_stages"][local_idx] = stage_probs[target_idx, :]
+
+                # Clear GPU memory after each frame
+                del processed, video_features, text_features
+                if state_features is not None:
+                    del state_features
+
+            torch.cuda.empty_cache()
+
+        # Interpolate predictions back to per-frame arrays for smooth visualization.
+        if stride > 1:
+            all_local = np.arange(num_frames)
+            for scheme in schemes_to_viz:
+                sd = scheme_data[scheme]
+
+                valid = np.isfinite(sd["viz_progress"])
+                valid_idx = np.where(valid)[0]
+                if valid_idx.size >= 1:
+                    sd["viz_progress"] = interpolate_progress(
+                        valid_idx, sd["viz_progress"][valid_idx], all_local
+                    )
+
+                    stage_interp = np.zeros_like(sd["viz_stages"], dtype=np.float32)
+                    for s in range(sd["num_stages"]):
+                        stage_interp[:, s] = interpolate_progress(
+                            valid_idx, sd["viz_stages"][valid_idx, s], all_local
+                        )
+
+                    stage_interp = np.clip(stage_interp, 0.0, 1.0)
+                    row_sums = stage_interp.sum(axis=1, keepdims=True)
+                    nz = row_sums.squeeze(-1) > 0
+                    stage_interp[nz] = stage_interp[nz] / row_sums[nz]
+                    sd["viz_stages"] = stage_interp
+                else:
+                    # No valid points: keep NaNs/zeros; visualization will be empty.
+                    sd["viz_stages"] = np.nan_to_num(sd["viz_stages"], nan=0.0)
+
+        # Generate visualization for each head
+        ordered_viz_frames = [viz_frames[idx] for idx in sorted(display_indices)]
+        for scheme in schemes_to_viz:
+            sd = scheme_data[scheme]
+            stage_labels = sd["subtask_names"] or [f"Stage {i + 1}" for i in range(sd["num_stages"])]
+            viz_path = output_dir / f"sarm_prediction_ep{episode_idx}_{scheme}.png"
+
+            visualize_episode(
+                frames=np.array(ordered_viz_frames),
+                progress_preds=sd["viz_progress"],
+                stage_preds=sd["viz_stages"],
+                title=f"{task} (Episode {episode_idx})",
+                output_path=viz_path,
+                stage_labels=stage_labels,
+                gt_progress=sd["viz_gt_progress"] if not np.all(np.isnan(sd["viz_gt_progress"])) else None,
+                gt_stages=sd["viz_gt_stages"] if not np.all(np.isnan(sd["viz_gt_stages"])) else None,
+            )
+
+        # Clear memory between episodes
+        torch.cuda.empty_cache()
+
+    logging.info(f"Visualizations saved to: {output_dir.absolute()}")
+
+
+def generate_all_frame_indices(ep_start: int, ep_end: int, frame_gap: int = 30) -> list[int]:
+    """Generate all frame indices, ordered by offset for cache-friendly access.
+
+    Orders frames as: [0, 30, 60...], [1, 31, 61...], ..., [29, 59, 89...]
+    This groups frames that share similar temporal windows together.
+    """
+    num_frames = ep_end - ep_start
+    indices = []
+    for offset in range(frame_gap):
+        for frame_rel in range(offset, num_frames, frame_gap):
+            indices.append(ep_start + frame_rel)
+    return indices
+
+
+def interpolate_progress(
+    computed_indices: np.ndarray,
+    computed_values: np.ndarray,
+    all_indices: np.ndarray,
+) -> np.ndarray:
+    """Linearly interpolate values to fill in gaps (robust to NaNs / edge cases)."""
+    computed_indices = np.asarray(computed_indices)
+    computed_values = np.asarray(computed_values)
+    all_indices = np.asarray(all_indices)
+
+    mask = np.isfinite(computed_values)
+    if mask.sum() == 0:
+        return np.full(all_indices.shape, np.nan, dtype=np.float32)
+    if mask.sum() == 1:
+        return np.full(all_indices.shape, float(computed_values[mask][0]), dtype=np.float32)
+
+    out = np.interp(all_indices, computed_indices[mask], computed_values[mask])
+    return out.astype(np.float32)
+
+
+def compute_sarm_progress(
+    dataset_repo_id: str,
+    reward_model_path: str,
+    output_path: str | None = None,
+    head_mode: str = "sparse",
+    device: str = "cuda",
+    num_visualizations: int = 5,
+    output_dir: str = "./sarm_viz",
+    stride: int = 1,
+):
+    """
+    Compute SARM progress predictions for all frames in a dataset.
+
+    Args:
+        dataset_repo_id: HuggingFace dataset repo ID or local path
+        reward_model_path: Path to pretrained SARM model
+        output_path: Path to save results. If None, saves to dataset's cache directory
+        head_mode: SARM head to use ("sparse", "dense", or "both")
+        device: Device to use for inference
+        num_visualizations: Number of episodes to visualize (0 to skip)
+        output_dir: Directory to save visualizations
+        stride: Compute progress every N frames, interpolate the rest (default: 1 = every frame)
+    """
+    dataset, reward_model, preprocess = load_sarm_resources(dataset_repo_id, reward_model_path, device)
+
+    # Set preprocessor to eval mode to disable augmentations
+    if hasattr(preprocess, "eval"):
+        preprocess.eval()
+    for step in preprocess.steps:
+        if hasattr(step, "eval"):
+            step.eval()
+
+    image_key = reward_model.config.image_key
+    state_key = reward_model.config.state_key
+    frame_gap = reward_model.config.frame_gap
+    num_episodes = dataset.num_episodes
+    total_frames = dataset.num_frames
+    logging.info(f"Processing {total_frames} frames across {num_episodes} episodes")
+
+    # Determine which heads to compute
+    dual_mode = reward_model.config.uses_dual_heads
+    compute_sparse = head_mode in ("sparse", "both") or not dual_mode
+    compute_dense = head_mode in ("dense", "both") and dual_mode
+
+    # Storage arrays
+    all_indices = []
+    all_episode_indices = []
+    all_frame_indices = []
+    all_progress_sparse = [] if compute_sparse else None
+    all_progress_dense = [] if compute_dense else None
+
+    if stride > 1:
+        logging.info(f"Using stride={stride}: computing every {stride} frames, interpolating the rest")
+
+    # Process all episodes
+    for episode_idx in tqdm(range(num_episodes), desc="Episodes"):
+        ep = dataset.meta.episodes[episode_idx]
+        ep_start = ep["dataset_from_index"]
+        ep_end = ep["dataset_to_index"]
+
+        # Get task description
+        task = dataset[ep_start].get("task", "perform the task")
+
+        # Generate frames to compute (with stride applied)
+        all_ep_indices = generate_all_frame_indices(ep_start, ep_end, frame_gap)
+        if stride > 1:
+            # Only compute every stride-th frame (relative to episode start)
+            compute_indices = [idx for idx in all_ep_indices if (idx - ep_start) % stride == 0]
+            # Always include last frame for better interpolation at episode end
+            last_frame = ep_end - 1
+            if last_frame not in compute_indices:
+                compute_indices.append(last_frame)
+            compute_indices = sorted(set(compute_indices))
+        else:
+            compute_indices = all_ep_indices
+
+        center_idx = reward_model.config.n_obs_steps // 2  # Center of bidirectional window
+
+        # Dictionary to collect results
+        frame_results = {}
+
+        for query_idx in tqdm(compute_indices, desc=f"  Ep {episode_idx}", leave=False):
+            try:
+                sample = dataset[query_idx]
+
+                batch = {
+                    image_key: sample[image_key],
+                    "task": task,
+                    "index": query_idx,
+                    "episode_index": episode_idx,
+                }
+                if state_key in sample:
+                    batch[state_key] = sample[state_key]
+
+                with torch.no_grad():
+                    processed = preprocess(batch)
+                    video_features = processed["video_features"].to(device)
+                    text_features = processed["text_features"].to(device)
+                    state_features = processed.get("state_features")
+                    if state_features is not None:
+                        state_features = state_features.to(device)
+                    lengths = processed.get("lengths")
+
+                    sparse_val = np.nan
+                    dense_val = np.nan
+
+                    # Compute sparse prediction for center frame
+                    if compute_sparse:
+                        sparse_progress = reward_model.calculate_rewards(
+                            text_embeddings=text_features,
+                            video_embeddings=video_features,
+                            state_features=state_features,
+                            lengths=lengths,
+                            return_all_frames=True,
+                            head_mode="sparse",
+                        )
+                        sparse_val = float(
+                            sparse_progress[0, center_idx]
+                            if sparse_progress.ndim == 2
+                            else sparse_progress[center_idx]
+                        )
+
+                    # Compute dense prediction for center frame
+                    if compute_dense:
+                        dense_progress = reward_model.calculate_rewards(
+                            text_embeddings=text_features,
+                            video_embeddings=video_features,
+                            state_features=state_features,
+                            lengths=lengths,
+                            return_all_frames=True,
+                            head_mode="dense",
+                        )
+                        dense_val = float(
+                            dense_progress[0, center_idx]
+                            if dense_progress.ndim == 2
+                            else dense_progress[center_idx]
+                        )
+
+                    frame_results[query_idx] = (sparse_val, dense_val)
+
+            except Exception as e:
+                logging.warning(f"Failed to process frame {query_idx}: {e}")
+
+        # Interpolate to get values for all frames
+        computed_indices = np.array(sorted(frame_results.keys()))
+        computed_sparse = (
+            np.array([frame_results[i][0] for i in computed_indices]) if compute_sparse else None
+        )
+        computed_dense = np.array([frame_results[i][1] for i in computed_indices]) if compute_dense else None
+
+        # All frame indices for this episode
+        all_frame_idx_array = np.arange(ep_start, ep_end)
+
+        if stride > 1 and len(computed_indices) > 1:
+            # Interpolate progress values
+            if compute_sparse:
+                interp_sparse = interpolate_progress(computed_indices, computed_sparse, all_frame_idx_array)
+            if compute_dense:
+                interp_dense = interpolate_progress(computed_indices, computed_dense, all_frame_idx_array)
+        else:
+            # No interpolation needed
+            interp_sparse = computed_sparse if compute_sparse else None
+            interp_dense = computed_dense if compute_dense else None
+
+        # Store results for all frames
+        for i, frame_idx in enumerate(all_frame_idx_array):
+            local_idx = frame_idx - ep_start
+            all_indices.append(frame_idx)
+            all_episode_indices.append(episode_idx)
+            all_frame_indices.append(local_idx)
+            if compute_sparse:
+                if stride > 1 and len(computed_indices) > 1:
+                    all_progress_sparse.append(float(interp_sparse[i]))
+                elif frame_idx in frame_results:
+                    all_progress_sparse.append(frame_results[frame_idx][0])
+                else:
+                    all_progress_sparse.append(np.nan)
+            if compute_dense:
+                if stride > 1 and len(computed_indices) > 1:
+                    all_progress_dense.append(float(interp_dense[i]))
+                elif frame_idx in frame_results:
+                    all_progress_dense.append(frame_results[frame_idx][1])
+                else:
+                    all_progress_dense.append(np.nan)
+
+    # Create output table
+    table_data = {
+        "index": np.array(all_indices, dtype=np.int64),
+        "episode_index": np.array(all_episode_indices, dtype=np.int64),
+        "frame_index": np.array(all_frame_indices, dtype=np.int64),
+    }
+    if compute_sparse:
+        table_data["progress_sparse"] = np.array(all_progress_sparse, dtype=np.float32)
+    if compute_dense:
+        table_data["progress_dense"] = np.array(all_progress_dense, dtype=np.float32)
+
+    # Sort by index
+    df = pa.table(table_data).to_pandas()
+    df = df.sort_values("index").reset_index(drop=True)
+    final_table = pa.Table.from_pandas(df, preserve_index=False)
+
+    # Add metadata with reward model path
+    metadata = {b"reward_model_path": reward_model_path.encode()}
+    final_table = final_table.replace_schema_metadata(metadata)
+
+    # Determine output path
+    output_path = Path(dataset.root) / "sarm_progress.parquet" if output_path is None else Path(output_path)
+
+    # Save
+    output_path.parent.mkdir(parents=True, exist_ok=True)
+    pq.write_table(final_table, output_path)
+    logging.info(f"Saved {len(final_table)} frame progress values to {output_path}")
+
+    # Print statistics
+    if "progress_sparse" in df.columns:
+        valid = df["progress_sparse"].dropna()
+        logging.info(
+            f"Sparse progress: mean={valid.mean():.4f}, std={valid.std():.4f}, "
+            f"min={valid.min():.4f}, max={valid.max():.4f}"
+        )
+
+    if "progress_dense" in df.columns:
+        valid = df["progress_dense"].dropna()
+        logging.info(
+            f"Dense progress: mean={valid.mean():.4f}, std={valid.std():.4f}, "
+            f"min={valid.min():.4f}, max={valid.max():.4f}"
+        )
+
+    # Visualize episodes after processing
+    if num_visualizations > 0:
+        viz_episodes = list(range(min(num_visualizations, num_episodes)))
+        logging.info(f"Generating {len(viz_episodes)} visualizations...")
+        visualize_sarm_predictions(
+            dataset=dataset,
+            reward_model=reward_model,
+            preprocess=preprocess,
+            episode_indices=viz_episodes,
+            head_mode=head_mode,
+            output_dir=Path(output_dir),
+            stride=stride,
+        )
+
+    return output_path
+
+
+def main():
+    parser = argparse.ArgumentParser(
+        description="Compute SARM progress values for RA-BC weighting or visualize SARM predictions",
+        formatter_class=argparse.RawDescriptionHelpFormatter,
+        epilog="""
+Examples:
+    # Full RA-BC computation with visualizations
+    python src/lerobot/policies/sarm/compute_rabc_weights.py \\
+        --dataset-repo-id lerobot/aloha_sim_insertion_human \\
+        --reward-model-path <USER>/sarm_single_uni4
+
+    # Visualize predictions only (no RA-BC computation)
+    python src/lerobot/policies/sarm/compute_rabc_weights.py \\
+        --dataset-repo-id lerobot/aloha_sim_insertion_human \\
+        --reward-model-path <USER>/sarm_single_uni4 \\
+        --visualize-only \\
+        --num-visualizations 10
+        """,
+    )
+    parser.add_argument(
+        "--dataset-repo-id",
+        type=str,
+        required=True,
+        help="HuggingFace dataset repo ID or local path",
+    )
+    parser.add_argument(
+        "--reward-model-path",
+        type=str,
+        default=None,
+        help="Path to pretrained SARM model (reads from existing parquet metadata if not provided)",
+    )
+    parser.add_argument(
+        "--output-path",
+        type=str,
+        default=None,
+        help="Output path for parquet. If not set, saves to dataset's cache directory",
+    )
+    parser.add_argument(
+        "--head-mode",
+        type=str,
+        default="sparse",
+        choices=["sparse", "dense", "both"],
+        help="SARM head to use (default: sparse)",
+    )
+    parser.add_argument(
+        "--device",
+        type=str,
+        default="cuda",
+        help="Device to use (default: cuda)",
+    )
+    # Visualization options
+    parser.add_argument(
+        "--visualize-only",
+        action="store_true",
+        help="Only visualize SARM predictions (no RA-BC computation)",
+    )
+    parser.add_argument(
+        "--num-visualizations",
+        type=int,
+        default=5,
+        help="Number of episodes to visualize (default: 5, set to 0 to skip)",
+    )
+    parser.add_argument(
+        "--output-dir",
+        type=str,
+        default="./sarm_viz",
+        help="Output directory for visualizations (default: ./sarm_viz)",
+    )
+    parser.add_argument(
+        "--push-to-hub",
+        action="store_true",
+        help="Upload progress file to the dataset repo on HuggingFace Hub",
+        default=True,
+    )
+    parser.add_argument(
+        "--stride",
+        type=int,
+        default=1,
+        help="Compute progress every N frames, interpolate the rest (default: 1 = every frame)",
+    )
+
+    args = parser.parse_args()
+
+    logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
+
+    # Try to get reward_model_path from parquet metadata if not provided
+    reward_model_path = args.reward_model_path
+    if reward_model_path is None:
+        # Load dataset to find parquet path
+        temp_dataset = LeRobotDataset(args.dataset_repo_id, download_videos=False)
+        parquet_path = Path(temp_dataset.root) / "sarm_progress.parquet"
+        reward_model_path = get_reward_model_path_from_parquet(parquet_path)
+        if reward_model_path:
+            logging.info(f"Using reward model from parquet metadata: {reward_model_path}")
+        else:
+            raise ValueError(
+                "--reward-model-path is required (no existing parquet with model metadata found)"
+            )
+
+    # Handle visualize-only mode
+    if args.visualize_only:
+        dataset, reward_model, preprocess = load_sarm_resources(
+            args.dataset_repo_id, reward_model_path, args.device
+        )
+        logging.info(f"Visualization-only mode: visualizing {args.num_visualizations} episodes")
+        viz_episodes = list(range(min(args.num_visualizations, dataset.num_episodes)))
+        visualize_sarm_predictions(
+            dataset=dataset,
+            reward_model=reward_model,
+            preprocess=preprocess,
+            episode_indices=viz_episodes,
+            head_mode=args.head_mode,
+            output_dir=Path(args.output_dir),
+            stride=args.stride,
+        )
+        print(f"\nVisualizations saved to: {Path(args.output_dir).absolute()}")
+        return
+
+    # Full RABC computation (compute_sarm_progress loads model/dataset itself)
+    output_path = compute_sarm_progress(
+        dataset_repo_id=args.dataset_repo_id,
+        reward_model_path=reward_model_path,
+        output_path=args.output_path,
+        head_mode=args.head_mode,
+        device=args.device,
+        num_visualizations=args.num_visualizations,
+        output_dir=args.output_dir,
+        stride=args.stride,
+    )
+
+    print(f"\nSARM progress values saved to: {output_path}")
+
+    # Upload to Hub if requested
+    if args.push_to_hub:
+        from huggingface_hub import HfApi
+
+        api = HfApi()
+        hub_path = "sarm_progress.parquet"
+
+        print(f"\nUploading to Hub: {args.dataset_repo_id}/{hub_path}")
+        api.upload_file(
+            path_or_fileobj=str(output_path),
+            path_in_repo=hub_path,
+            repo_id=args.dataset_repo_id,
+            repo_type="dataset",
+        )
+        print(
+            f"Successfully uploaded to: https://huggingface.co/datasets/{args.dataset_repo_id}/blob/main/{hub_path}"
+        )
+
+        print("\nTo use in training, add to your config:")
+        print("  use_rabc: true")
+        print(f"  rabc_progress_path: hf://datasets/{args.dataset_repo_id}/{hub_path}")
+        print("  rabc_head_mode: sparse  # or dense")
+    else:
+        print("\nTo use in training, add to your config:")
+        print("  use_rabc: true")
+        print(f"  rabc_progress_path: {output_path}")
+        print("  rabc_head_mode: sparse  # or dense")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/policies/sarm/configuration_sarm.py b/lerobot/src/lerobot/policies/sarm/configuration_sarm.py
new file mode 100644
index 0000000000000000000000000000000000000000..673422fe2998fa973a81428e22615bd5804194db
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sarm/configuration_sarm.py
@@ -0,0 +1,249 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Qianzhong Chen, Justin Yu, Mac Schwager, Pieter Abbeel, Yide Shentu, Philipp Wu
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+SARM: Stage-Aware Reward Modeling for Long Horizon Robot Manipulation.
+Paper: https://arxiv.org/abs/2509.25358
+"""
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.optim.optimizers import AdamWConfig
+from lerobot.optim.schedulers import CosineDecayWithWarmupSchedulerConfig
+from lerobot.utils.constants import OBS_IMAGES, OBS_STATE
+
+
+@PreTrainedConfig.register_subclass("sarm")
+@dataclass
+class SARMConfig(PreTrainedConfig):
+    """Configuration class for SARM (Stage-Aware Reward Modeling).
+
+    Supports three annotation modes:
+
+    1. single_stage (default): No annotations needed. Uses the episode's task description
+       as a single stage covering the entire episode.
+
+    2. dense_only: Uses dense (fine-grained) annotations from VLM, with an auto-generated
+       single sparse "task" stage covering the full episode. The dense head learns detailed
+       subtask progression while sparse provides overall task completion.
+
+    3. dual: Full dual-head mode with both sparse (high-level) and dense (fine-grained)
+       annotations from VLM. Both heads are trained on their respective annotations.
+
+    The annotation_mode determines how sparse_temporal_proportions and dense_temporal_proportions
+    are loaded/generated during model initialization.
+    """
+
+    annotation_mode: str = "single_stage"  # "single_stage", "dense_only", or "dual"
+    n_obs_steps: int = 8  # Number of observation history steps
+    frame_gap: int = 30  # Frame gap between frames (at 30 fps = 1 second)
+    max_rewind_steps: int = 4  # Maximum rewind steps for temporal augmentation
+
+    # Total frames = 1 + n_obs_steps + max_rewind_steps (computed in property)
+    # During training with rewind: [obs_frames] + [rewind_frames]
+    # During inference: [obs_frames] only
+
+    # Architecture params
+    image_dim: int = 512
+    text_dim: int = 512
+    hidden_dim: int = 768
+    num_heads: int = 12
+    num_layers: int = 8
+    max_state_dim: int = 32
+    drop_n_last_frames: int = 1
+    batch_size: int = 64
+    clip_batch_size: int = 64
+    dropout: float = 0.1
+    stage_loss_weight: float = 1.0  # Weight for stage classification loss when using subtask annotations
+
+    rewind_probability: float = 0.8
+    language_perturbation_probability: float = 0.2
+
+    # Sparse annotations (high-level stages)
+    num_sparse_stages: int = 1
+    sparse_subtask_names: list | None = None
+    sparse_temporal_proportions: list | None = None
+
+    # Dense annotations (fine-grained stages)
+    num_dense_stages: int | None = None
+    dense_subtask_names: list | None = None
+    dense_temporal_proportions: list | None = None
+
+    pretrained_model_path: str | None = None
+    device: str | None = None
+    image_key: str = OBS_IMAGES + ".top"  # Key for image used from the dataset
+    state_key: str = OBS_STATE
+
+    # Populated by the processor (video_features, state_features, text_features)
+    input_features: dict = field(default_factory=lambda: {})
+
+    # Output features (updated in __post_init__)
+    output_features: dict = field(
+        default_factory=lambda: {
+            "stage": PolicyFeature(shape=(9, 5), type=FeatureType.REWARD),
+            "progress": PolicyFeature(shape=(9, 1), type=FeatureType.REWARD),
+        }
+    )
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.MEAN_STD,
+            "LANGUAGE": NormalizationMode.IDENTITY,
+            "REWARD": NormalizationMode.IDENTITY,
+        }
+    )
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        if self.annotation_mode not in ["single_stage", "dense_only", "dual"]:
+            raise ValueError(
+                f"annotation_mode must be 'single_stage', 'dense_only', or 'dual', got {self.annotation_mode}"
+            )
+
+        if self.annotation_mode == "single_stage":
+            # Use task description as stage name, full episode as one stage
+            self.num_sparse_stages = 1
+            self.sparse_subtask_names = ["task"]
+            self.sparse_temporal_proportions = [1.0]
+            self.num_dense_stages = None
+            self.dense_subtask_names = None
+            self.dense_temporal_proportions = None
+
+        elif self.annotation_mode == "dense_only":
+            self.num_sparse_stages = 1
+            self.sparse_subtask_names = ["task"]
+            self.sparse_temporal_proportions = [1.0]
+
+        self.input_features = {}
+        self.output_features = {}
+
+        if self.image_key:
+            self.input_features[self.image_key] = PolicyFeature(shape=(480, 640, 3), type=FeatureType.VISUAL)
+
+        self.input_features[self.state_key] = PolicyFeature(
+            shape=(self.max_state_dim,),
+            type=FeatureType.STATE,
+        )
+
+        # Update output features based on annotation_mode
+        if self.annotation_mode in ["dense_only", "dual"]:
+            self.output_features["sparse_stage"] = PolicyFeature(
+                shape=(self.num_frames, self.num_sparse_stages), type=FeatureType.REWARD
+            )
+            self.output_features["sparse_progress"] = PolicyFeature(
+                shape=(self.num_frames, 1), type=FeatureType.REWARD
+            )
+            dense_stages = self.num_dense_stages or self.num_sparse_stages
+            self.output_features["dense_stage"] = PolicyFeature(
+                shape=(self.num_frames, dense_stages), type=FeatureType.REWARD
+            )
+            self.output_features["dense_progress"] = PolicyFeature(
+                shape=(self.num_frames, 1), type=FeatureType.REWARD
+            )
+        else:
+            self.output_features["sparse_stage"] = PolicyFeature(
+                shape=(self.num_frames, self.num_sparse_stages), type=FeatureType.REWARD
+            )
+            self.output_features["sparse_progress"] = PolicyFeature(
+                shape=(self.num_frames, 1), type=FeatureType.REWARD
+            )
+
+        if self.max_rewind_steps >= self.n_obs_steps:
+            raise ValueError(
+                f"max_rewind_steps ({self.max_rewind_steps}) must be less than n_obs_steps ({self.n_obs_steps})"
+            )
+        if self.num_sparse_stages < 1:
+            raise ValueError(f"num_sparse_stages must be at least 1, got {self.num_sparse_stages}")
+        if (
+            self.annotation_mode in ["dense_only", "dual"]
+            and self.num_dense_stages is not None
+            and self.num_dense_stages < 2
+        ):
+            raise ValueError(f"num_dense_stages must be at least 2, got {self.num_dense_stages}")
+
+    def get_optimizer_preset(self) -> AdamWConfig:
+        """Get default optimizer configuration for SARM training."""
+        return AdamWConfig(
+            lr=5e-5,
+            weight_decay=1e-3,
+            betas=(0.9, 0.999),
+            eps=1e-8,
+        )
+
+    def get_scheduler_preset(self) -> CosineDecayWithWarmupSchedulerConfig:
+        """Get default learning rate scheduler configuration."""
+        return CosineDecayWithWarmupSchedulerConfig(
+            peak_lr=5e-5,
+            decay_lr=5e-6,
+            num_warmup_steps=500,
+            num_decay_steps=50000,
+        )
+
+    def validate_features(self) -> None:
+        pass
+
+    @property
+    def uses_dual_heads(self) -> bool:
+        """Whether the model uses dual heads (dense_only or dual annotation modes)."""
+        return self.annotation_mode in ["dense_only", "dual"]
+
+    @property
+    def num_frames(self) -> int:
+        """Total number of frames in sequence.
+
+        For training: 1 + n_obs_steps + max_rewind_steps
+        The sequence is: [obs_frames (n_obs_steps + 1)] + [rewind_frames (max_rewind_steps)]
+        """
+        return 1 + self.n_obs_steps + self.max_rewind_steps
+
+    @property
+    def max_length(self) -> int:
+        return self.num_frames
+
+    @property
+    def observation_delta_indices(self) -> list[int]:
+        """Bidirectional frame sampling centered on target frame.
+
+        Example with n_obs_steps=8, gap=30:
+        Before: [-120, -90, -60, -30]  (4 frames)
+        Current: [0]                   (1 frame)
+        After:  [30, 60, 90, 120]      (4 frames)
+        Total: 9 frames
+        """
+        half_steps = self.n_obs_steps // 2
+
+        past_deltas = [-self.frame_gap * i for i in range(half_steps, 0, -1)]
+        future_deltas = [self.frame_gap * i for i in range(1, half_steps + 1)]
+        obs_deltas = past_deltas + [0] + future_deltas
+
+        # Rewind placeholders
+        rewind_deltas = [-self.frame_gap * (i + 1) for i in range(self.max_rewind_steps)]
+
+        return obs_deltas + rewind_deltas
+
+    @property
+    def action_delta_indices(self) -> None:
+        """SARM is a reward model, not an action policy."""
+        return None
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/sarm/modeling_sarm.py b/lerobot/src/lerobot/policies/sarm/modeling_sarm.py
new file mode 100644
index 0000000000000000000000000000000000000000..6051d90f8ef51d5905a5a3d6693e24eecb74a560
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sarm/modeling_sarm.py
@@ -0,0 +1,794 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Qianzhong Chen, Justin Yu, Mac Schwager, Pieter Abbeel, Yide Shentu, Philipp Wu
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+SARM: Stage-Aware Reward Modeling for Long Horizon Robot Manipulation.
+
+Paper: https://arxiv.org/abs/2509.25358
+
+- StageTransformer: Predicts stage classification (sparse/dense)
+- SubtaskTransformer: Predicts within-stage progress (tau) conditioned on stage
+"""
+
+import json
+import logging
+import random
+
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.nn.functional as F  # noqa: N812
+from torch import Tensor
+
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.sarm.configuration_sarm import SARMConfig
+from lerobot.policies.sarm.sarm_utils import (
+    normalize_stage_tau,
+    pad_state_to_max_dim,
+)
+from lerobot.utils.constants import OBS_STR
+
+
+class StageTransformer(nn.Module):
+    """
+    Stage classification transformer for SARM.
+
+    Predicts which stage/subtask the current frame belongs to.
+    Supports both sparse (high-level) and dense (fine-grained) annotation schemes.
+
+    Input streams: [vis_proj, lang_proj, state_proj] concatenated -> (B, N+2, T, D)
+    Output: stage logits (B, T, num_classes)
+    """
+
+    def __init__(
+        self,
+        d_model: int = 512,
+        vis_emb_dim: int = 512,
+        text_emb_dim: int = 512,
+        state_dim: int = 32,
+        n_layers: int = 6,
+        n_heads: int = 8,
+        dropout: float = 0.1,
+        num_cameras: int = 1,
+        num_classes_sparse: int = 4,
+        num_classes_dense: int = 8,
+    ):
+        super().__init__()
+        self.d_model = d_model
+        self.num_cameras = num_cameras
+
+        # Projections
+        self.lang_proj = nn.Linear(text_emb_dim, d_model)
+        self.visual_proj = nn.Linear(vis_emb_dim, d_model)
+        self.state_proj = nn.Linear(state_dim, d_model)
+
+        # Encoder
+        enc_layer = nn.TransformerEncoderLayer(d_model, n_heads, 4 * d_model, dropout, batch_first=True)
+        self.transformer = nn.TransformerEncoder(enc_layer, n_layers)
+
+        # Positional bias on first visual frame
+        self.first_pos = nn.Parameter(torch.zeros(1, d_model))
+
+        # Shared fusion MLP
+        # Fuses (num_cameras + 2) streams: cameras + lang + state
+        fused_in = d_model * (num_cameras + 2)
+        self.fusion_backbone = nn.Sequential(
+            nn.LayerNorm(fused_in),
+            nn.Linear(fused_in, d_model),
+            nn.ReLU(),
+        )
+
+        # Scheme-specific heads
+        self.heads = nn.ModuleDict(
+            {
+                "sparse": nn.Linear(d_model, num_classes_sparse),
+                "dense": nn.Linear(d_model, num_classes_dense),
+            }
+        )
+
+    def _prep_lang(self, lang_emb: torch.Tensor, B: int, T: int, D: int) -> torch.Tensor:  # noqa: N803
+        """
+        Prepare language embeddings for fusion.
+
+        Accepts lang_emb of shape:
+          - (B, text_emb_dim) -> broadcast across time
+          - (B, T, text_emb_dim) -> per-timestep (dense annotation mode)
+
+        Returns: (B, 1, T, D)
+        """
+        if lang_emb.dim() == 3:
+            # (B, T, E) -> (B, T, D) -> (B, 1, T, D)
+            lang_proj = self.lang_proj(lang_emb).unsqueeze(1)
+        else:
+            # (B, E) -> (B, 1, 1, D) -> expand to (B, 1, T, D)
+            lang_proj = self.lang_proj(lang_emb).unsqueeze(1).unsqueeze(2).expand(B, 1, T, D)
+        return lang_proj
+
+    def forward(
+        self,
+        img_seq: torch.Tensor,  # (B, N, T, vis_emb_dim)
+        lang_emb: torch.Tensor,  # (B, E) or (B, T, E)
+        state: torch.Tensor,  # (B, T, state_dim)
+        lengths: torch.Tensor,  # (B,) - valid sequence lengths
+        scheme: str = "sparse",  # "sparse" or "dense"
+    ) -> torch.Tensor:
+        """
+        Forward pass for stage classification.
+
+        Args:
+            img_seq: Image embeddings (B, N, T, vis_emb_dim) where N=num_cameras
+            lang_emb: Language embeddings (B, E) or (B, T, E) for dense
+            state: State features (B, T, state_dim)
+            lengths: Valid sequence lengths (B,) for masking
+            scheme: "sparse" or "dense" for head selection
+
+        Returns:
+            Stage logits (B, T, num_classes)
+        """
+        assert scheme in self.heads, f"Unknown scheme '{scheme}'. Use one of {list(self.heads.keys())}."
+
+        B, N, T, _ = img_seq.shape  # noqa: N806
+        D = self.d_model  # noqa: N806
+        device = img_seq.device
+
+        # Project inputs
+        vis_proj = self.visual_proj(img_seq)  # (B, N, T, D)
+        state_proj = self.state_proj(state).unsqueeze(1)  # (B, 1, T, D)
+        lang_proj = self._prep_lang(lang_emb, B, T, D)  # (B, 1, T, D)
+
+        # Concatenate streams
+        # cameras + lang + state -> (B, N+2, T, D)
+        x = torch.cat([vis_proj, lang_proj, state_proj], dim=1)
+
+        # Add positional bias to first visual frame
+        x[:, :N, 0, :] = x[:, :N, 0, :] + self.first_pos
+
+        # Flatten to tokens for Transformer
+        x_tokens = x.view(B, (N + 2) * T, D)
+        L = x_tokens.size(1)  # noqa: N806
+
+        # Create padding mask
+        base_mask = torch.arange(T, device=device).expand(B, T) >= lengths.unsqueeze(1)  # (B, T)
+        mask = base_mask.unsqueeze(1).expand(B, N + 2, T).reshape(B, (N + 2) * T)
+
+        # Create causal mask
+        causal_mask = torch.triu(torch.ones(L, L, device=device, dtype=torch.bool), diagonal=1)
+
+        # Encode
+        h = self.transformer(x_tokens, mask=causal_mask, src_key_padding_mask=mask, is_causal=True)
+
+        # Reshape and fuse
+        h = h.view(B, N + 2, T, D).permute(0, 2, 1, 3).reshape(B, T, (N + 2) * D)
+        fused = self.fusion_backbone(h)  # (B, T, D)
+
+        # Scheme-specific logits
+        logits = self.heads[scheme](fused)  # (B, T, num_classes)
+        return logits
+
+
+class SubtaskTransformer(nn.Module):
+    """
+    Subtask progress regression transformer for SARM.
+
+    Predicts within-stage normalized progress (tau) conditioned on stage prior.
+    The stage prior is a one-hot encoding passed from StageTransformer predictions.
+
+    Input streams: [vis_proj, lang_proj, state_proj, stage_emb] -> (B, N+3, T, D)
+    Output: tau predictions (B, T) in [0, 1]
+    """
+
+    def __init__(
+        self,
+        d_model: int = 512,
+        vis_emb_dim: int = 512,
+        text_emb_dim: int = 512,
+        state_dim: int = 32,
+        n_layers: int = 6,
+        n_heads: int = 8,
+        dropout: float = 0.1,
+        num_cameras: int = 1,
+    ):
+        super().__init__()
+        self.d_model = d_model
+        self.num_cameras = num_cameras
+
+        # Projections
+        self.lang_proj = nn.Linear(text_emb_dim, d_model)
+        self.visual_proj = nn.Linear(vis_emb_dim, d_model)
+        self.state_proj = nn.Linear(state_dim, d_model)
+
+        # Encoder
+        enc = nn.TransformerEncoderLayer(d_model, n_heads, 4 * d_model, dropout, batch_first=True)
+        self.transformer = nn.TransformerEncoder(enc, n_layers)
+
+        # Learned bias on first visual frame
+        self.first_pos = nn.Parameter(torch.zeros(1, d_model))
+
+        # Shared fusion backbone
+        # Fuses (num_cameras + 3) streams: cameras + lang + state + stage_emb
+        fused_in = d_model * (num_cameras + 3)
+        self.fusion_backbone = nn.Sequential(
+            nn.LayerNorm(fused_in),
+            nn.Linear(fused_in, d_model),
+            nn.ReLU(),
+        )
+
+        # Scheme-specific regression heads
+        self.heads = nn.ModuleDict(
+            {
+                "sparse": nn.Linear(d_model, 1),
+                "dense": nn.Linear(d_model, 1),
+            }
+        )
+
+    def _prep_lang(self, lang_emb: torch.Tensor, B: int, T: int, D: int) -> torch.Tensor:  # noqa: N803
+        """
+        Prepare language embeddings for fusion.
+        """
+        if lang_emb.dim() == 3:
+            # (B, T, E) -> (B, T, D) -> (B, 1, T, D)
+            return self.lang_proj(lang_emb).unsqueeze(1)
+        else:
+            # (B, E) -> (B, 1, 1, D) -> (B, 1, T, D)
+            return self.lang_proj(lang_emb).unsqueeze(1).unsqueeze(2).expand(B, 1, T, D)
+
+    def _stage_to_dmodel(self, stage_prior: torch.Tensor) -> torch.Tensor:
+        """
+        Deterministic projection of one-hot stage to d_model by pad/truncate.
+
+        Args:
+            stage_prior: One-hot stage embedding (B, 1, T, C)
+
+        Returns:
+            Projected stage embedding (B, 1, T, d_model)
+        """
+        B, one, T, C = stage_prior.shape  # noqa: N806
+        D = self.d_model  # noqa: N806
+        if D == C:
+            return stage_prior
+        elif D > C:
+            pad = torch.zeros(B, one, T, D - C, device=stage_prior.device, dtype=stage_prior.dtype)
+            return torch.cat([stage_prior, pad], dim=-1)
+        else:
+            return stage_prior[..., :D]
+
+    def forward(
+        self,
+        img_seq: torch.Tensor,  # (B, N, T, vis_emb_dim)
+        lang_emb: torch.Tensor,  # (B, E) or (B, T, E)
+        state: torch.Tensor,  # (B, T, state_dim)
+        lengths: torch.Tensor,  # (B,) - valid sequence lengths
+        stage_prior: torch.Tensor,  # (B, 1, T, C) one-hot from gen_stage_emb
+        scheme: str = "sparse",  # "sparse" or "dense"
+    ) -> torch.Tensor:
+        """
+        Forward pass for subtask progress regression.
+
+        Args:
+            img_seq: Image embeddings (B, N, T, vis_emb_dim)
+            lang_emb: Language embeddings (B, E) or (B, T, E)
+            state: State features (B, T, state_dim)
+            lengths: Valid sequence lengths (B,) for masking
+            stage_prior: One-hot stage prior (B, 1, T, num_classes)
+            scheme: "sparse" or "dense" for head selection
+
+        Returns:
+            Tau predictions (B, T) in [0, 1] via sigmoid
+        """
+        assert scheme in self.heads, f"Unknown scheme '{scheme}'. Use one of {list(self.heads.keys())}."
+
+        B, N, T, _ = img_seq.shape  # noqa: N806
+        D = self.d_model  # noqa: N806
+        device = img_seq.device
+
+        # Project inputs
+        vis_proj = self.visual_proj(img_seq)  # (B, N, T, D)
+        state_proj = self.state_proj(state).unsqueeze(1)  # (B, 1, T, D)
+        lang_proj = self._prep_lang(lang_emb, B, T, D)  # (B, 1, T, D)
+        stage_emb = self._stage_to_dmodel(stage_prior)  # (B, 1, T, D)
+
+        # Concatenate all streams
+        # cameras + lang + state + stage_emb -> (B, N+3, T, D)
+        x = torch.cat([vis_proj, lang_proj, state_proj, stage_emb], dim=1)
+
+        # Add positional bias to first visual frame
+        x[:, :N, 0, :] = x[:, :N, 0, :] + self.first_pos
+
+        # Flatten to tokens
+        x_tokens = x.view(B, (N + 3) * T, D)
+        L = x_tokens.size(1)  # noqa: N806
+
+        # Create padding mask
+        base_mask = torch.arange(T, device=device).expand(B, T) >= lengths.unsqueeze(1)
+        mask = base_mask.unsqueeze(1).expand(B, N + 3, T).reshape(B, (N + 3) * T)
+
+        # Create causal mask
+        causal_mask = torch.triu(torch.ones(L, L, device=device, dtype=torch.bool), diagonal=1)
+
+        # Encode
+        h = self.transformer(x_tokens, mask=causal_mask, src_key_padding_mask=mask, is_causal=True)
+
+        # Reshape and fuse
+        h = h.view(B, N + 3, T, D)
+        h_flat = h.permute(0, 2, 1, 3).reshape(B, T, (N + 3) * D)
+        fused = self.fusion_backbone(h_flat)  # (B, T, D)
+
+        # Scheme-specific regression head -> sigmoid
+        r = torch.sigmoid(self.heads[scheme](fused)).squeeze(-1)  # (B, T)
+        return r
+
+
+def gen_stage_emb(num_classes: int, targets: torch.Tensor) -> torch.Tensor:
+    """
+    Generate one-hot stage embeddings from targets.
+
+    Args:
+        num_classes: Number of stage classes
+        targets: Target values (B, T) where integer part is stage index
+
+    Returns:
+        One-hot stage embedding (B, 1, T, num_classes)
+    """
+    # Integer part of float targets -> [0, C-1]
+    idx = targets.long().clamp(min=0, max=num_classes - 1)  # (B, T)
+    C = num_classes  # noqa: N806
+    # Identity-lookup one-hot
+    stage_onehot = torch.eye(C, device=targets.device)[idx]  # (B, T, C)
+    stage_onehot = stage_onehot.unsqueeze(1)  # (B, 1, T, C)
+    return stage_onehot
+
+
+class SARMRewardModel(PreTrainedPolicy):
+    """
+    SARM Reward Model for stage-aware task completion rewards.
+
+    Uses two separate transformer models:
+    - StageTransformer: Classifies which stage/subtask
+    - SubtaskTransformer: Predicts within-stage progress (tau)
+
+    Training uses 75%/25% GT/predicted stage conditioning (teacher forcing).
+    """
+
+    name = "sarm"
+    config_class = SARMConfig
+
+    def __init__(self, config: SARMConfig, dataset_stats: dict | None = None, dataset_meta=None):
+        super().__init__(config, dataset_stats)
+        config.validate_features()
+        self.config = config
+        self.dataset_stats = dataset_stats
+        self.device = torch.device(
+            config.device if config.device else "cuda" if torch.cuda.is_available() else "cpu"
+        )
+
+        # Load temporal proportions based on annotation_mode
+        if config.annotation_mode == "single_stage":
+            logging.info(f"Using single_stage mode: sparse_subtask_names={config.sparse_subtask_names}")
+        elif dataset_meta is not None:
+            self._load_temporal_proportions(dataset_meta)
+
+        # Create two separate models
+        self.stage_model = StageTransformer(
+            d_model=config.hidden_dim,
+            vis_emb_dim=config.image_dim,
+            text_emb_dim=config.text_dim,
+            state_dim=config.max_state_dim,
+            n_layers=config.num_layers,
+            n_heads=config.num_heads,
+            dropout=config.dropout,
+            num_cameras=1,  # Single camera for now
+            num_classes_sparse=config.num_sparse_stages,
+            num_classes_dense=config.num_dense_stages or config.num_sparse_stages,
+        )
+
+        self.subtask_model = SubtaskTransformer(
+            d_model=config.hidden_dim,
+            vis_emb_dim=config.image_dim,
+            text_emb_dim=config.text_dim,
+            state_dim=config.max_state_dim,
+            n_layers=config.num_layers,
+            n_heads=config.num_heads,
+            dropout=config.dropout,
+            num_cameras=1,
+        )
+
+        self.stage_model.to(self.device)
+        self.subtask_model.to(self.device)
+
+        # GT/predicted stage ratio for teacher forcing
+        self.gt_stage_ratio = 0.75
+
+        if config.uses_dual_heads:
+            logging.info(
+                f"SARM initialized with dual heads: {config.num_sparse_stages} sparse stages, "
+                f"{config.num_dense_stages} dense stages"
+            )
+        else:
+            logging.info(f"SARM initialized with sparse head only: {config.num_sparse_stages} stages")
+
+        logging.info(f"SARM initialized on {self.device}")
+
+    def _load_proportions_from_json(self, path, annotation_type: str) -> tuple[list[str], list[float]]:
+        """Load temporal proportions from a JSON file (preserving order)."""
+        if not path.exists():
+            raise ValueError(
+                f"{annotation_type.capitalize()} temporal proportions not found at {path}. "
+                f"Run the subtask annotation tool with --{annotation_type}-subtasks to generate annotations."
+            )
+        with open(path) as f:
+            proportions_dict = json.load(f)
+        names = list(proportions_dict.keys())
+        logging.info(f"Loaded {len(names)} {annotation_type} subtasks: {names}")
+        logging.info(f"{annotation_type.capitalize()} temporal proportions: {proportions_dict}")
+        return names, [proportions_dict[name] for name in names]
+
+    def _load_temporal_proportions(self, dataset_meta) -> None:
+        """Load temporal proportions based on annotation_mode."""
+        meta_path = dataset_meta.root / "meta"
+
+        if self.config.annotation_mode == "dual":
+            names, props = self._load_proportions_from_json(
+                meta_path / "temporal_proportions_sparse.json", "sparse"
+            )
+            (
+                self.config.num_sparse_stages,
+                self.config.sparse_subtask_names,
+                self.config.sparse_temporal_proportions,
+            ) = len(names), names, props
+
+        if self.config.annotation_mode in ["dense_only", "dual"]:
+            names, props = self._load_proportions_from_json(
+                meta_path / "temporal_proportions_dense.json", "dense"
+            )
+            (
+                self.config.num_dense_stages,
+                self.config.dense_subtask_names,
+                self.config.dense_temporal_proportions,
+            ) = len(names), names, props
+            if self.config.annotation_mode == "dense_only":
+                logging.info(f"Using auto-generated sparse 'task' stage: {self.config.sparse_subtask_names}")
+
+    def to(self, device):
+        """Override to method to ensure all components move together."""
+        super().to(device)
+        self.device = device if isinstance(device, torch.device) else torch.device(device)
+        self.stage_model.to(device)
+        self.subtask_model.to(device)
+        return self
+
+    @torch.no_grad()
+    def calculate_rewards(
+        self,
+        text_embeddings: np.ndarray | torch.Tensor,
+        video_embeddings: np.ndarray | torch.Tensor,
+        state_features: np.ndarray | torch.Tensor | None = None,
+        lengths: np.ndarray | torch.Tensor | None = None,
+        return_all_frames: bool = False,
+        return_stages: bool = False,
+        return_confidence: bool = False,
+        head_mode: str | None = "sparse",
+        frame_index: int | None = None,
+    ) -> np.ndarray | tuple:
+        """
+        Calculate rewards for given text, video, and state representations.
+
+        This is the canonical method for SARM reward computation, used for:
+        - Inference/visualization
+        - RA-BC weight computation
+
+        Args:
+            text_embeddings: Encoded text representations (batch_size, 512)
+            video_embeddings: Encoded video representations (batch_size, num_frames, 512)
+            state_features: Joint state features (batch_size, num_frames, state_dim)
+            lengths: Valid sequence lengths (batch_size,)
+            return_all_frames: If True, return rewards for all frames
+            return_stages: If True, also return stage predictions
+            return_confidence: If True, also return stage confidence
+            head_mode: Which head to use ("sparse" or "dense")
+            frame_index: Index of the target frame to extract (default: n_obs_steps).
+
+        Returns:
+            Rewards and optionally stage probs/confidence.
+        """
+        if isinstance(text_embeddings, np.ndarray):
+            text_embeddings = torch.tensor(text_embeddings, dtype=torch.float32)
+        if isinstance(video_embeddings, np.ndarray):
+            video_embeddings = torch.tensor(video_embeddings, dtype=torch.float32)
+        if state_features is not None and isinstance(state_features, np.ndarray):
+            state_features = torch.tensor(state_features, dtype=torch.float32)
+
+        # Handle single sample case
+        if text_embeddings.dim() == 1:
+            text_embeddings = text_embeddings.unsqueeze(0)
+            video_embeddings = video_embeddings.unsqueeze(0)
+            if state_features is not None:
+                state_features = state_features.unsqueeze(0)
+            single_sample = True
+        else:
+            single_sample = False
+
+        batch_size = video_embeddings.shape[0]
+        seq_len = video_embeddings.shape[1]
+
+        scheme = head_mode
+
+        # Default lengths if not provided
+        if lengths is None:
+            lengths = torch.full((batch_size,), seq_len, dtype=torch.int32)
+        elif isinstance(lengths, np.ndarray):
+            lengths = torch.tensor(lengths, dtype=torch.int32)
+
+        # Reshape video to (B, N, T, D) for multi-camera format
+        # Currently single camera: (B, T, D) -> (B, 1, T, D)
+        img_seq = video_embeddings.unsqueeze(1).to(self.device)
+        lang_emb = text_embeddings.to(self.device)
+        state = (
+            state_features.to(self.device)
+            if state_features is not None
+            else torch.zeros(batch_size, seq_len, self.config.max_state_dim, device=self.device)
+        )
+        lens = lengths.to(self.device)
+
+        # Pad state to max_state_dim
+        state = pad_state_to_max_dim(state, self.config.max_state_dim)
+
+        # Get num_classes for this scheme
+        num_classes = self.config.num_sparse_stages if scheme == "sparse" else self.config.num_dense_stages
+
+        # Run stage model
+        stage_logits = self.stage_model(img_seq, lang_emb, state, lens, scheme=scheme)
+        stage_probs = F.softmax(stage_logits, dim=-1)  # (B, T, num_classes)
+        stage_idx = stage_probs.argmax(dim=-1)  # (B, T)
+        stage_conf = stage_probs.gather(-1, stage_idx.unsqueeze(-1)).squeeze(-1)  # (B, T)
+
+        # Create one-hot stage prior
+        stage_onehot = F.one_hot(stage_idx, num_classes=num_classes).float()  # (B, T, C)
+        stage_emb = stage_onehot.unsqueeze(1)  # (B, 1, T, C)
+
+        # Run subtask model
+        tau_pred = self.subtask_model(img_seq, lang_emb, state, lens, stage_emb, scheme=scheme)
+
+        # Compute final reward: stage + tau
+        raw_reward = stage_idx.float() + tau_pred  # (B, T)
+
+        # Normalize to [0, 1] using temporal proportions for proper weighting
+        if scheme == "sparse":
+            normalized_reward = normalize_stage_tau(
+                raw_reward,
+                num_stages=num_classes,
+                temporal_proportions=self.config.sparse_temporal_proportions,
+                subtask_names=self.config.sparse_subtask_names,
+            )
+        else:
+            normalized_reward = normalize_stage_tau(
+                raw_reward,
+                num_stages=num_classes,
+                temporal_proportions=self.config.dense_temporal_proportions,
+                subtask_names=self.config.dense_subtask_names,
+            )
+
+        # Default frame index is n_obs_steps (last observation frame)
+        if frame_index is None:
+            frame_index = self.config.n_obs_steps
+
+        # Prepare outputs (batch mode or no smoothing)
+        if return_all_frames:
+            rewards = normalized_reward.cpu().numpy()
+        else:
+            rewards = normalized_reward[:, frame_index].cpu().numpy()
+
+        if single_sample:
+            rewards = rewards[0] if not return_all_frames else rewards[0]
+
+        outputs = [rewards]
+        if return_stages:
+            probs = stage_probs.cpu().numpy()
+            if single_sample:
+                probs = probs[0]
+            outputs.append(probs)
+        if return_confidence:
+            conf = stage_conf.cpu().numpy()
+            if single_sample:
+                conf = conf[0]
+            outputs.append(conf)
+
+        return outputs[0] if len(outputs) == 1 else tuple(outputs)
+
+    def train(self, mode: bool = True):
+        """Set training mode for both models."""
+        super().train(mode)
+        self.stage_model.train(mode)
+        self.subtask_model.train(mode)
+        return self
+
+    def eval(self):
+        """Set evaluation mode for both models."""
+        return self.train(False)
+
+    def parameters(self):
+        """Override to return trainable parameters from both models."""
+        from itertools import chain
+
+        return chain(self.stage_model.parameters(), self.subtask_model.parameters())
+
+    def get_optim_params(self):
+        """Override to return optimizer parameters from both models."""
+        return self.parameters()
+
+    def reset(self):
+        """Required by PreTrainedPolicy but not used for reward models."""
+        pass
+
+    def predict_action_chunk(self, batch: dict[str, Tensor]) -> Tensor:
+        """Required by PreTrainedPolicy but not used for reward models."""
+        raise NotImplementedError("SARM model does not predict action chunks")
+
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Required by PreTrainedPolicy but not used for SARM."""
+        raise NotImplementedError("SARM model does not select actions")
+
+    def _train_step(
+        self,
+        img_emb: torch.Tensor,  # (B, N, T, D)
+        lang_emb: torch.Tensor,  # (B, E) or (B, T, E)
+        state: torch.Tensor,  # (B, T, state_dim)
+        lengths: torch.Tensor,  # (B,)
+        targets: torch.Tensor,  # (B, T) - format: stage.tau
+        scheme: str,
+    ) -> dict[str, torch.Tensor]:
+        """
+        Single training step for one annotation scheme.
+
+        Implements 75%/25% GT/predicted stage conditioning.
+
+        Args:
+            img_emb: Image embeddings (B, N, T, D)
+            lang_emb: Language embeddings
+            state: State features
+            lengths: Valid sequence lengths
+            targets: Target values where floor=stage, remainder=tau
+            scheme: "sparse" or "dense"
+
+        Returns:
+            Dict with stage_loss, subtask_loss, total_loss
+        """
+        num_classes = self.config.num_sparse_stages if scheme == "sparse" else self.config.num_dense_stages
+
+        # Ground truth: stage (integer) and tau (fractional)
+        # Clamp stage indices to valid range [0, num_classes-1] to handle edge cases
+        # where targets may exceed expected range (e.g., frames between subtasks)
+        gt_stage = torch.floor(targets).long().clamp(0, num_classes - 1)  # (B, T)
+        gt_tau = torch.remainder(targets, 1.0)  # (B, T)
+
+        # Run stage model
+        stage_pred = self.stage_model(img_emb, lang_emb, state, lengths, scheme=scheme)
+
+        # 75%/25% GT/predicted stage conditioning
+        if random.random() < self.gt_stage_ratio:
+            # Mode 1: Use ground truth stage -> one-hot
+            stage_emb = gen_stage_emb(num_classes, targets)  # (B, 1, T, C)
+        else:
+            # Mode 2: Use predicted stage argmax -> one-hot
+            stage_idx = stage_pred.argmax(dim=-1)  # (B, T)
+            stage_onehot = F.one_hot(stage_idx, num_classes=num_classes).float()  # (B, T, C)
+            stage_emb = stage_onehot.unsqueeze(1)  # (B, 1, T, C)
+
+        # Run subtask model with stage prior
+        tau_pred = self.subtask_model(img_emb, lang_emb, state, lengths, stage_emb, scheme=scheme)
+
+        # Compute losses
+        stage_loss = F.cross_entropy(stage_pred.view(-1, num_classes), gt_stage.view(-1), reduction="mean")
+        subtask_loss = F.mse_loss(tau_pred, gt_tau, reduction="mean")
+
+        return {
+            "stage_loss": stage_loss,
+            "subtask_loss": subtask_loss,
+            "total_loss": stage_loss + subtask_loss,
+        }
+
+    def forward(self, batch):
+        """
+        Forward pass for SARM reward model training.
+
+        Uses stage+tau target format where:
+        - Integer part = stage index
+        - Fractional part = within-stage progress (tau)
+
+        Training uses 75%/25% GT/predicted stage conditioning.
+
+        Args:
+            batch: Dictionary with 'observation' containing:
+                - 'video_features': (B, T, 512) pre-encoded video features
+                - 'text_features': (B, 512) or (B, T, 512) text features
+                - 'state_features': (B, T, state_dim) joint state features
+                - 'lengths': (B,) valid sequence lengths
+                - 'sparse_targets': (B, T) sparse targets (stage.tau format)
+                - 'dense_targets': (B, T) dense targets (optional, for dual mode)
+
+        Returns:
+            Tuple of (total_loss, output_dict with loss components)
+        """
+        observation = batch.get(OBS_STR, batch)
+
+        # Extract features
+        video_features = observation["video_features"].to(self.device)
+        text_features = observation["text_features"].to(self.device)
+        state_features = observation.get("state_features")
+        if state_features is not None:
+            state_features = state_features.to(self.device)
+
+        batch_size = video_features.shape[0]
+        seq_len = video_features.shape[1]
+
+        # Get lengths (default to full sequence)
+        lengths = observation.get("lengths")
+        if lengths is None:
+            lengths = torch.full((batch_size,), seq_len, dtype=torch.int32, device=self.device)
+        else:
+            lengths = lengths.to(self.device)
+
+        # Reshape video to (B, N, T, D) - single camera
+        img_emb = video_features.unsqueeze(1)
+
+        # Pad state to max_state_dim
+        if state_features is None:
+            state_features = torch.zeros(batch_size, seq_len, self.config.max_state_dim, device=self.device)
+        else:
+            state_features = pad_state_to_max_dim(state_features, self.config.max_state_dim)
+
+        output_dict = {}
+        total_loss = torch.tensor(0.0, device=self.device)
+
+        # Sparse training (always)
+        sparse_targets = observation.get("sparse_targets")
+        if sparse_targets is None:
+            # Try legacy format
+            sparse_targets = observation.get("targets")
+        if sparse_targets is None:
+            raise ValueError("sparse_targets (or targets) is required for SARM training")
+        sparse_targets = sparse_targets.to(self.device)
+
+        sparse_result = self._train_step(
+            img_emb, text_features, state_features, lengths, sparse_targets, scheme="sparse"
+        )
+        output_dict["sparse_stage_loss"] = sparse_result["stage_loss"].item()
+        output_dict["sparse_subtask_loss"] = sparse_result["subtask_loss"].item()
+        total_loss = total_loss + sparse_result["total_loss"]
+
+        # Dense training (if dual mode)
+        if self.config.uses_dual_heads:
+            dense_targets = observation.get("dense_targets")
+            if dense_targets is not None:
+                dense_targets = dense_targets.to(self.device)
+                dense_result = self._train_step(
+                    img_emb, text_features, state_features, lengths, dense_targets, scheme="dense"
+                )
+                output_dict["dense_stage_loss"] = dense_result["stage_loss"].item()
+                output_dict["dense_subtask_loss"] = dense_result["subtask_loss"].item()
+                total_loss = total_loss + dense_result["total_loss"]
+
+        output_dict["total_loss"] = total_loss.item()
+        return total_loss, output_dict
+
+
+def compute_stage_loss(stage_logits: torch.Tensor, target_stages: torch.Tensor) -> torch.Tensor:
+    """Compute cross-entropy loss for stage classification."""
+    _, _, num_stages = stage_logits.shape
+    stage_logits_flat = stage_logits.reshape(-1, num_stages)
+    # Clamp target stage indices to valid range [0, num_stages-1]
+    target_stages_flat = target_stages.reshape(-1).clamp(0, num_stages - 1)
+    return F.cross_entropy(stage_logits_flat, target_stages_flat)
diff --git a/lerobot/src/lerobot/policies/sarm/processor_sarm.py b/lerobot/src/lerobot/policies/sarm/processor_sarm.py
new file mode 100644
index 0000000000000000000000000000000000000000..f377a7ffad02f9dbced84da79005e60db439be82
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sarm/processor_sarm.py
@@ -0,0 +1,516 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""SARM Processor for encoding images/text and generating stage+tau targets."""
+
+import random
+from typing import Any
+
+import numpy as np
+import pandas as pd
+import torch
+from faker import Faker
+from PIL import Image
+from transformers import CLIPModel, CLIPProcessor
+
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.policies.sarm.configuration_sarm import SARMConfig
+from lerobot.policies.sarm.sarm_utils import (
+    apply_rewind_augmentation,
+    compute_absolute_indices,
+    find_stage_and_tau,
+    pad_state_to_max_dim,
+)
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    ProcessorStep,
+    RenameObservationsProcessorStep,
+)
+from lerobot.processor.converters import (
+    from_tensor_to_numpy,
+    policy_action_to_transition,
+    transition_to_policy_action,
+)
+from lerobot.processor.pipeline import PipelineFeatureType
+from lerobot.types import EnvTransition, TransitionKey
+from lerobot.utils.constants import POLICY_POSTPROCESSOR_DEFAULT_NAME, POLICY_PREPROCESSOR_DEFAULT_NAME
+
+
+class SARMEncodingProcessorStep(ProcessorStep):
+    """ProcessorStep that encodes images and text with CLIP and generates stage and progress labels for SARM."""
+
+    def __init__(
+        self,
+        config: SARMConfig,
+        image_key: str | None = None,
+        dataset_meta=None,
+        dataset_stats: dict | None = None,
+    ):
+        super().__init__()
+        self.config = config
+        self.image_key = image_key or config.image_key
+        self.dataset_meta = dataset_meta
+        self.dataset_stats = dataset_stats
+        self.annotation_mode = config.annotation_mode
+
+        # Helper to create temporal proportions dict
+        def make_props_dict(names, props):
+            return dict(zip(names, props, strict=True)) if names and props else None
+
+        # Sparse annotations (always needed)
+        self.sparse_temporal_proportions = make_props_dict(
+            config.sparse_subtask_names, config.sparse_temporal_proportions
+        )
+        self.sparse_subtask_names = config.sparse_subtask_names
+
+        # Dense annotations (only for dual mode)
+        self.dense_subtask_names = config.dense_subtask_names if config.uses_dual_heads else None
+        self.dense_temporal_proportions = (
+            make_props_dict(config.dense_subtask_names, config.dense_temporal_proportions)
+            if config.uses_dual_heads
+            else None
+        )
+
+        self.device = torch.device(
+            self.config.device if self.config.device else "cuda" if torch.cuda.is_available() else "cpu"
+        )
+
+        self.clip_model = CLIPModel.from_pretrained("openai/clip-vit-base-patch32")
+        self.clip_processor = CLIPProcessor.from_pretrained("openai/clip-vit-base-patch32", use_fast=True)
+        self.clip_model.to(self.device)
+        self.clip_model.eval()
+
+        self.verbs = ["move", "grasp", "rotate", "push", "pull", "slide", "lift", "place"]
+        self.fake = Faker()
+
+    def _find_episode_for_frame(self, frame_idx: int) -> int:
+        """Find the episode index for a given frame index."""
+        for ep_idx in range(len(self.dataset_meta.episodes)):
+            ep_start = self.dataset_meta.episodes[ep_idx]["dataset_from_index"]
+            ep_end = self.dataset_meta.episodes[ep_idx]["dataset_to_index"]
+            if ep_start <= frame_idx < ep_end:
+                return ep_idx
+        return 0
+
+    def _get_episode_indices(self, frame_indices: np.ndarray, episode_index) -> np.ndarray:
+        """Get episode indices for each frame index."""
+        if episode_index is None:
+            return np.array([self._find_episode_for_frame(int(f)) for f in frame_indices])
+
+        episode_indices = np.atleast_1d(np.asarray(from_tensor_to_numpy(episode_index)))
+
+        # If single episode but multiple frames, compute episode for each frame
+        if len(episode_indices) == 1 and len(frame_indices) > 1:
+            return np.array([self._find_episode_for_frame(int(f)) for f in frame_indices])
+
+        return episode_indices
+
+    def _generate_perturbed_task(self) -> str:
+        """Generate a random perturbed task string for language perturbation."""
+        num_words = random.randint(1, 5)
+        verb = random.choice(self.verbs)
+        phrase = " ".join([verb] + self.fake.words(nb=num_words))
+        return phrase
+
+    def _get_annotation_config(self, annotation_type: str) -> tuple[list[str], dict[str, float] | None]:
+        """Get global subtask names and temporal proportions for an annotation type."""
+        if annotation_type == "dense":
+            return self.dense_subtask_names, self.dense_temporal_proportions
+        return self.sparse_subtask_names, self.sparse_temporal_proportions
+
+    def _load_episode_annotations(
+        self,
+        ep_idx: int,
+        episodes_df: pd.DataFrame | None,
+        annotation_type: str,
+        global_names: list[str],
+    ) -> tuple[list | None, list | None, list | None]:
+        """Load subtask annotations for an episode from DataFrame."""
+        # Single-stage mode: (linear progress 0→1)
+        if episodes_df is None or len(global_names) == 1:
+            return None, None, None
+
+        # Resolve column name with fallback
+        def col(suffix):
+            prefixed = f"{annotation_type}_{suffix}"
+            return prefixed if prefixed in episodes_df.columns else suffix
+
+        col_names = col("subtask_names")
+        if col_names not in episodes_df.columns or ep_idx >= len(episodes_df):
+            return None, None, None
+
+        subtask_names = episodes_df.loc[ep_idx, col_names]
+        if subtask_names is None or (isinstance(subtask_names, float) and pd.isna(subtask_names)):
+            return None, None, None
+
+        return (
+            subtask_names,
+            episodes_df.loc[ep_idx, col("subtask_start_frames")],
+            episodes_df.loc[ep_idx, col("subtask_end_frames")],
+        )
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """
+        Encode images, text, and normalize states in the transition.
+
+        Implements SARM training data preparation:
+        - Applies language perturbation (20% probability)
+        - Applies rewind augmentation (80% probability)
+        - Generates stage+tau targets for all frames
+        - Outputs lengths tensor for valid sequence masking
+        """
+        new_transition = transition.copy() if hasattr(transition, "copy") else dict(transition)
+        observation = new_transition.get(TransitionKey.OBSERVATION)
+        comp_data = new_transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+
+        frame_index = comp_data.get("index")
+        episode_index = comp_data.get("episode_index")
+
+        if frame_index is None:
+            raise ValueError("Frame index ('index') not found in COMPLEMENTARY_DATA")
+        if episode_index is None:
+            raise ValueError("Episode index ('episode_index') not found in COMPLEMENTARY_DATA")
+
+        frame_indices = np.atleast_1d(np.asarray(from_tensor_to_numpy(frame_index)))
+        episode_indices = self._get_episode_indices(frame_indices, episode_index)
+
+        image = observation.get(self.image_key)
+        if isinstance(image, torch.Tensor):
+            image = image.cpu().numpy()
+
+        # If 4D (T, C, H, W) from delta_timestamps, add batch dim
+        # If 3D (C, H, W) single frame, add batch and time dims
+        if image.ndim == 4:
+            image = image[np.newaxis, ...]  # (T, C, H, W) -> (1, T, C, H, W)
+        elif image.ndim == 3:
+            image = image[np.newaxis, np.newaxis, ...]  # (C, H, W) -> (1, 1, C, H, W)
+
+        batch_size = image.shape[0]
+        total_frames = image.shape[1]  # Should be 13: 9 obs + 4 rewind placeholders
+        n_obs_steps = self.config.n_obs_steps
+        max_rewind_steps = self.config.max_rewind_steps
+        n_obs_frames = 1 + n_obs_steps  # 9 observation frames (including current)
+
+        # Rewind augmentation
+        rewind_steps = torch.zeros(batch_size, dtype=torch.int32)
+        apply_rewind = self.training and random.random() < self.config.rewind_probability
+
+        if apply_rewind and self.dataset_meta is not None:
+            for b_idx, (ep_idx, frame_idx) in enumerate(
+                zip(episode_indices.tolist(), frame_indices.tolist(), strict=True)
+            ):
+                ep_idx, frame_idx = int(ep_idx), int(frame_idx)
+                ep_start = self.dataset_meta.episodes[ep_idx]["dataset_from_index"]
+
+                rewind_step, _ = apply_rewind_augmentation(
+                    frame_idx, ep_start, n_obs_steps, max_rewind_steps, frame_gap=self.config.frame_gap
+                )
+                rewind_steps[b_idx] = rewind_step
+
+        # Compute valid lengths: n_obs_frames + rewind_steps
+        lengths = n_obs_frames + rewind_steps  # (B,)
+
+        # Apply rewind masking to images
+        # For frames beyond valid length, we mask with zeros (or copy last valid frame)
+        for b_idx in range(batch_size):
+            valid_len = lengths[b_idx].item()
+            if valid_len < total_frames:
+                image[b_idx, valid_len:] = 0  # Zero out frames beyond valid length
+
+        # Encode images with CLIP
+        video_features = self._encode_images_batch(image)
+        observation["video_features"] = video_features
+
+        state_key = self.config.state_key
+        state_data = observation.get(state_key)
+
+        if isinstance(state_data, torch.Tensor):
+            state_tensor = state_data.float()
+        else:
+            state_tensor = torch.tensor(state_data, dtype=torch.float32)
+
+        if state_tensor.ndim == 2:
+            state_tensor = state_tensor.unsqueeze(0)  # (T, D) -> (1, T, D)
+        elif state_tensor.ndim == 1:
+            state_tensor = state_tensor.unsqueeze(0).unsqueeze(0)  # (D,) -> (1, 1, D)
+
+        # Apply same rewind masking to state
+        for b_idx in range(batch_size):
+            valid_len = lengths[b_idx].item()
+            if valid_len < state_tensor.shape[1]:
+                state_tensor[b_idx, valid_len:] = 0  # Zero out frames beyond valid length
+
+        observation["state_features"] = pad_state_to_max_dim(state_tensor, self.config.max_state_dim)
+
+        task = comp_data.get("task")
+        if isinstance(task, list):
+            task = task[0] if task else ""
+
+        # Apply language perturbation during training (20% probability)
+        # When perturbed, targets will be zeroed to train model to output low values for irrelevant text
+        apply_perturbation = self.training and random.random() < self.config.language_perturbation_probability
+        if apply_perturbation:
+            task = self._generate_perturbed_task()
+
+        # Encode text with CLIP
+        observation["text_features"] = self._encode_text_clip(task, batch_size)
+
+        # Store lengths for model
+        observation["lengths"] = lengths
+
+        # When language is perturbed, targets are zero so perturbed samples don't contribute to progress loss
+        if self.dataset_meta is not None:
+            episodes_df = self.dataset_meta.episodes.to_pandas()
+
+            # Generate sparse targets
+            if self.sparse_temporal_proportions is not None:
+                if apply_perturbation:
+                    # Zero targets when language is perturbed
+                    sparse_targets = torch.zeros(batch_size, total_frames, dtype=torch.float32)
+                else:
+                    sparse_targets = self._compute_batch_targets(
+                        frame_indices, episode_indices, lengths, rewind_steps, episodes_df, "sparse"
+                    )
+                observation["sparse_targets"] = sparse_targets
+
+            # Generate dense targets (for dual mode)
+            if self.config.uses_dual_heads and self.dense_temporal_proportions is not None:
+                if apply_perturbation:
+                    # Zero targets when language is perturbed
+                    dense_targets = torch.zeros(batch_size, total_frames, dtype=torch.float32)
+                else:
+                    dense_targets = self._compute_batch_targets(
+                        frame_indices, episode_indices, lengths, rewind_steps, episodes_df, "dense"
+                    )
+                observation["dense_targets"] = dense_targets
+
+        new_transition[TransitionKey.OBSERVATION] = observation
+        return new_transition
+
+    def _compute_batch_targets(
+        self,
+        frame_indices: np.ndarray,
+        episode_indices: np.ndarray,
+        lengths: torch.Tensor,
+        rewind_steps: torch.Tensor,
+        episodes_df: pd.DataFrame | None,
+        annotation_type: str,
+    ) -> torch.Tensor:
+        """Compute stage+tau targets for a batch of samples."""
+        batch_size = len(frame_indices)
+        n_obs_steps = self.config.n_obs_steps
+        max_rewind_steps = self.config.max_rewind_steps
+        total_frames = 1 + n_obs_steps + max_rewind_steps
+        frame_gap = self.config.frame_gap
+
+        global_names, temporal_props = self._get_annotation_config(annotation_type)
+        targets = torch.zeros(batch_size, total_frames, dtype=torch.float32)
+
+        for b_idx in range(batch_size):
+            ep_idx = int(episode_indices[b_idx])
+            frame_idx = int(frame_indices[b_idx])
+
+            ep_start = self.dataset_meta.episodes[ep_idx]["dataset_from_index"]
+            ep_end = self.dataset_meta.episodes[ep_idx]["dataset_to_index"]
+            ep_length = ep_end - ep_start
+
+            subtask_names, subtask_start_frames, subtask_end_frames = self._load_episode_annotations(
+                ep_idx, episodes_df, annotation_type, global_names
+            )
+
+            # Compute observation frame indices
+            obs_indices, _ = compute_absolute_indices(
+                frame_idx, ep_start, ep_end, n_obs_steps, frame_gap=frame_gap
+            )
+            obs_indices = obs_indices.tolist()
+
+            # Compute targets for observation frames
+            for t_idx, abs_idx in enumerate(obs_indices):
+                rel_frame = abs_idx - ep_start
+                targets[b_idx, t_idx] = find_stage_and_tau(
+                    rel_frame,
+                    ep_length,
+                    subtask_names,
+                    subtask_start_frames,
+                    subtask_end_frames,
+                    global_names,
+                    temporal_props,
+                    return_combined=True,
+                )
+
+            # Compute targets for rewind frames (if any)
+            rewind_step = rewind_steps[b_idx].item()
+            if rewind_step > 0:
+                _, rewind_indices = apply_rewind_augmentation(
+                    frame_idx,
+                    ep_start,
+                    n_obs_steps,
+                    max_rewind_steps,
+                    frame_gap=frame_gap,
+                    rewind_step=rewind_step,
+                )
+
+                for r_idx, abs_idx in enumerate(rewind_indices[:rewind_step]):
+                    rel_frame = max(0, abs_idx - ep_start)
+                    targets[b_idx, n_obs_steps + 1 + r_idx] = find_stage_and_tau(
+                        rel_frame,
+                        ep_length,
+                        subtask_names,
+                        subtask_start_frames,
+                        subtask_end_frames,
+                        global_names,
+                        temporal_props,
+                        return_combined=True,
+                    )
+
+        return targets
+
+    @property
+    def training(self) -> bool:
+        return getattr(self, "_training_mode", True)
+
+    def train(self, mode: bool = True):
+        """Set training mode for augmentation decisions."""
+        self._training_mode = mode
+        return self
+
+    def eval(self):
+        """Set evaluation mode (disable augmentations)."""
+        return self.train(False)
+
+    @torch.no_grad()
+    def _encode_images_batch(self, images: np.ndarray) -> torch.Tensor:
+        """Encode a batch of images using CLIP.
+
+        Args:
+            images: Batched images with shape: (B, T, C, H, W)
+
+        Returns:
+            Encoded feature vectors with shape (B, T, 512)
+        """
+
+        batch_size, seq_length = images.shape[0], images.shape[1]
+        images = images.reshape(batch_size * seq_length, *images.shape[2:])
+
+        num_frames = images.shape[0]
+        images_list = []
+        for i in range(num_frames):
+            img = images[i]
+            if img.shape[0] in [1, 3]:  # Channel first (C, H, W)
+                img = img.transpose(1, 2, 0)
+
+            # Handle single channel
+            if img.shape[-1] == 1:
+                img = np.repeat(img, 3, axis=-1)
+
+            if img.dtype != np.uint8:
+                img = (img * 255).astype(np.uint8) if img.max() <= 1.0 else img.astype(np.uint8)
+
+            images_list.append(Image.fromarray(img))
+
+        all_embeddings = []
+        for i in range(0, num_frames, self.config.clip_batch_size):
+            batch_imgs = images_list[i : i + self.config.clip_batch_size]
+
+            inputs = self.clip_processor(images=batch_imgs, return_tensors="pt")
+            inputs = {k: v.to(self.device) for k, v in inputs.items()}
+
+            # Get image embeddings
+            embeddings = self.clip_model.get_image_features(**inputs).detach().cpu()
+
+            # Handle single frame case
+            if embeddings.dim() == 1:
+                embeddings = embeddings.unsqueeze(0)
+
+            all_embeddings.append(embeddings)
+
+        all_embeddings = torch.cat(all_embeddings)  # (B*T, 512)
+        all_embeddings = all_embeddings.reshape(batch_size, seq_length, -1)  # (B, T, 512)
+
+        return all_embeddings
+
+    @torch.no_grad()
+    def _encode_text_clip(self, text: str, batch_size: int) -> torch.Tensor:
+        """Encode text using CLIP text encoder (per SARM paper A.4).
+
+        Args:
+            text: Task description text to encode
+            batch_size: Batch size to replicate for
+
+        Returns:
+            Encoded text features with shape (B, 512)
+        """
+        inputs = self.clip_processor.tokenizer([text], return_tensors="pt", padding=True, truncation=True)
+        inputs = {k: v.to(self.device) for k, v in inputs.items()}
+
+        text_embedding = self.clip_model.get_text_features(**inputs).detach().cpu()
+        text_embedding = text_embedding.expand(batch_size, -1)
+
+        return text_embedding
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """Add encoded features to the observation features."""
+        features[PipelineFeatureType.OBSERVATION]["video_features"] = PolicyFeature(
+            type=FeatureType.VISUAL, shape=(self.config.num_frames, self.config.image_dim)
+        )
+        features[PipelineFeatureType.OBSERVATION]["text_features"] = PolicyFeature(
+            type=FeatureType.LANGUAGE, shape=(self.config.text_dim,)
+        )
+        features[PipelineFeatureType.OBSERVATION]["state_features"] = PolicyFeature(
+            type=FeatureType.STATE, shape=(self.config.num_frames, self.config.max_state_dim)
+        )
+        return features
+
+
+def make_sarm_pre_post_processors(
+    config: SARMConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+    dataset_meta=None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """Create pre-processor and post-processor pipelines for SARM."""
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=[
+                AddBatchDimensionProcessorStep(),
+                RenameObservationsProcessorStep(rename_map={}),
+                NormalizerProcessorStep(
+                    features={**config.input_features, **config.output_features},
+                    norm_map=config.normalization_mapping,
+                    stats=dataset_stats,
+                ),
+                SARMEncodingProcessorStep(
+                    config=config, dataset_meta=dataset_meta, dataset_stats=dataset_stats
+                ),
+                DeviceProcessorStep(device=config.device),
+            ],
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=[DeviceProcessorStep(device="cpu")],
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/sarm/sarm_utils.py b/lerobot/src/lerobot/policies/sarm/sarm_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..5b6955d3864f227d51c5f8a39c771107b3c5e5dd
--- /dev/null
+++ b/lerobot/src/lerobot/policies/sarm/sarm_utils.py
@@ -0,0 +1,295 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import random
+
+import numpy as np
+import torch
+import torch.nn.functional as F  # noqa: N812
+
+
+def find_stage_and_tau(
+    current_frame: int,
+    episode_length: int,
+    subtask_names: list | None,
+    subtask_start_frames: list | None,
+    subtask_end_frames: list | None,
+    global_subtask_names: list,
+    temporal_proportions: dict,
+    return_combined: bool = False,
+) -> tuple[int, float] | float:
+    """Find stage and within-stage progress (tau) for a frame.
+
+    Args:
+        current_frame: Frame index relative to episode start
+        episode_length: Total frames in episode
+        subtask_names: Subtask names for this episode (None for single_stage)
+        subtask_start_frames: Subtask start frames
+        subtask_end_frames: Subtask end frames
+        global_subtask_names: Global list of all subtask names
+        temporal_proportions: Dict of temporal proportions
+        return_combined: If True, return stage+tau as float; else (stage_idx, tau) tuple
+
+    Returns:
+        Float (stage.tau) if return_combined, else (stage_idx, tau) tuple
+    """
+    stage_idx, tau = 0, 0.0
+    num_stages = len(global_subtask_names)
+
+    # Single-stage mode: linear progress from 0 to 1
+    if num_stages == 1:
+        tau = min(1.0, max(0.0, current_frame / max(episode_length - 1, 1)))
+    elif subtask_names is None:
+        pass  # stage_idx=0, tau=0.0
+    elif current_frame < subtask_start_frames[0]:
+        pass  # Before first subtask: stage_idx=0, tau=0.0
+    elif current_frame > subtask_end_frames[-1]:
+        stage_idx, tau = num_stages - 1, 0.999  # After last subtask
+    else:
+        # Find which subtask this frame belongs to
+        found = False
+        for name, start, end in zip(subtask_names, subtask_start_frames, subtask_end_frames, strict=True):
+            if start <= current_frame <= end:
+                stage_idx = global_subtask_names.index(name) if name in global_subtask_names else 0
+                tau = compute_tau(current_frame, start, end)
+                found = True
+                break
+        # Frame between subtasks - use previous subtask's end state
+        if not found:
+            for j in range(len(subtask_names) - 1):
+                if subtask_end_frames[j] < current_frame < subtask_start_frames[j + 1]:
+                    name = subtask_names[j]
+                    stage_idx = global_subtask_names.index(name) if name in global_subtask_names else j
+                    tau = 1.0
+                    break
+
+    if return_combined:
+        # Clamp to avoid overflow at end
+        if stage_idx >= num_stages - 1 and tau >= 1.0:
+            return num_stages - 1 + 0.999
+        return stage_idx + tau
+    return stage_idx, tau
+
+
+def compute_absolute_indices(
+    frame_idx: int,
+    ep_start: int,
+    ep_end: int,
+    n_obs_steps: int,
+    frame_gap: int = 30,
+) -> tuple[torch.Tensor, torch.Tensor]:
+    """Compute absolute frame indices with clamping for bidirectional observation sequence.
+
+    Bidirectional sampling centered on target frame:
+    - Before: [-frame_gap * half_steps, ..., -frame_gap] (half_steps frames)
+    - Current: [0] (1 frame)
+    - After: [frame_gap, ..., frame_gap * half_steps] (half_steps frames)
+    - Total: n_obs_steps + 1 frames
+
+    Out-of-bounds frames are clamped (duplicated from boundary).
+
+    Args:
+        frame_idx: Target frame index (center frame of sequence)
+        ep_start: Episode start index
+        ep_end: Episode end index (exclusive)
+        n_obs_steps: Number of observation steps (must be even for symmetric sampling)
+        frame_gap: Gap between observation frames
+
+    Returns:
+        Tuple of (indices, out_of_bounds_flags)
+    """
+    half_steps = n_obs_steps // 2
+
+    # Bidirectional deltas: past + current + future
+    past_deltas = [-frame_gap * i for i in range(half_steps, 0, -1)]
+    future_deltas = [frame_gap * i for i in range(1, half_steps + 1)]
+    delta_indices = past_deltas + [0] + future_deltas
+
+    frames = []
+    out_of_bounds = []
+
+    for delta in delta_indices:
+        target_idx = frame_idx + delta
+        # Clamp to episode bounds (duplicate boundary frames for out-of-bounds)
+        clamped_idx = max(ep_start, min(ep_end - 1, target_idx))
+        frames.append(clamped_idx)
+        # Flag as out-of-bounds if clamping occurred
+        out_of_bounds.append(1 if target_idx != clamped_idx else 0)
+
+    return torch.tensor(frames), torch.tensor(out_of_bounds)
+
+
+def apply_rewind_augmentation(
+    frame_idx: int,
+    ep_start: int,
+    n_obs_steps: int,
+    max_rewind_steps: int,
+    frame_gap: int = 30,
+    rewind_step: int | None = None,
+) -> tuple[int, list[int]]:
+    """
+    Generate rewind frame indices for temporal augmentation.
+
+    Rewind simulates going backwards through previously seen frames,
+    starting from before the earliest observation frame (for bidirectional sampling).
+    Appends reversed frames after the observation sequence.
+
+    Args:
+        frame_idx: Target frame index (center of bidirectional observation window)
+        ep_start: Episode start index
+        n_obs_steps: Number of observation steps
+        max_rewind_steps: Maximum rewind steps
+        frame_gap: Gap between frames
+        rewind_step: If provided, use this exact rewind step (for deterministic behavior).
+                     If None, sample randomly.
+
+    Returns:
+        Tuple of (rewind_step, rewind_indices)
+    """
+    # For bidirectional sampling, earliest obs frame is at frame_idx - half_steps * frame_gap
+    half_steps = n_obs_steps // 2
+    earliest_obs_frame = frame_idx - half_steps * frame_gap
+
+    # Required history: frames before earliest observation frame
+    if earliest_obs_frame <= ep_start:
+        return 0, []  # No history before observation window
+
+    # Max valid rewind steps based on available history before earliest obs frame
+    available_history = earliest_obs_frame - ep_start
+    max_valid_step = available_history // frame_gap
+    max_rewind = min(max_rewind_steps, max(0, max_valid_step))
+
+    if max_rewind <= 0:
+        return 0, []
+
+    # Sample rewind steps if not provided
+    rewind_step = random.randint(1, max_rewind) if rewind_step is None else min(rewind_step, max_rewind)
+
+    if rewind_step == 0:
+        return 0, []
+
+    # Generate rewind indices going backwards from earliest obs frame
+    # rewind_indices[0] is closest to obs window, rewind_indices[-1] is furthest back
+    rewind_indices = []
+    for i in range(1, rewind_step + 1):
+        idx = earliest_obs_frame - i * frame_gap
+        idx = max(ep_start, idx)  # Clamp to episode start
+        rewind_indices.append(idx)
+
+    return rewind_step, rewind_indices
+
+
+def compute_tau(current_frame: int | float, subtask_start: int | float, subtask_end: int | float) -> float:
+    """Compute τ_t = (t - s_k) / (e_k - s_k) ∈ [0, 1]. Returns 1.0 for zero-duration subtasks."""
+    duration = subtask_end - subtask_start
+    if duration <= 0:
+        return 1.0
+    return float(np.clip((current_frame - subtask_start) / duration, 0.0, 1.0))
+
+
+def pad_state_to_max_dim(state: torch.Tensor, max_state_dim: int) -> torch.Tensor:
+    """Pad the state tensor's last dimension to max_state_dim with zeros."""
+    current_dim = state.shape[-1]
+    if current_dim >= max_state_dim:
+        return state[..., :max_state_dim]  # Truncate if larger
+
+    # Pad with zeros on the right
+    padding = (0, max_state_dim - current_dim)  # (left, right) for last dim
+    return F.pad(state, padding, mode="constant", value=0)
+
+
+def temporal_proportions_to_breakpoints(
+    temporal_proportions: dict[str, float] | list[float] | None,
+    subtask_names: list[str] | None = None,
+) -> list[float] | None:
+    """Convert temporal proportions to cumulative breakpoints for normalization."""
+    if temporal_proportions is None:
+        return None
+
+    if isinstance(temporal_proportions, dict):
+        if subtask_names is not None:
+            proportions = [temporal_proportions.get(name, 0.0) for name in subtask_names]
+        else:
+            proportions = list(temporal_proportions.values())
+    else:
+        proportions = list(temporal_proportions)
+
+    total = sum(proportions)
+    if total > 0 and abs(total - 1.0) > 1e-6:
+        proportions = [p / total for p in proportions]
+
+    breakpoints = [0.0]
+    cumsum = 0.0
+    for prop in proportions:
+        cumsum += prop
+        breakpoints.append(cumsum)
+    breakpoints[-1] = 1.0
+
+    return breakpoints
+
+
+def normalize_stage_tau(
+    x: float | torch.Tensor,
+    num_stages: int | None = None,
+    breakpoints: list[float] | None = None,
+    temporal_proportions: dict[str, float] | list[float] | None = None,
+    subtask_names: list[str] | None = None,
+) -> float | torch.Tensor:
+    """
+    Normalize stage+tau reward to [0, 1] with custom breakpoints.
+
+    Maps stage index + within-stage tau to normalized progress [0, 1].
+    The breakpoints are designed to give appropriate weight to each stage
+    based on their importance in the task (using temporal proportions).
+
+    Priority: breakpoints > temporal_proportions > linear fallback
+
+    Args:
+        x: Raw reward value (stage index + tau) where stage ∈ [0, num_stages-1] and tau ∈ [0, 1)
+        num_stages: Number of stages (required if breakpoints/proportions not provided)
+        breakpoints: Optional custom breakpoints list of length num_stages + 1.
+        temporal_proportions: Optional temporal proportions dict/list to compute breakpoints.
+        subtask_names: Optional ordered list of subtask names (for dict proportions)
+
+    Returns:
+        Normalized progress value ∈ [0, 1]
+    """
+    if breakpoints is not None:
+        num_stages = len(breakpoints) - 1
+    elif temporal_proportions is not None:
+        breakpoints = temporal_proportions_to_breakpoints(temporal_proportions, subtask_names)
+        num_stages = len(breakpoints) - 1
+    elif num_stages is not None:
+        breakpoints = [i / num_stages for i in range(num_stages + 1)]
+    else:
+        raise ValueError("Either num_stages, breakpoints, or temporal_proportions must be provided")
+
+    if isinstance(x, torch.Tensor):
+        result = torch.zeros_like(x)
+        for i in range(num_stages):
+            mask = (x >= i) & (x < i + 1)
+            tau_in_stage = x - i
+            result[mask] = breakpoints[i] + tau_in_stage[mask] * (breakpoints[i + 1] - breakpoints[i])
+        result[x >= num_stages] = 1.0
+        return result.clamp(0.0, 1.0)
+    else:
+        if x < 0:
+            return 0.0
+        if x >= num_stages:
+            return 1.0
+        stage = int(x)
+        tau = x - stage
+        return breakpoints[stage] + tau * (breakpoints[stage + 1] - breakpoints[stage])
diff --git a/lerobot/src/lerobot/policies/smolvla/README.md b/lerobot/src/lerobot/policies/smolvla/README.md
new file mode 120000
index 0000000000000000000000000000000000000000..f8de40269b8c6c93dc9d7c048102932c5e26cfa3
--- /dev/null
+++ b/lerobot/src/lerobot/policies/smolvla/README.md
@@ -0,0 +1 @@
+../../../../docs/source/policy_smolvla_README.md
\ No newline at end of file
diff --git a/lerobot/src/lerobot/policies/smolvla/configuration_smolvla.py b/lerobot/src/lerobot/policies/smolvla/configuration_smolvla.py
new file mode 100644
index 0000000000000000000000000000000000000000..b861b856b7f7a6536bca2c3eef8d2cb474c2d1bb
--- /dev/null
+++ b/lerobot/src/lerobot/policies/smolvla/configuration_smolvla.py
@@ -0,0 +1,162 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.optim.optimizers import AdamWConfig
+from lerobot.optim.schedulers import (
+    CosineDecayWithWarmupSchedulerConfig,
+)
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+from lerobot.utils.constants import OBS_IMAGES
+
+
+@PreTrainedConfig.register_subclass("smolvla")
+@dataclass
+class SmolVLAConfig(PreTrainedConfig):
+    # Input / output structure.
+    n_obs_steps: int = 1
+    chunk_size: int = 50
+    n_action_steps: int = 50
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.MEAN_STD,
+            "ACTION": NormalizationMode.MEAN_STD,
+        }
+    )
+
+    # Shorter state and action vectors will be padded
+    max_state_dim: int = 32
+    max_action_dim: int = 32
+
+    # Image preprocessing
+    resize_imgs_with_padding: tuple[int, int] = (512, 512)
+
+    # Add empty images. Used by smolvla_aloha_sim which adds the empty
+    # left and right wrist cameras in addition to the top camera.
+    empty_cameras: int = 0
+
+    # Converts the joint and gripper values from the standard Aloha space to
+    # the space used by the pi internal runtime which was used to train the base model.
+    adapt_to_pi_aloha: bool = False
+
+    # Converts joint dimensions to deltas with respect to the current state before passing to the model.
+    # Gripper dimensions will remain in absolute values.
+    use_delta_joint_actions_aloha: bool = False
+
+    # Tokenizer
+    tokenizer_max_length: int = 48
+
+    # Decoding
+    num_steps: int = 10
+
+    # Attention utils
+    use_cache: bool = True
+
+    # Finetuning settings
+    freeze_vision_encoder: bool = True
+    train_expert_only: bool = True
+    train_state_proj: bool = True
+
+    # Training presets
+    optimizer_lr: float = 1e-4
+    optimizer_betas: tuple[float, float] = (0.9, 0.95)
+    optimizer_eps: float = 1e-8
+    optimizer_weight_decay: float = 1e-10
+    optimizer_grad_clip_norm: float = 10
+
+    scheduler_warmup_steps: int = 1_000
+    scheduler_decay_steps: int = 30_000
+    scheduler_decay_lr: float = 2.5e-6
+
+    vlm_model_name: str = "HuggingFaceTB/SmolVLM2-500M-Video-Instruct"  # Select the VLM backbone.
+    load_vlm_weights: bool = False  # Set to False in case of training the expert from scratch. True when init from pretrained SmolVLA weights
+
+    add_image_special_tokens: bool = False  # Whether to use special image tokens around image features.
+
+    attention_mode: str = "cross_attn"
+
+    prefix_length: int = -1
+
+    pad_language_to: str = "longest"  # "max_length"
+
+    num_expert_layers: int = -1  # Less or equal to 0 is the default where the action expert has the same number of layers of VLM. Otherwise the expert have less layers.
+    num_vlm_layers: int = 16  # Number of layers used in the VLM (first num_vlm_layers layers)
+    self_attn_every_n_layers: int = 2  # Interleave SA layers each self_attn_every_n_layers
+    expert_width_multiplier: float = 0.75  # The action expert hidden size (wrt to the VLM)
+
+    min_period: float = 4e-3  # sensitivity range for the timestep used in sine-cosine positional encoding
+    max_period: float = 4.0
+
+    # Real-Time Chunking (RTC) configuration
+    rtc_config: RTCConfig | None = None
+
+    compile_model: bool = False  # Whether to use torch.compile for model optimization
+    compile_mode: str = "max-autotune"  # Torch compile mode
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        """Input validation (not exhaustive)."""
+        if self.n_action_steps > self.chunk_size:
+            raise ValueError(
+                f"The chunk size is the upper bound for the number of action steps per model invocation. Got "
+                f"{self.n_action_steps} for `n_action_steps` and {self.chunk_size} for `chunk_size`."
+            )
+        if self.use_delta_joint_actions_aloha:
+            raise NotImplementedError(
+                "`use_delta_joint_actions_aloha` is used by smolvla for aloha real models. It is not ported yet in LeRobot."
+            )
+
+    def validate_features(self) -> None:
+        for i in range(self.empty_cameras):
+            key = f"{OBS_IMAGES}.empty_camera_{i}"
+            empty_camera = PolicyFeature(
+                type=FeatureType.VISUAL,
+                shape=(3, 480, 640),
+            )
+            self.input_features[key] = empty_camera
+
+    def get_optimizer_preset(self) -> AdamWConfig:
+        return AdamWConfig(
+            lr=self.optimizer_lr,
+            betas=self.optimizer_betas,
+            eps=self.optimizer_eps,
+            weight_decay=self.optimizer_weight_decay,
+            grad_clip_norm=self.optimizer_grad_clip_norm,
+        )
+
+    def get_scheduler_preset(self):
+        return CosineDecayWithWarmupSchedulerConfig(
+            peak_lr=self.optimizer_lr,
+            decay_lr=self.scheduler_decay_lr,
+            num_warmup_steps=self.scheduler_warmup_steps,
+            num_decay_steps=self.scheduler_decay_steps,
+        )
+
+    @property
+    def observation_delta_indices(self) -> list:
+        return [0]
+
+    @property
+    def action_delta_indices(self) -> list:
+        return list(range(self.chunk_size))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/smolvla/modeling_smolvla.py b/lerobot/src/lerobot/policies/smolvla/modeling_smolvla.py
new file mode 100644
index 0000000000000000000000000000000000000000..7110ba7d2019d3a309026ef314ee924da765017d
--- /dev/null
+++ b/lerobot/src/lerobot/policies/smolvla/modeling_smolvla.py
@@ -0,0 +1,905 @@
+#!/usr/bin/env python
+
+# Copyright 2025 HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+SmolVLA:
+
+[Paper](https://huggingface.co/papers/2506.01844)
+
+Designed by Hugging Face.
+
+Install smolvla extra dependencies:
+```bash
+pip install -e ".[smolvla]"
+```
+
+Example of finetuning the smolvla pretrained model (`smolvla_base`):
+```bash
+lerobot-train \
+--policy.path=lerobot/smolvla_base \
+--dataset.repo_id=<USER>/svla_so100_task1_v3 \
+--batch_size=64 \
+--steps=200000
+```
+
+Example of finetuning a smolVLA. SmolVLA is composed of a pretrained VLM,
+and an action expert.
+```bash
+lerobot-train \
+--policy.type=smolvla \
+--dataset.repo_id=<USER>/svla_so100_task1_v3 \
+--batch_size=64 \
+--steps=200000
+```
+
+Example of using the smolvla pretrained model outside LeRobot training framework:
+```python
+policy = SmolVLAPolicy.from_pretrained("lerobot/smolvla_base")
+```
+
+"""
+
+import math
+from collections import deque
+from typing import TypedDict, Unpack
+
+import torch
+import torch.nn.functional as F  # noqa: N812
+from torch import Tensor, nn
+
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.rtc.modeling_rtc import RTCProcessor
+from lerobot.policies.smolvla.configuration_smolvla import SmolVLAConfig
+from lerobot.policies.smolvla.smolvlm_with_expert import SmolVLMWithExpertModel
+from lerobot.policies.utils import (
+    populate_queues,
+)
+from lerobot.utils.constants import ACTION, OBS_LANGUAGE_ATTENTION_MASK, OBS_LANGUAGE_TOKENS, OBS_STATE
+from lerobot.utils.device_utils import get_safe_dtype
+
+
+class ActionSelectKwargs(TypedDict, total=False):
+    inference_delay: int | None
+    prev_chunk_left_over: Tensor | None
+    execution_horizon: int | None
+
+
+def create_sinusoidal_pos_embedding(
+    time: torch.tensor, dimension: int, min_period: float, max_period: float, device="cpu"
+) -> Tensor:
+    """Computes sine-cosine positional embedding vectors for scalar positions."""
+    if dimension % 2 != 0:
+        raise ValueError(f"dimension ({dimension}) must be divisible by 2")
+
+    if time.ndim != 1:
+        raise ValueError("The time tensor is expected to be of shape `(batch_size, )`.")
+
+    dtype = get_safe_dtype(torch.float64, device.type)
+    fraction = torch.linspace(0.0, 1.0, dimension // 2, dtype=dtype, device=device)
+    period = min_period * (max_period / min_period) ** fraction
+
+    # Compute the outer product
+    scaling_factor = 1.0 / period * 2 * math.pi
+    sin_input = scaling_factor[None, :] * time[:, None]
+    pos_emb = torch.cat([torch.sin(sin_input), torch.cos(sin_input)], dim=1)
+    return pos_emb
+
+
+def make_att_2d_masks(pad_masks, att_masks):
+    """Copied from big_vision.
+
+    Tokens can attend to valid inputs tokens which have a cumulative mask_ar
+    smaller or equal to theirs. This way `mask_ar` int[B, N] can be used to
+    setup several types of attention, for example:
+
+      [[1 1 1 1 1 1]]: pure causal attention.
+
+      [[0 0 0 1 1 1]]: prefix-lm attention. The first 3 tokens can attend between
+          themselves and the last 3 tokens have a causal attention. The first
+          entry could also be a 1 without changing behaviour.
+
+      [[1 0 1 0 1 0 0 1 0 0]]: causal attention between 4 blocks. Tokens of a
+          block can attend all previous blocks and all tokens on the same block.
+
+    Args:
+      input_mask: bool[B, N] true if its part of the input, false if padding.
+      mask_ar: int32[B, N] mask that's 1 where previous tokens cannot depend on
+        it and 0 where it shares the same attention mask as the previous token.
+    """
+    if att_masks.ndim != 2:
+        raise ValueError(att_masks.ndim)
+    if pad_masks.ndim != 2:
+        raise ValueError(pad_masks.ndim)
+
+    cumsum = torch.cumsum(att_masks, dim=1)
+    att_2d_masks = cumsum[:, None, :] <= cumsum[:, :, None]
+    pad_2d_masks = pad_masks[:, None, :] * pad_masks[:, :, None]
+    att_2d_masks = att_2d_masks & pad_2d_masks
+    return att_2d_masks
+
+
+def resize_with_pad(img, width, height, pad_value=-1):
+    # assume no-op when width height fits already
+    if img.ndim != 4:
+        raise ValueError(f"(b,c,h,w) expected, but {img.shape}")
+
+    cur_height, cur_width = img.shape[2:]
+
+    ratio = max(cur_width / width, cur_height / height)
+    resized_height = int(cur_height / ratio)
+    resized_width = int(cur_width / ratio)
+    resized_img = F.interpolate(
+        img, size=(resized_height, resized_width), mode="bilinear", align_corners=False
+    )
+
+    pad_height = max(0, int(height - resized_height))
+    pad_width = max(0, int(width - resized_width))
+
+    # pad on left and top of image
+    padded_img = F.pad(resized_img, (pad_width, 0, pad_height, 0), value=pad_value)
+    return padded_img
+
+
+def pad_vector(vector, new_dim):
+    """Can be (batch_size x sequence_length x features_dimension)
+    or (batch_size x features_dimension)
+    """
+    if vector.shape[-1] == new_dim:
+        return vector
+    shape = list(vector.shape)
+    current_dim = shape[-1]
+    shape[-1] = new_dim
+    new_vector = torch.zeros(*shape, dtype=vector.dtype, device=vector.device)
+    new_vector[..., :current_dim] = vector
+    return new_vector
+
+
+def normalize(x, min_val, max_val):
+    return (x - min_val) / (max_val - min_val)
+
+
+def unnormalize(x, min_val, max_val):
+    return x * (max_val - min_val) + min_val
+
+
+def safe_arcsin(value):
+    # This ensures that the input stays within
+    # [−1,1] to avoid invalid values for arcsin
+    return torch.arcsin(torch.clamp(value, -1.0, 1.0))
+
+
+def aloha_gripper_to_angular(value):
+    # Aloha transforms the gripper positions into a linear space. The following code
+    # reverses this transformation to be consistent with smolvla which is pretrained in
+    # angular space.
+    #
+    # These values are coming from the Aloha code:
+    # PUPPET_GRIPPER_POSITION_OPEN, PUPPET_GRIPPER_POSITION_CLOSED
+    value = unnormalize(value, min_val=0.01844, max_val=0.05800)
+
+    # This is the inverse of the angular to linear transformation inside the Interbotix code.
+    def linear_to_radian(linear_position, arm_length, horn_radius):
+        value = (horn_radius**2 + linear_position**2 - arm_length**2) / (2 * horn_radius * linear_position)
+        return safe_arcsin(value)
+
+    # The constants are taken from the Interbotix code.
+    value = linear_to_radian(value, arm_length=0.036, horn_radius=0.022)
+
+    # Normalize to [0, 1].
+    # The values 0.4 and 1.5 were measured on an actual Trossen robot.
+    return normalize(value, min_val=0.4, max_val=1.5)
+
+
+def aloha_gripper_from_angular(value):
+    # Convert from the gripper position used by smolvla to the gripper position that is used by Aloha.
+    # Note that the units are still angular but the range is different.
+
+    # The values 0.4 and 1.5 were measured on an actual Trossen robot.
+    value = unnormalize(value, min_val=0.4, max_val=1.5)
+
+    # These values are coming from the Aloha code:
+    # PUPPET_GRIPPER_JOINT_OPEN, PUPPET_GRIPPER_JOINT_CLOSE
+    return normalize(value, min_val=-0.6213, max_val=1.4910)
+
+
+def aloha_gripper_from_angular_inv(value):
+    # Directly inverts the gripper_from_angular function.
+    value = unnormalize(value, min_val=-0.6213, max_val=1.4910)
+    return normalize(value, min_val=0.4, max_val=1.5)
+
+
+class SmolVLAPolicy(PreTrainedPolicy):
+    """Wrapper class around VLAFlowMatching model to train and run inference within LeRobot."""
+
+    config_class = SmolVLAConfig
+    name = "smolvla"
+
+    def __init__(
+        self,
+        config: SmolVLAConfig,
+        **kwargs,
+    ):
+        """
+        Args:
+            config: Policy configuration class instance or None, in which case the default instantiation of
+                    the configuration class is used.
+        """
+
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+        self.init_rtc_processor()
+        self.model = VLAFlowMatching(config, rtc_processor=self.rtc_processor)
+        self.reset()
+
+    def reset(self):
+        """This should be called whenever the environment is reset."""
+        self._queues = {
+            ACTION: deque(maxlen=self.config.n_action_steps),
+        }
+
+    def init_rtc_processor(self):
+        """Initialize RTC processor if RTC is enabled in config."""
+        self.rtc_processor = None
+
+        # Lets create processor if the config provided
+        # If RTC is not enabled - we still can track the denoising data
+        if self.config.rtc_config is not None:
+            self.rtc_processor = RTCProcessor(self.config.rtc_config)
+
+            # In case of calling init_rtc_processor after the model is created
+            # We need to set the rtc_processor to the model
+            # During the normal initialization process the model is not created yet
+            model_value = getattr(self, "model", None)
+            if model_value is not None:
+                model_value.rtc_processor = self.rtc_processor
+
+    def get_optim_params(self) -> dict:
+        return self.parameters()
+
+    def _get_action_chunk(
+        self, batch: dict[str, Tensor], noise: Tensor | None = None, **kwargs: Unpack[ActionSelectKwargs]
+    ) -> Tensor:
+        # TODO: Check if this for loop is needed.
+        # Context: In fact, self.queues contains only ACTION field, and in inference, we don't have action in the batch
+        # In the case of offline inference, we have the action in the batch
+        # that why without the k != ACTION check, it will raise an error because we are trying to stack
+        # on an empty container.
+        for k in batch:
+            if k in self._queues and k != ACTION:
+                batch[k] = torch.stack(list(self._queues[k]), dim=1)
+
+        images, img_masks = self.prepare_images(batch)
+        state = self.prepare_state(batch)
+        lang_tokens = batch[f"{OBS_LANGUAGE_TOKENS}"]
+        lang_masks = batch[f"{OBS_LANGUAGE_ATTENTION_MASK}"]
+
+        actions = self.model.sample_actions(
+            images, img_masks, lang_tokens, lang_masks, state, noise=noise, **kwargs
+        )
+
+        # Unpad actions
+        original_action_dim = self.config.action_feature.shape[0]
+        actions = actions[:, :, :original_action_dim]
+
+        if self.config.adapt_to_pi_aloha:
+            actions = self._pi_aloha_encode_actions(actions)
+
+        return actions
+
+    def _prepare_batch(self, batch: dict[str, Tensor]) -> dict[str, Tensor]:
+        if self.config.adapt_to_pi_aloha:
+            batch[OBS_STATE] = self._pi_aloha_decode_state(batch[OBS_STATE])
+
+        return batch
+
+    @torch.no_grad()
+    def predict_action_chunk(
+        self, batch: dict[str, Tensor], noise: Tensor | None = None, **kwargs: Unpack[ActionSelectKwargs]
+    ) -> Tensor:
+        self.eval()
+
+        batch = self._prepare_batch(batch)
+        self._queues = populate_queues(self._queues, batch, exclude_keys=[ACTION])
+
+        actions = self._get_action_chunk(batch, noise, **kwargs)
+        return actions
+
+    @torch.no_grad()
+    def select_action(
+        self, batch: dict[str, Tensor], noise: Tensor | None = None, **kwargs: Unpack[ActionSelectKwargs]
+    ) -> Tensor:
+        """Select a single action given environment observations.
+
+        This method wraps `select_actions` in order to return one action at a time for execution in the
+        environment. It works by managing the actions in a queue and only calling `select_actions` when the
+        queue is empty.
+        """
+
+        assert not self._rtc_enabled(), (
+            "RTC is not supported for select_action, use it with predict_action_chunk"
+        )
+
+        self.eval()
+        batch = self._prepare_batch(batch)
+        self._queues = populate_queues(self._queues, batch, exclude_keys=[ACTION])
+
+        if self._check_get_actions_condition():
+            actions = self._get_action_chunk(batch, noise)
+
+            # `self.predict_action_chunk` returns a (batch_size, n_action_steps, action_dim) tensor, but the queue
+            # effectively has shape (n_action_steps, batch_size, *), hence the transpose.
+            self._queues[ACTION].extend(actions.transpose(0, 1)[: self.config.n_action_steps])
+
+        return self._queues[ACTION].popleft()
+
+    def _check_get_actions_condition(self) -> bool:
+        return len(self._queues[ACTION]) == 0
+
+    def _rtc_enabled(self) -> bool:
+        return self.config.rtc_config is not None and self.config.rtc_config.enabled
+
+    def forward(
+        self, batch: dict[str, Tensor], noise=None, time=None, reduction: str = "mean"
+    ) -> dict[str, Tensor]:
+        """Do a full training forward pass to compute the loss.
+
+        Args:
+            batch: Training batch containing observations and actions.
+            noise: Optional noise tensor for flow matching.
+            time: Optional time tensor for flow matching.
+            reduction: How to reduce the loss. Options:
+                - "mean": Return scalar mean loss (default, backward compatible)
+                - "none": Return per-sample losses of shape (batch_size,) for RA-BC weighting
+        """
+        if self.config.adapt_to_pi_aloha:
+            batch[OBS_STATE] = self._pi_aloha_decode_state(batch[OBS_STATE])
+            batch[ACTION] = self._pi_aloha_encode_actions_inv(batch[ACTION])
+
+        images, img_masks = self.prepare_images(batch)
+        state = self.prepare_state(batch)
+        lang_tokens = batch[f"{OBS_LANGUAGE_TOKENS}"]
+        lang_masks = batch[f"{OBS_LANGUAGE_ATTENTION_MASK}"]
+        actions = self.prepare_action(batch)
+        actions_is_pad = batch.get("action_is_pad")
+        loss_dict = {}
+        losses = self.model.forward(images, img_masks, lang_tokens, lang_masks, state, actions, noise, time)
+        original_action_dim = self.config.action_feature.shape[0]
+        losses = losses[:, :, :original_action_dim]
+        loss_dict["losses_after_forward"] = losses.clone().mean().item()
+
+        if actions_is_pad is not None:
+            in_episode_bound = ~actions_is_pad
+            losses = losses * in_episode_bound.unsqueeze(-1)
+            loss_dict["losses_after_in_ep_bound"] = losses.clone().mean().item()
+
+        # Remove padding
+        losses = losses[:, :, : self.config.max_action_dim]
+        loss_dict["losses_after_rm_padding"] = losses.clone().mean().item()
+
+        if reduction == "none":
+            # Return per-sample losses (B,) by averaging over time and action dims
+            per_sample_loss = losses.mean(dim=(1, 2))
+            loss_dict["loss"] = per_sample_loss.mean().item()
+            return per_sample_loss, loss_dict
+        else:
+            # Default: return scalar mean loss
+            loss = losses.mean()
+            loss_dict["loss"] = loss.item()
+            return loss, loss_dict
+
+    def prepare_images(self, batch):
+        """Apply SmolVLA preprocessing to the images, like resizing to 224x224 and padding to keep aspect ratio, and
+        convert pixel range from [0.0, 1.0] to [-1.0, 1.0] as requested by SigLIP.
+        """
+        images = []
+        img_masks = []
+        present_img_keys = [key for key in self.config.image_features if key in batch]
+        missing_img_keys = [key for key in self.config.image_features if key not in batch]
+
+        if len(present_img_keys) == 0:
+            raise ValueError(
+                f"All image features are missing from the batch. At least one expected. (batch: {batch.keys()}) (image_features:{self.config.image_features})"
+            )
+        # Preprocess image features present in the batch
+        for key in present_img_keys:
+            img = batch[key][:, -1, :, :, :] if batch[key].ndim == 5 else batch[key]
+            if self.config.resize_imgs_with_padding is not None:
+                img = resize_with_pad(img, *self.config.resize_imgs_with_padding, pad_value=0)
+
+            # Normalize from range [0,1] to [-1,1] as expacted by siglip
+            img = img * 2.0 - 1.0
+
+            bsize = img.shape[0]
+            device = img.device
+            if f"{key}_padding_mask" in batch:
+                mask = batch[f"{key}_padding_mask"].bool()
+            else:
+                mask = torch.ones(bsize, dtype=torch.bool, device=device)
+            images.append(img)
+            img_masks.append(mask)
+
+        # Create image features not present in the batch
+        # as fully 0 padded images.
+        for num_empty_cameras in range(len(missing_img_keys)):
+            if num_empty_cameras >= self.config.empty_cameras:
+                break
+            img = torch.ones_like(img) * -1
+            mask = torch.zeros_like(mask)
+            images.append(img)
+            img_masks.append(mask)
+        return images, img_masks
+
+    def _pi_aloha_decode_state(self, state):
+        # Flip the joints.
+        for motor_idx in [1, 2, 8, 9]:
+            state[:, motor_idx] *= -1
+        # Reverse the gripper transformation that is being applied by the Aloha runtime.
+        for motor_idx in [6, 13]:
+            state[:, motor_idx] = aloha_gripper_to_angular(state[:, motor_idx])
+        return state
+
+    def _pi_aloha_encode_actions(self, actions):
+        # Flip the joints.
+        for motor_idx in [1, 2, 8, 9]:
+            actions[:, :, motor_idx] *= -1
+        # Reverse the gripper transformation that is being applied by the Aloha runtime.
+        for motor_idx in [6, 13]:
+            actions[:, :, motor_idx] = aloha_gripper_from_angular(actions[:, :, motor_idx])
+        return actions
+
+    def _pi_aloha_encode_actions_inv(self, actions):
+        # Flip the joints again.
+        for motor_idx in [1, 2, 8, 9]:
+            actions[:, :, motor_idx] *= -1
+        # Reverse the gripper transformation that is being applied by the Aloha runtime.
+        for motor_idx in [6, 13]:
+            actions[:, :, motor_idx] = aloha_gripper_from_angular_inv(actions[:, :, motor_idx])
+        return actions
+
+    def prepare_state(self, batch):
+        """Pad state"""
+        state = batch[OBS_STATE][:, -1, :] if batch[OBS_STATE].ndim > 2 else batch[OBS_STATE]
+        state = pad_vector(state, self.config.max_state_dim)
+        return state
+
+    def prepare_action(self, batch):
+        """Pad action"""
+        actions = pad_vector(batch[ACTION], self.config.max_action_dim)
+        return actions
+
+    def _get_default_peft_targets(self) -> dict[str, any]:
+        """Return default PEFT target modules for SmolVLA fine-tuning."""
+        common_projections = (
+            "state_proj|action_in_proj|action_out_proj|action_time_mlp_in|action_time_mlp_out"
+        )
+        target_modules = rf"(model\.vlm_with_expert\.lm_expert\..*\.(q|v)_proj|model\.({common_projections}))"
+        return {
+            "target_modules": target_modules,
+            "modules_to_save": [],
+        }
+
+    def _validate_peft_config(self, peft_config) -> None:
+        """Validate PEFT configuration for SmolVLA."""
+        super()._validate_peft_config(peft_config)
+        if not self.config.load_vlm_weights:
+            import logging
+
+            logging.warning(
+                "Training SmolVLA from scratch using PEFT. This is unlikely to yield good results. "
+                "Set `load_vlm_weights=True` to fine-tune the existing policy."
+            )
+
+
+def pad_tensor(tensor, max_len, pad_value=0):
+    """
+    Efficiently pads a tensor along sequence dimension to match max_len.
+
+    Args:
+        tensor (torch.Tensor): Shape (B, L, ...) or (B, L).
+        max_len (int): Fixed sequence length.
+        pad_value (int/float): Value for padding.
+
+    Returns:
+        torch.Tensor: Shape (B, max_len, ...) or (B, max_len).
+    """
+    b, d = tensor.shape[:2]
+
+    # Create a padded tensor of max_len and copy the existing values
+    padded_tensor = torch.full(
+        (b, max_len, *tensor.shape[2:]), pad_value, dtype=tensor.dtype, device=tensor.device
+    )
+    padded_tensor[:, :d] = tensor  # Efficient in-place copy
+
+    return padded_tensor
+
+
+class VLAFlowMatching(nn.Module):
+    """
+    SmolVLA
+
+    [Paper]()
+
+    Designed by Hugging Face.
+    ┌──────────────────────────────┐
+    │                 actions      │
+    │                    ▲         │
+    │ ┌─────────┐      ┌─|────┐    │
+    │ |         │────► │      │    │
+    │ |         │ kv   │      │    │
+    │ |         │────► │Action│    │
+    │ |   VLM   │cache │Expert│    |
+    │ │         │────► |      │    │
+    │ │         │      │      │    │
+    │ └▲──▲───▲─┘      └───▲──┘    |
+    │  │  |   |            │       |
+    │  |  |   |          noise     │
+    │  │  │ state                  │
+    │  │ language tokens           │
+    │  image(s)                    │
+    └──────────────────────────────┘
+    """
+
+    def __init__(self, config: SmolVLAConfig, rtc_processor: RTCProcessor | None = None):
+        super().__init__()
+        self.config = config
+
+        self.vlm_with_expert = SmolVLMWithExpertModel(
+            model_id=self.config.vlm_model_name,
+            freeze_vision_encoder=self.config.freeze_vision_encoder,
+            train_expert_only=self.config.train_expert_only,
+            load_vlm_weights=self.config.load_vlm_weights,
+            attention_mode=self.config.attention_mode,
+            num_expert_layers=self.config.num_expert_layers,
+            num_vlm_layers=self.config.num_vlm_layers,
+            self_attn_every_n_layers=self.config.self_attn_every_n_layers,
+            expert_width_multiplier=self.config.expert_width_multiplier,
+            device=self.config.device if self.config.device is not None else "auto",
+        )
+        self.state_proj = nn.Linear(
+            self.config.max_state_dim, self.vlm_with_expert.config.text_config.hidden_size
+        )
+        self.action_in_proj = nn.Linear(self.config.max_action_dim, self.vlm_with_expert.expert_hidden_size)
+        self.action_out_proj = nn.Linear(self.vlm_with_expert.expert_hidden_size, self.config.max_action_dim)
+
+        self.action_time_mlp_in = nn.Linear(
+            self.vlm_with_expert.expert_hidden_size * 2, self.vlm_with_expert.expert_hidden_size
+        )
+        self.action_time_mlp_out = nn.Linear(
+            self.vlm_with_expert.expert_hidden_size, self.vlm_with_expert.expert_hidden_size
+        )
+
+        self.set_requires_grad()
+        self.fake_image_token = self.vlm_with_expert.processor.tokenizer.fake_image_token_id
+        self.global_image_token = self.vlm_with_expert.processor.tokenizer.global_image_token_id
+        self.global_image_start_token = torch.tensor(
+            [self.fake_image_token, self.global_image_token], dtype=torch.long
+        )
+
+        self.add_image_special_tokens = self.config.add_image_special_tokens
+        self.image_end_token = torch.tensor([self.fake_image_token], dtype=torch.long)
+        self.prefix_length = self.config.prefix_length
+        self.rtc_processor = rtc_processor
+
+        # Compile model if requested
+        if config.compile_model:
+            torch.set_float32_matmul_precision("high")
+            self.sample_actions = torch.compile(self.sample_actions, mode=config.compile_mode)
+            self.forward = torch.compile(self.forward, mode=config.compile_mode)
+
+    def _rtc_enabled(self):
+        return self.config.rtc_config is not None and self.config.rtc_config.enabled
+
+    def set_requires_grad(self):
+        for params in self.state_proj.parameters():
+            params.requires_grad = self.config.train_state_proj
+
+    def sample_noise(self, shape, device):
+        noise = torch.normal(
+            mean=0.0,
+            std=1.0,
+            size=shape,
+            dtype=torch.float32,
+            device=device,
+        )
+        return noise
+
+    def sample_time(self, bsize, device):
+        beta_dist = torch.distributions.Beta(concentration1=1.5, concentration0=1.0)
+        time_beta = beta_dist.sample((bsize,)).to(device=device, dtype=torch.float32)
+        time = time_beta * 0.999 + 0.001
+        return time
+
+    def embed_prefix(
+        self, images, img_masks, lang_tokens, lang_masks, state: torch.Tensor = None
+    ) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
+        """Embed images with SigLIP and language tokens with embedding layer to prepare
+        for SmolVLM transformer processing.
+        """
+        embs = []
+        pad_masks = []
+        att_masks = []
+        for _img_idx, (
+            img,
+            img_mask,
+        ) in enumerate(zip(images, img_masks, strict=False)):
+            if self.add_image_special_tokens:
+                image_start_token = (
+                    self.vlm_with_expert.embed_language_tokens(
+                        self.global_image_start_token.to(device=self.vlm_with_expert.vlm.device)
+                    )
+                    .unsqueeze(0)
+                    .expand(img.shape[0], -1, -1)
+                )
+                image_start_mask = torch.ones_like(
+                    image_start_token[:, :, 0], dtype=torch.bool, device=image_start_token.device
+                )
+                att_masks += [0] * (image_start_mask.shape[-1])
+                embs.append(image_start_token)
+                pad_masks.append(image_start_mask)
+
+            img_emb = self.vlm_with_expert.embed_image(img)
+            img_emb = img_emb
+
+            # Normalize image embeddings
+            img_emb_dim = img_emb.shape[-1]
+            img_emb = img_emb * torch.tensor(img_emb_dim**0.5, dtype=img_emb.dtype, device=img_emb.device)
+
+            bsize, num_img_embs = img_emb.shape[:2]
+            img_mask = img_mask[:, None].expand(bsize, num_img_embs)
+
+            embs.append(img_emb)
+            pad_masks.append(img_mask)
+
+            att_masks += [0] * (num_img_embs)
+            if self.add_image_special_tokens:
+                image_end_token = (
+                    self.vlm_with_expert.embed_language_tokens(
+                        self.image_end_token.to(device=self.vlm_with_expert.vlm.device)
+                    )
+                    .unsqueeze(0)
+                    .expand(img.shape[0], -1, -1)
+                )
+                image_end_mask = torch.ones_like(
+                    image_end_token[:, :, 0], dtype=torch.bool, device=image_end_token.device
+                )
+                embs.append(image_end_token)
+                pad_masks.append(image_end_mask)
+                att_masks += [0] * (image_end_mask.shape[1])
+        lang_emb = self.vlm_with_expert.embed_language_tokens(lang_tokens)
+        # Normalize language embeddings
+        lang_emb_dim = lang_emb.shape[-1]
+        lang_emb = lang_emb * math.sqrt(lang_emb_dim)
+
+        embs.append(lang_emb)
+        pad_masks.append(lang_masks)
+
+        num_lang_embs = lang_emb.shape[1]
+        att_masks += [0] * num_lang_embs
+
+        state_emb = self.state_proj(state)
+        state_emb = state_emb[:, None, :] if state_emb.ndim == 2 else state_emb
+        embs.append(state_emb)
+        bsize = state_emb.shape[0]
+        device = state_emb.device
+
+        states_seq_len = state_emb.shape[1]
+        state_mask = torch.ones(bsize, states_seq_len, dtype=torch.bool, device=device)
+        pad_masks.append(state_mask)
+
+        # Set attention masks so that image and language inputs do not attend to state or actions
+        att_masks += [1] * (states_seq_len)
+        embs = torch.cat(embs, dim=1)
+        pad_masks = torch.cat(pad_masks, dim=1)
+        att_masks = torch.tensor(att_masks, dtype=torch.bool, device=pad_masks.device)
+        att_masks = att_masks[None, :]
+
+        seq_len = pad_masks.shape[1]
+        if seq_len < self.prefix_length:
+            embs = pad_tensor(embs, self.prefix_length, pad_value=0)
+            pad_masks = pad_tensor(pad_masks, self.prefix_length, pad_value=0)
+            att_masks = pad_tensor(att_masks, self.prefix_length, pad_value=0)
+
+        att_masks = att_masks.expand(bsize, -1)
+
+        return embs, pad_masks, att_masks
+
+    def embed_suffix(self, noisy_actions, timestep):
+        """Embed state, noisy_actions, timestep to prepare for Expert Gemma processing."""
+        embs = []
+        pad_masks = []
+        att_masks = []
+
+        # Fuse timestep + action information using an MLP
+        action_emb = self.action_in_proj(noisy_actions)
+        device = action_emb.device
+        bsize = action_emb.shape[0]
+        dtype = action_emb.dtype
+        # Embed timestep using sine-cosine positional encoding with sensitivity in the range [0, 1]
+        time_emb = create_sinusoidal_pos_embedding(
+            timestep,
+            self.vlm_with_expert.expert_hidden_size,
+            self.config.min_period,
+            self.config.max_period,
+            device=device,
+        )
+        time_emb = time_emb.type(dtype=dtype)
+
+        time_emb = time_emb[:, None, :].expand_as(action_emb)
+        action_time_emb = torch.cat([action_emb, time_emb], dim=2)
+
+        action_time_emb = self.action_time_mlp_in(action_time_emb)
+        action_time_emb = F.silu(action_time_emb)  # swish == silu
+        action_time_emb = self.action_time_mlp_out(action_time_emb)
+
+        # Add to input tokens
+        embs.append(action_time_emb)
+
+        bsize, action_time_dim = action_time_emb.shape[:2]
+        action_time_mask = torch.ones(bsize, action_time_dim, dtype=torch.bool, device=device)
+        pad_masks.append(action_time_mask)
+
+        # Set attention masks so that image, language and state inputs do not attend to action tokens
+        att_masks += [1] * self.config.chunk_size
+        embs = torch.cat(embs, dim=1)
+        pad_masks = torch.cat(pad_masks, dim=1)
+        att_masks = torch.tensor(att_masks, dtype=embs.dtype, device=embs.device)
+        att_masks = att_masks[None, :].expand(bsize, len(att_masks))
+        return embs, pad_masks, att_masks
+
+    def forward(
+        self, images, img_masks, lang_tokens, lang_masks, state, actions, noise=None, time=None
+    ) -> Tensor:
+        """Do a full training forward pass and compute the loss (batch_size x num_steps x num_motors)"""
+        if noise is None:
+            noise = self.sample_noise(actions.shape, actions.device)
+
+        if time is None:
+            time = self.sample_time(actions.shape[0], actions.device)
+
+        time_expanded = time[:, None, None]
+        x_t = time_expanded * noise + (1 - time_expanded) * actions
+        u_t = noise - actions
+        prefix_embs, prefix_pad_masks, prefix_att_masks = self.embed_prefix(
+            images, img_masks, lang_tokens, lang_masks, state=state
+        )
+        suffix_embs, suffix_pad_masks, suffix_att_masks = self.embed_suffix(x_t, time)
+
+        pad_masks = torch.cat([prefix_pad_masks, suffix_pad_masks], dim=1)
+        att_masks = torch.cat([prefix_att_masks, suffix_att_masks], dim=1)
+
+        att_2d_masks = make_att_2d_masks(pad_masks, att_masks)
+        position_ids = torch.cumsum(pad_masks, dim=1) - 1
+        (_, suffix_out), _ = self.vlm_with_expert.forward(
+            attention_mask=att_2d_masks,
+            position_ids=position_ids,
+            past_key_values=None,
+            inputs_embeds=[prefix_embs, suffix_embs],
+            use_cache=False,
+            fill_kv_cache=False,
+        )
+        suffix_out = suffix_out[:, -self.config.chunk_size :]
+        # Original openpi code, upcast attention output
+        suffix_out = suffix_out.to(dtype=torch.float32)
+        v_t = self.action_out_proj(suffix_out)
+        losses = F.mse_loss(u_t, v_t, reduction="none")
+        return losses
+
+    def sample_actions(
+        self,
+        images,
+        img_masks,
+        lang_tokens,
+        lang_masks,
+        state,
+        noise=None,
+        **kwargs: Unpack[ActionSelectKwargs],
+    ) -> Tensor:
+        """Do a full inference forward and compute the action (batch_size x num_steps x num_motors)"""
+        bsize = state.shape[0]
+        device = state.device
+
+        if noise is None:
+            actions_shape = (bsize, self.config.chunk_size, self.config.max_action_dim)
+            noise = self.sample_noise(actions_shape, device)
+
+        prefix_embs, prefix_pad_masks, prefix_att_masks = self.embed_prefix(
+            images, img_masks, lang_tokens, lang_masks, state=state
+        )
+        prefix_att_2d_masks = make_att_2d_masks(prefix_pad_masks, prefix_att_masks)
+        prefix_position_ids = torch.cumsum(prefix_pad_masks, dim=1) - 1
+        # Compute image and language key value cache
+        _, past_key_values = self.vlm_with_expert.forward(
+            attention_mask=prefix_att_2d_masks,
+            position_ids=prefix_position_ids,
+            past_key_values=None,
+            inputs_embeds=[prefix_embs, None],
+            use_cache=self.config.use_cache,
+            fill_kv_cache=True,
+        )
+        num_steps = self.config.num_steps
+        dt = -1.0 / num_steps
+
+        x_t = noise
+        for step in range(num_steps):
+            time = 1.0 + step * dt
+            time_tensor = torch.tensor(time, dtype=torch.float32, device=device).expand(bsize)
+
+            def denoise_step_partial_call(input_x_t, current_timestep=time_tensor):
+                return self.denoise_step(
+                    x_t=input_x_t,
+                    prefix_pad_masks=prefix_pad_masks,
+                    past_key_values=past_key_values,
+                    timestep=current_timestep,
+                )
+
+            if self._rtc_enabled():
+                inference_delay = kwargs.get("inference_delay")
+                prev_chunk_left_over = kwargs.get("prev_chunk_left_over")
+                execution_horizon = kwargs.get("execution_horizon")
+
+                v_t = self.rtc_processor.denoise_step(
+                    x_t=x_t,
+                    prev_chunk_left_over=prev_chunk_left_over,
+                    inference_delay=inference_delay,
+                    time=time,
+                    original_denoise_step_partial=denoise_step_partial_call,
+                    execution_horizon=execution_horizon,
+                )
+            else:
+                v_t = denoise_step_partial_call(x_t)
+
+            x_t = x_t + dt * v_t
+
+            if self.rtc_processor is not None and self.rtc_processor.is_debug_enabled():
+                self.rtc_processor.track(time=time, x_t=x_t, v_t=v_t)
+
+        return x_t
+
+    def denoise_step(
+        self,
+        prefix_pad_masks,
+        past_key_values,
+        x_t,
+        timestep,
+    ):
+        """Apply one denoising step of the noise `x_t` at a given timestep."""
+        suffix_embs, suffix_pad_masks, suffix_att_masks = self.embed_suffix(x_t, timestep)
+
+        suffix_len = suffix_pad_masks.shape[1]
+        batch_size = prefix_pad_masks.shape[0]
+        prefix_len = prefix_pad_masks.shape[1]
+        prefix_pad_2d_masks = prefix_pad_masks[:, None, :].expand(batch_size, suffix_len, prefix_len)
+
+        suffix_att_2d_masks = make_att_2d_masks(suffix_pad_masks, suffix_att_masks)
+
+        full_att_2d_masks = torch.cat([prefix_pad_2d_masks, suffix_att_2d_masks], dim=2)
+        prefix_offsets = torch.sum(prefix_pad_masks, dim=-1)[:, None]
+        position_ids = prefix_offsets + torch.cumsum(suffix_pad_masks, dim=1) - 1
+
+        outputs_embeds, _ = self.vlm_with_expert.forward(
+            attention_mask=full_att_2d_masks,
+            position_ids=position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=[None, suffix_embs],
+            use_cache=self.config.use_cache,
+            fill_kv_cache=False,
+        )
+        suffix_out = outputs_embeds[1]
+        suffix_out = suffix_out[:, -self.config.chunk_size :]
+        suffix_out = suffix_out.to(dtype=torch.float32)
+        v_t = self.action_out_proj(suffix_out)
+        return v_t
diff --git a/lerobot/src/lerobot/policies/smolvla/processor_smolvla.py b/lerobot/src/lerobot/policies/smolvla/processor_smolvla.py
new file mode 100644
index 0000000000000000000000000000000000000000..3fc130aa1d034bf86e64696f250a8eb50972e697
--- /dev/null
+++ b/lerobot/src/lerobot/policies/smolvla/processor_smolvla.py
@@ -0,0 +1,141 @@
+#!/usr/bin/env python
+
+# Copyright 2025 HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from typing import Any
+
+import torch
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.policies.smolvla.configuration_smolvla import SmolVLAConfig
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    ComplementaryDataProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    ProcessorStepRegistry,
+    RenameObservationsProcessorStep,
+    TokenizerProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.utils.constants import POLICY_POSTPROCESSOR_DEFAULT_NAME, POLICY_PREPROCESSOR_DEFAULT_NAME
+
+
+def make_smolvla_pre_post_processors(
+    config: SmolVLAConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for the SmolVLA policy.
+
+    The pre-processing pipeline prepares input data for the model by:
+    1.  Renaming features to match pretrained configurations.
+    2.  Normalizing input and output features based on dataset statistics.
+    3.  Adding a batch dimension.
+    4.  Ensuring the language task description ends with a newline character.
+    5.  Tokenizing the language task description.
+    6.  Moving all data to the specified device.
+
+    The post-processing pipeline handles the model's output by:
+    1.  Moving data to the CPU.
+    2.  Unnormalizing the output actions to their original scale.
+
+    Args:
+        config: The configuration object for the SmolVLA policy.
+        dataset_stats: A dictionary of statistics for normalization.
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines.
+    """
+
+    input_steps = [
+        RenameObservationsProcessorStep(rename_map={}),  # To mimic the same processor as pretrained one
+        AddBatchDimensionProcessorStep(),
+        SmolVLANewLineProcessor(),
+        TokenizerProcessorStep(
+            tokenizer_name=config.vlm_model_name,
+            padding=config.pad_language_to,
+            padding_side="right",
+            max_length=config.tokenizer_max_length,
+        ),
+        DeviceProcessorStep(device=config.device),
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+    ]
+    output_steps = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
+
+
+@ProcessorStepRegistry.register(name="smolvla_new_line_processor")
+class SmolVLANewLineProcessor(ComplementaryDataProcessorStep):
+    """
+    A processor step that ensures the 'task' description ends with a newline character.
+
+    This step is necessary for certain tokenizers (e.g., PaliGemma) that expect a
+    newline at the end of the prompt. It handles both single string tasks and lists
+    of string tasks.
+    """
+
+    def complementary_data(self, complementary_data):
+        if "task" not in complementary_data:
+            return complementary_data
+
+        task = complementary_data["task"]
+        if task is None:
+            return complementary_data
+
+        new_complementary_data = dict(complementary_data)
+
+        # Handle both string and list of strings
+        if isinstance(task, str):
+            # Single string: add newline if not present
+            if not task.endswith("\n"):
+                new_complementary_data["task"] = f"{task}\n"
+        elif isinstance(task, list) and all(isinstance(t, str) for t in task):
+            # List of strings: add newline to each if not present
+            new_complementary_data["task"] = [t if t.endswith("\n") else f"{t}\n" for t in task]
+        # If task is neither string nor list of strings, leave unchanged
+
+        return new_complementary_data
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
diff --git a/lerobot/src/lerobot/policies/smolvla/smolvlm_with_expert.py b/lerobot/src/lerobot/policies/smolvla/smolvlm_with_expert.py
new file mode 100644
index 0000000000000000000000000000000000000000..caca41dabb2cdbf1522570745b8d3c8fbf61519e
--- /dev/null
+++ b/lerobot/src/lerobot/policies/smolvla/smolvlm_with_expert.py
@@ -0,0 +1,549 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import copy
+
+import torch
+from torch import nn
+from transformers import (
+    AutoConfig,
+    AutoModel,
+    AutoModelForImageTextToText,
+    AutoProcessor,
+    SmolVLMForConditionalGeneration,
+)
+
+
+def apply_rope(x, positions, max_wavelength=10_000):
+    """
+    Applies RoPE positions [B, L] to x [B, L, H, D].
+    """
+    d_half = x.shape[-1] // 2
+    device = x.device
+    dtype = x.dtype
+    x = x.to(torch.float32)
+
+    freq_exponents = (2.0 / x.shape[-1]) * torch.arange(d_half, dtype=torch.float32, device=device)
+    timescale = max_wavelength**freq_exponents
+    radians = positions[..., None].to(torch.float32) / timescale[None, None, :].to(torch.float32)
+
+    radians = radians[..., None, :]
+
+    sin = torch.sin(radians)  # .to(dtype=dtype)
+    cos = torch.cos(radians)  # .to(dtype=dtype)
+
+    x1, x2 = x.split(d_half, dim=-1)
+    res = torch.empty_like(x)
+    res[..., :d_half] = x1 * cos - x2 * sin
+    res[..., d_half:] = x2 * cos + x1 * sin
+
+    return res.to(dtype)
+
+
+def get_intermediate_size(hidden_dim, ffn_dim_multiplier=4, multiple_of=256):
+    hidden_dim = int(2 * hidden_dim / 3)
+    hidden_dim = int(ffn_dim_multiplier * hidden_dim)
+    hidden_dim = multiple_of * ((hidden_dim + multiple_of - 1) // multiple_of)
+    return hidden_dim
+
+
+class SmolVLMWithExpertModel(nn.Module):
+    def __init__(
+        self,
+        model_id: str = "HuggingFaceTB/SmolVLM2-500M-Video-Instruct",
+        load_vlm_weights: bool = True,
+        train_expert_only: bool = True,
+        freeze_vision_encoder: bool = False,
+        attention_mode: str = "self_attn",
+        num_expert_layers: int = -1,
+        num_vlm_layers: int = -1,
+        self_attn_every_n_layers: int = -1,
+        expert_width_multiplier: float = 0.5,
+        device: str = "auto",
+    ):
+        super().__init__()
+        if load_vlm_weights:
+            print(f"Loading  {model_id} weights ...")
+            self.vlm = AutoModelForImageTextToText.from_pretrained(
+                model_id,
+                torch_dtype="bfloat16",
+                low_cpu_mem_usage=True,
+            )
+            config = self.vlm.config
+        else:
+            config = AutoConfig.from_pretrained(model_id)
+            self.vlm = SmolVLMForConditionalGeneration(config=config)
+        self.processor = AutoProcessor.from_pretrained(model_id)
+        if num_vlm_layers > 0:
+            print(f"Reducing the number of VLM layers to {num_vlm_layers} ...")
+            self.get_vlm_model().text_model.layers = self.get_vlm_model().text_model.layers[:num_vlm_layers]
+        self.num_vlm_layers = len(self.get_vlm_model().text_model.layers)
+        self.config = config
+        # Smaller lm expert
+        lm_expert_config = copy.deepcopy(config.text_config)
+        hidden_size = lm_expert_config.hidden_size
+        lm_expert_config.hidden_size = int(hidden_size * expert_width_multiplier)  # hidden_size // 2
+        lm_expert_config.intermediate_size = get_intermediate_size(int(hidden_size * expert_width_multiplier))
+        lm_expert_config.num_hidden_layers = self.num_vlm_layers
+        if num_expert_layers > 0:
+            assert len(self.get_vlm_model().text_model.layers) % num_expert_layers == 0, (
+                f"Number of layers in the VLM {len(self.get_vlm_model().text_model.layers)} are not multiple of num_expert_layers {num_expert_layers}"
+            )
+            lm_expert_config.num_hidden_layers = num_expert_layers
+        self.lm_expert = AutoModel.from_config(lm_expert_config)
+
+        self.num_expert_layers = len(self.lm_expert.layers)
+        self.self_attn_every_n_layers = self_attn_every_n_layers
+        if "cross" in attention_mode:
+            # Reshape qkv projections to have the same input dimension as the vlm
+            for layer_idx in range(len(self.lm_expert.layers)):
+                if self.self_attn_every_n_layers > 0 and layer_idx % self.self_attn_every_n_layers == 0:
+                    continue
+                self.lm_expert.layers[layer_idx].self_attn.k_proj = nn.Linear(
+                    config.text_config.num_key_value_heads * config.text_config.head_dim,
+                    lm_expert_config.num_key_value_heads * lm_expert_config.head_dim,
+                    bias=lm_expert_config.attention_bias,
+                )
+                self.lm_expert.layers[layer_idx].self_attn.v_proj = nn.Linear(
+                    config.text_config.num_key_value_heads * config.text_config.head_dim,
+                    lm_expert_config.num_key_value_heads * lm_expert_config.head_dim,
+                    bias=lm_expert_config.attention_bias,
+                )
+        # Remove unused embed_tokens
+        self.lm_expert.embed_tokens = None
+
+        self.num_attention_heads = self.config.text_config.num_attention_heads
+        self.num_key_value_heads = self.config.text_config.num_key_value_heads
+
+        self.freeze_vision_encoder = freeze_vision_encoder
+        self.train_expert_only = train_expert_only
+        self.attention_mode = attention_mode
+        self.expert_hidden_size = lm_expert_config.hidden_size
+        self.set_requires_grad()
+
+    def get_vlm_model(self):
+        return self.vlm.model
+
+    def set_requires_grad(self):
+        if self.freeze_vision_encoder:
+            self.get_vlm_model().vision_model.eval()
+            for params in self.get_vlm_model().vision_model.parameters():
+                params.requires_grad = False
+        if self.train_expert_only:
+            self.vlm.eval()
+            for params in self.vlm.parameters():
+                params.requires_grad = False
+        else:
+            # To avoid unused params issue with distributed training
+            last_layers = [self.num_vlm_layers - 1]
+            if (
+                self.num_vlm_layers != self.num_expert_layers
+                and self.num_vlm_layers % self.num_expert_layers == 0
+            ):
+                last_layers.append(self.num_vlm_layers - 2)
+            frozen_layers = [
+                "lm_head",
+                "text_model.model.norm.weight",
+            ]
+            for layer in last_layers:
+                frozen_layers.append(f"text_model.model.layers.{layer}.")
+
+            for name, params in self.vlm.named_parameters():
+                if any(k in name for k in frozen_layers):
+                    params.requires_grad = False
+        # To avoid unused params issue with distributed training
+        for name, params in self.lm_expert.named_parameters():
+            if "lm_head" in name:
+                params.requires_grad = False
+
+    def train(self, mode: bool = True):
+        super().train(mode)
+
+        if self.freeze_vision_encoder:
+            self.get_vlm_model().vision_model.eval()
+
+        if self.train_expert_only:
+            self.vlm.eval()
+
+    def embed_image(self, image: torch.Tensor):
+        patch_attention_mask = None
+        # Get sequence from the vision encoder
+        image_hidden_states = (
+            self.get_vlm_model()
+            .vision_model(
+                pixel_values=image.to(dtype=self.get_vlm_model().vision_model.dtype),
+                patch_attention_mask=patch_attention_mask,
+            )
+            .last_hidden_state
+        )
+        # Modality projection & resampling
+        image_hidden_states = self.get_vlm_model().connector(image_hidden_states)
+        return image_hidden_states
+
+    def embed_language_tokens(self, tokens: torch.Tensor):
+        return self.get_vlm_model().text_model.get_input_embeddings()(tokens)
+
+    def forward_attn_layer(
+        self,
+        model_layers,
+        inputs_embeds,
+        layer_idx,
+        position_ids,
+        attention_mask,
+        batch_size,
+        head_dim,
+        use_cache: bool = True,
+        fill_kv_cache: bool = True,
+        past_key_values=None,
+    ) -> list[torch.Tensor]:
+        query_states = []
+        key_states = []
+        value_states = []
+        for i, hidden_states in enumerate(inputs_embeds):
+            layer = model_layers[i][layer_idx]
+            if hidden_states is None or layer is None:
+                continue
+            hidden_states = layer.input_layernorm(hidden_states)
+
+            input_shape = hidden_states.shape[:-1]
+            hidden_shape = (*input_shape, -1, layer.self_attn.head_dim)
+
+            hidden_states = hidden_states.to(dtype=layer.self_attn.q_proj.weight.dtype)
+            query_state = layer.self_attn.q_proj(hidden_states).view(hidden_shape)
+            key_state = layer.self_attn.k_proj(hidden_states).view(hidden_shape)
+            value_state = layer.self_attn.v_proj(hidden_states).view(hidden_shape)
+
+            query_states.append(query_state)
+            key_states.append(key_state)
+            value_states.append(value_state)
+
+        # B,L,H,D with L sequence length, H number of heads, D head dim
+        # concatenate on the number of embeddings/tokens
+        query_states = torch.cat(query_states, dim=1)
+        key_states = torch.cat(key_states, dim=1)
+        value_states = torch.cat(value_states, dim=1)
+        seq_len = query_states.shape[1]
+        if seq_len < position_ids.shape[1]:
+            _position_ids = position_ids[:, :seq_len]
+            _attention_mask = attention_mask[:, :seq_len, :seq_len]
+        else:
+            _position_ids = position_ids
+            _attention_mask = attention_mask
+
+        attention_mask_ = _attention_mask
+        position_ids_ = _position_ids
+
+        query_states = apply_rope(query_states, position_ids_)
+        key_states = apply_rope(key_states, position_ids_)
+
+        if use_cache and past_key_values is None:
+            past_key_values = {}
+
+        if use_cache:
+            if fill_kv_cache:
+                past_key_values[layer_idx] = {
+                    "key_states": key_states,
+                    "value_states": value_states,
+                }
+            else:
+                # TODO here, some optimization can be done - similar to a `StaticCache` we can declare the `max_len` before.
+                # so we create an empty cache, with just one cuda malloc, and if (in autoregressive case) we reach
+                # the max len, then we (for instance) double the cache size. This implementation already exists
+                # in `transformers`. (molbap)
+                key_states = torch.cat([past_key_values[layer_idx]["key_states"], key_states], dim=1)
+                value_states = torch.cat([past_key_values[layer_idx]["value_states"], value_states], dim=1)
+
+        attention_interface = self.get_attention_interface()
+
+        att_output = attention_interface(
+            attention_mask_, batch_size, head_dim, query_states, key_states, value_states
+        )
+        return [att_output], past_key_values
+
+    def forward_cross_attn_layer(
+        self,
+        model_layers,
+        inputs_embeds,
+        layer_idx,
+        position_ids,
+        attention_mask,
+        batch_size,
+        head_dim,
+        use_cache: bool = True,
+        fill_kv_cache: bool = True,
+        past_key_values=None,
+    ) -> list[torch.Tensor]:
+        attention_interface = self.get_attention_interface()
+
+        att_outputs = []
+        assert len(inputs_embeds) == 2 or (use_cache and past_key_values is not None and not fill_kv_cache), (
+            f"Both len(inputs_embeds) == {len(inputs_embeds)} and past_key_values is {past_key_values}"
+        )
+
+        if len(inputs_embeds) == 2 and not past_key_values:
+            # Prefix attention
+            seq_len = inputs_embeds[0].shape[1]
+            position_id, expert_position_id = position_ids[:, :seq_len], position_ids[:, seq_len:]
+            prefix_attention_mask = attention_mask[:, :seq_len, :seq_len]
+
+            layer = model_layers[0][layer_idx]
+
+            hidden_states = layer.input_layernorm(inputs_embeds[0])
+
+            input_shape = hidden_states.shape[:-1]
+            hidden_shape = (*input_shape, -1, layer.self_attn.head_dim)
+
+            hidden_states = hidden_states.to(dtype=layer.self_attn.q_proj.weight.dtype)
+            query_state = layer.self_attn.q_proj(hidden_states).view(hidden_shape)
+            key_state = layer.self_attn.k_proj(hidden_states).view(hidden_shape)
+            value_states = layer.self_attn.v_proj(hidden_states).view(hidden_shape)
+
+            # B,L,H,D with L sequence length, H number of heads, D head dim
+            query_states = apply_rope(query_state, position_id)
+            key_states = apply_rope(key_state, position_id)
+
+            att_output = attention_interface(
+                prefix_attention_mask, batch_size, head_dim, query_states, key_states, value_states
+            )
+            att_outputs.append(att_output)
+        else:
+            expert_position_id = position_ids
+
+        if use_cache and past_key_values is None:
+            past_key_values = {}
+
+        if use_cache:
+            if fill_kv_cache:
+                past_key_values[layer_idx] = {
+                    "key_states": key_states,
+                    "value_states": value_states,
+                }
+            else:
+                # TODO here, some optimization can be done - similar to a `StaticCache` we can declare the `max_len` before.
+                # so we create an empty cache, with just one cuda malloc, and if (in autoregressive case) we reach
+                # the max len, then we (for instance) double the cache size. This implementation already exists
+                # in `transformers`. (molbap)
+                key_states = past_key_values[layer_idx]["key_states"]
+                value_states = past_key_values[layer_idx]["value_states"]
+
+        # Expert
+        expert_layer = model_layers[1][layer_idx]
+        if expert_layer is not None:
+            expert_hidden_states = expert_layer.input_layernorm(inputs_embeds[1])
+
+            expert_input_shape = expert_hidden_states.shape[:-1]
+            expert_hidden_shape = (*expert_input_shape, -1, expert_layer.self_attn.head_dim)
+
+            expert_hidden_states = expert_hidden_states.to(dtype=expert_layer.self_attn.q_proj.weight.dtype)
+            expert_query_state = expert_layer.self_attn.q_proj(expert_hidden_states).view(expert_hidden_shape)
+
+            _key_states = key_states.to(dtype=expert_layer.self_attn.k_proj.weight.dtype).view(
+                *key_states.shape[:2], -1
+            )
+            expert_key_states = expert_layer.self_attn.k_proj(_key_states).view(
+                *_key_states.shape[:-1], -1, expert_layer.self_attn.head_dim
+            )  # k_proj should have same dim as kv
+
+            _value_states = value_states.to(dtype=expert_layer.self_attn.v_proj.weight.dtype).view(
+                *value_states.shape[:2], -1
+            )
+            expert_value_states = expert_layer.self_attn.v_proj(_value_states).view(
+                *_value_states.shape[:-1], -1, expert_layer.self_attn.head_dim
+            )
+
+            expert_position_id = (
+                expert_position_id - torch.min(expert_position_id, dim=1, keepdim=True).values
+            )  # start from 0
+            expert_attention_mask = attention_mask[
+                :, -inputs_embeds[1].shape[1] :, : expert_key_states.shape[1] :
+            ]  # take into account kv
+
+            expert_query_states = apply_rope(expert_query_state, expert_position_id)
+
+            att_output = attention_interface(
+                expert_attention_mask,
+                batch_size,
+                head_dim,
+                expert_query_states,
+                expert_key_states,
+                expert_value_states,
+            )
+            att_outputs.append(att_output)
+        else:
+            att_outputs.append(None)
+
+        # att_output = att_output.to(dtype=models[i].dtype)
+        return att_outputs, past_key_values
+
+    def get_model_layers(self, models: list) -> list:
+        vlm_layers = []
+        expert_layers = []
+        multiple_of = self.num_vlm_layers // self.num_expert_layers
+        for i in range(self.num_vlm_layers):
+            if multiple_of > 0 and i > 0 and i % multiple_of != 0:
+                expert_layer = None
+            else:
+                expert_layer_index = i // multiple_of if multiple_of > 0 else i
+                expert_layer = models[1].layers[expert_layer_index]
+            vlm_layers.append(models[0].layers[i])
+            expert_layers.append(expert_layer)
+        return [vlm_layers, expert_layers]
+
+    def forward(
+        self,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: list[torch.FloatTensor] = None,
+        use_cache: bool | None = None,
+        fill_kv_cache: bool | None = None,
+    ):
+        models = [self.get_vlm_model().text_model, self.lm_expert]
+        model_layers = self.get_model_layers(models)
+        for hidden_states in inputs_embeds:
+            # TODO this is very inefficient
+            # dtype is always the same, batch size too (if > 1 len)
+            # device could be trickier in multi gpu edge cases but that's it
+            if hidden_states is None:
+                continue
+            batch_size = hidden_states.shape[0]
+
+        # RMSNorm
+        num_layers = self.num_vlm_layers
+        head_dim = self.vlm.config.text_config.head_dim
+        for layer_idx in range(num_layers):
+            if (
+                fill_kv_cache
+                or "cross" not in self.attention_mode
+                or (self.self_attn_every_n_layers > 0 and layer_idx % self.self_attn_every_n_layers == 0)
+            ):
+                att_outputs, past_key_values = self.forward_attn_layer(
+                    model_layers,
+                    inputs_embeds,
+                    layer_idx,
+                    position_ids,
+                    attention_mask,
+                    batch_size,
+                    head_dim,
+                    use_cache=use_cache,
+                    fill_kv_cache=fill_kv_cache,
+                    past_key_values=past_key_values,
+                )
+            else:
+                att_outputs, past_key_values = self.forward_cross_attn_layer(
+                    model_layers,
+                    inputs_embeds,
+                    layer_idx,
+                    position_ids,
+                    attention_mask,
+                    batch_size,
+                    head_dim,
+                    use_cache=use_cache,
+                    fill_kv_cache=fill_kv_cache,
+                    past_key_values=past_key_values,
+                )
+            outputs_embeds = []
+            start = 0
+            for i, hidden_states in enumerate(inputs_embeds):
+                layer = model_layers[i][layer_idx]
+                att_output = (
+                    att_outputs[i] if i < len(att_outputs) else att_outputs[0]
+                )  # in case of self_attn
+                if hidden_states is not None:
+                    if layer is None:
+                        outputs_embeds.append(hidden_states)
+                        continue
+                    end = start + hidden_states.shape[1]
+
+                    if att_output.dtype != layer.self_attn.o_proj.weight.dtype:
+                        att_output = att_output.to(layer.self_attn.o_proj.weight.dtype)
+                    att_out = att_output[:, start:end]
+                    out_emb = layer.self_attn.o_proj(att_out)
+
+                    out_emb += hidden_states
+                    after_first_residual = out_emb.clone()
+
+                    out_emb = layer.post_attention_layernorm(out_emb)
+                    out_emb = layer.mlp(out_emb)
+
+                    out_emb += after_first_residual
+
+                    outputs_embeds.append(out_emb)
+
+                    start = end if len(att_outputs) == 1 else 0
+                else:
+                    outputs_embeds.append(None)
+
+            inputs_embeds = outputs_embeds
+
+        # final norm
+        outputs_embeds = []
+        for i, hidden_states in enumerate(inputs_embeds):
+            if hidden_states is not None:
+                out_emb = models[i].norm(hidden_states)
+                outputs_embeds.append(out_emb)
+            else:
+                outputs_embeds.append(None)
+        return outputs_embeds, past_key_values
+
+    def get_attention_interface(self):
+        attention_interface = self.eager_attention_forward
+        return attention_interface
+
+    def eager_attention_forward(
+        self, attention_mask, batch_size, head_dim, query_states, key_states, value_states
+    ):
+        num_att_heads = self.num_attention_heads
+        num_key_value_heads = self.num_key_value_heads
+        num_key_value_groups = num_att_heads // num_key_value_heads
+
+        sequence_length = key_states.shape[1]
+
+        key_states = key_states[:, :, :, None, :].expand(
+            batch_size, sequence_length, num_key_value_heads, num_key_value_groups, head_dim
+        )
+        key_states = key_states.reshape(
+            batch_size, sequence_length, num_key_value_heads * num_key_value_groups, head_dim
+        )
+
+        value_states = value_states[:, :, :, None, :].expand(
+            batch_size, sequence_length, num_key_value_heads, num_key_value_groups, head_dim
+        )
+        value_states = value_states.reshape(
+            batch_size, sequence_length, num_key_value_heads * num_key_value_groups, head_dim
+        )
+
+        # Attention here is upcasted to float32 to match the original eager implementation.
+        query_states = query_states.to(dtype=torch.float32)
+        key_states = key_states.to(dtype=torch.float32)
+
+        query_states = query_states.transpose(1, 2)
+        key_states = key_states.transpose(1, 2)
+
+        att_weights = torch.matmul(query_states, key_states.transpose(2, 3))
+        att_weights *= head_dim**-0.5
+
+        att_weights = att_weights.to(dtype=torch.float32)
+        big_neg = torch.finfo(att_weights.dtype).min  # -2.3819763e38  # See gemma/modules.py
+        masked_att_weights = torch.where(attention_mask[:, None, :, :], att_weights, big_neg)
+        probs = nn.functional.softmax(masked_att_weights, dim=-1)
+        probs = probs.to(dtype=value_states.dtype)
+
+        att_output = torch.matmul(probs, value_states.permute(0, 2, 1, 3))
+
+        att_output = att_output.permute(0, 2, 1, 3)
+        # we use -1 because sequence length can change
+        att_output = att_output.reshape(batch_size, -1, num_key_value_heads * num_key_value_groups * head_dim)
+
+        return att_output
diff --git a/lerobot/src/lerobot/policies/tdmpc/README.md b/lerobot/src/lerobot/policies/tdmpc/README.md
new file mode 120000
index 0000000000000000000000000000000000000000..413ea87b8aa47343500041a62de97e6241a8b6cf
--- /dev/null
+++ b/lerobot/src/lerobot/policies/tdmpc/README.md
@@ -0,0 +1 @@
+../../../../docs/source/policy_tdmpc_README.md
\ No newline at end of file
diff --git a/lerobot/src/lerobot/policies/tdmpc/configuration_tdmpc.py b/lerobot/src/lerobot/policies/tdmpc/configuration_tdmpc.py
new file mode 100644
index 0000000000000000000000000000000000000000..3ec4934728ee5bad5c339873a53f7973e887992d
--- /dev/null
+++ b/lerobot/src/lerobot/policies/tdmpc/configuration_tdmpc.py
@@ -0,0 +1,208 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Nicklas Hansen, Xiaolong Wang, Hao Su,
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import NormalizationMode
+from lerobot.optim.optimizers import AdamConfig
+
+
+@PreTrainedConfig.register_subclass("tdmpc")
+@dataclass
+class TDMPCConfig(PreTrainedConfig):
+    """Configuration class for TDMPCPolicy.
+
+    Defaults are configured for training with xarm_lift_medium_replay providing proprioceptive and single
+    camera observations.
+
+    The parameters you will most likely need to change are the ones which depend on the environment / sensors.
+    Those are: `input_features`, `output_features`, and perhaps `max_random_shift_ratio`.
+
+    Args:
+        n_action_repeats: The number of times to repeat the action returned by the planning. (hint: Google
+            action repeats in Q-learning or ask your favorite chatbot)
+        horizon: Horizon for model predictive control.
+        n_action_steps: Number of action steps to take from the plan given by model predictive control. This
+            is an alternative to using action repeats. If this is set to more than 1, then we require
+            `n_action_repeats == 1`, `use_mpc == True` and `n_action_steps <= horizon`. Note that this
+            approach of using multiple steps from the plan is not in the original implementation.
+        input_features: A dictionary defining the PolicyFeature of the input data for the policy. The key represents
+            the input data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        output_features: A dictionary defining the PolicyFeature of the output data for the policy. The key represents
+            the output data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        normalization_mapping: A dictionary that maps from a str value of FeatureType (e.g., "STATE", "VISUAL") to
+            a corresponding NormalizationMode (e.g., NormalizationMode.MIN_MAX)
+        image_encoder_hidden_dim: Number of channels for the convolutional layers used for image encoding.
+        state_encoder_hidden_dim: Hidden dimension for MLP used for state vector encoding.
+        latent_dim: Observation's latent embedding dimension.
+        q_ensemble_size: Number of Q function estimators to use in an ensemble for uncertainty estimation.
+        mlp_dim: Hidden dimension of MLPs used for modelling the dynamics encoder, reward function, policy
+            (π), Q ensemble, and V.
+        discount: Discount factor (γ) to use for the reinforcement learning formalism.
+        use_mpc: Whether to use model predictive control. The alternative is to just sample the policy model
+            (π) for each step.
+        cem_iterations: Number of iterations for the MPPI/CEM loop in MPC.
+        max_std: Maximum standard deviation for actions sampled from the gaussian PDF in CEM.
+        min_std: Minimum standard deviation for noise applied to actions sampled from the policy model (π).
+            Doubles up as the minimum standard deviation for actions sampled from the gaussian PDF in CEM.
+        n_gaussian_samples: Number of samples to draw from the gaussian distribution every CEM iteration. Must
+            be non-zero.
+        n_pi_samples: Number of samples to draw from the policy / world model rollout every CEM iteration. Can
+            be zero.
+        uncertainty_regularizer_coeff: Coefficient for the uncertainty regularization used when estimating
+            trajectory values (this is the λ coefficient in eqn 4 of FOWM).
+        n_elites: The number of elite samples to use for updating the gaussian parameters every CEM iteration.
+        elite_weighting_temperature: The temperature to use for softmax weighting (by trajectory value) of the
+            elites, when updating the gaussian parameters for CEM.
+        gaussian_mean_momentum: Momentum (α) used for EMA updates of the mean parameter μ of the gaussian
+            parameters optimized in CEM. Updates are calculated as μ⁻ ← αμ⁻ + (1-α)μ.
+        max_random_shift_ratio: Maximum random shift (as a proportion of the image size) to apply to the
+            image(s) (in units of pixels) for training-time augmentation. If set to 0, no such augmentation
+            is applied. Note that the input images are assumed to be square for this augmentation.
+        reward_coeff: Loss weighting coefficient for the reward regression loss.
+        expectile_weight: Weighting (τ) used in expectile regression for the state value function (V).
+            v_pred < v_target is weighted by τ and v_pred >= v_target is weighted by (1-τ). τ is expected to
+            be in [0, 1]. Setting τ closer to 1 results in a more "optimistic" V. This is sensible to do
+            because v_target is obtained by evaluating the learned state-action value functions (Q) with
+            in-sample actions that may not be always optimal.
+        value_coeff: Loss weighting coefficient for both the state-action value (Q) TD loss, and the state
+            value (V) expectile regression loss.
+        consistency_coeff: Loss weighting coefficient for the consistency loss.
+        advantage_scaling: A factor by which the advantages are scaled prior to exponentiation for advantage
+            weighted regression of the policy (π) estimator parameters. Note that the exponentiated advantages
+            are clamped at 100.0.
+        pi_coeff: Loss weighting coefficient for the action regression loss.
+        temporal_decay_coeff: Exponential decay coefficient for decaying the loss coefficient for future time-
+            steps. Hint: each loss computation involves `horizon` steps worth of actions starting from the
+            current time step.
+        target_model_momentum: Momentum (α) used for EMA updates of the target models. Updates are calculated
+            as ϕ ← αϕ + (1-α)θ where ϕ are the parameters of the target model and θ are the parameters of the
+            model being trained.
+    """
+
+    # Input / output structure.
+    n_obs_steps: int = 1
+    n_action_repeats: int = 2
+    horizon: int = 5
+    n_action_steps: int = 1
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.IDENTITY,
+            "ENV": NormalizationMode.IDENTITY,
+            "ACTION": NormalizationMode.MIN_MAX,
+        }
+    )
+
+    # Architecture / modeling.
+    # Neural networks.
+    image_encoder_hidden_dim: int = 32
+    state_encoder_hidden_dim: int = 256
+    latent_dim: int = 50
+    q_ensemble_size: int = 5
+    mlp_dim: int = 512
+    # Reinforcement learning.
+    discount: float = 0.9
+
+    # Inference.
+    use_mpc: bool = True
+    cem_iterations: int = 6
+    max_std: float = 2.0
+    min_std: float = 0.05
+    n_gaussian_samples: int = 512
+    n_pi_samples: int = 51
+    uncertainty_regularizer_coeff: float = 1.0
+    n_elites: int = 50
+    elite_weighting_temperature: float = 0.5
+    gaussian_mean_momentum: float = 0.1
+
+    # Training and loss computation.
+    max_random_shift_ratio: float = 0.0476
+    # Loss coefficients.
+    reward_coeff: float = 0.5
+    expectile_weight: float = 0.9
+    value_coeff: float = 0.1
+    consistency_coeff: float = 20.0
+    advantage_scaling: float = 3.0
+    pi_coeff: float = 0.5
+    temporal_decay_coeff: float = 0.5
+    # Target model.
+    target_model_momentum: float = 0.995
+
+    # Training presets
+    optimizer_lr: float = 3e-4
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        """Input validation (not exhaustive)."""
+        if self.n_gaussian_samples <= 0:
+            raise ValueError(
+                f"The number of gaussian samples for CEM should be non-zero. Got `{self.n_gaussian_samples=}`"
+            )
+        if self.normalization_mapping["ACTION"] is not NormalizationMode.MIN_MAX:
+            raise ValueError(
+                "TD-MPC assumes the action space dimensions to all be in [-1, 1]. Therefore it is strongly "
+                f"advised that you stick with the default. See {self.__class__.__name__} docstring for more "
+                "information."
+            )
+        if self.n_obs_steps != 1:
+            raise ValueError(
+                f"Multiple observation steps not handled yet. Got `nobs_steps={self.n_obs_steps}`"
+            )
+        if self.n_action_steps > 1:
+            if self.n_action_repeats != 1:
+                raise ValueError(
+                    "If `n_action_steps > 1`, `n_action_repeats` must be left to its default value of 1."
+                )
+            if not self.use_mpc:
+                raise ValueError("If `n_action_steps > 1`, `use_mpc` must be set to `True`.")
+            if self.n_action_steps > self.horizon:
+                raise ValueError("`n_action_steps` must be less than or equal to `horizon`.")
+
+    def get_optimizer_preset(self) -> AdamConfig:
+        return AdamConfig(lr=self.optimizer_lr)
+
+    def get_scheduler_preset(self) -> None:
+        return None
+
+    def validate_features(self) -> None:
+        # There should only be one image key.
+        if len(self.image_features) > 1:
+            raise ValueError(
+                f"{self.__class__.__name__} handles at most one image for now. Got image keys {self.image_features}."
+            )
+
+        if len(self.image_features) > 0:
+            image_ft = next(iter(self.image_features.values()))
+            if image_ft.shape[-2] != image_ft.shape[-1]:
+                # TODO(alexander-soare): This limitation is solely because of code in the random shift
+                # augmentation. It should be able to be removed.
+                raise ValueError(f"Only square images are handled now. Got image shape {image_ft.shape}.")
+
+    @property
+    def observation_delta_indices(self) -> list:
+        return list(range(self.horizon + 1))
+
+    @property
+    def action_delta_indices(self) -> list:
+        return list(range(self.horizon))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return list(range(self.horizon))
diff --git a/lerobot/src/lerobot/policies/tdmpc/modeling_tdmpc.py b/lerobot/src/lerobot/policies/tdmpc/modeling_tdmpc.py
new file mode 100644
index 0000000000000000000000000000000000000000..f83c82e2187e59b4a5710d0bd498b439faf5120c
--- /dev/null
+++ b/lerobot/src/lerobot/policies/tdmpc/modeling_tdmpc.py
@@ -0,0 +1,830 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Nicklas Hansen, Xiaolong Wang, Hao Su,
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Implementation of Finetuning Offline World Models in the Real World.
+
+The comments in this code may sometimes refer to these references:
+    TD-MPC paper: Temporal Difference Learning for Model Predictive Control (https://huggingface.co/papers/2203.04955)
+    FOWM paper: Finetuning Offline World Models in the Real World (https://huggingface.co/papers/2310.16029)
+"""
+
+# ruff: noqa: N806
+
+from collections import deque
+from collections.abc import Callable
+from copy import deepcopy
+from functools import partial
+
+import einops
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.nn.functional as F  # noqa: N812
+from torch import Tensor
+
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.tdmpc.configuration_tdmpc import TDMPCConfig
+from lerobot.policies.utils import get_device_from_parameters, get_output_shape, populate_queues
+from lerobot.utils.constants import ACTION, OBS_ENV_STATE, OBS_IMAGE, OBS_PREFIX, OBS_STATE, OBS_STR, REWARD
+
+
+class TDMPCPolicy(PreTrainedPolicy):
+    """Implementation of TD-MPC learning + inference.
+
+    Please note several warnings for this policy.
+        - Evaluation of pretrained weights created with the original FOWM code
+            (https://github.com/fyhMer/fowm) works as expected. To be precise: we trained and evaluated a
+            model with the FOWM code for the xarm_lift_medium_replay dataset. We ported the weights across
+            to LeRobot, and were able to evaluate with the same success metric. BUT, we had to use inter-
+            process communication to use the xarm environment from FOWM. This is because our xarm
+            environment uses newer dependencies and does not match the environment in FOWM. See
+            https://github.com/huggingface/lerobot/pull/103 for implementation details.
+        - We have NOT checked that training on LeRobot reproduces the results from FOWM.
+        - Nevertheless, we have verified that we can train TD-MPC for PushT. See
+          `lerobot/configs/policy/tdmpc_pusht_keypoints.yaml`.
+        - Our current xarm datasets were generated using the environment from FOWM. Therefore they do not
+          match our xarm environment.
+    """
+
+    config_class = TDMPCConfig
+    name = "tdmpc"
+
+    def __init__(
+        self,
+        config: TDMPCConfig,
+        **kwargs,
+    ):
+        """
+        Args:
+            config: Policy configuration class instance or None, in which case the default instantiation of
+                the configuration class is used.
+        """
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        self.model = TDMPCTOLD(config)
+        self.model_target = deepcopy(self.model)
+        for param in self.model_target.parameters():
+            param.requires_grad = False
+
+        self.reset()
+
+    def get_optim_params(self) -> dict:
+        return self.parameters()
+
+    def reset(self):
+        """
+        Clear observation and action queues. Clear previous means for warm starting of MPPI/CEM. Should be
+        called on `env.reset()`
+        """
+        self._queues = {
+            OBS_STATE: deque(maxlen=1),
+            ACTION: deque(maxlen=max(self.config.n_action_steps, self.config.n_action_repeats)),
+        }
+        if self.config.image_features:
+            self._queues[OBS_IMAGE] = deque(maxlen=1)
+        if self.config.env_state_feature:
+            self._queues[OBS_ENV_STATE] = deque(maxlen=1)
+        # Previous mean obtained from the cross-entropy method (CEM) used during MPC. It is used to warm start
+        # CEM for the next step.
+        self._prev_mean: torch.Tensor | None = None
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor]) -> Tensor:
+        """Predict a chunk of actions given environment observations."""
+        batch = {key: torch.stack(list(self._queues[key]), dim=1) for key in batch if key in self._queues}
+
+        # Remove the time dimensions as it is not handled yet.
+        for key in batch:
+            assert batch[key].shape[1] == 1
+            batch[key] = batch[key][:, 0]
+
+        # NOTE: Order of observations matters here.
+        encode_keys = []
+        if self.config.image_features:
+            encode_keys.append(OBS_IMAGE)
+        if self.config.env_state_feature:
+            encode_keys.append(OBS_ENV_STATE)
+        encode_keys.append(OBS_STATE)
+        z = self.model.encode({k: batch[k] for k in encode_keys})
+        if self.config.use_mpc:  # noqa: SIM108
+            actions = self.plan(z)  # (horizon, batch, action_dim)
+        else:
+            # Plan with the policy (π) alone. This always returns one action so unsqueeze to get a
+            # sequence dimension like in the MPC branch.
+            actions = self.model.pi(z).unsqueeze(0)
+
+        actions = torch.clamp(actions, -1, +1)
+
+        return actions
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select a single action given environment observations."""
+        # NOTE: for offline evaluation, we have action in the batch, so we need to pop it out
+        if ACTION in batch:
+            batch.pop(ACTION)
+
+        if self.config.image_features:
+            batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
+            batch[OBS_IMAGE] = batch[next(iter(self.config.image_features))]
+        # NOTE: for offline evaluation, we have action in the batch, so we need to pop it out
+        if ACTION in batch:
+            batch.pop(ACTION)
+
+        self._queues = populate_queues(self._queues, batch)
+
+        # When the action queue is depleted, populate it again by querying the policy.
+        if len(self._queues[ACTION]) == 0:
+            actions = self.predict_action_chunk(batch)
+
+            if self.config.n_action_repeats > 1:
+                for _ in range(self.config.n_action_repeats):
+                    self._queues[ACTION].append(actions[0])
+            else:
+                # Action queue is (n_action_steps, batch_size, action_dim), so we transpose the action.
+                self._queues[ACTION].extend(actions[: self.config.n_action_steps])
+
+        action = self._queues[ACTION].popleft()
+        return action
+
+    @torch.no_grad()
+    def plan(self, z: Tensor) -> Tensor:
+        """Plan sequence of actions using TD-MPC inference.
+
+        Args:
+            z: (batch, latent_dim,) tensor for the initial state.
+        Returns:
+            (horizon, batch, action_dim,) tensor for the planned trajectory of actions.
+        """
+        device = get_device_from_parameters(self)
+
+        batch_size = z.shape[0]
+
+        # Sample Nπ trajectories from the policy.
+        pi_actions = torch.empty(
+            self.config.horizon,
+            self.config.n_pi_samples,
+            batch_size,
+            self.config.action_feature.shape[0],
+            device=device,
+        )
+        if self.config.n_pi_samples > 0:
+            _z = einops.repeat(z, "b d -> n b d", n=self.config.n_pi_samples)
+            for t in range(self.config.horizon):
+                # Note: Adding a small amount of noise here doesn't hurt during inference and may even be
+                # helpful for CEM.
+                pi_actions[t] = self.model.pi(_z, self.config.min_std)
+                _z = self.model.latent_dynamics(_z, pi_actions[t])
+
+        # In the CEM loop we will need this for a call to estimate_value with the gaussian sampled
+        # trajectories.
+        z = einops.repeat(z, "b d -> n b d", n=self.config.n_gaussian_samples + self.config.n_pi_samples)
+
+        # Model Predictive Path Integral (MPPI) with the cross-entropy method (CEM) as the optimization
+        # algorithm.
+        # The initial mean and standard deviation for the cross-entropy method (CEM).
+        mean = torch.zeros(
+            self.config.horizon, batch_size, self.config.action_feature.shape[0], device=device
+        )
+        # Maybe warm start CEM with the mean from the previous step.
+        if self._prev_mean is not None:
+            mean[:-1] = self._prev_mean[1:]
+        std = self.config.max_std * torch.ones_like(mean)
+
+        for _ in range(self.config.cem_iterations):
+            # Randomly sample action trajectories for the gaussian distribution.
+            std_normal_noise = torch.randn(
+                self.config.horizon,
+                self.config.n_gaussian_samples,
+                batch_size,
+                self.config.action_feature.shape[0],
+                device=std.device,
+            )
+            gaussian_actions = torch.clamp(mean.unsqueeze(1) + std.unsqueeze(1) * std_normal_noise, -1, 1)
+
+            # Compute elite actions.
+            actions = torch.cat([gaussian_actions, pi_actions], dim=1)
+            value = self.estimate_value(z, actions).nan_to_num_(0)
+            elite_idxs = torch.topk(value, self.config.n_elites, dim=0).indices  # (n_elites, batch)
+            elite_value = value.take_along_dim(elite_idxs, dim=0)  # (n_elites, batch)
+            # (horizon, n_elites, batch, action_dim)
+            elite_actions = actions.take_along_dim(einops.rearrange(elite_idxs, "n b -> 1 n b 1"), dim=1)
+
+            # Update gaussian PDF parameters to be the (weighted) mean and standard deviation of the elites.
+            max_value = elite_value.max(0, keepdim=True)[0]  # (1, batch)
+            # The weighting is a softmax over trajectory values. Note that this is not the same as the usage
+            # of Ω in eqn 4 of the TD-MPC paper. Instead it is the normalized version of it: s = Ω/ΣΩ. This
+            # makes the equations: μ = Σ(s⋅Γ), σ = Σ(s⋅(Γ-μ)²).
+            score = torch.exp(self.config.elite_weighting_temperature * (elite_value - max_value))
+            score /= score.sum(axis=0, keepdim=True)
+            # (horizon, batch, action_dim)
+            _mean = torch.sum(einops.rearrange(score, "n b -> n b 1") * elite_actions, dim=1)
+            _std = torch.sqrt(
+                torch.sum(
+                    einops.rearrange(score, "n b -> n b 1")
+                    * (elite_actions - einops.rearrange(_mean, "h b d -> h 1 b d")) ** 2,
+                    dim=1,
+                )
+            )
+            # Update mean with an exponential moving average, and std with a direct replacement.
+            mean = (
+                self.config.gaussian_mean_momentum * mean + (1 - self.config.gaussian_mean_momentum) * _mean
+            )
+            std = _std.clamp_(self.config.min_std, self.config.max_std)
+
+        # Keep track of the mean for warm-starting subsequent steps.
+        self._prev_mean = mean
+
+        # Randomly select one of the elite actions from the last iteration of MPPI/CEM using the softmax
+        # scores from the last iteration.
+        actions = elite_actions[:, torch.multinomial(score.T, 1).squeeze(), torch.arange(batch_size)]
+
+        return actions
+
+    @torch.no_grad()
+    def estimate_value(self, z: Tensor, actions: Tensor):
+        """Estimates the value of a trajectory as per eqn 4 of the FOWM paper.
+
+        Args:
+            z: (batch, latent_dim) tensor of initial latent states.
+            actions: (horizon, batch, action_dim) tensor of action trajectories.
+        Returns:
+            (batch,) tensor of values.
+        """
+        # Initialize return and running discount factor.
+        G, running_discount = 0, 1
+        # Iterate over the actions in the trajectory to simulate the trajectory using the latent dynamics
+        # model. Keep track of return.
+        for t in range(actions.shape[0]):
+            # We will compute the reward in a moment. First compute the uncertainty regularizer from eqn 4
+            # of the FOWM paper.
+            if self.config.uncertainty_regularizer_coeff > 0:
+                regularization = -(
+                    self.config.uncertainty_regularizer_coeff * self.model.Qs(z, actions[t]).std(0)
+                )
+            else:
+                regularization = 0
+            # Estimate the next state (latent) and reward.
+            z, reward = self.model.latent_dynamics_and_reward(z, actions[t])
+            # Update the return and running discount.
+            G += running_discount * (reward + regularization)
+            running_discount *= self.config.discount
+        # Add the estimated value of the final state (using the minimum for a conservative estimate).
+        # Do so by predicting the next action, then taking a minimum over the ensemble of state-action value
+        # estimators.
+        # Note: This small amount of added noise seems to help a bit at inference time as observed by success
+        # metrics over 50 episodes of xarm_lift_medium_replay.
+        next_action = self.model.pi(z, self.config.min_std)  # (batch, action_dim)
+        terminal_values = self.model.Qs(z, next_action)  # (ensemble, batch)
+        # Randomly choose 2 of the Qs for terminal value estimation (as in App C. of the FOWM paper).
+        if self.config.q_ensemble_size > 2:
+            G += (
+                running_discount
+                * torch.min(terminal_values[torch.randint(0, self.config.q_ensemble_size, size=(2,))], dim=0)[
+                    0
+                ]
+            )
+        else:
+            G += running_discount * torch.min(terminal_values, dim=0)[0]
+        # Finally, also regularize the terminal value.
+        if self.config.uncertainty_regularizer_coeff > 0:
+            G -= running_discount * self.config.uncertainty_regularizer_coeff * terminal_values.std(0)
+        return G
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, dict]:
+        """Run the batch through the model and compute the loss.
+
+        Returns a dictionary with loss as a tensor, and other information as native floats.
+        """
+        device = get_device_from_parameters(self)
+
+        if self.config.image_features:
+            batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
+            batch[OBS_IMAGE] = batch[next(iter(self.config.image_features))]
+
+        info = {}
+
+        # (b, t) -> (t, b)
+        for key in batch:
+            if isinstance(batch[key], torch.Tensor) and batch[key].ndim > 1:
+                batch[key] = batch[key].transpose(1, 0)
+
+        action = batch[ACTION]  # (t, b, action_dim)
+        reward = batch[REWARD]  # (t, b)
+        observations = {k: v for k, v in batch.items() if k.startswith(OBS_PREFIX)}
+
+        # Apply random image augmentations.
+        if self.config.image_features and self.config.max_random_shift_ratio > 0:
+            observations[OBS_IMAGE] = flatten_forward_unflatten(
+                partial(random_shifts_aug, max_random_shift_ratio=self.config.max_random_shift_ratio),
+                observations[OBS_IMAGE],
+            )
+
+        # Get the current observation for predicting trajectories, and all future observations for use in
+        # the latent consistency loss and TD loss.
+        current_observation, next_observations = {}, {}
+        for k in observations:
+            current_observation[k] = observations[k][0]
+            next_observations[k] = observations[k][1:]
+        horizon, batch_size = next_observations[
+            OBS_IMAGE if self.config.image_features else OBS_ENV_STATE
+        ].shape[:2]
+
+        # Run latent rollout using the latent dynamics model and policy model.
+        # Note this has shape `horizon+1` because there are `horizon` actions and a current `z`. Each action
+        # gives us a next `z`.
+        batch_size = batch["index"].shape[0]
+        z_preds = torch.empty(horizon + 1, batch_size, self.config.latent_dim, device=device)
+        z_preds[0] = self.model.encode(current_observation)
+        reward_preds = torch.empty_like(reward, device=device)
+        for t in range(horizon):
+            z_preds[t + 1], reward_preds[t] = self.model.latent_dynamics_and_reward(z_preds[t], action[t])
+
+        # Compute Q and V value predictions based on the latent rollout.
+        q_preds_ensemble = self.model.Qs(z_preds[:-1], action)  # (ensemble, horizon, batch)
+        v_preds = self.model.V(z_preds[:-1])
+        info.update({"Q": q_preds_ensemble.mean().item(), "V": v_preds.mean().item()})
+
+        # Compute various targets with stopgrad.
+        with torch.no_grad():
+            # Latent state consistency targets.
+            z_targets = self.model_target.encode(next_observations)
+            # State-action value targets (or TD targets) as in eqn 3 of the FOWM. Unlike TD-MPC which uses the
+            # learned state-action value function in conjunction with the learned policy: Q(z, π(z)), FOWM
+            # uses a learned state value function: V(z). This means the TD targets only depend on in-sample
+            # actions (not actions estimated by π).
+            # Note: Here we do not use self.model_target, but self.model. This is to follow the original code
+            # and the FOWM paper.
+            q_targets = reward + self.config.discount * self.model.V(self.model.encode(next_observations))
+            # From eqn 3 of FOWM. These appear as Q(z, a). Here we call them v_targets to emphasize that we
+            # are using them to compute loss for V.
+            v_targets = self.model_target.Qs(z_preds[:-1].detach(), action, return_min=True)
+
+        # Compute losses.
+        # Exponentially decay the loss weight with respect to the timestep. Steps that are more distant in the
+        # future have less impact on the loss. Note: unsqueeze will let us broadcast to (seq, batch).
+        temporal_loss_coeffs = torch.pow(
+            self.config.temporal_decay_coeff, torch.arange(horizon, device=device)
+        ).unsqueeze(-1)
+        # Compute consistency loss as MSE loss between latents predicted from the rollout and latents
+        # predicted from the (target model's) observation encoder.
+        consistency_loss = (
+            (
+                temporal_loss_coeffs
+                * F.mse_loss(z_preds[1:], z_targets, reduction="none").mean(dim=-1)
+                # `z_preds` depends on the current observation and the actions.
+                * ~batch[f"{OBS_STR}.state_is_pad"][0]
+                * ~batch["action_is_pad"]
+                # `z_targets` depends on the next observation.
+                * ~batch[f"{OBS_STR}.state_is_pad"][1:]
+            )
+            .sum(0)
+            .mean()
+        )
+        # Compute the reward loss as MSE loss between rewards predicted from the rollout and the dataset
+        # rewards.
+        reward_loss = (
+            (
+                temporal_loss_coeffs
+                * F.mse_loss(reward_preds, reward, reduction="none")
+                * ~batch["next.reward_is_pad"]
+                # `reward_preds` depends on the current observation and the actions.
+                * ~batch[f"{OBS_STR}.state_is_pad"][0]
+                * ~batch["action_is_pad"]
+            )
+            .sum(0)
+            .mean()
+        )
+        # Compute state-action value loss (TD loss) for all of the Q functions in the ensemble.
+        q_value_loss = (
+            (
+                temporal_loss_coeffs
+                * F.mse_loss(
+                    q_preds_ensemble,
+                    einops.repeat(q_targets, "t b -> e t b", e=q_preds_ensemble.shape[0]),
+                    reduction="none",
+                ).sum(0)  # sum over ensemble
+                # `q_preds_ensemble` depends on the first observation and the actions.
+                * ~batch[f"{OBS_STR}.state_is_pad"][0]
+                * ~batch["action_is_pad"]
+                # q_targets depends on the reward and the next observations.
+                * ~batch["next.reward_is_pad"]
+                * ~batch[f"{OBS_STR}.state_is_pad"][1:]
+            )
+            .sum(0)
+            .mean()
+        )
+        # Compute state value loss as in eqn 3 of FOWM.
+        diff = v_targets - v_preds
+        # Expectile loss penalizes:
+        #   - `v_preds <  v_targets` with weighting `expectile_weight`
+        #   - `v_preds >= v_targets` with weighting `1 - expectile_weight`
+        raw_v_value_loss = torch.where(
+            diff > 0, self.config.expectile_weight, (1 - self.config.expectile_weight)
+        ) * (diff**2)
+        v_value_loss = (
+            (
+                temporal_loss_coeffs
+                * raw_v_value_loss
+                # `v_targets` depends on the first observation and the actions, as does `v_preds`.
+                * ~batch[f"{OBS_STR}.state_is_pad"][0]
+                * ~batch["action_is_pad"]
+            )
+            .sum(0)
+            .mean()
+        )
+
+        # Calculate the advantage weighted regression loss for π as detailed in FOWM 3.1.
+        # We won't need these gradients again so detach.
+        z_preds = z_preds.detach()
+        # Use stopgrad for the advantage calculation.
+        with torch.no_grad():
+            advantage = self.model_target.Qs(z_preds[:-1], action, return_min=True) - self.model.V(
+                z_preds[:-1]
+            )
+            info["advantage"] = advantage[0]
+            # (t, b)
+            exp_advantage = torch.clamp(torch.exp(advantage * self.config.advantage_scaling), max=100.0)
+        action_preds = self.model.pi(z_preds[:-1])  # (t, b, a)
+        # Calculate the MSE between the actions and the action predictions.
+        # Note: FOWM's original code calculates the log probability (wrt to a unit standard deviation
+        # gaussian) and sums over the action dimension. Computing the (negative) log probability amounts to
+        # multiplying the MSE by 0.5 and adding a constant offset (the log(2*pi)/2 term, times the action
+        # dimension). Here we drop the constant offset as it doesn't change the optimization step, and we drop
+        # the 0.5 as we instead make a configuration parameter for it (see below where we compute the total
+        # loss).
+        mse = F.mse_loss(action_preds, action, reduction="none").sum(-1)  # (t, b)
+        # NOTE: The original implementation does not take the sum over the temporal dimension like with the
+        # other losses.
+        # TODO(alexander-soare): Take the sum over the temporal dimension and check that training still works
+        # as well as expected.
+        pi_loss = (
+            exp_advantage
+            * mse
+            * temporal_loss_coeffs
+            # `action_preds` depends on the first observation and the actions.
+            * ~batch[f"{OBS_STR}.state_is_pad"][0]
+            * ~batch["action_is_pad"]
+        ).mean()
+
+        loss = (
+            self.config.consistency_coeff * consistency_loss
+            + self.config.reward_coeff * reward_loss
+            + self.config.value_coeff * q_value_loss
+            + self.config.value_coeff * v_value_loss
+            + self.config.pi_coeff * pi_loss
+        )
+
+        info.update(
+            {
+                "consistency_loss": consistency_loss.item(),
+                "reward_loss": reward_loss.item(),
+                "Q_value_loss": q_value_loss.item(),
+                "V_value_loss": v_value_loss.item(),
+                "pi_loss": pi_loss.item(),
+                "sum_loss": loss.item() * self.config.horizon,
+            }
+        )
+
+        # Undo (b, t) -> (t, b).
+        for key in batch:
+            if isinstance(batch[key], torch.Tensor) and batch[key].ndim > 1:
+                batch[key] = batch[key].transpose(1, 0)
+
+        return loss, info
+
+    def update(self):
+        """Update the target model's parameters with an EMA step."""
+        # Note a minor variation with respect to the original FOWM code. Here they do this based on an EMA
+        # update frequency parameter which is set to 2 (every 2 steps an update is done). To simplify the code
+        # we update every step and adjust the decay parameter `alpha` accordingly (0.99 -> 0.995)
+        update_ema_parameters(self.model_target, self.model, self.config.target_model_momentum)
+
+
+class TDMPCTOLD(nn.Module):
+    """Task-Oriented Latent Dynamics (TOLD) model used in TD-MPC."""
+
+    def __init__(self, config: TDMPCConfig):
+        super().__init__()
+        self.config = config
+        self._encoder = TDMPCObservationEncoder(config)
+        self._dynamics = nn.Sequential(
+            nn.Linear(config.latent_dim + config.action_feature.shape[0], config.mlp_dim),
+            nn.LayerNorm(config.mlp_dim),
+            nn.Mish(),
+            nn.Linear(config.mlp_dim, config.mlp_dim),
+            nn.LayerNorm(config.mlp_dim),
+            nn.Mish(),
+            nn.Linear(config.mlp_dim, config.latent_dim),
+            nn.LayerNorm(config.latent_dim),
+            nn.Sigmoid(),
+        )
+        self._reward = nn.Sequential(
+            nn.Linear(config.latent_dim + config.action_feature.shape[0], config.mlp_dim),
+            nn.LayerNorm(config.mlp_dim),
+            nn.Mish(),
+            nn.Linear(config.mlp_dim, config.mlp_dim),
+            nn.LayerNorm(config.mlp_dim),
+            nn.Mish(),
+            nn.Linear(config.mlp_dim, 1),
+        )
+        self._pi = nn.Sequential(
+            nn.Linear(config.latent_dim, config.mlp_dim),
+            nn.LayerNorm(config.mlp_dim),
+            nn.Mish(),
+            nn.Linear(config.mlp_dim, config.mlp_dim),
+            nn.LayerNorm(config.mlp_dim),
+            nn.Mish(),
+            nn.Linear(config.mlp_dim, config.action_feature.shape[0]),
+        )
+        self._Qs = nn.ModuleList(
+            [
+                nn.Sequential(
+                    nn.Linear(config.latent_dim + config.action_feature.shape[0], config.mlp_dim),
+                    nn.LayerNorm(config.mlp_dim),
+                    nn.Tanh(),
+                    nn.Linear(config.mlp_dim, config.mlp_dim),
+                    nn.ELU(),
+                    nn.Linear(config.mlp_dim, 1),
+                )
+                for _ in range(config.q_ensemble_size)
+            ]
+        )
+        self._V = nn.Sequential(
+            nn.Linear(config.latent_dim, config.mlp_dim),
+            nn.LayerNorm(config.mlp_dim),
+            nn.Tanh(),
+            nn.Linear(config.mlp_dim, config.mlp_dim),
+            nn.ELU(),
+            nn.Linear(config.mlp_dim, 1),
+        )
+        self._init_weights()
+
+    def _init_weights(self):
+        """Initialize model weights.
+
+        Orthogonal initialization for all linear and convolutional layers' weights (apart from final layers
+        of reward network and Q networks which get zero initialization).
+        Zero initialization for all linear and convolutional layers' biases.
+        """
+
+        def _apply_fn(m):
+            if isinstance(m, nn.Linear):
+                nn.init.orthogonal_(m.weight.data)
+                if m.bias is not None:
+                    nn.init.zeros_(m.bias)
+            elif isinstance(m, nn.Conv2d):
+                gain = nn.init.calculate_gain("relu")
+                nn.init.orthogonal_(m.weight.data, gain)
+                if m.bias is not None:
+                    nn.init.zeros_(m.bias)
+
+        self.apply(_apply_fn)
+        for m in [self._reward, *self._Qs]:
+            assert isinstance(m[-1], nn.Linear), (
+                "Sanity check. The last linear layer needs 0 initialization on weights."
+            )
+            nn.init.zeros_(m[-1].weight)
+            nn.init.zeros_(m[-1].bias)  # this has already been done, but keep this line here for good measure
+
+    def encode(self, obs: dict[str, Tensor]) -> Tensor:
+        """Encodes an observation into its latent representation."""
+        return self._encoder(obs)
+
+    def latent_dynamics_and_reward(self, z: Tensor, a: Tensor) -> tuple[Tensor, Tensor]:
+        """Predict the next state's latent representation and the reward given a current latent and action.
+
+        Args:
+            z: (*, latent_dim) tensor for the current state's latent representation.
+            a: (*, action_dim) tensor for the action to be applied.
+        Returns:
+            A tuple containing:
+                - (*, latent_dim) tensor for the next state's latent representation.
+                - (*,) tensor for the estimated reward.
+        """
+        x = torch.cat([z, a], dim=-1)
+        return self._dynamics(x), self._reward(x).squeeze(-1)
+
+    def latent_dynamics(self, z: Tensor, a: Tensor) -> Tensor:
+        """Predict the next state's latent representation given a current latent and action.
+
+        Args:
+            z: (*, latent_dim) tensor for the current state's latent representation.
+            a: (*, action_dim) tensor for the action to be applied.
+        Returns:
+            (*, latent_dim) tensor for the next state's latent representation.
+        """
+        x = torch.cat([z, a], dim=-1)
+        return self._dynamics(x)
+
+    def pi(self, z: Tensor, std: float = 0.0) -> Tensor:
+        """Samples an action from the learned policy.
+
+        The policy can also have added (truncated) Gaussian noise injected for encouraging exploration when
+        generating rollouts for online training.
+
+        Args:
+            z: (*, latent_dim) tensor for the current state's latent representation.
+            std: The standard deviation of the injected noise.
+        Returns:
+            (*, action_dim) tensor for the sampled action.
+        """
+        action = torch.tanh(self._pi(z))
+        if std > 0:
+            std = torch.ones_like(action) * std
+            action += torch.randn_like(action) * std
+        return action
+
+    def V(self, z: Tensor) -> Tensor:  # noqa: N802
+        """Predict state value (V).
+
+        Args:
+            z: (*, latent_dim) tensor for the current state's latent representation.
+        Returns:
+            (*,) tensor of estimated state values.
+        """
+        return self._V(z).squeeze(-1)
+
+    def Qs(self, z: Tensor, a: Tensor, return_min: bool = False) -> Tensor:  # noqa: N802
+        """Predict state-action value for all of the learned Q functions.
+
+        Args:
+            z: (*, latent_dim) tensor for the current state's latent representation.
+            a: (*, action_dim) tensor for the action to be applied.
+            return_min: Set to true for implementing the detail in App. C of the FOWM paper: randomly select
+                2 of the Qs and return the minimum
+        Returns:
+            (q_ensemble, *) tensor for the value predictions of each learned Q function in the ensemble OR
+            (*,) tensor if return_min=True.
+        """
+        x = torch.cat([z, a], dim=-1)
+        if not return_min:
+            return torch.stack([q(x).squeeze(-1) for q in self._Qs], dim=0)
+        else:
+            if len(self._Qs) > 2:  # noqa: SIM108
+                Qs = [self._Qs[i] for i in np.random.choice(len(self._Qs), size=2)]
+            else:
+                Qs = self._Qs
+            return torch.stack([q(x).squeeze(-1) for q in Qs], dim=0).min(dim=0)[0]
+
+
+class TDMPCObservationEncoder(nn.Module):
+    """Encode image and/or state vector observations."""
+
+    def __init__(self, config: TDMPCConfig):
+        """
+        Creates encoders for pixel and/or state modalities.
+        TODO(alexander-soare): The original work allows for multiple images by concatenating them along the
+            channel dimension. Re-implement this capability.
+        """
+        super().__init__()
+        self.config = config
+
+        if config.image_features:
+            self.image_enc_layers = nn.Sequential(
+                nn.Conv2d(
+                    next(iter(config.image_features.values())).shape[0],
+                    config.image_encoder_hidden_dim,
+                    7,
+                    stride=2,
+                ),
+                nn.ReLU(),
+                nn.Conv2d(config.image_encoder_hidden_dim, config.image_encoder_hidden_dim, 5, stride=2),
+                nn.ReLU(),
+                nn.Conv2d(config.image_encoder_hidden_dim, config.image_encoder_hidden_dim, 3, stride=2),
+                nn.ReLU(),
+                nn.Conv2d(config.image_encoder_hidden_dim, config.image_encoder_hidden_dim, 3, stride=2),
+                nn.ReLU(),
+            )
+            dummy_shape = (1, *next(iter(config.image_features.values())).shape)
+            out_shape = get_output_shape(self.image_enc_layers, dummy_shape)[1:]
+            self.image_enc_layers.extend(
+                nn.Sequential(
+                    nn.Flatten(),
+                    nn.Linear(np.prod(out_shape), config.latent_dim),
+                    nn.LayerNorm(config.latent_dim),
+                    nn.Sigmoid(),
+                )
+            )
+
+        if config.robot_state_feature:
+            self.state_enc_layers = nn.Sequential(
+                nn.Linear(config.robot_state_feature.shape[0], config.state_encoder_hidden_dim),
+                nn.ELU(),
+                nn.Linear(config.state_encoder_hidden_dim, config.latent_dim),
+                nn.LayerNorm(config.latent_dim),
+                nn.Sigmoid(),
+            )
+
+        if config.env_state_feature:
+            self.env_state_enc_layers = nn.Sequential(
+                nn.Linear(config.env_state_feature.shape[0], config.state_encoder_hidden_dim),
+                nn.ELU(),
+                nn.Linear(config.state_encoder_hidden_dim, config.latent_dim),
+                nn.LayerNorm(config.latent_dim),
+                nn.Sigmoid(),
+            )
+
+    def forward(self, obs_dict: dict[str, Tensor]) -> Tensor:
+        """Encode the image and/or state vector.
+
+        Each modality is encoded into a feature vector of size (latent_dim,) and then a uniform mean is taken
+        over all features.
+        """
+        feat = []
+        # NOTE: Order of observations matters here.
+        if self.config.image_features:
+            feat.append(
+                flatten_forward_unflatten(
+                    self.image_enc_layers, obs_dict[next(iter(self.config.image_features))]
+                )
+            )
+        if self.config.env_state_feature:
+            feat.append(self.env_state_enc_layers(obs_dict[OBS_ENV_STATE]))
+        if self.config.robot_state_feature:
+            feat.append(self.state_enc_layers(obs_dict[OBS_STATE]))
+        return torch.stack(feat, dim=0).mean(0)
+
+
+def random_shifts_aug(x: Tensor, max_random_shift_ratio: float) -> Tensor:
+    """Randomly shifts images horizontally and vertically.
+
+    Adapted from https://github.com/facebookresearch/drqv2
+    """
+    b, _, h, w = x.size()
+    assert h == w, "non-square images not handled yet"
+    pad = int(round(max_random_shift_ratio * h))
+    x = F.pad(x, tuple([pad] * 4), "replicate")
+    eps = 1.0 / (h + 2 * pad)
+    arange = torch.linspace(
+        -1.0 + eps,
+        1.0 - eps,
+        h + 2 * pad,
+        device=x.device,
+        dtype=torch.float32,
+    )[:h]
+    arange = einops.repeat(arange, "w -> h w 1", h=h)
+    base_grid = torch.cat([arange, arange.transpose(1, 0)], dim=2)
+    base_grid = einops.repeat(base_grid, "h w c -> b h w c", b=b)
+    # A random shift in units of pixels and within the boundaries of the padding.
+    shift = torch.randint(
+        0,
+        2 * pad + 1,
+        size=(b, 1, 1, 2),
+        device=x.device,
+        dtype=torch.float32,
+    )
+    shift *= 2.0 / (h + 2 * pad)
+    grid = base_grid + shift
+    return F.grid_sample(x, grid, padding_mode="zeros", align_corners=False)
+
+
+def update_ema_parameters(ema_net: nn.Module, net: nn.Module, alpha: float):
+    """Update EMA parameters in place with ema_param <- alpha * ema_param + (1 - alpha) * param."""
+    for ema_module, module in zip(ema_net.modules(), net.modules(), strict=True):
+        for (n_p_ema, p_ema), (n_p, p) in zip(
+            ema_module.named_parameters(recurse=False), module.named_parameters(recurse=False), strict=True
+        ):
+            assert n_p_ema == n_p, "Parameter names don't match for EMA model update"
+            if isinstance(p, dict):
+                raise RuntimeError("Dict parameter not supported")
+            if isinstance(module, nn.modules.batchnorm._BatchNorm) or not p.requires_grad:
+                # Copy BatchNorm parameters, and non-trainable parameters directly.
+                p_ema.copy_(p.to(dtype=p_ema.dtype).data)
+            with torch.no_grad():
+                p_ema.mul_(alpha)
+                p_ema.add_(p.to(dtype=p_ema.dtype).data, alpha=1 - alpha)
+
+
+def flatten_forward_unflatten(fn: Callable[[Tensor], Tensor], image_tensor: Tensor) -> Tensor:
+    """Helper to temporarily flatten extra dims at the start of the image tensor.
+
+    Args:
+        fn: Callable that the image tensor will be passed to. It should accept (B, C, H, W) and return
+            (B, *), where * is any number of dimensions.
+        image_tensor: An image tensor of shape (**, C, H, W), where ** is any number of dimensions, generally
+            different from *.
+    Returns:
+        A return value from the callable reshaped to (**, *).
+    """
+    if image_tensor.ndim == 4:
+        return fn(image_tensor)
+    start_dims = image_tensor.shape[:-3]
+    inp = torch.flatten(image_tensor, end_dim=-4)
+    flat_out = fn(inp)
+    return torch.reshape(flat_out, (*start_dims, *flat_out.shape[1:]))
diff --git a/lerobot/src/lerobot/policies/tdmpc/processor_tdmpc.py b/lerobot/src/lerobot/policies/tdmpc/processor_tdmpc.py
new file mode 100644
index 0000000000000000000000000000000000000000..9b6f97e50020a6702104020b19de9d0112ca11ad
--- /dev/null
+++ b/lerobot/src/lerobot/policies/tdmpc/processor_tdmpc.py
@@ -0,0 +1,90 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Nicklas Hansen, Xiaolong Wang, Hao Su,
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from typing import Any
+
+import torch
+
+from lerobot.policies.tdmpc.configuration_tdmpc import TDMPCConfig
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    RenameObservationsProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.utils.constants import POLICY_POSTPROCESSOR_DEFAULT_NAME, POLICY_PREPROCESSOR_DEFAULT_NAME
+
+
+def make_tdmpc_pre_post_processors(
+    config: TDMPCConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for the TDMPC policy.
+
+    The pre-processing pipeline prepares input data for the model by:
+    1. Renaming features to match pretrained configurations.
+    2. Normalizing input and output features based on dataset statistics.
+    3. Adding a batch dimension.
+    4. Moving all data to the specified device.
+
+    The post-processing pipeline handles the model's output by:
+    1. Moving data to the CPU.
+    2. Unnormalizing the output features to their original scale.
+
+    Args:
+        config: The configuration object for the TDMPC policy.
+        dataset_stats: A dictionary of statistics for normalization.
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines.
+    """
+
+    input_steps = [
+        RenameObservationsProcessorStep(rename_map={}),
+        AddBatchDimensionProcessorStep(),
+        DeviceProcessorStep(device=config.device),
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+    ]
+    output_steps = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/utils.py b/lerobot/src/lerobot/policies/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..82ab510056c3a40c6a1b5d77894698fe8eb33061
--- /dev/null
+++ b/lerobot/src/lerobot/policies/utils.py
@@ -0,0 +1,249 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+from collections import deque
+
+import numpy as np
+import torch
+from torch import nn
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.datasets.feature_utils import build_dataset_frame
+from lerobot.types import PolicyAction, RobotAction, RobotObservation
+from lerobot.utils.constants import ACTION, OBS_STR
+
+
+def populate_queues(
+    queues: dict[str, deque], batch: dict[str, torch.Tensor], exclude_keys: list[str] | None = None
+):
+    if exclude_keys is None:
+        exclude_keys = []
+    for key in batch:
+        # Ignore keys not in the queues already (leaving the responsibility to the caller to make sure the
+        # queues have the keys they want).
+        if key not in queues or key in exclude_keys:
+            continue
+        if len(queues[key]) != queues[key].maxlen:
+            # initialize by copying the first observation several times until the queue is full
+            while len(queues[key]) != queues[key].maxlen:
+                queues[key].append(batch[key])
+        else:
+            # add latest observation to the queue
+            queues[key].append(batch[key])
+    return queues
+
+
+def get_device_from_parameters(module: nn.Module) -> torch.device:
+    """Get a module's device by checking one of its parameters.
+
+    Note: assumes that all parameters have the same device
+    """
+    return next(iter(module.parameters())).device
+
+
+def get_dtype_from_parameters(module: nn.Module) -> torch.dtype:
+    """Get a module's parameter dtype by checking one of its parameters.
+
+    Note: assumes that all parameters have the same dtype.
+    """
+    return next(iter(module.parameters())).dtype
+
+
+def get_output_shape(module: nn.Module, input_shape: tuple) -> tuple:
+    """
+    Calculates the output shape of a PyTorch module given an input shape.
+
+    Args:
+        module (nn.Module): a PyTorch module
+        input_shape (tuple): A tuple representing the input shape, e.g., (batch_size, channels, height, width)
+
+    Returns:
+        tuple: The output shape of the module.
+    """
+    dummy_input = torch.zeros(size=input_shape)
+    with torch.inference_mode():
+        output = module(dummy_input)
+    return tuple(output.shape)
+
+
+def log_model_loading_keys(missing_keys: list[str], unexpected_keys: list[str]) -> None:
+    """Log missing and unexpected keys when loading a model.
+
+    Args:
+        missing_keys (list[str]): Keys that were expected but not found.
+        unexpected_keys (list[str]): Keys that were found but not expected.
+    """
+    if missing_keys:
+        logging.warning(f"Missing key(s) when loading model: {missing_keys}")
+    if unexpected_keys:
+        logging.warning(f"Unexpected key(s) when loading model: {unexpected_keys}")
+
+
+# TODO(Steven): Move this function to a proper preprocessor step
+def prepare_observation_for_inference(
+    observation: dict[str, np.ndarray],
+    device: torch.device,
+    task: str | None = None,
+    robot_type: str | None = None,
+) -> RobotObservation:
+    """Converts observation data to model-ready PyTorch tensors.
+
+    This function takes a dictionary of NumPy arrays, performs necessary
+    preprocessing, and prepares it for model inference. The steps include:
+    1. Converting NumPy arrays to PyTorch tensors.
+    2. Normalizing and permuting image data (if any).
+    3. Adding a batch dimension to each tensor.
+    4. Moving all tensors to the specified compute device.
+    5. Adding task and robot type information to the dictionary.
+
+    Args:
+        observation: A dictionary mapping observation names (str) to NumPy
+            array data. For images, the format is expected to be (H, W, C).
+        device: The PyTorch device (e.g., 'cpu' or 'cuda') to which the
+            tensors will be moved.
+        task: An optional string identifier for the current task.
+        robot_type: An optional string identifier for the robot being used.
+
+    Returns:
+        A dictionary where values are PyTorch tensors preprocessed for
+        inference, residing on the target device. Image tensors are reshaped
+        to (C, H, W) and normalized to a [0, 1] range.
+    """
+    for name in observation:
+        observation[name] = torch.from_numpy(observation[name])
+        if "image" in name:
+            observation[name] = observation[name].type(torch.float32) / 255
+            observation[name] = observation[name].permute(2, 0, 1).contiguous()
+        observation[name] = observation[name].unsqueeze(0)
+        observation[name] = observation[name].to(device)
+
+    observation["task"] = task if task else ""
+    observation["robot_type"] = robot_type if robot_type else ""
+
+    return observation
+
+
+def build_inference_frame(
+    observation: RobotObservation,
+    device: torch.device,
+    ds_features: dict[str, dict],
+    task: str | None = None,
+    robot_type: str | None = None,
+) -> RobotObservation:
+    """Constructs a model-ready observation tensor dict from a raw observation.
+
+    This utility function orchestrates the process of converting a raw,
+    unstructured observation from an environment into a structured,
+    tensor-based format suitable for passing to a policy model.
+
+    Args:
+        observation: The raw observation dictionary, which may contain
+            superfluous keys.
+        device: The target PyTorch device for the final tensors.
+        ds_features: A configuration dictionary that specifies which features
+            to extract from the raw observation.
+        task: An optional string identifier for the current task.
+        robot_type: An optional string identifier for the robot being used.
+
+    Returns:
+        A dictionary of preprocessed tensors ready for model inference.
+    """
+    # Extracts the correct keys from the incoming raw observation
+    observation = build_dataset_frame(ds_features, observation, prefix=OBS_STR)
+
+    # Performs the necessary conversions to the observation
+    observation = prepare_observation_for_inference(observation, device, task, robot_type)
+
+    return observation
+
+
+def make_robot_action(action_tensor: PolicyAction, ds_features: dict[str, dict]) -> RobotAction:
+    """Converts a policy's output tensor into a dictionary of named actions.
+
+    This function translates the numerical output from a policy model into a
+    human-readable and robot-consumable format, where each dimension of the
+    action tensor is mapped to a named motor or actuator command.
+
+    Args:
+        action_tensor: A PyTorch tensor representing the policy's action,
+            typically with a batch dimension (e.g., shape [1, action_dim]).
+        ds_features: A configuration dictionary containing metadata, including
+            the names corresponding to each index of the action tensor.
+
+    Returns:
+        A dictionary mapping action names (e.g., "joint_1_motor") to their
+        corresponding floating-point values, ready to be sent to a robot
+        controller.
+    """
+    # TODO(Steven): Check if these steps are already in all postprocessor policies
+    action_tensor = action_tensor.squeeze(0)
+    action_tensor = action_tensor.to("cpu")
+
+    action_names = ds_features[ACTION]["names"]
+    act_processed_policy: RobotAction = {
+        f"{name}": float(action_tensor[i]) for i, name in enumerate(action_names)
+    }
+    return act_processed_policy
+
+
+def raise_feature_mismatch_error(
+    provided_features: set[str],
+    expected_features: set[str],
+) -> None:
+    """
+    Raises a standardized ValueError for feature mismatches between dataset/environment and policy config.
+    """
+    missing = expected_features - provided_features
+    extra = provided_features - expected_features
+    # TODO (jadechoghari): provide a dynamic rename map suggestion to the user.
+    raise ValueError(
+        f"Feature mismatch between dataset/environment and policy config.\n"
+        f"- Missing features: {sorted(missing) if missing else 'None'}\n"
+        f"- Extra features: {sorted(extra) if extra else 'None'}\n\n"
+        f"Please ensure your dataset and policy use consistent feature names.\n"
+        f"If your dataset uses different observation keys (e.g., cameras named differently), "
+        f"use the `--rename_map` argument, for example:\n"
+        f'  --rename_map=\'{{"observation.images.left": "observation.images.camera1", '
+        f'"observation.images.top": "observation.images.camera2"}}\''
+    )
+
+
+def validate_visual_features_consistency(
+    cfg: PreTrainedConfig,
+    features: dict[str, PolicyFeature],
+) -> None:
+    """
+    Validates visual feature consistency between a policy config and provided dataset/environment features.
+
+    Validation passes if EITHER:
+    - Policy's expected visuals are a subset of dataset (policy uses some cameras, dataset has more)
+    - Dataset's provided visuals are a subset of policy (policy declares extras for flexibility)
+
+    Args:
+        cfg (PreTrainedConfig): The model or policy configuration containing input_features and type.
+        features (Dict[str, PolicyFeature]): A mapping of feature names to PolicyFeature objects.
+    """
+    expected_visuals = {k for k, v in cfg.input_features.items() if v.type == FeatureType.VISUAL}
+    provided_visuals = {k for k, v in features.items() if v.type == FeatureType.VISUAL}
+
+    # Accept if either direction is a subset
+    policy_subset_of_dataset = expected_visuals.issubset(provided_visuals)
+    dataset_subset_of_policy = provided_visuals.issubset(expected_visuals)
+
+    if not (policy_subset_of_dataset or dataset_subset_of_policy):
+        raise_feature_mismatch_error(provided_visuals, expected_visuals)
diff --git a/lerobot/src/lerobot/policies/vqbet/README.md b/lerobot/src/lerobot/policies/vqbet/README.md
new file mode 120000
index 0000000000000000000000000000000000000000..a4ae9291acc141f92748c46c158738f135feeb2d
--- /dev/null
+++ b/lerobot/src/lerobot/policies/vqbet/README.md
@@ -0,0 +1 @@
+../../../../docs/source/policy_vqbet_README.md
\ No newline at end of file
diff --git a/lerobot/src/lerobot/policies/vqbet/configuration_vqbet.py b/lerobot/src/lerobot/policies/vqbet/configuration_vqbet.py
new file mode 100644
index 0000000000000000000000000000000000000000..32906e52804ae63ef985332eabba66c86f315327
--- /dev/null
+++ b/lerobot/src/lerobot/policies/vqbet/configuration_vqbet.py
@@ -0,0 +1,189 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Seungjae Lee and Yibin Wang and Haritheja Etukuru
+# and H. Jin Kim and Nur Muhammad Mahi Shafiullah and Lerrel Pinto
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import NormalizationMode
+from lerobot.optim.optimizers import AdamConfig
+from lerobot.optim.schedulers import VQBeTSchedulerConfig
+
+
+@PreTrainedConfig.register_subclass("vqbet")
+@dataclass
+class VQBeTConfig(PreTrainedConfig):
+    """Configuration class for VQ-BeT.
+
+    Defaults are configured for training with PushT providing proprioceptive and single camera observations.
+
+    The parameters you will most likely need to change are the ones which depend on the environment / sensors.
+    Those are: `input_features` and `output_features`.
+
+    Notes on the inputs and outputs:
+        - "observation.state" is required as an input key.
+        - At least one key starting with "observation.image is required as an input.
+        - If there are multiple keys beginning with "observation.image" they are treated as multiple camera
+          views. Right now we only support all images having the same shape.
+        - "action" is required as an output key.
+
+    Args:
+        n_obs_steps: Number of environment steps worth of observations to pass to the policy (takes the
+            current step and additional steps going back).
+        n_action_pred_token: Total number of current token and future tokens that VQ-BeT predicts.
+        action_chunk_size: Action chunk size of each action prediction token.
+        input_features: A dictionary defining the PolicyFeature of the input data for the policy. The key represents
+            the input data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        output_features: A dictionary defining the PolicyFeature of the output data for the policy. The key represents
+            the output data name, and the value is PolicyFeature, which consists of FeatureType and shape attributes.
+        normalization_mapping: A dictionary that maps from a str value of FeatureType (e.g., "STATE", "VISUAL") to
+            a corresponding NormalizationMode (e.g., NormalizationMode.MIN_MAX)
+        vision_backbone: Name of the torchvision resnet backbone to use for encoding images.
+        crop_shape: (H, W) shape to crop images to as a preprocessing step for the vision backbone. Must fit
+            within the image size. If None, no cropping is done.
+        crop_is_random: Whether the crop should be random at training time (it's always a center crop in eval
+            mode).
+        pretrained_backbone_weights: Pretrained weights from torchvision to initialize the backbone.
+            `None` means no pretrained weights.
+        use_group_norm: Whether to replace batch normalization with group normalization in the backbone.
+            The group sizes are set to be about 16 (to be precise, feature_dim // 16).
+        spatial_softmax_num_keypoints: Number of keypoints for SpatialSoftmax.
+        n_vqvae_training_steps: Number of optimization steps for training Residual VQ.
+        vqvae_n_embed: Number of embedding vectors in the RVQ dictionary (each layer).
+        vqvae_embedding_dim: Dimension of each embedding vector in the RVQ dictionary.
+        vqvae_enc_hidden_dim: Size of hidden dimensions of Encoder / Decoder part of Residaul VQ-VAE
+        gpt_block_size: Max block size of minGPT (should be larger than the number of input tokens)
+        gpt_input_dim: Size of output input of GPT. This is also used as the dimension of observation features.
+        gpt_output_dim: Size of output dimension of GPT. This is also used as a input dimension of offset / bin prediction headers.
+        gpt_n_layer: Number of layers of GPT
+        gpt_n_head: Number of headers of GPT
+        gpt_hidden_dim: Size of hidden dimensions of GPT
+        dropout: Dropout rate for GPT
+        offset_loss_weight:  A constant that is multiplied to the offset loss
+        primary_code_loss_weight: A constant that is multiplied to the primary code prediction loss
+        secondary_code_loss_weight: A constant that is multiplied to the secondary code prediction loss
+        bet_softmax_temperature: Sampling temperature of code for rollout with VQ-BeT
+        sequentially_select: Whether select code of primary / secondary as sequentially (pick primary code,
+            and then select secodnary code), or at the same time.
+    """
+
+    # Inputs / output structure.
+    n_obs_steps: int = 5
+    n_action_pred_token: int = 3
+    action_chunk_size: int = 5
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.MIN_MAX,
+            "ACTION": NormalizationMode.MIN_MAX,
+        }
+    )
+
+    # Architecture / modeling.
+    # Vision backbone.
+    vision_backbone: str = "resnet18"
+    crop_shape: tuple[int, int] | None = (84, 84)
+    crop_is_random: bool = True
+    pretrained_backbone_weights: str | None = None
+    use_group_norm: bool = True
+    spatial_softmax_num_keypoints: int = 32
+    # VQ-VAE
+    n_vqvae_training_steps: int = 20000
+    vqvae_n_embed: int = 16
+    vqvae_embedding_dim: int = 256
+    vqvae_enc_hidden_dim: int = 128
+    # VQ-BeT
+    gpt_block_size: int = 500
+    gpt_input_dim: int = 512
+    gpt_output_dim: int = 512
+    gpt_n_layer: int = 8
+    gpt_n_head: int = 8
+    gpt_hidden_dim: int = 512
+    dropout: float = 0.1
+    offset_loss_weight: float = 10000.0
+    primary_code_loss_weight: float = 5.0
+    secondary_code_loss_weight: float = 0.5
+    bet_softmax_temperature: float = 0.1
+    sequentially_select: bool = False
+
+    # Training presets
+    optimizer_lr: float = 1e-4
+    optimizer_betas: tuple = (0.95, 0.999)
+    optimizer_eps: float = 1e-8
+    optimizer_weight_decay: float = 1e-6
+    optimizer_vqvae_lr: float = 1e-3
+    optimizer_vqvae_weight_decay: float = 1e-4
+    scheduler_warmup_steps: int = 500
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        """Input validation (not exhaustive)."""
+        if not self.vision_backbone.startswith("resnet"):
+            raise ValueError(
+                f"`vision_backbone` must be one of the ResNet variants. Got {self.vision_backbone}."
+            )
+
+    def get_optimizer_preset(self) -> AdamConfig:
+        return AdamConfig(
+            lr=self.optimizer_lr,
+            betas=self.optimizer_betas,
+            eps=self.optimizer_eps,
+            weight_decay=self.optimizer_weight_decay,
+        )
+
+    def get_scheduler_preset(self) -> VQBeTSchedulerConfig:
+        return VQBeTSchedulerConfig(
+            num_warmup_steps=self.scheduler_warmup_steps,
+            num_vqvae_training_steps=self.n_vqvae_training_steps,
+        )
+
+    def validate_features(self) -> None:
+        # Note: this check was previously performed inside VQBeTRgbEncoder in the form of
+        # assert len(image_keys) == 1
+        if not len(self.image_features) == 1:
+            raise ValueError("You must provide only one image among the inputs.")
+
+        if self.crop_shape is not None:
+            for key, image_ft in self.image_features.items():
+                if self.crop_shape[0] > image_ft.shape[1] or self.crop_shape[1] > image_ft.shape[2]:
+                    raise ValueError(
+                        f"`crop_shape` should fit within the images shapes. Got {self.crop_shape} "
+                        f"for `crop_shape` and {image_ft.shape} for "
+                        f"`{key}`."
+                    )
+
+        # Check that all input images have the same shape.
+        first_image_key, first_image_ft = next(iter(self.image_features.items()))
+        for key, image_ft in self.image_features.items():
+            if image_ft.shape != first_image_ft.shape:
+                raise ValueError(
+                    f"`{key}` does not match `{first_image_key}`, but we expect all image shapes to match."
+                )
+
+    @property
+    def observation_delta_indices(self) -> list:
+        return list(range(1 - self.n_obs_steps, 1))
+
+    @property
+    def action_delta_indices(self) -> list:
+        return list(range(1 - self.n_obs_steps, self.n_action_pred_token + self.action_chunk_size - 1))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/vqbet/modeling_vqbet.py b/lerobot/src/lerobot/policies/vqbet/modeling_vqbet.py
new file mode 100644
index 0000000000000000000000000000000000000000..6d3976b7932951cc8b222508eea467056096f3fb
--- /dev/null
+++ b/lerobot/src/lerobot/policies/vqbet/modeling_vqbet.py
@@ -0,0 +1,904 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Seungjae Lee and Yibin Wang and Haritheja Etukuru
+# and H. Jin Kim and Nur Muhammad Mahi Shafiullah and Lerrel Pinto
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import warnings
+from collections import deque
+from collections.abc import Callable
+
+import einops
+import numpy as np
+import torch
+import torch.nn.functional as F  # noqa: N812
+import torchvision
+from torch import Tensor, nn
+
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.utils import get_device_from_parameters, get_output_shape, populate_queues
+from lerobot.policies.vqbet.configuration_vqbet import VQBeTConfig
+from lerobot.policies.vqbet.vqbet_utils import GPT, ResidualVQ
+from lerobot.utils.constants import ACTION, OBS_IMAGES, OBS_STATE
+
+# ruff: noqa: N806
+
+
+class VQBeTPolicy(PreTrainedPolicy):
+    """
+    VQ-BeT Policy as per "Behavior Generation with Latent Actions"
+    """
+
+    config_class = VQBeTConfig
+    name = "vqbet"
+
+    def __init__(
+        self,
+        config: VQBeTConfig | None = None,
+        **kwargs,
+    ):
+        """
+        Args:
+            config: Policy configuration class instance or None, in which case the default instantiation of
+                the configuration class is used.
+            dataset_stats: Dataset statistics to be used for normalization. If not passed here, it is expected
+                that they will be passed with a call to `load_state_dict` before the policy is used.
+        """
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        self.vqbet = VQBeTModel(config)
+
+        self.reset()
+
+    def get_optim_params(self) -> dict:
+        vqvae_params = (
+            list(self.vqbet.action_head.vqvae_model.encoder.parameters())
+            + list(self.vqbet.action_head.vqvae_model.decoder.parameters())
+            + list(self.vqbet.action_head.vqvae_model.vq_layer.parameters())
+        )
+        decay_params, no_decay_params = self.vqbet.policy.configure_parameters()
+        decay_params = (
+            decay_params
+            + list(self.vqbet.rgb_encoder.parameters())
+            + list(self.vqbet.state_projector.parameters())
+            + list(self.vqbet.rgb_feature_projector.parameters())
+            + [self.vqbet.action_token]
+            + list(self.vqbet.action_head.map_to_cbet_preds_offset.parameters())
+        )
+
+        if self.config.sequentially_select:
+            decay_params = (
+                decay_params
+                + list(self.vqbet.action_head.map_to_cbet_preds_primary_bin.parameters())
+                + list(self.vqbet.action_head.map_to_cbet_preds_secondary_bin.parameters())
+            )
+        else:
+            decay_params = decay_params + list(self.vqbet.action_head.map_to_cbet_preds_bin.parameters())
+
+        return [
+            {
+                "params": decay_params,
+            },
+            {
+                "params": vqvae_params,
+                "weight_decay": self.config.optimizer_vqvae_weight_decay,
+                "lr": self.config.optimizer_vqvae_lr,
+            },
+            {
+                "params": no_decay_params,
+                "weight_decay": 0.0,
+            },
+        ]
+
+    def reset(self):
+        """
+        Clear observation and action queues. Should be called on `env.reset()`
+        queues are populated during rollout of the policy, they contain the n latest observations and actions
+        """
+        self._queues = {
+            OBS_IMAGES: deque(maxlen=self.config.n_obs_steps),
+            OBS_STATE: deque(maxlen=self.config.n_obs_steps),
+            ACTION: deque(maxlen=self.config.action_chunk_size),
+        }
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor]) -> Tensor:
+        batch = {k: torch.stack(list(self._queues[k]), dim=1) for k in batch if k in self._queues}
+        actions = self.vqbet(batch, rollout=True)[:, : self.config.action_chunk_size]
+        return actions
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select a single action given environment observations.
+
+        This method wraps `select_actions` in order to return one action at a time for execution in the
+        environment. It works by managing the actions in a queue and only calling `select_actions` when the
+        queue is empty.
+        """
+        # NOTE: for offline evaluation, we have action in the batch, so we need to pop it out
+        if ACTION in batch:
+            batch.pop(ACTION)
+        batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
+        # NOTE: It's important that this happens after stacking the images into a single key.
+        batch[OBS_IMAGES] = torch.stack([batch[key] for key in self.config.image_features], dim=-4)
+        # NOTE: for offline evaluation, we have action in the batch, so we need to pop it out
+        if ACTION in batch:
+            batch.pop(ACTION)
+
+        self._queues = populate_queues(self._queues, batch)
+
+        if not self.vqbet.action_head.vqvae_model.discretized.item():
+            warnings.warn(
+                "To evaluate in the environment, your VQ-BeT model should contain a pretrained Residual VQ.",
+                stacklevel=1,
+            )
+
+        if len(self._queues[ACTION]) == 0:
+            actions = self.predict_action_chunk(batch)
+            # since the data in the action queue's dimension is (action_chunk_size, batch_size, action_dim), we transpose the action and fill the queue
+            self._queues[ACTION].extend(actions.transpose(0, 1))
+
+        action = self._queues[ACTION].popleft()
+        return action
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, dict]:
+        """Run the batch through the model and compute the loss for training or validation."""
+        batch = dict(batch)  # shallow copy so that adding a key doesn't modify the original
+        batch[OBS_IMAGES] = torch.stack([batch[key] for key in self.config.image_features], dim=-4)
+        # VQ-BeT discretizes action using VQ-VAE before training BeT (please refer to section 3.2 in the VQ-BeT paper https://huggingface.co/papers/2403.03181)
+        if not self.vqbet.action_head.vqvae_model.discretized.item():
+            # loss: total loss of training RVQ
+            # n_different_codes: how many of the total possible VQ codes are being used in single batch (how many of them have at least one encoder embedding as a nearest neighbor). This can be at most `vqvae_n_embed * number of layers of RVQ (=2)`.
+            # n_different_combinations: how many different code combinations are being used out of all possible combinations in single batch. This can be at most `vqvae_n_embed ^ number of layers of RVQ (=2)` (hint consider the RVQ as a decision tree).
+            loss, n_different_codes, n_different_combinations, recon_l1_error = (
+                self.vqbet.action_head.discretize(self.config.n_vqvae_training_steps, batch[ACTION])
+            )
+            return loss, {
+                "n_different_codes": n_different_codes,
+                "n_different_combinations": n_different_combinations,
+                "recon_l1_error": recon_l1_error,
+            }
+        # if Residual VQ is already trained, VQ-BeT trains its GPT and bin prediction head / offset prediction head parts.
+        _, loss_dict = self.vqbet(batch, rollout=False)
+        loss = loss_dict.pop("loss")
+
+        return loss, loss_dict
+
+
+class SpatialSoftmax(nn.Module):
+    """
+    Spatial Soft Argmax operation described in "Deep Spatial Autoencoders for Visuomotor Learning" by Finn et al.
+    (https://huggingface.co/papers/1509.06113). A minimal port of the robomimic implementation.
+
+    At a high level, this takes 2D feature maps (from a convnet/ViT) and returns the "center of mass"
+    of activations of each channel, i.e., keypoints in the image space for the policy to focus on.
+
+    Example: take feature maps of size (512x10x12). We generate a grid of normalized coordinates (10x12x2):
+    -----------------------------------------------------
+    | (-1., -1.)   | (-0.82, -1.)   | ... | (1., -1.)   |
+    | (-1., -0.78) | (-0.82, -0.78) | ... | (1., -0.78) |
+    | ...          | ...            | ... | ...         |
+    | (-1., 1.)    | (-0.82, 1.)    | ... | (1., 1.)    |
+    -----------------------------------------------------
+    This is achieved by applying channel-wise softmax over the activations (512x120) and computing the dot
+    product with the coordinates (120x2) to get expected points of maximal activation (512x2).
+
+    The example above results in 512 keypoints (corresponding to the 512 input channels). We can optionally
+    provide num_kp != None to control the number of keypoints. This is achieved by a first applying a learnable
+    linear mapping (in_channels, H, W) -> (num_kp, H, W).
+    """
+
+    def __init__(self, input_shape, num_kp=None):
+        """
+        Args:
+            input_shape (list): (C, H, W) input feature map shape.
+            num_kp (int): number of keypoints in output. If None, output will have the same number of channels as input.
+        """
+        super().__init__()
+
+        assert len(input_shape) == 3
+        self._in_c, self._in_h, self._in_w = input_shape
+
+        if num_kp is not None:
+            self.nets = torch.nn.Conv2d(self._in_c, num_kp, kernel_size=1)
+            self._out_c = num_kp
+        else:
+            self.nets = None
+            self._out_c = self._in_c
+
+        # we could use torch.linspace directly but that seems to behave slightly differently than numpy
+        # and causes a small degradation in pc_success of pre-trained models.
+        pos_x, pos_y = np.meshgrid(np.linspace(-1.0, 1.0, self._in_w), np.linspace(-1.0, 1.0, self._in_h))
+        pos_x = torch.from_numpy(pos_x.reshape(self._in_h * self._in_w, 1)).float()
+        pos_y = torch.from_numpy(pos_y.reshape(self._in_h * self._in_w, 1)).float()
+        # register as buffer so it's moved to the correct device.
+        self.register_buffer("pos_grid", torch.cat([pos_x, pos_y], dim=1))
+
+    def forward(self, features: Tensor) -> Tensor:
+        """
+        Args:
+            features: (B, C, H, W) input feature maps.
+        Returns:
+            (B, K, 2) image-space coordinates of keypoints.
+        """
+        if self.nets is not None:
+            features = self.nets(features)
+
+        # [B, K, H, W] -> [B * K, H * W] where K is number of keypoints
+        features = features.reshape(-1, self._in_h * self._in_w)
+        # 2d softmax normalization
+        attention = F.softmax(features, dim=-1)
+        # [B * K, H * W] x [H * W, 2] -> [B * K, 2] for spatial coordinate mean in x and y dimensions
+        expected_xy = attention @ self.pos_grid
+        # reshape to [B, K, 2]
+        feature_keypoints = expected_xy.view(-1, self._out_c, 2)
+
+        return feature_keypoints
+
+
+class VQBeTModel(nn.Module):
+    """VQ-BeT: The underlying neural network for VQ-BeT
+
+    Note: In this code we use the terms `rgb_encoder`, 'policy', `action_head`. The meanings are as follows.
+        - The `rgb_encoder` process rgb-style image observations to one-dimensional embedding vectors
+        - A `policy` is a minGPT architecture, that takes observation sequences and action query tokens to generate `features`.
+        - These `features` pass through the action head, which passes through the code prediction, offset prediction head,
+        and finally generates a prediction for the action chunks.
+
+        -------------------------------** legend **-------------------------------
+        │   n = n_obs_steps, p = n_action_pred_token, c = action_chunk_size)   │
+        │   o_{t} : visual observation at timestep {t}                           │
+        │   s_{t} : state observation at timestep {t}                            │
+        │   a_{t} : action at timestep {t}                                       │
+        │   A_Q : action_query_token                                             │
+        --------------------------------------------------------------------------
+
+
+        Training Phase 1. Discretize action using Residual VQ (for config.n_vqvae_training_steps steps)
+
+
+        ┌─────────────────┐            ┌─────────────────┐            ┌─────────────────┐
+        │                 │            │                 │            │                 │
+        │   RVQ encoder   │    ─►      │     Residual    │    ─►      │   RVQ Decoder   │
+        │ (a_{t}~a_{t+p}) │            │  Code Quantizer │            │                 │
+        │                 │            │                 │            │                 │
+        └─────────────────┘            └─────────────────┘            └─────────────────┘
+
+        Training Phase 2.
+
+          timestep {t-n+1}   timestep {t-n+2}                timestep {t}
+            ┌─────┴─────┐     ┌─────┴─────┐                 ┌─────┴─────┐
+
+        o_{t-n+1}         o_{t-n+2}           ...         o_{t}
+            │                 │                             │
+            │ s_{t-n+1}       │ s_{t-n+2}         ...       │   s_{t}           p
+            │     │           │     │                       │     │     ┌───────┴───────┐
+            │     │    A_Q    │     │    A_Q          ...   │     │    A_Q     ...     A_Q
+            │     │     │     │     │     │                 │     │     │               │
+        ┌───▼─────▼─────▼─────▼─────▼─────▼─────────────────▼─────▼─────▼───────────────▼───┐
+        │                                                                                   │
+        │                                       GPT                                         │       =>    policy
+        │                                                                                   │
+        └───────────────▼─────────────────▼─────────────────────────────▼───────────────▼───┘
+                        │                 │                             │               │
+                    ┌───┴───┐         ┌───┴───┐                     ┌───┴───┐       ┌───┴───┐
+                  code    offset    code    offset                code    offset  code    offset
+                    ▼       │         ▼       │                     ▼       │       ▼       │       =>    action_head
+               RVQ Decoder  │    RVQ Decoder  │                RVQ Decoder  │  RVQ Decoder  │
+                    └── + ──┘         └── + ──┘                     └── + ──┘       └── + ──┘
+                        ▼                 ▼                             ▼               ▼
+                   action chunk      action chunk                  action chunk     action chunk
+                    a_{t-n+1} ~       a_{t-n+2} ~                   a_{t} ~     ...  a_{t+p-1} ~
+                     a_{t-n+c}         a_{t-n+c+1}                   a_{t+c-1}        a_{t+p+c-1}
+
+                                                                        ▼
+                                                      ONLY this chunk is used in rollout!
+    """
+
+    def __init__(self, config: VQBeTConfig):
+        super().__init__()
+        self.config = config
+
+        self.rgb_encoder = VQBeTRgbEncoder(config)
+        self.num_images = len(self.config.image_features)
+        # This action query token is used as a prompt for querying action chunks. Please refer to "A_Q" in the image above.
+        # Note: During the forward pass, this token is repeated as many times as needed. The authors also experimented with initializing the necessary number of tokens independently and observed inferior results.
+        self.action_token = nn.Parameter(torch.randn(1, 1, self.config.gpt_input_dim))
+
+        # To input state and observation features into GPT layers, we first project the features to fit the shape of input size of GPT.
+        self.state_projector = MLP(
+            config.robot_state_feature.shape[0], hidden_channels=[self.config.gpt_input_dim]
+        )
+        self.rgb_feature_projector = MLP(
+            self.rgb_encoder.feature_dim, hidden_channels=[self.config.gpt_input_dim]
+        )
+
+        # GPT part of VQ-BeT
+        self.policy = GPT(config)
+        # bin prediction head / offset prediction head part of VQ-BeT
+        self.action_head = VQBeTHead(config)
+
+        # Action tokens for: each observation step, the current action token, and all future action tokens.
+        num_tokens = self.config.n_action_pred_token + self.config.n_obs_steps - 1
+        self.register_buffer(
+            "select_target_actions_indices",
+            torch.row_stack([torch.arange(i, i + self.config.action_chunk_size) for i in range(num_tokens)]),
+        )
+
+    def forward(self, batch: dict[str, Tensor], rollout: bool) -> tuple[dict, dict]:
+        # Input validation.
+        assert set(batch).issuperset({OBS_STATE, OBS_IMAGES})
+        batch_size, n_obs_steps = batch[OBS_STATE].shape[:2]
+        assert n_obs_steps == self.config.n_obs_steps
+
+        # Extract image feature (first combine batch and sequence dims).
+        img_features = self.rgb_encoder(einops.rearrange(batch[OBS_IMAGES], "b s n ... -> (b s n) ..."))
+        # Separate batch and sequence dims.
+        img_features = einops.rearrange(
+            img_features, "(b s n) ... -> b s n ...", b=batch_size, s=n_obs_steps, n=self.num_images
+        )
+
+        # Arrange prior and current observation step tokens as shown in the class docstring.
+        # First project features to token dimension.
+        rgb_tokens = self.rgb_feature_projector(
+            img_features
+        )  # (batch, obs_step, number of different cameras, projection dims)
+        input_tokens = [rgb_tokens[:, :, i] for i in range(rgb_tokens.size(2))]
+        input_tokens.append(self.state_projector(batch[OBS_STATE]))  # (batch, obs_step, projection dims)
+        input_tokens.append(einops.repeat(self.action_token, "1 1 d -> b n d", b=batch_size, n=n_obs_steps))
+        # Interleave tokens by stacking and rearranging.
+        input_tokens = torch.stack(input_tokens, dim=2)
+        input_tokens = einops.rearrange(input_tokens, "b n t d -> b (n t) d")
+
+        len_additional_action_token = self.config.n_action_pred_token - 1
+        future_action_tokens = self.action_token.repeat(batch_size, len_additional_action_token, 1)
+
+        # add additional action query tokens for predicting future action chunks
+        input_tokens = torch.cat([input_tokens, future_action_tokens], dim=1)
+
+        # get action features (pass through GPT)
+        features = self.policy(input_tokens)
+        # len(self.config.input_features) is the number of different observation modes.
+        # this line gets the index of action prompt tokens.
+        historical_act_pred_index = np.arange(0, n_obs_steps) * (len(self.config.input_features) + 1) + len(
+            self.config.input_features
+        )
+
+        # only extract the output tokens at the position of action query:
+        # Behavior Transformer (BeT), and VQ-BeT are both sequence-to-sequence prediction models,
+        # mapping sequential observation to sequential action (please refer to section 2.2 in BeT paper https://huggingface.co/papers/2206.11251).
+        # Thus, it predicts a historical action sequence, in addition to current and future actions (predicting future actions : optional).
+        if len_additional_action_token > 0:
+            features = torch.cat(
+                [features[:, historical_act_pred_index], features[:, -len_additional_action_token:]], dim=1
+            )
+        else:
+            features = features[:, historical_act_pred_index]
+        # pass through action head
+        action_head_output = self.action_head(features)
+        # if rollout, VQ-BeT don't calculate loss
+        if rollout:
+            return action_head_output["predicted_action"][:, n_obs_steps - 1, :].reshape(
+                batch_size, self.config.action_chunk_size, -1
+            )
+        # else, it calculate overall loss (bin prediction loss, and offset loss)
+        else:
+            output = batch[ACTION][:, self.select_target_actions_indices]
+            loss = self.action_head.loss_fn(action_head_output, output, reduction="mean")
+            return action_head_output, loss
+
+
+class VQBeTHead(nn.Module):
+    def __init__(self, config: VQBeTConfig):
+        """
+        VQBeTHead takes output of GPT layers, and pass the feature through bin prediction head (`self.map_to_cbet_preds_bin`), and offset prediction head (`self.map_to_cbet_preds_offset`)
+
+        self.map_to_cbet_preds_bin: outputs probability of each code (for each layer).
+            The input dimension of `self.map_to_cbet_preds_bin` is same with the output of GPT,
+            and the output dimension of `self.map_to_cbet_preds_bin` is `self.vqvae_model.vqvae_num_layers (=fixed as 2) * self.config.vqvae_n_embed`.
+            if the agent select the code sequentially, we use self.map_to_cbet_preds_primary_bin and self.map_to_cbet_preds_secondary_bin instead of self._map_to_cbet_preds_bin.
+
+        self.map_to_cbet_preds_offset: output the predicted offsets for all the codes in all the layers.
+            The input dimension of ` self.map_to_cbet_preds_offset` is same with the output of GPT,
+            and the output dimension of ` self.map_to_cbet_preds_offset` is `self.vqvae_model.vqvae_num_layers (=fixed as 2) * self.config.vqvae_n_embed * config.action_chunk_size * config.action_feature.shape[0]`.
+        """
+
+        super().__init__()
+        self.config = config
+        # init vqvae
+        self.vqvae_model = VqVae(config)
+        if config.sequentially_select:
+            self.map_to_cbet_preds_primary_bin = MLP(
+                in_channels=config.gpt_output_dim,
+                hidden_channels=[self.config.vqvae_n_embed],
+            )
+            self.map_to_cbet_preds_secondary_bin = MLP(
+                in_channels=config.gpt_output_dim + self.config.vqvae_n_embed,
+                hidden_channels=[self.config.vqvae_n_embed],
+            )
+        else:
+            self.map_to_cbet_preds_bin = MLP(
+                in_channels=config.gpt_output_dim,
+                hidden_channels=[self.vqvae_model.vqvae_num_layers * self.config.vqvae_n_embed],
+            )
+        self.map_to_cbet_preds_offset = MLP(
+            in_channels=config.gpt_output_dim,
+            hidden_channels=[
+                self.vqvae_model.vqvae_num_layers
+                * self.config.vqvae_n_embed
+                * config.action_chunk_size
+                * config.action_feature.shape[0],
+            ],
+        )
+        # loss
+        self._focal_loss_fn = FocalLoss(gamma=2.0)
+
+    def discretize(self, n_vqvae_training_steps, actions):
+        # Resize the action sequence data to fit the action chunk size using a sliding window approach.
+        actions = torch.cat(
+            [
+                actions[:, j : j + self.config.action_chunk_size, :]
+                for j in range(actions.shape[1] + 1 - self.config.action_chunk_size)
+            ],
+            dim=0,
+        )
+        # `actions` is a tensor of shape (new_batch, action_chunk_size, action_dim) where new_batch is the number of possible chunks created from the original sequences using the sliding window.
+
+        loss, metric = self.vqvae_model.vqvae_forward(actions)
+        n_different_codes = sum(
+            [len(torch.unique(metric[2][:, i])) for i in range(self.vqvae_model.vqvae_num_layers)]
+        )
+        n_different_combinations = len(torch.unique(metric[2], dim=0))
+        recon_l1_error = metric[0].detach().cpu().item()
+        self.vqvae_model.optimized_steps += 1
+        # if we updated RVQ more than `n_vqvae_training_steps` steps, we freeze the RVQ part.
+        if self.vqvae_model.optimized_steps >= n_vqvae_training_steps:
+            self.vqvae_model.discretized.fill_(True)
+            self.vqvae_model.vq_layer.freeze_codebook.fill_(True)
+            print("Finished discretizing action data!")
+            self.vqvae_model.eval()
+            for param in self.vqvae_model.vq_layer.parameters():
+                param.requires_grad = False
+        return loss, n_different_codes, n_different_combinations, recon_l1_error
+
+    def forward(self, x, **kwargs) -> dict:
+        # N is the batch size, and T is number of action query tokens, which are process through same GPT
+        N, T, _ = x.shape
+        # we calculate N and T side parallelly. Thus, the dimensions would be
+        # (batch size * number of action query tokens, action chunk size, action dimension)
+        x = einops.rearrange(x, "N T WA -> (N T) WA")
+
+        # sample offsets
+        cbet_offsets = self.map_to_cbet_preds_offset(x)
+        cbet_offsets = einops.rearrange(
+            cbet_offsets,
+            "(NT) (G C WA) -> (NT) G C WA",
+            G=self.vqvae_model.vqvae_num_layers,
+            C=self.config.vqvae_n_embed,
+        )
+        # if self.config.sequentially_select is True, bin prediction head first sample the primary code, and then sample secondary code
+        if self.config.sequentially_select:
+            cbet_primary_logits = self.map_to_cbet_preds_primary_bin(x)
+
+            # select primary bin first
+            cbet_primary_probs = torch.softmax(
+                cbet_primary_logits / self.config.bet_softmax_temperature, dim=-1
+            )
+            NT, choices = cbet_primary_probs.shape
+            sampled_primary_centers = einops.rearrange(
+                torch.multinomial(cbet_primary_probs.view(-1, choices), num_samples=1),
+                "(NT) 1 -> NT",
+                NT=NT,
+            )
+
+            cbet_secondary_logits = self.map_to_cbet_preds_secondary_bin(
+                torch.cat(
+                    (x, F.one_hot(sampled_primary_centers, num_classes=self.config.vqvae_n_embed)),
+                    axis=1,
+                )
+            )
+            cbet_secondary_probs = torch.softmax(
+                cbet_secondary_logits / self.config.bet_softmax_temperature, dim=-1
+            )
+            sampled_secondary_centers = einops.rearrange(
+                torch.multinomial(cbet_secondary_probs.view(-1, choices), num_samples=1),
+                "(NT) 1 -> NT",
+                NT=NT,
+            )
+            sampled_centers = torch.stack((sampled_primary_centers, sampled_secondary_centers), axis=1)
+            cbet_logits = torch.stack([cbet_primary_logits, cbet_secondary_logits], dim=1)
+        # if self.config.sequentially_select is False, bin prediction head samples primary and secondary code at once.
+        else:
+            cbet_logits = self.map_to_cbet_preds_bin(x)
+            cbet_logits = einops.rearrange(
+                cbet_logits, "(NT) (G C) -> (NT) G C", G=self.vqvae_model.vqvae_num_layers
+            )
+            cbet_probs = torch.softmax(cbet_logits / self.config.bet_softmax_temperature, dim=-1)
+            NT, G, choices = cbet_probs.shape
+            sampled_centers = einops.rearrange(
+                torch.multinomial(cbet_probs.view(-1, choices), num_samples=1),
+                "(NT G) 1 -> NT G",
+                NT=NT,
+            )
+
+        device = get_device_from_parameters(self)
+        indices = (
+            torch.arange(NT, device=device).unsqueeze(1),
+            torch.arange(self.vqvae_model.vqvae_num_layers, device=device).unsqueeze(0),
+            sampled_centers,
+        )
+        # Use advanced indexing to sample the values (Extract the only offsets corresponding to the sampled codes.)
+        sampled_offsets = cbet_offsets[indices]
+        # Then, sum the offsets over the RVQ layers to get a net offset for the bin prediction
+        sampled_offsets = sampled_offsets.sum(dim=1)
+        with torch.no_grad():
+            # Get the centroids (= vectors corresponding to the codes) of each layer to pass it through RVQ decoder
+            return_decoder_input = self.vqvae_model.get_embeddings_from_code(sampled_centers).clone().detach()
+            # pass the centroids through decoder to get actions.
+            decoded_action = self.vqvae_model.get_action_from_latent(return_decoder_input).clone().detach()
+        # reshaped extracted offset to match with decoded centroids
+        sampled_offsets = einops.rearrange(
+            sampled_offsets, "NT (W A) -> NT W A", W=self.config.action_chunk_size
+        )
+        # add offset and decoded centroids
+        predicted_action = decoded_action + sampled_offsets
+        predicted_action = einops.rearrange(
+            predicted_action,
+            "(N T) W A -> N T (W A)",
+            N=N,
+            T=T,
+            W=self.config.action_chunk_size,
+        )
+
+        return {
+            "cbet_logits": cbet_logits,
+            "predicted_action": predicted_action,
+            "sampled_centers": sampled_centers,
+            "decoded_action": decoded_action,
+        }
+
+    def loss_fn(self, pred, target, **kwargs):
+        """
+        for given ground truth action values (target), and prediction (pred) this function calculates the overall loss.
+
+        predicted_action: predicted action chunk (offset + decoded centroids)
+        sampled_centers: sampled centroids (code of RVQ)
+        decoded_action: decoded action, which is produced by passing sampled_centers through RVQ decoder
+        NT: batch size * T
+        T: number of action query tokens, which are process through same GPT
+        cbet_logits: probability of all codes in each layer
+        """
+        action_seq = target
+        predicted_action = pred["predicted_action"]
+        sampled_centers = pred["sampled_centers"]
+        decoded_action = pred["decoded_action"]
+        NT = predicted_action.shape[0] * predicted_action.shape[1]
+
+        cbet_logits = pred["cbet_logits"]
+
+        predicted_action = einops.rearrange(
+            predicted_action, "N T (W A) -> (N T) W A", W=self.config.action_chunk_size
+        )
+
+        action_seq = einops.rearrange(action_seq, "N T W A -> (N T) W A")
+        # Figure out the loss for the actions.
+        # First, we need to find the closest cluster center for each ground truth action.
+        with torch.no_grad():
+            state_vq, action_bins = self.vqvae_model.get_code(action_seq)  # action_bins: NT, G
+
+        # Now we can compute the loss.
+
+        # offset loss is L1 distance between the predicted action and ground truth action
+        offset_loss = F.l1_loss(action_seq, predicted_action)
+
+        # calculate primary code prediction loss
+        cbet_loss1 = self._focal_loss_fn(
+            cbet_logits[:, 0, :],
+            action_bins[:, 0],
+        )
+        # calculate secondary code prediction loss
+        cbet_loss2 = self._focal_loss_fn(
+            cbet_logits[:, 1, :],
+            action_bins[:, 1],
+        )
+        # add all the prediction loss
+        cbet_loss = (
+            cbet_loss1 * self.config.primary_code_loss_weight
+            + cbet_loss2 * self.config.secondary_code_loss_weight
+        )
+
+        equal_primary_code_rate = torch.sum((action_bins[:, 0] == sampled_centers[:, 0]).int()) / (NT)
+        equal_secondary_code_rate = torch.sum((action_bins[:, 1] == sampled_centers[:, 1]).int()) / (NT)
+
+        action_mse_error = torch.mean((action_seq - predicted_action) ** 2)
+        vq_action_error = torch.mean(torch.abs(action_seq - decoded_action))
+        offset_action_error = torch.mean(torch.abs(action_seq - predicted_action))
+        action_error_max = torch.max(torch.abs(action_seq - predicted_action))
+
+        loss = cbet_loss + self.config.offset_loss_weight * offset_loss
+
+        loss_dict = {
+            "loss": loss,
+            "classification_loss": cbet_loss.detach().cpu().item(),
+            "offset_loss": offset_loss.detach().cpu().item(),
+            "equal_primary_code_rate": equal_primary_code_rate.detach().cpu().item(),
+            "equal_secondary_code_rate": equal_secondary_code_rate.detach().cpu().item(),
+            "vq_action_error": vq_action_error.detach().cpu().item(),
+            "offset_action_error": offset_action_error.detach().cpu().item(),
+            "action_error_max": action_error_max.detach().cpu().item(),
+            "action_mse_error": action_mse_error.detach().cpu().item(),
+        }
+        return loss_dict
+
+
+class VQBeTRgbEncoder(nn.Module):
+    """Encode an RGB image into a 1D feature vector.
+
+    Includes the ability to normalize and crop the image first.
+
+    Same with DiffusionRgbEncoder from modeling_diffusion.py
+    """
+
+    def __init__(self, config: VQBeTConfig):
+        super().__init__()
+        # Set up optional preprocessing.
+        if config.crop_shape is not None:
+            self.do_crop = True
+            # Always use center crop for eval
+            self.center_crop = torchvision.transforms.CenterCrop(config.crop_shape)
+            if config.crop_is_random:
+                self.maybe_random_crop = torchvision.transforms.RandomCrop(config.crop_shape)
+            else:
+                self.maybe_random_crop = self.center_crop
+        else:
+            self.do_crop = False
+
+        # Set up backbone.
+        backbone_model = getattr(torchvision.models, config.vision_backbone)(
+            weights=config.pretrained_backbone_weights
+        )
+        # Note: This assumes that the layer4 feature map is children()[-3]
+        # TODO(alexander-soare): Use a safer alternative.
+        self.backbone = nn.Sequential(*(list(backbone_model.children())[:-2]))
+        if config.use_group_norm:
+            if config.pretrained_backbone_weights:
+                raise ValueError(
+                    "You can't replace BatchNorm in a pretrained model without ruining the weights!"
+                )
+            self.backbone = _replace_submodules(
+                root_module=self.backbone,
+                predicate=lambda x: isinstance(x, nn.BatchNorm2d),
+                func=lambda x: nn.GroupNorm(num_groups=x.num_features // 16, num_channels=x.num_features),
+            )
+
+        # Set up pooling and final layers.
+        # Use a dry run to get the feature map shape.
+        # The dummy input should take the number of image channels from `config.image_features` and it should
+        # use the height and width from `config.crop_shape` if it is provided, otherwise it should use the
+        # height and width from `config.image_features`.
+
+        images_shape = next(iter(config.image_features.values())).shape
+        dummy_shape_h_w = config.crop_shape if config.crop_shape is not None else images_shape[1:]
+        dummy_shape = (1, images_shape[0], *dummy_shape_h_w)
+        feature_map_shape = get_output_shape(self.backbone, dummy_shape)[1:]
+
+        self.pool = SpatialSoftmax(feature_map_shape, num_kp=config.spatial_softmax_num_keypoints)
+        self.feature_dim = config.spatial_softmax_num_keypoints * 2
+        self.out = nn.Linear(config.spatial_softmax_num_keypoints * 2, self.feature_dim)
+        self.relu = nn.ReLU()
+
+    def forward(self, x: Tensor) -> Tensor:
+        """
+        Args:
+            x: (B, C, H, W) image tensor with pixel values in [0, 1].
+        Returns:
+            (B, D) image feature.
+        """
+        # Preprocess: maybe crop (if it was set up in the __init__).
+        if self.do_crop:
+            if self.training:  # noqa: SIM108
+                x = self.maybe_random_crop(x)
+            else:
+                # Always use center crop for eval.
+                x = self.center_crop(x)
+        # Extract backbone feature.
+        x = torch.flatten(self.pool(self.backbone(x)), start_dim=1)
+        # Final linear layer with non-linearity.
+        x = self.relu(self.out(x))
+        return x
+
+
+def _replace_submodules(
+    root_module: nn.Module, predicate: Callable[[nn.Module], bool], func: Callable[[nn.Module], nn.Module]
+) -> nn.Module:
+    """
+    Args:
+        root_module: The module for which the submodules need to be replaced
+        predicate: Takes a module as an argument and must return True if the that module is to be replaced.
+        func: Takes a module as an argument and returns a new module to replace it with.
+    Returns:
+        The root module with its submodules replaced.
+    """
+    if predicate(root_module):
+        return func(root_module)
+
+    replace_list = [k.split(".") for k, m in root_module.named_modules(remove_duplicate=True) if predicate(m)]
+    for *parents, k in replace_list:
+        parent_module = root_module
+        if len(parents) > 0:
+            parent_module = root_module.get_submodule(".".join(parents))
+        if isinstance(parent_module, nn.Sequential):
+            src_module = parent_module[int(k)]
+        else:
+            src_module = getattr(parent_module, k)
+        tgt_module = func(src_module)
+        if isinstance(parent_module, nn.Sequential):
+            parent_module[int(k)] = tgt_module
+        else:
+            setattr(parent_module, k, tgt_module)
+    # verify that all BN are replaced
+    assert not any(predicate(m) for _, m in root_module.named_modules(remove_duplicate=True))
+    return root_module
+
+
+class VqVae(nn.Module):
+    def __init__(
+        self,
+        config: VQBeTConfig,
+    ):
+        """
+        VQ-VAE is composed of three parts: encoder, vq_layer, and decoder.
+        Encoder and decoder are MLPs consisting of an input, output layer, and hidden layer, respectively.
+        The vq_layer uses residual VQs.
+
+        This class contains functions for training the encoder and decoder along with the residual VQ layer (for training phase 1),
+        as well as functions to help BeT training part in training phase 2.
+        """
+
+        super().__init__()
+        self.config = config
+        # 'discretized' indicates whether the Residual VQ part is trained or not. (After finishing the training, we set discretized=True)
+        self.register_buffer("discretized", torch.tensor(False))
+        self.optimized_steps = 0
+        # we use the fixed number of layers for Residual VQ across all environments.
+        self.vqvae_num_layers = 2
+
+        self.vq_layer = ResidualVQ(
+            dim=config.vqvae_embedding_dim,
+            num_quantizers=self.vqvae_num_layers,
+            codebook_size=config.vqvae_n_embed,
+        )
+
+        self.encoder = MLP(
+            in_channels=self.config.action_feature.shape[0] * self.config.action_chunk_size,
+            hidden_channels=[
+                config.vqvae_enc_hidden_dim,
+                config.vqvae_enc_hidden_dim,
+                config.vqvae_embedding_dim,
+            ],
+        )
+        self.decoder = MLP(
+            in_channels=config.vqvae_embedding_dim,
+            hidden_channels=[
+                config.vqvae_enc_hidden_dim,
+                config.vqvae_enc_hidden_dim,
+                self.config.action_feature.shape[0] * self.config.action_chunk_size,
+            ],
+        )
+
+    def get_embeddings_from_code(self, encoding_indices):
+        # This function gets code indices as inputs, and outputs embedding vectors corresponding to the code indices.
+        with torch.no_grad():
+            z_embed = self.vq_layer.get_codebook_vector_from_indices(encoding_indices)
+            # since the RVQ has multiple layers, it adds the vectors in the axis of layers to provide a vector for that code combination.
+            z_embed = z_embed.sum(dim=0)
+        return z_embed
+
+    def get_action_from_latent(self, latent):
+        # given latent vector, this function outputs the decoded action.
+        output = self.decoder(latent)
+        if self.config.action_chunk_size == 1:
+            return einops.rearrange(output, "N (T A) -> N T A", A=self.config.action_feature.shape[0])
+        else:
+            return einops.rearrange(output, "N (T A) -> N T A", A=self.config.action_feature.shape[0])
+
+    def get_code(self, state):
+        # in phase 2 of VQ-BeT training, we need a `ground truth labels of action data` to calculate the Focal loss for code prediction head. (please refer to section 3.3 in the paper https://huggingface.co/papers/2403.03181)
+        # this function outputs the `GT code` of given action using frozen encoder and quantization layers. (please refer to Figure 2. in the paper https://huggingface.co/papers/2403.03181)
+        state = einops.rearrange(state, "N T A -> N (T A)")
+        with torch.no_grad():
+            state_rep = self.encoder(state)
+            state_rep_shape = state_rep.shape[:-1]
+            state_rep_flat = state_rep.view(state_rep.size(0), -1, state_rep.size(1))
+            state_rep_flat, vq_code, vq_loss_state = self.vq_layer(state_rep_flat)
+            state_vq = state_rep_flat.view(*state_rep_shape, -1)
+            vq_code = vq_code.view(*state_rep_shape, -1)
+            vq_loss_state = torch.sum(vq_loss_state)
+            return state_vq, vq_code
+
+    def vqvae_forward(self, state):
+        # This function passes the given data through Residual VQ with Encoder and Decoder. Please refer to section 3.2 in the paper https://huggingface.co/papers/2403.03181).
+        state = einops.rearrange(state, "N T A -> N (T A)")
+        # We start with passing action (or action chunk) at:t+n through the encoder ϕ.
+        state_rep = self.encoder(state)
+        state_rep_shape = state_rep.shape[:-1]
+        state_rep_flat = state_rep.view(state_rep.size(0), -1, state_rep.size(1))
+        # The resulting latent embedding vector x = ϕ(at:t+n) is then mapped to an embedding vector in the codebook of the RVQ layers by the nearest neighbor look-up.
+        state_rep_flat, vq_code, vq_loss_state = self.vq_layer(state_rep_flat)
+        state_vq = state_rep_flat.view(*state_rep_shape, -1)
+        vq_code = vq_code.view(*state_rep_shape, -1)
+        # since the RVQ has multiple layers, it adds the vectors in the axis of layers to provide a vector for that code combination.
+        vq_loss_state = torch.sum(vq_loss_state)
+        # Then, the discretized vector zq(x) is reconstructed as ψ(zq(x)) by passing through the decoder ψ.
+        dec_out = self.decoder(state_vq)
+        # Calculate L1 reconstruction loss
+        encoder_loss = (state - dec_out).abs().mean()
+        # add encoder reconstruction loss and commitment loss
+        rep_loss = encoder_loss + vq_loss_state * 5
+
+        metric = (
+            encoder_loss.clone().detach(),
+            vq_loss_state.clone().detach(),
+            vq_code,
+            rep_loss.item(),
+        )
+        return rep_loss, metric
+
+
+class FocalLoss(nn.Module):
+    """
+    From https://github.com/notmahi/miniBET/blob/main/behavior_transformer/bet.py
+    """
+
+    def __init__(self, gamma: float = 0, size_average: bool = True):
+        super().__init__()
+        self.gamma = gamma
+        self.size_average = size_average
+
+    def forward(self, input, target):
+        if len(input.shape) == 3:
+            N, T, _ = input.shape
+            logpt = F.log_softmax(input, dim=-1)
+            logpt = logpt.gather(-1, target.view(N, T, 1)).view(N, T)
+        elif len(input.shape) == 2:
+            logpt = F.log_softmax(input, dim=-1)
+            logpt = logpt.gather(-1, target.view(-1, 1)).view(-1)
+        pt = logpt.exp()
+
+        loss = -1 * (1 - pt) ** self.gamma * logpt
+        if self.size_average:
+            return loss.mean()
+        else:
+            return loss.sum()
+
+
+class MLP(torch.nn.Sequential):
+    def __init__(
+        self,
+        in_channels: int,
+        hidden_channels: list[int],
+    ):
+        layers = []
+        in_dim = in_channels
+        for hidden_dim in hidden_channels[:-1]:
+            layers.append(torch.nn.Linear(in_dim, hidden_dim))
+            layers.append(torch.nn.ReLU())
+            in_dim = hidden_dim
+
+        layers.append(torch.nn.Linear(in_dim, hidden_channels[-1]))
+
+        super().__init__(*layers)
diff --git a/lerobot/src/lerobot/policies/vqbet/processor_vqbet.py b/lerobot/src/lerobot/policies/vqbet/processor_vqbet.py
new file mode 100644
index 0000000000000000000000000000000000000000..1e19ff779cbde9c5d962c92c13b95421d93bb2c6
--- /dev/null
+++ b/lerobot/src/lerobot/policies/vqbet/processor_vqbet.py
@@ -0,0 +1,91 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Seungjae Lee and Yibin Wang and Haritheja Etukuru
+# and H. Jin Kim and Nur Muhammad Mahi Shafiullah and Lerrel Pinto
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from typing import Any
+
+import torch
+
+from lerobot.policies.vqbet.configuration_vqbet import VQBeTConfig
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    RenameObservationsProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.utils.constants import POLICY_POSTPROCESSOR_DEFAULT_NAME, POLICY_PREPROCESSOR_DEFAULT_NAME
+
+
+def make_vqbet_pre_post_processors(
+    config: VQBeTConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for the VQ-BeT policy.
+
+    The pre-processing pipeline prepares input data for the model by:
+    1. Renaming features, allowing customization to match pretrained configurations.
+    2. Normalizing input and output features based on dataset statistics.
+    3. Adding a batch dimension.
+    4. Moving all data to the specified device.
+
+    The post-processing pipeline handles the model's output by:
+    1. Moving data to the CPU.
+    2. Unnormalizing the output features to their original scale.
+
+    Args:
+        config: The configuration object for the VQ-BeT policy.
+        dataset_stats: A dictionary of statistics for normalization.
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines.
+    """
+
+    input_steps = [
+        RenameObservationsProcessorStep(rename_map={}),  # Let the possibility to the user to rename the keys
+        AddBatchDimensionProcessorStep(),
+        DeviceProcessorStep(device=config.device),
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+    ]
+    output_steps = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/vqbet/vqbet_utils.py b/lerobot/src/lerobot/policies/vqbet/vqbet_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..7b13577f629ca0b427e6a1f6f559c29f10366636
--- /dev/null
+++ b/lerobot/src/lerobot/policies/vqbet/vqbet_utils.py
@@ -0,0 +1,1450 @@
+#!/usr/bin/env python
+
+# Copyright 2024 Seungjae Lee and Yibin Wang and Haritheja Etukuru
+# and H. Jin Kim and Nur Muhammad Mahi Shafiullah and Lerrel Pinto
+# and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import math
+from collections.abc import Callable
+from functools import partial
+from math import ceil
+from random import randrange
+
+import torch
+import torch.distributed as distributed
+import torch.nn.functional as F  # noqa: N812
+from einops import pack, rearrange, reduce, repeat, unpack
+from torch import einsum, nn
+from torch.cuda.amp import autocast
+from torch.optim import Optimizer
+
+from lerobot.policies.vqbet.configuration_vqbet import VQBeTConfig
+
+# ruff: noqa: N806
+
+"""
+This file is part of a VQ-BeT that utilizes code from the following repositories:
+
+    - Vector Quantize PyTorch code is licensed under the MIT License:
+        Original source: https://github.com/lucidrains/vector-quantize-pytorch
+
+    - nanoGPT part is an adaptation of Andrej Karpathy's nanoGPT implementation in PyTorch.
+        Original source: https://github.com/karpathy/nanoGPT
+
+We also made some changes to the original code to adapt it to our needs. The changes are described in the code below.
+"""
+
+"""
+This is a part for nanoGPT that utilizes code from the following repository:
+
+    - Andrej Karpathy's nanoGPT implementation in PyTorch.
+        Original source: https://github.com/karpathy/nanoGPT
+
+    - The nanoGPT code is licensed under the MIT License:
+
+    MIT License
+
+    Copyright (c) 2022 Andrej Karpathy
+
+    Permission is hereby granted, free of charge, to any person obtaining a copy
+    of this software and associated documentation files (the "Software"), to deal
+    in the Software without restriction, including without limitation the rights
+    to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+    copies of the Software, and to permit persons to whom the Software is
+    furnished to do so, subject to the following conditions:
+
+    The above copyright notice and this permission notice shall be included in all
+    copies or substantial portions of the Software.
+
+    THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+    IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+    FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+    AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+    LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+    OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+    SOFTWARE.
+
+    - We've made some changes to the original code to adapt it to our needs.
+
+        Changed variable names:
+            - n_head -> gpt_n_head
+            - n_embd -> gpt_hidden_dim
+            - block_size -> gpt_block_size
+            - n_layer -> gpt_n_layer
+
+
+        class GPT(nn.Module):
+            - removed unused functions `def generate`, `def estimate_mfu`, and `def from_pretrained`
+            - changed the `configure_optimizers` to `def configure_parameters` and made it to return only the parameters of the model: we use an external optimizer in our training loop.
+            - in the function `forward`, we removed target loss calculation parts, since it will be calculated in the training loop (after passing through bin prediction and offset prediction heads).
+
+"""
+
+
+class CausalSelfAttention(nn.Module):
+    def __init__(self, config):
+        super().__init__()
+        assert config.gpt_hidden_dim % config.gpt_n_head == 0
+        # key, query, value projections for all heads, but in a batch
+        self.c_attn = nn.Linear(config.gpt_hidden_dim, 3 * config.gpt_hidden_dim)
+        # output projection
+        self.c_proj = nn.Linear(config.gpt_hidden_dim, config.gpt_hidden_dim)
+        # regularization
+        self.attn_dropout = nn.Dropout(config.dropout)
+        self.resid_dropout = nn.Dropout(config.dropout)
+        # causal mask to ensure that attention is only applied to the left in the input sequence
+        self.register_buffer(
+            "bias",
+            torch.tril(torch.ones(config.gpt_block_size, config.gpt_block_size)).view(
+                1, 1, config.gpt_block_size, config.gpt_block_size
+            ),
+        )
+        self.gpt_n_head = config.gpt_n_head
+        self.gpt_hidden_dim = config.gpt_hidden_dim
+
+    def forward(self, x):
+        (
+            B,
+            T,
+            C,
+        ) = x.size()  # batch size, sequence length, embedding dimensionality (gpt_hidden_dim)
+
+        # calculate query, key, values for all heads in batch and move head forward to be the batch dim
+        q, k, v = self.c_attn(x).split(self.gpt_hidden_dim, dim=2)
+        k = k.view(B, T, self.gpt_n_head, C // self.gpt_n_head).transpose(1, 2)  # (B, nh, T, hs)
+        q = q.view(B, T, self.gpt_n_head, C // self.gpt_n_head).transpose(1, 2)  # (B, nh, T, hs)
+        v = v.view(B, T, self.gpt_n_head, C // self.gpt_n_head).transpose(1, 2)  # (B, nh, T, hs)
+
+        # causal self-attention; Self-attend: (B, nh, T, hs) x (B, nh, hs, T) -> (B, nh, T, T)
+        att = (q @ k.transpose(-2, -1)) * (1.0 / math.sqrt(k.size(-1)))
+        att = att.masked_fill(self.bias[:, :, :T, :T] == 0, float("-inf"))
+        att = F.softmax(att, dim=-1)
+        att = self.attn_dropout(att)
+        y = att @ v  # (B, nh, T, T) x (B, nh, T, hs) -> (B, nh, T, hs)
+        y = y.transpose(1, 2).contiguous().view(B, T, C)  # re-assemble all head outputs side by side
+
+        # output projection
+        y = self.resid_dropout(self.c_proj(y))
+        return y
+
+
+class Block(nn.Module):
+    # causual self-attention block for GPT
+    def __init__(self, config):
+        super().__init__()
+        self.ln_1 = nn.LayerNorm(config.gpt_hidden_dim)
+        self.attn = CausalSelfAttention(config)
+        self.ln_2 = nn.LayerNorm(config.gpt_hidden_dim)
+        self.mlp = nn.Sequential(
+            nn.Linear(config.gpt_hidden_dim, 4 * config.gpt_hidden_dim),
+            nn.GELU(),
+            nn.Linear(4 * config.gpt_hidden_dim, config.gpt_hidden_dim),
+            nn.Dropout(config.dropout),
+        )
+
+    def forward(self, x):
+        x = x + self.attn(self.ln_1(x))
+        x = x + self.mlp(self.ln_2(x))
+        return x
+
+
+class GPT(nn.Module):
+    """
+    Original comments:
+    Full definition of a GPT Language Model, all of it in this single file.
+    References:
+    1) the official GPT-2 TensorFlow implementation released by OpenAI:
+    https://github.com/openai/gpt-2/blob/master/src/model.py
+    2) huggingface/transformers PyTorch implementation:
+    https://github.com/huggingface/transformers/blob/main/src/transformers/models/gpt2/modeling_gpt2.py
+    """
+
+    def __init__(self, config: VQBeTConfig):
+        """
+        GPT model gets hyperparameters from a config object. Please refer configuration_vqbet.py for more details.
+        """
+        super().__init__()
+        assert config.gpt_output_dim is not None
+        assert config.gpt_block_size is not None
+        self.config = config
+
+        self.transformer = nn.ModuleDict(
+            {
+                "wte": nn.Linear(config.gpt_input_dim, config.gpt_hidden_dim),
+                "wpe": nn.Embedding(config.gpt_block_size, config.gpt_hidden_dim),
+                "drop": nn.Dropout(config.dropout),
+                "h": nn.ModuleList([Block(config) for _ in range(config.gpt_n_layer)]),
+                "ln_f": nn.LayerNorm(config.gpt_hidden_dim),
+            }
+        )
+        self.lm_head = nn.Linear(config.gpt_hidden_dim, config.gpt_output_dim, bias=False)
+        # init all weights, and apply a special scaled init to the residual projections, per GPT-2 paper
+        self.apply(self._init_weights)
+        for pn, p in self.named_parameters():
+            if pn.endswith("c_proj.weight"):
+                torch.nn.init.normal_(p, mean=0.0, std=0.02 / math.sqrt(2 * config.gpt_n_layer))
+
+        # report number of parameters
+        n_params = sum(p.numel() for p in self.parameters())
+        print(f"number of parameters: {n_params / 1e6:.2f}M")
+
+    def forward(self, input, targets=None):
+        device = input.device
+        b, t, d = input.size()
+        assert t <= self.config.gpt_block_size, (
+            f"Cannot forward sequence of length {t}, block size is only {self.config.gpt_block_size}"
+        )
+
+        # positional encodings that are added to the input embeddings
+        pos = torch.arange(0, t, dtype=torch.long, device=device).unsqueeze(0)  # shape (1, t)
+
+        # forward the GPT model itself
+        tok_emb = self.transformer.wte(input)  # token embeddings of shape (b, t, gpt_hidden_dim)
+        pos_emb = self.transformer.wpe(pos)  # position embeddings of shape (1, t, gpt_hidden_dim)
+        x = self.transformer.drop(tok_emb + pos_emb)
+        for block in self.transformer.h:
+            x = block(x)
+        x = self.transformer.ln_f(x)
+        logits = self.lm_head(x)
+        return logits
+
+    def _init_weights(self, module):
+        if isinstance(module, nn.Linear):
+            torch.nn.init.normal_(module.weight, mean=0.0, std=0.02)
+            if module.bias is not None:
+                torch.nn.init.zeros_(module.bias)
+        elif isinstance(module, nn.Embedding):
+            torch.nn.init.normal_(module.weight, mean=0.0, std=0.02)
+        elif isinstance(module, nn.LayerNorm):
+            torch.nn.init.zeros_(module.bias)
+            torch.nn.init.ones_(module.weight)
+
+    def configure_parameters(self):
+        """
+        This long function is unfortunately doing something very simple and is being very defensive:
+        We are separating out all parameters of the model into two buckets: those that will experience
+        weight decay for regularization and those that won't (biases, and layernorm/embedding weights).
+        """
+
+        # separate out all parameters to those that will and won't experience regularizing weight decay
+        decay = set()
+        no_decay = set()
+        whitelist_weight_modules = (torch.nn.Linear,)
+        blacklist_weight_modules = (torch.nn.LayerNorm, torch.nn.Embedding)
+        for mn, m in self.named_modules():
+            for pn, _p in m.named_parameters():
+                fpn = f"{mn}.{pn}" if mn else pn  # full param name
+                if pn.endswith("bias"):
+                    # all biases will not be decayed
+                    no_decay.add(fpn)
+                elif pn.endswith("weight") and isinstance(m, whitelist_weight_modules):
+                    # weights of whitelist modules will be weight decayed
+                    decay.add(fpn)
+                elif pn.endswith("weight") and isinstance(m, blacklist_weight_modules):
+                    # weights of blacklist modules will NOT be weight decayed
+                    no_decay.add(fpn)
+
+        # validate that we considered every parameter
+        param_dict = dict(self.named_parameters())
+        inter_params = decay & no_decay
+        union_params = decay | no_decay
+        assert len(inter_params) == 0, (
+            f"parameters {str(inter_params)} made it into both decay/no_decay sets!"
+        )
+        assert len(param_dict.keys() - union_params) == 0, (
+            f"parameters {str(param_dict.keys() - union_params)} were not separated into either decay/no_decay set!"
+        )
+
+        decay = [param_dict[pn] for pn in sorted(decay)]
+        no_decay = [param_dict[pn] for pn in sorted(no_decay)]
+        # return the parameters that require weight decay, and the parameters that don't separately.
+        return decay, no_decay
+
+
+"""
+This file is a part for Residual Vector Quantization that utilizes code from the following repository:
+
+    - Phil Wang's vector-quantize-pytorch implementation in PyTorch.
+        Original source: https://github.com/lucidrains/vector-quantize-pytorch
+
+    - The vector-quantize-pytorch code is licensed under the MIT License:
+
+        MIT License
+
+        Copyright (c) 2020 Phil Wang
+
+        Permission is hereby granted, free of charge, to any person obtaining a copy
+        of this software and associated documentation files (the "Software"), to deal
+        in the Software without restriction, including without limitation the rights
+        to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+        copies of the Software, and to permit persons to whom the Software is
+        furnished to do so, subject to the following conditions:
+
+        The above copyright notice and this permission notice shall be included in all
+        copies or substantial portions of the Software.
+
+        THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+        IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+        FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+        AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+        LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+        OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+        SOFTWARE.
+
+    - We've made some changes to the original code to adapt it to our needs.
+
+        class ResidualVQ(nn.Module):
+            - added `self.register_buffer('freeze_codebook', torch.tensor(False))` to the __init__ method:
+                This enables the user to save an indicator whether the codebook is frozen or not.
+            - changed the name of function `get_codes_from_indices` → `get_codebook_vector_from_indices`:
+                This is to make the function name more descriptive.
+
+        class VectorQuantize(nn.Module):
+            - removed the `use_cosine_sim` and `layernorm_after_project_in` parameters from the __init__ method:
+                These parameters are not used in the code.
+            - changed the name of function `get_codes_from_indices` → `get_codebook_vector_from_indices`:
+                This is to make the function name more descriptive.
+
+"""
+
+
+class ResidualVQ(nn.Module):
+    """
+    Residual VQ is composed of multiple VectorQuantize layers.
+
+    Follows Algorithm 1. in https://huggingface.co/papers/2107.03312
+        "Residual Vector Quantizer (a.k.a. multi-stage vector quantizer [36]) cascades Nq layers of VQ as follows. The unquantized input vector is
+        passed through a first VQ and quantization residuals are computed. The residuals are then iteratively quantized by a sequence of additional
+        Nq -1 vector quantizers, as described in Algorithm 1."
+
+
+    self.project_in: function for projecting input to codebook dimension
+    self.project_out: function for projecting codebook dimension to output dimension
+    self.layers: nn.ModuleList of VectorQuantize layers that contains Nq layers of VQ as described in the paper.
+    self.freeze_codebook: buffer to save an indicator whether the codebook is frozen or not. VQ-BeT will check this to determine whether to update the codebook or not.
+    """
+
+    def __init__(
+        self,
+        *,
+        dim,
+        num_quantizers,
+        codebook_dim=None,
+        shared_codebook=False,
+        heads=1,
+        quantize_dropout=False,
+        quantize_dropout_cutoff_index=0,
+        quantize_dropout_multiple_of=1,
+        accept_image_fmap=False,
+        **kwargs,
+    ):
+        super().__init__()
+        assert heads == 1, "residual vq is not compatible with multi-headed codes"
+        codebook_dim = codebook_dim if (codebook_dim is not None) else dim
+        codebook_input_dim = codebook_dim * heads
+
+        requires_projection = codebook_input_dim != dim
+        self.project_in = nn.Linear(dim, codebook_input_dim) if requires_projection else nn.Identity()
+        self.project_out = nn.Linear(codebook_input_dim, dim) if requires_projection else nn.Identity()
+
+        self.num_quantizers = num_quantizers
+
+        self.accept_image_fmap = accept_image_fmap
+        self.layers = nn.ModuleList(
+            [
+                VectorQuantize(
+                    dim=codebook_dim, codebook_dim=codebook_dim, accept_image_fmap=accept_image_fmap, **kwargs
+                )
+                for _ in range(num_quantizers)
+            ]
+        )
+
+        self.quantize_dropout = quantize_dropout and num_quantizers > 1
+
+        assert quantize_dropout_cutoff_index >= 0
+
+        self.register_buffer("freeze_codebook", torch.tensor(False))
+        self.quantize_dropout_cutoff_index = quantize_dropout_cutoff_index
+        self.quantize_dropout_multiple_of = quantize_dropout_multiple_of  # encodec paper proposes structured dropout, believe this was set to 4
+
+        if not shared_codebook:
+            return
+
+        first_vq, *rest_vq = self.layers
+        codebook = first_vq._codebook
+
+        for vq in rest_vq:
+            vq._codebook = codebook
+
+    @property
+    def codebooks(self):
+        codebooks = [layer._codebook.embed for layer in self.layers]
+        codebooks = torch.stack(codebooks, dim=0)
+        codebooks = rearrange(codebooks, "q 1 c d -> q c d")
+        return codebooks
+
+    def get_codebook_vector_from_indices(self, indices):
+        # this function will return the codes from all codebooks across layers corresponding to the indices
+        batch, quantize_dim = indices.shape[0], indices.shape[-1]
+
+        # may also receive indices in the shape of 'b h w q' (accept_image_fmap)
+
+        indices, ps = pack([indices], "b * q")
+
+        # because of quantize dropout, one can pass in indices that are coarse
+        # and the network should be able to reconstruct
+
+        if quantize_dim < self.num_quantizers:
+            assert self.quantize_dropout > 0.0, (
+                "quantize dropout must be greater than 0 if you wish to reconstruct from a signal with less fine quantizations"
+            )
+            indices = F.pad(indices, (0, self.num_quantizers - quantize_dim), value=-1)
+
+        # get ready for gathering
+
+        codebooks = repeat(self.codebooks, "q c d -> q b c d", b=batch)
+        gather_indices = repeat(indices, "b n q -> q b n d", d=codebooks.shape[-1])
+
+        # take care of quantizer dropout
+
+        mask = gather_indices == -1.0
+        gather_indices = gather_indices.masked_fill(
+            mask, 0
+        )  # have it fetch a dummy code to be masked out later
+
+        all_codes = codebooks.gather(2, gather_indices)  # gather all codes
+
+        # mask out any codes that were dropout-ed
+
+        all_codes = all_codes.masked_fill(mask, 0.0)
+
+        # if (accept_image_fmap = True) then return shape (quantize, batch, height, width, dimension)
+
+        (all_codes,) = unpack(all_codes, ps, "q b * d")
+
+        return all_codes
+
+    def forward(self, x, indices=None, return_all_codes=False, sample_codebook_temp=None):
+        """
+        For given input tensor x, this function will return the quantized output, the indices of the quantized output, and the loss.
+        First, the input tensor x is projected to the codebook dimension. Then, the input tensor x is passed through Nq layers of VectorQuantize.
+        The residual value of each layer is fed to the next layer.
+        """
+        num_quant, quant_dropout_multiple_of, return_loss, device = (
+            self.num_quantizers,
+            self.quantize_dropout_multiple_of,
+            (indices is not None),
+            x.device,
+        )
+
+        x = self.project_in(x)
+
+        assert not (self.accept_image_fmap and (indices is not None))
+
+        quantized_out = 0.0
+        residual = x
+
+        all_losses = []
+        all_indices = []
+
+        if return_loss:
+            assert not torch.any(indices == -1), (
+                "some of the residual vq indices were dropped out. please use indices derived when the module is in eval mode to derive cross entropy loss"
+            )
+            ce_losses = []
+
+        should_quantize_dropout = self.training and self.quantize_dropout and not return_loss
+
+        # sample a layer index at which to dropout further residual quantization
+        # also prepare null indices and loss
+
+        if should_quantize_dropout:
+            rand_quantize_dropout_index = randrange(self.quantize_dropout_cutoff_index, num_quant)
+
+            if quant_dropout_multiple_of != 1:
+                rand_quantize_dropout_index = (
+                    ceil((rand_quantize_dropout_index + 1) / quant_dropout_multiple_of)
+                    * quant_dropout_multiple_of
+                    - 1
+                )
+
+            null_indices_shape = (x.shape[0], *x.shape[-2:]) if self.accept_image_fmap else tuple(x.shape[:2])
+            null_indices = torch.full(null_indices_shape, -1.0, device=device, dtype=torch.long)
+            null_loss = torch.full((1,), 0.0, device=device, dtype=x.dtype)
+
+        # go through the layers
+
+        for quantizer_index, layer in enumerate(self.layers):
+            if should_quantize_dropout and quantizer_index > rand_quantize_dropout_index:
+                all_indices.append(null_indices)
+                all_losses.append(null_loss)
+                continue
+
+            layer_indices = None
+            if return_loss:
+                layer_indices = indices[..., quantizer_index]
+
+            quantized, *rest = layer(
+                residual,
+                indices=layer_indices,
+                sample_codebook_temp=sample_codebook_temp,
+                freeze_codebook=self.freeze_codebook,
+            )
+
+            residual = residual - quantized.detach()
+            quantized_out = quantized_out + quantized
+
+            if return_loss:
+                ce_loss = rest[0]
+                ce_losses.append(ce_loss)
+                continue
+
+            embed_indices, loss = rest
+
+            all_indices.append(embed_indices)
+            all_losses.append(loss)
+
+        # project out, if needed
+
+        quantized_out = self.project_out(quantized_out)
+
+        # whether to early return the cross entropy loss
+
+        if return_loss:
+            return quantized_out, sum(ce_losses)
+
+        # stack all losses and indices
+
+        all_losses, all_indices = map(partial(torch.stack, dim=-1), (all_losses, all_indices))
+
+        ret = (quantized_out, all_indices, all_losses)
+
+        if return_all_codes:
+            # whether to return all codes from all codebooks across layers
+            all_codes = self.get_codebook_vector_from_indices(all_indices)
+
+            # will return all codes in shape (quantizer, batch, sequence length, codebook dimension)
+            ret = (*ret, all_codes)
+
+        return ret
+
+
+class VectorQuantize(nn.Module):
+    def __init__(
+        self,
+        dim,
+        codebook_size,
+        codebook_dim=None,
+        heads=1,
+        separate_codebook_per_head=False,
+        decay=0.8,
+        eps=1e-5,
+        kmeans_init=False,
+        kmeans_iters=10,
+        sync_kmeans=True,
+        threshold_ema_dead_code=0,
+        channel_last=True,
+        accept_image_fmap=False,
+        commitment_weight=1.0,
+        commitment_use_cross_entropy_loss=False,
+        orthogonal_reg_weight=0.0,
+        orthogonal_reg_active_codes_only=False,
+        orthogonal_reg_max_codes=None,
+        stochastic_sample_codes=False,
+        sample_codebook_temp=1.0,
+        straight_through=False,
+        reinmax=False,  # using reinmax for improved straight-through, assuming straight through helps at all
+        sync_codebook=None,
+        sync_affine_param=False,
+        ema_update=True,
+        learnable_codebook=False,
+        in_place_codebook_optimizer: Callable[
+            ..., Optimizer
+        ] = None,  # Optimizer used to update the codebook embedding if using learnable_codebook
+        affine_param=False,
+        affine_param_batch_decay=0.99,
+        affine_param_codebook_decay=0.9,
+        sync_update_v=0.0,  # the v that controls optimistic vs pessimistic update for synchronous update rule (21) https://minyoungg.github.io/vqtorch/assets/draft_050523.pdf
+    ):
+        super().__init__()
+        self.dim = dim
+        self.heads = heads
+        self.separate_codebook_per_head = separate_codebook_per_head
+
+        codebook_dim = codebook_dim if (codebook_dim is not None) else dim
+        codebook_input_dim = codebook_dim * heads
+
+        requires_projection = codebook_input_dim != dim
+        self.project_in = nn.Linear(dim, codebook_input_dim) if requires_projection else nn.Identity()
+        self.project_out = nn.Linear(codebook_input_dim, dim) if requires_projection else nn.Identity()
+
+        self.eps = eps
+        self.commitment_weight = commitment_weight
+        self.commitment_use_cross_entropy_loss = commitment_use_cross_entropy_loss  # whether to use cross entropy loss to codebook as commitment loss
+
+        self.learnable_codebook = learnable_codebook
+
+        has_codebook_orthogonal_loss = orthogonal_reg_weight > 0
+        self.has_codebook_orthogonal_loss = has_codebook_orthogonal_loss
+        self.orthogonal_reg_weight = orthogonal_reg_weight
+        self.orthogonal_reg_active_codes_only = orthogonal_reg_active_codes_only
+        self.orthogonal_reg_max_codes = orthogonal_reg_max_codes
+
+        assert not (ema_update and learnable_codebook), "learnable codebook not compatible with EMA update"
+
+        assert 0 <= sync_update_v <= 1.0
+        assert not (sync_update_v > 0.0 and not learnable_codebook), "learnable codebook must be turned on"
+
+        self.sync_update_v = sync_update_v
+
+        gumbel_sample_fn = partial(
+            gumbel_sample,
+            stochastic=stochastic_sample_codes,
+            reinmax=reinmax,
+            straight_through=straight_through,
+        )
+
+        if sync_codebook is None:
+            sync_codebook = distributed.is_initialized() and distributed.get_world_size() > 1
+
+        codebook_kwargs = {
+            "dim": codebook_dim,
+            "num_codebooks": heads if separate_codebook_per_head else 1,
+            "codebook_size": codebook_size,
+            "kmeans_init": kmeans_init,
+            "kmeans_iters": kmeans_iters,
+            "sync_kmeans": sync_kmeans,
+            "decay": decay,
+            "eps": eps,
+            "threshold_ema_dead_code": threshold_ema_dead_code,
+            "use_ddp": sync_codebook,
+            "learnable_codebook": has_codebook_orthogonal_loss or learnable_codebook,
+            "sample_codebook_temp": sample_codebook_temp,
+            "gumbel_sample": gumbel_sample_fn,
+            "ema_update": ema_update,
+        }
+
+        if affine_param:
+            codebook_kwargs = dict(
+                **codebook_kwargs,
+                affine_param=True,
+                sync_affine_param=sync_affine_param,
+                affine_param_batch_decay=affine_param_batch_decay,
+                affine_param_codebook_decay=affine_param_codebook_decay,
+            )
+
+        self._codebook = EuclideanCodebook(**codebook_kwargs)
+
+        self.in_place_codebook_optimizer = (
+            in_place_codebook_optimizer(self._codebook.parameters())
+            if (in_place_codebook_optimizer is not None)
+            else None
+        )
+
+        self.codebook_size = codebook_size
+
+        self.accept_image_fmap = accept_image_fmap
+        self.channel_last = channel_last
+
+    @property
+    def codebook(self):
+        codebook = self._codebook.embed
+
+        if self.separate_codebook_per_head:
+            return codebook
+
+        return rearrange(codebook, "1 ... -> ...")
+
+    @codebook.setter
+    def codebook(self, codes):
+        if not self.separate_codebook_per_head:
+            codes = rearrange(codes, "... -> 1 ...")
+
+        self._codebook.embed.copy_(codes)
+
+    def get_codebook_vector_from_indices(self, indices):
+        codebook = self.codebook
+        is_multiheaded = codebook.ndim > 2
+
+        if not is_multiheaded:
+            codes = codebook[indices]
+            return rearrange(codes, "... h d -> ... (h d)")
+
+        indices, ps = pack_one(indices, "b * h")
+        indices = rearrange(indices, "b n h -> b h n")
+
+        indices = repeat(indices, "b h n -> b h n d", d=codebook.shape[-1])
+        codebook = repeat(codebook, "h n d -> b h n d", b=indices.shape[0])
+
+        codes = codebook.gather(2, indices)
+        codes = rearrange(codes, "b h n d -> b n (h d)")
+        codes = unpack_one(codes, ps, "b * d")
+        return codes
+
+    def forward(
+        self,
+        x,
+        indices=None,
+        mask=None,
+        sample_codebook_temp=None,
+        freeze_codebook=False,
+    ):
+        orig_input = x
+
+        only_one = x.ndim == 2
+
+        if only_one:
+            assert mask is None
+            x = rearrange(x, "b d -> b 1 d")
+
+        shape, device, heads, is_multiheaded, _codebook_size, return_loss = (
+            x.shape,
+            x.device,
+            self.heads,
+            self.heads > 1,
+            self.codebook_size,
+            (indices is not None),
+        )
+
+        need_transpose = not self.channel_last and not self.accept_image_fmap
+        should_inplace_optimize = self.in_place_codebook_optimizer is not None
+
+        # rearrange inputs
+
+        if self.accept_image_fmap:
+            height, width = x.shape[-2:]
+            x = rearrange(x, "b c h w -> b (h w) c")
+
+        if need_transpose:
+            x = rearrange(x, "b d n -> b n d")
+
+        # project input
+
+        x = self.project_in(x)
+
+        # handle multi-headed separate codebooks
+
+        if is_multiheaded:
+            ein_rhs_eq = "h b n d" if self.separate_codebook_per_head else "1 (b h) n d"
+            x = rearrange(x, f"b n (h d) -> {ein_rhs_eq}", h=heads)
+
+        # l2norm for cosine sim, otherwise identity
+
+        x = self._codebook.transform_input(x)
+
+        # codebook forward kwargs
+
+        codebook_forward_kwargs = {
+            "sample_codebook_temp": sample_codebook_temp,
+            "mask": mask,
+            "freeze_codebook": freeze_codebook,
+        }
+
+        # quantize
+
+        quantize, embed_ind, distances = self._codebook(x, **codebook_forward_kwargs)
+
+        # one step in-place update
+
+        if should_inplace_optimize and self.training and not freeze_codebook:
+            if mask is not None:
+                loss = F.mse_loss(quantize, x.detach(), reduction="none")
+
+                loss_mask = mask
+                if is_multiheaded:
+                    loss_mask = repeat(
+                        mask,
+                        "b n -> c (b h) n",
+                        c=loss.shape[0],
+                        h=loss.shape[1] // mask.shape[0],
+                    )
+
+                loss = loss[loss_mask].mean()
+
+            else:
+                loss = F.mse_loss(quantize, x.detach())
+
+            loss.backward()
+            self.in_place_codebook_optimizer.step()
+            self.in_place_codebook_optimizer.zero_grad()
+
+            # quantize again
+
+            quantize, embed_ind, distances = self._codebook(x, **codebook_forward_kwargs)
+
+        if self.training:
+            # determine code to use for commitment loss
+            maybe_detach = torch.detach if not self.learnable_codebook or freeze_codebook else identity
+
+            commit_quantize = maybe_detach(quantize)
+
+            # straight through
+
+            quantize = x + (quantize - x).detach()
+
+            if self.sync_update_v > 0.0:
+                # (21) in https://minyoungg.github.io/vqtorch/assets/draft_050523.pdf
+                quantize = quantize + self.sync_update_v * (quantize - quantize.detach())
+
+        # function for calculating cross entropy loss to distance matrix
+        # used for (1) naturalspeech2 training residual vq latents to be close to the correct codes and (2) cross-entropy based commitment loss
+
+        def calculate_ce_loss(codes):
+            if not is_multiheaded:
+                dist_einops_eq = "1 b n l -> b l n"
+            elif self.separate_codebook_per_head:
+                dist_einops_eq = "c b n l -> b l n c"
+            else:
+                dist_einops_eq = "1 (b h) n l -> b l n h"
+
+            ce_loss = F.cross_entropy(
+                rearrange(distances, dist_einops_eq, b=shape[0]), codes, ignore_index=-1
+            )
+
+            return ce_loss
+
+        # if returning cross entropy loss on codes that were passed in
+
+        if return_loss:
+            return quantize, calculate_ce_loss(indices)
+
+        # transform embedding indices
+
+        if is_multiheaded:
+            if self.separate_codebook_per_head:
+                embed_ind = rearrange(embed_ind, "h b n -> b n h", h=heads)
+            else:
+                embed_ind = rearrange(embed_ind, "1 (b h) n -> b n h", h=heads)
+
+        if self.accept_image_fmap:
+            embed_ind = rearrange(embed_ind, "b (h w) ... -> b h w ...", h=height, w=width)
+
+        if only_one:
+            embed_ind = rearrange(embed_ind, "b 1 -> b")
+
+        # aggregate loss
+
+        loss = torch.tensor([0.0], device=device, requires_grad=self.training)
+
+        if self.training:
+            if self.commitment_weight > 0:
+                if self.commitment_use_cross_entropy_loss:
+                    if mask is not None:
+                        ce_loss_mask = mask
+                        if is_multiheaded:
+                            ce_loss_mask = repeat(ce_loss_mask, "b n -> b n h", h=heads)
+
+                        embed_ind.masked_fill_(~ce_loss_mask, -1)
+
+                    commit_loss = calculate_ce_loss(embed_ind)
+                else:
+                    if mask is not None:
+                        # with variable lengthed sequences
+                        commit_loss = F.mse_loss(commit_quantize, x, reduction="none")
+
+                        loss_mask = mask
+                        if is_multiheaded:
+                            loss_mask = repeat(
+                                loss_mask,
+                                "b n -> c (b h) n",
+                                c=commit_loss.shape[0],
+                                h=commit_loss.shape[1] // mask.shape[0],
+                            )
+
+                        commit_loss = commit_loss[loss_mask].mean()
+                    else:
+                        commit_loss = F.mse_loss(commit_quantize, x)
+
+                loss = loss + commit_loss * self.commitment_weight
+
+            if self.has_codebook_orthogonal_loss:
+                codebook = self._codebook.embed
+
+                # only calculate orthogonal loss for the activated codes for this batch
+
+                if self.orthogonal_reg_active_codes_only:
+                    assert not (is_multiheaded and self.separate_codebook_per_head), (
+                        "orthogonal regularization for only active codes not compatible with multi-headed with separate codebooks yet"
+                    )
+                    unique_code_ids = torch.unique(embed_ind)
+                    codebook = codebook[:, unique_code_ids]
+
+                num_codes = codebook.shape[-2]
+
+                if (self.orthogonal_reg_max_codes is not None) and num_codes > self.orthogonal_reg_max_codes:
+                    rand_ids = torch.randperm(num_codes, device=device)[: self.orthogonal_reg_max_codes]
+                    codebook = codebook[:, rand_ids]
+
+                orthogonal_reg_loss = orthogonal_loss_fn(codebook)
+                loss = loss + orthogonal_reg_loss * self.orthogonal_reg_weight
+
+        # handle multi-headed quantized embeddings
+
+        if is_multiheaded:
+            if self.separate_codebook_per_head:
+                quantize = rearrange(quantize, "h b n d -> b n (h d)", h=heads)
+            else:
+                quantize = rearrange(quantize, "1 (b h) n d -> b n (h d)", h=heads)
+
+        # project out
+
+        quantize = self.project_out(quantize)
+
+        # rearrange quantized embeddings
+
+        if need_transpose:
+            quantize = rearrange(quantize, "b n d -> b d n")
+
+        if self.accept_image_fmap:
+            quantize = rearrange(quantize, "b (h w) c -> b c h w", h=height, w=width)
+
+        if only_one:
+            quantize = rearrange(quantize, "b 1 d -> b d")
+
+        # if masking, only return quantized for where mask has True
+
+        if mask is not None:
+            quantize = torch.where(rearrange(mask, "... -> ... 1"), quantize, orig_input)
+
+        return quantize, embed_ind, loss
+
+
+def noop(*args, **kwargs):
+    pass
+
+
+def identity(t):
+    return t
+
+
+def cdist(x, y):
+    x2 = reduce(x**2, "b n d -> b n", "sum")
+    y2 = reduce(y**2, "b n d -> b n", "sum")
+    xy = einsum("b i d, b j d -> b i j", x, y) * -2
+    return (rearrange(x2, "b i -> b i 1") + rearrange(y2, "b j -> b 1 j") + xy).sqrt()
+
+
+def log(t, eps=1e-20):
+    return torch.log(t.clamp(min=eps))
+
+
+def ema_inplace(old, new, decay):
+    is_mps = str(old.device).startswith("mps:")
+
+    if not is_mps:
+        old.lerp_(new, 1 - decay)
+    else:
+        old.mul_(decay).add_(new * (1 - decay))
+
+
+def pack_one(t, pattern):
+    return pack([t], pattern)
+
+
+def unpack_one(t, ps, pattern):
+    return unpack(t, ps, pattern)[0]
+
+
+def uniform_init(*shape):
+    t = torch.empty(shape)
+    nn.init.kaiming_uniform_(t)
+    return t
+
+
+def gumbel_noise(t):
+    noise = torch.zeros_like(t).uniform_(0, 1)
+    return -log(-log(noise))
+
+
+def gumbel_sample(
+    logits,
+    temperature=1.0,
+    stochastic=False,
+    straight_through=False,
+    reinmax=False,
+    dim=-1,
+    training=True,
+):
+    dtype, size = logits.dtype, logits.shape[dim]
+
+    if training and stochastic and temperature > 0:
+        sampling_logits = (logits / temperature) + gumbel_noise(logits)
+    else:
+        sampling_logits = logits
+
+    ind = sampling_logits.argmax(dim=dim)
+    one_hot = F.one_hot(ind, size).type(dtype)
+
+    assert not (reinmax and not straight_through), (
+        "reinmax can only be turned on if using straight through gumbel softmax"
+    )
+
+    if not straight_through or temperature <= 0.0 or not training:
+        return ind, one_hot
+
+    # use reinmax for better second-order accuracy - https://huggingface.co/papers/2304.08612
+    # algorithm 2
+
+    if reinmax:
+        π0 = logits.softmax(dim=dim)
+        π1 = (one_hot + (logits / temperature).softmax(dim=dim)) / 2
+        π1 = ((log(π1) - logits).detach() + logits).softmax(dim=1)
+        π2 = 2 * π1 - 0.5 * π0
+        one_hot = π2 - π2.detach() + one_hot
+    else:
+        π1 = (logits / temperature).softmax(dim=dim)
+        one_hot = one_hot + π1 - π1.detach()
+
+    return ind, one_hot
+
+
+def laplace_smoothing(x, n_categories, eps=1e-5, dim=-1):
+    denom = x.sum(dim=dim, keepdim=True)
+    return (x + eps) / (denom + n_categories * eps)
+
+
+def sample_vectors(samples, num):
+    num_samples, device = samples.shape[0], samples.device
+    if num_samples >= num:
+        indices = torch.randperm(num_samples, device=device)[:num]
+    else:
+        indices = torch.randint(0, num_samples, (num,), device=device)
+
+    return samples[indices]
+
+
+def batched_sample_vectors(samples, num):
+    return torch.stack([sample_vectors(sample, num) for sample in samples.unbind(dim=0)], dim=0)
+
+
+def pad_shape(shape, size, dim=0):
+    return [size if i == dim else s for i, s in enumerate(shape)]
+
+
+def sample_multinomial(total_count, probs):
+    device = probs.device
+    probs = probs.cpu()
+
+    total_count = probs.new_full((), total_count)
+    remainder = probs.new_ones(())
+    sample = torch.empty_like(probs, dtype=torch.long)
+
+    for i, p in enumerate(probs):
+        s = torch.binomial(total_count, p / remainder)
+        sample[i] = s
+        total_count -= s
+        remainder -= p
+
+    return sample.to(device)
+
+
+def all_gather_sizes(x, dim):
+    size = torch.tensor(x.shape[dim], dtype=torch.long, device=x.device)
+    all_sizes = [torch.empty_like(size) for _ in range(distributed.get_world_size())]
+    distributed.all_gather(all_sizes, size)
+    return torch.stack(all_sizes)
+
+
+def all_gather_variably_sized(x, sizes, dim=0):
+    rank = distributed.get_rank()
+    all_x = []
+
+    for i, size in enumerate(sizes):
+        t = x if i == rank else x.new_empty(pad_shape(x.shape, size, dim))
+        distributed.broadcast(t, src=i, async_op=True)
+        all_x.append(t)
+
+    distributed.barrier()
+    return all_x
+
+
+def sample_vectors_distributed(local_samples, num):
+    local_samples = rearrange(local_samples, "1 ... -> ...")
+
+    rank = distributed.get_rank()
+    all_num_samples = all_gather_sizes(local_samples, dim=0)
+
+    if rank == 0:
+        samples_per_rank = sample_multinomial(num, all_num_samples / all_num_samples.sum())
+    else:
+        samples_per_rank = torch.empty_like(all_num_samples)
+
+    distributed.broadcast(samples_per_rank, src=0)
+    samples_per_rank = samples_per_rank.tolist()
+
+    local_samples = sample_vectors(local_samples, samples_per_rank[rank])
+    all_samples = all_gather_variably_sized(local_samples, samples_per_rank, dim=0)
+    out = torch.cat(all_samples, dim=0)
+
+    return rearrange(out, "... -> 1 ...")
+
+
+def batched_bincount(x, *, minlength):
+    batch, dtype, device = x.shape[0], x.dtype, x.device
+    target = torch.zeros(batch, minlength, dtype=dtype, device=device)
+    values = torch.ones_like(x)
+    target.scatter_add_(-1, x, values)
+    return target
+
+
+def kmeans(
+    samples,
+    num_clusters,
+    num_iters=10,
+    sample_fn=batched_sample_vectors,
+    all_reduce_fn=noop,
+):
+    num_codebooks, dim, dtype, _device = (
+        samples.shape[0],
+        samples.shape[-1],
+        samples.dtype,
+        samples.device,
+    )
+
+    means = sample_fn(samples, num_clusters)
+
+    for _ in range(num_iters):
+        dists = -torch.cdist(samples, means, p=2)
+
+        buckets = torch.argmax(dists, dim=-1)
+        bins = batched_bincount(buckets, minlength=num_clusters)
+        all_reduce_fn(bins)
+
+        zero_mask = bins == 0
+        bins_min_clamped = bins.masked_fill(zero_mask, 1)
+
+        new_means = buckets.new_zeros(num_codebooks, num_clusters, dim, dtype=dtype)
+
+        new_means.scatter_add_(1, repeat(buckets, "h n -> h n d", d=dim), samples)
+        new_means = new_means / rearrange(bins_min_clamped, "... -> ... 1")
+        all_reduce_fn(new_means)
+
+        means = torch.where(rearrange(zero_mask, "... -> ... 1"), means, new_means)
+
+    return means, bins
+
+
+def batched_embedding(indices, embeds):
+    batch, dim = indices.shape[1], embeds.shape[-1]
+    indices = repeat(indices, "h b n -> h b n d", d=dim)
+    embeds = repeat(embeds, "h c d -> h b c d", b=batch)
+    return embeds.gather(2, indices)
+
+
+def orthogonal_loss_fn(t):
+    # eq (2) from https://huggingface.co/papers/2112.00384
+    h, n = t.shape[:2]
+    normed_codes = F.normalize(t, p=2, dim=-1)
+    cosine_sim = einsum("h i d, h j d -> h i j", normed_codes, normed_codes)
+    return (cosine_sim**2).sum() / (h * n**2) - (1 / n)
+
+
+class EuclideanCodebook(nn.Module):
+    def __init__(
+        self,
+        dim,
+        codebook_size,
+        num_codebooks=1,
+        kmeans_init=False,
+        kmeans_iters=10,
+        sync_kmeans=True,
+        decay=0.8,
+        eps=1e-5,
+        threshold_ema_dead_code=2,
+        reset_cluster_size=None,
+        use_ddp=False,
+        learnable_codebook=False,
+        gumbel_sample=gumbel_sample,
+        sample_codebook_temp=1.0,
+        ema_update=True,
+        affine_param=False,
+        sync_affine_param=False,
+        affine_param_batch_decay=0.99,
+        affine_param_codebook_decay=0.9,
+    ):
+        super().__init__()
+        self.transform_input = identity
+
+        self.decay = decay
+        self.ema_update = ema_update
+
+        init_fn = uniform_init if not kmeans_init else torch.zeros
+        embed = init_fn(num_codebooks, codebook_size, dim)
+
+        self.codebook_size = codebook_size
+        self.num_codebooks = num_codebooks
+
+        self.kmeans_iters = kmeans_iters
+        self.eps = eps
+        self.threshold_ema_dead_code = threshold_ema_dead_code
+        self.reset_cluster_size = (
+            reset_cluster_size if (reset_cluster_size is not None) else threshold_ema_dead_code
+        )
+
+        assert callable(gumbel_sample)
+        self.gumbel_sample = gumbel_sample
+        self.sample_codebook_temp = sample_codebook_temp
+
+        assert not (use_ddp and num_codebooks > 1 and kmeans_init), (
+            "kmeans init is not compatible with multiple codebooks in distributed environment for now"
+        )
+
+        self.sample_fn = sample_vectors_distributed if use_ddp and sync_kmeans else batched_sample_vectors
+        self.kmeans_all_reduce_fn = distributed.all_reduce if use_ddp and sync_kmeans else noop
+        self.all_reduce_fn = distributed.all_reduce if use_ddp else noop
+
+        self.register_buffer("initted", torch.Tensor([not kmeans_init]))
+        self.register_buffer("cluster_size", torch.zeros(num_codebooks, codebook_size))
+        self.register_buffer("embed_avg", embed.clone())
+
+        self.learnable_codebook = learnable_codebook
+        if learnable_codebook:
+            self.embed = nn.Parameter(embed)
+        else:
+            self.register_buffer("embed", embed)
+
+        # affine related params
+
+        self.affine_param = affine_param
+        self.sync_affine_param = sync_affine_param
+
+        if not affine_param:
+            return
+
+        self.affine_param_batch_decay = affine_param_batch_decay
+        self.affine_param_codebook_decay = affine_param_codebook_decay
+
+        self.register_buffer("batch_mean", None)
+        self.register_buffer("batch_variance", None)
+
+        self.register_buffer("codebook_mean_needs_init", torch.Tensor([True]))
+        self.register_buffer("codebook_mean", torch.empty(num_codebooks, 1, dim))
+        self.register_buffer("codebook_variance_needs_init", torch.Tensor([True]))
+        self.register_buffer("codebook_variance", torch.empty(num_codebooks, 1, dim))
+
+    @torch.jit.ignore
+    def init_embed_(self, data, mask=None):
+        if self.initted:
+            return
+
+        if mask is not None:
+            c = data.shape[0]
+            data = rearrange(data[mask], "(c n) d -> c n d", c=c)
+
+        embed, cluster_size = kmeans(
+            data,
+            self.codebook_size,
+            self.kmeans_iters,
+            sample_fn=self.sample_fn,
+            all_reduce_fn=self.kmeans_all_reduce_fn,
+        )
+
+        embed_sum = embed * rearrange(cluster_size, "... -> ... 1")
+
+        self.embed.data.copy_(embed)
+        self.embed_avg.data.copy_(embed_sum)
+        self.cluster_size.data.copy_(cluster_size)
+        self.initted.data.copy_(torch.Tensor([True]))
+
+    @torch.jit.ignore
+    def update_with_decay(self, buffer_name, new_value, decay):
+        old_value = getattr(self, buffer_name)
+
+        needs_init = getattr(self, buffer_name + "_needs_init", False)
+
+        if needs_init:
+            self.register_buffer(buffer_name + "_needs_init", torch.Tensor([False]))
+
+        if not (old_value is not None) or needs_init:
+            self.register_buffer(buffer_name, new_value.detach())
+
+            return
+
+        value = old_value * decay + new_value.detach() * (1 - decay)
+        self.register_buffer(buffer_name, value)
+
+    @torch.jit.ignore
+    def update_affine(self, data, embed, mask=None):
+        assert self.affine_param
+
+        var_fn = partial(torch.var, unbiased=False)
+
+        # calculate codebook mean and variance
+
+        embed = rearrange(embed, "h ... d -> h (...) d")
+
+        if self.training:
+            self.update_with_decay(
+                "codebook_mean",
+                reduce(embed, "h n d -> h 1 d", "mean"),
+                self.affine_param_codebook_decay,
+            )
+            self.update_with_decay(
+                "codebook_variance",
+                reduce(embed, "h n d -> h 1 d", var_fn),
+                self.affine_param_codebook_decay,
+            )
+
+        # prepare batch data, which depends on whether it has masking
+
+        data = rearrange(data, "h ... d -> h (...) d")
+
+        if mask is not None:
+            c = data.shape[0]
+            data = rearrange(data[mask], "(c n) d -> c n d", c=c)
+
+        # calculate batch mean and variance
+
+        if not self.sync_affine_param:
+            self.update_with_decay(
+                "batch_mean",
+                reduce(data, "h n d -> h 1 d", "mean"),
+                self.affine_param_batch_decay,
+            )
+            self.update_with_decay(
+                "batch_variance",
+                reduce(data, "h n d -> h 1 d", var_fn),
+                self.affine_param_batch_decay,
+            )
+            return
+
+        num_vectors, device, dtype = data.shape[-2], data.device, data.dtype
+
+        # number of vectors, for denominator
+
+        num_vectors = torch.tensor([num_vectors], device=device, dtype=dtype)
+        distributed.all_reduce(num_vectors)
+
+        # calculate distributed mean
+
+        batch_sum = reduce(data, "h n d -> h 1 d", "sum")
+        distributed.all_reduce(batch_sum)
+        batch_mean = batch_sum / num_vectors
+
+        self.update_with_decay("batch_mean", batch_mean, self.affine_param_batch_decay)
+
+        # calculate distributed variance
+
+        variance_number = reduce((data - batch_mean) ** 2, "h n d -> h 1 d", "sum")
+        distributed.all_reduce(variance_number)
+        batch_variance = variance_number / num_vectors
+
+        self.update_with_decay("batch_variance", batch_variance, self.affine_param_batch_decay)
+
+    def replace(self, batch_samples, batch_mask):
+        for ind, (samples, mask) in enumerate(
+            zip(batch_samples.unbind(dim=0), batch_mask.unbind(dim=0), strict=False)
+        ):
+            if not torch.any(mask):
+                continue
+
+            sampled = self.sample_fn(rearrange(samples, "... -> 1 ..."), mask.sum().item())
+            sampled = rearrange(sampled, "1 ... -> ...")
+
+            self.embed.data[ind][mask] = sampled
+
+            self.cluster_size.data[ind][mask] = self.reset_cluster_size
+            self.embed_avg.data[ind][mask] = sampled * self.reset_cluster_size
+
+    def expire_codes_(self, batch_samples):
+        if self.threshold_ema_dead_code == 0:
+            return
+
+        expired_codes = self.cluster_size < self.threshold_ema_dead_code
+
+        if not torch.any(expired_codes):
+            return
+
+        batch_samples = rearrange(batch_samples, "h ... d -> h (...) d")
+        self.replace(batch_samples, batch_mask=expired_codes)
+
+    @autocast(enabled=False)
+    def forward(self, x, sample_codebook_temp=None, mask=None, freeze_codebook=False):
+        needs_codebook_dim = x.ndim < 4
+        sample_codebook_temp = (
+            sample_codebook_temp if (sample_codebook_temp is not None) else self.sample_codebook_temp
+        )
+
+        x = x.float()
+
+        if needs_codebook_dim:
+            x = rearrange(x, "... -> 1 ...")
+
+        flatten, ps = pack_one(x, "h * d")
+
+        if mask is not None:
+            mask = repeat(
+                mask,
+                "b n -> c (b h n)",
+                c=flatten.shape[0],
+                h=flatten.shape[-2] // (mask.shape[0] * mask.shape[1]),
+            )
+
+        self.init_embed_(flatten, mask=mask)
+
+        if self.affine_param:
+            self.update_affine(flatten, self.embed, mask=mask)
+
+        embed = self.embed if self.learnable_codebook else self.embed.detach()
+
+        if self.affine_param:
+            codebook_std = self.codebook_variance.clamp(min=1e-5).sqrt()
+            batch_std = self.batch_variance.clamp(min=1e-5).sqrt()
+            embed = (embed - self.codebook_mean) * (batch_std / codebook_std) + self.batch_mean
+
+        dist = -cdist(flatten, embed)
+
+        embed_ind, embed_onehot = self.gumbel_sample(
+            dist, dim=-1, temperature=sample_codebook_temp, training=self.training
+        )
+
+        embed_ind = unpack_one(embed_ind, ps, "h *")
+
+        if self.training:
+            unpacked_onehot = unpack_one(embed_onehot, ps, "h * c")
+            quantize = einsum("h b n c, h c d -> h b n d", unpacked_onehot, embed)
+        else:
+            quantize = batched_embedding(embed_ind, embed)
+
+        if self.training and self.ema_update and not freeze_codebook:
+            if self.affine_param:
+                flatten = (flatten - self.batch_mean) * (codebook_std / batch_std) + self.codebook_mean
+
+            if mask is not None:
+                embed_onehot[~mask] = 0.0
+
+            cluster_size = embed_onehot.sum(dim=1)
+
+            self.all_reduce_fn(cluster_size)
+            ema_inplace(self.cluster_size.data, cluster_size, self.decay)
+
+            embed_sum = einsum("h n d, h n c -> h c d", flatten, embed_onehot)
+            self.all_reduce_fn(embed_sum.contiguous())
+            ema_inplace(self.embed_avg.data, embed_sum, self.decay)
+
+            cluster_size = laplace_smoothing(
+                self.cluster_size, self.codebook_size, self.eps
+            ) * self.cluster_size.sum(dim=-1, keepdim=True)
+
+            embed_normalized = self.embed_avg / rearrange(cluster_size, "... -> ... 1")
+            self.embed.data.copy_(embed_normalized)
+            self.expire_codes_(x)
+
+        if needs_codebook_dim:
+            quantize, embed_ind = tuple(rearrange(t, "1 ... -> ...") for t in (quantize, embed_ind))
+
+        dist = unpack_one(dist, ps, "h * d")
+
+        return quantize, embed_ind, dist
diff --git a/lerobot/src/lerobot/policies/wall_x/README.md b/lerobot/src/lerobot/policies/wall_x/README.md
new file mode 120000
index 0000000000000000000000000000000000000000..4426967dec8c249524d044299fcb012afa92f428
--- /dev/null
+++ b/lerobot/src/lerobot/policies/wall_x/README.md
@@ -0,0 +1 @@
+../../../../docs/source/policy_walloss_README.md
\ No newline at end of file
diff --git a/lerobot/src/lerobot/policies/wall_x/__init__.py b/lerobot/src/lerobot/policies/wall_x/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..d80c27bda89b6c61a1d6d5bce6d68c5b58899c26
--- /dev/null
+++ b/lerobot/src/lerobot/policies/wall_x/__init__.py
@@ -0,0 +1,19 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configuration_wall_x import WallXConfig
+
+__all__ = ["WallXConfig", "WallXPolicy", "make_wall_x_pre_post_processors"]
diff --git a/lerobot/src/lerobot/policies/wall_x/configuration_wall_x.py b/lerobot/src/lerobot/policies/wall_x/configuration_wall_x.py
new file mode 100644
index 0000000000000000000000000000000000000000..5269c4e10be88518f3211af45b479146fad77099
--- /dev/null
+++ b/lerobot/src/lerobot/policies/wall_x/configuration_wall_x.py
@@ -0,0 +1,166 @@
+# Copyright 2025 HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.optim.optimizers import AdamWConfig
+from lerobot.optim.schedulers import CosineDecayWithWarmupSchedulerConfig
+from lerobot.utils.constants import ACTION, OBS_STATE
+
+
+@PreTrainedConfig.register_subclass("wall_x")
+@dataclass
+class WallXConfig(PreTrainedConfig):
+    """
+    Configuration class for Wall-X policy.
+
+    Wall-X is based on Qwen2.5-VL with action prediction capabilities using flow matching.
+    It supports cross-embodiment robotic control through unified action representations.
+
+    This config supports multi-modal learning with vision, language, and action data.
+    """
+
+    # ==================== Input / Output Structure ====================
+    n_obs_steps: int = 1
+    chunk_size: int = 32  # action_horizon in wall-x
+    n_action_steps: int = 32
+
+    # Action dimension - wall-x uses 20
+    max_action_dim: int = 20
+    max_state_dim: int = 20  # For proprioception
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.MEAN_STD,
+            "ACTION": NormalizationMode.MEAN_STD,
+        }
+    )
+
+    # ==================== Action Prediction ====================
+    # Pretrained model paths
+    pretrained_name_or_path: str = "x-square-robot/wall-oss-flow"
+
+    # Tokenizer settings
+    action_tokenizer_path: str | None = "lerobot/fast-action-tokenizer"
+
+    # Action prediction mode: "diffusion" or "fast"
+    prediction_mode: str = "diffusion"
+
+    # Attention Implementation, options: "eager", "flash_attention_2", "sdpa"
+    # NOTE: flash-attn==2.7.4.post1 is required for flash_attention_2 implementation
+    attn_implementation: str = "eager"
+
+    # ==================== Optimizer Presets ====================
+    optimizer_lr: float = 2e-5
+    optimizer_betas: tuple[float, float] = (0.9, 0.95)
+    optimizer_eps: float = 1e-8
+    optimizer_weight_decay: float = 0.01
+    optimizer_grad_clip_norm: float = 1.0
+
+    scheduler_warmup_steps: int = 1000
+    scheduler_decay_steps: int = 100000
+    scheduler_decay_lr: float = 1e-6
+
+    def __post_init__(self):
+        super().__post_init__()
+
+        # Input validation
+        if self.n_action_steps > self.chunk_size:
+            raise ValueError(
+                f"The chunk size is the upper bound for the number of action steps per model invocation. Got "
+                f"{self.n_action_steps} for `n_action_steps` and {self.chunk_size} for `chunk_size`."
+            )
+
+        if self.prediction_mode not in ["diffusion", "fast"]:
+            raise ValueError(f"prediction_mode must be 'diffusion' or 'fast', got {self.prediction_mode}")
+
+        # Assign use_fast_tokenizer based on prediction_mode
+        if self.prediction_mode == "fast":
+            self.use_fast_tokenizer = True
+        elif self.prediction_mode == "diffusion":
+            self.use_fast_tokenizer = False
+            self.action_tokenizer_path = None  # disable action tokenizer for diffusion mode
+        else:
+            raise ValueError(f"prediction_mode must be 'diffusion' or 'fast', got {self.prediction_mode}")
+
+    def validate_features(self) -> None:
+        """Validate and set up input/output features."""
+        image_features = [key for key, feat in self.input_features.items() if feat.type == FeatureType.VISUAL]
+        if not image_features:
+            raise ValueError(
+                "Wall-X policy requires at least one visual input feature. "
+                "No features of type FeatureType.VISUAL found in input_features."
+            )
+
+        if OBS_STATE not in self.input_features:
+            state_feature = PolicyFeature(
+                type=FeatureType.STATE,
+                shape=(self.max_state_dim,),  # Padded to max_state_dim
+            )
+            self.input_features[OBS_STATE] = state_feature
+        else:
+            state_shape = self.input_features[OBS_STATE].shape
+            state_dim = state_shape[0] if state_shape else 0
+            if state_dim > self.max_state_dim:
+                raise ValueError(
+                    f"State dimension {state_dim} exceeds max_state_dim {self.max_state_dim}. "
+                    f"Either reduce state dimension or increase max_state_dim in config."
+                )
+
+        if ACTION not in self.output_features:
+            action_feature = PolicyFeature(
+                type=FeatureType.ACTION,
+                shape=(self.max_action_dim,),  # Padded to max_action_dim
+            )
+            self.output_features[ACTION] = action_feature
+        else:
+            action_shape = self.output_features[ACTION].shape
+            action_dim = action_shape[0] if action_shape else 0
+            if action_dim > self.max_action_dim:
+                raise ValueError(
+                    f"Action dimension {action_dim} exceeds max_action_dim {self.max_action_dim}. "
+                    f"Either reduce action dimension or increase max_action_dim in config."
+                )
+
+    def get_optimizer_preset(self) -> AdamWConfig:
+        return AdamWConfig(
+            lr=self.optimizer_lr,
+            betas=self.optimizer_betas,
+            eps=self.optimizer_eps,
+            weight_decay=self.optimizer_weight_decay,
+            grad_clip_norm=self.optimizer_grad_clip_norm,
+        )
+
+    def get_scheduler_preset(self):
+        return CosineDecayWithWarmupSchedulerConfig(
+            peak_lr=self.optimizer_lr,
+            decay_lr=self.scheduler_decay_lr,
+            num_warmup_steps=self.scheduler_warmup_steps,
+            num_decay_steps=self.scheduler_decay_steps,
+        )
+
+    @property
+    def observation_delta_indices(self) -> list:
+        return None
+
+    @property
+    def action_delta_indices(self) -> list:
+        return list(range(self.chunk_size))
+
+    @property
+    def reward_delta_indices(self) -> None:
+        return None
diff --git a/lerobot/src/lerobot/policies/wall_x/constant.py b/lerobot/src/lerobot/policies/wall_x/constant.py
new file mode 100644
index 0000000000000000000000000000000000000000..43e5e7fb60464c4e700302d3ea800ba8fb0fb8fd
--- /dev/null
+++ b/lerobot/src/lerobot/policies/wall_x/constant.py
@@ -0,0 +1,41 @@
+#!/usr/bin/env python
+
+# Copyright 2025 HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Wall-X Constants and Configuration Data.
+"""
+
+CAMERA_NAME_MAPPING = {
+    "face_view": "front view",
+    "left_wrist_view": "left wrist view",
+    "right_wrist_view": "right wrist view",
+    "move1_view": "move view",
+    "move2_view": "move view",
+    "wall_view": "wall view",
+    "top_view": "top view",
+}
+
+RESOLUTION = 256
+
+# Parameters for preprocessing
+MAX_PIXELS = 16384 * 28 * 28
+MIN_PIXELS = 4 * 28 * 28
+IMAGE_FACTOR = 28
+PRIORITY_ORDER = None
+GENERATE_SUBTASK_RATIO = 0.0
+MODEL_TYPE = "qwen2_5"
+
+TOKENIZER_MAX_LENGTH = 768
diff --git a/lerobot/src/lerobot/policies/wall_x/modeling_wall_x.py b/lerobot/src/lerobot/policies/wall_x/modeling_wall_x.py
new file mode 100644
index 0000000000000000000000000000000000000000..84ee05743666a18d8b08b69e1305315a471c8e29
--- /dev/null
+++ b/lerobot/src/lerobot/policies/wall_x/modeling_wall_x.py
@@ -0,0 +1,2018 @@
+#!/usr/bin/env python
+
+# Copyright 2025 HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Wall-X: Cross-embodiment robotic control using Qwen2.5-VL with flow matching.
+
+[Paper](https://github.com/x2-robot/wall-x)
+
+Install wall-x extra dependencies:
+```bash
+pip install -e ".[wall_x]"
+```
+
+Example of finetuning a wall-x model:
+```bash
+lerobot-train \
+--policy.type=wall_x \
+--dataset.repo_id=your/dataset \
+--batch_size=32 \
+--steps=100000
+```
+"""
+
+import math
+from collections import deque
+from os import PathLike
+from typing import Any
+
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from peft import LoraConfig, get_peft_model
+from PIL import Image
+from qwen_vl_utils.vision_process import smart_resize
+from torch import Tensor
+from torch.distributions import Beta
+from torch.nn import CrossEntropyLoss
+from torchdiffeq import odeint
+from transformers import AutoProcessor, BatchFeature
+from transformers.cache_utils import (
+    StaticCache,
+)
+from transformers.models.qwen2_5_vl.modeling_qwen2_5_vl import (
+    Qwen2_5_VLForConditionalGeneration,
+)
+from transformers.utils import is_torchdynamo_compiling, logging
+
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.utils import populate_queues
+from lerobot.policies.wall_x.configuration_wall_x import WallXConfig
+from lerobot.policies.wall_x.constant import (
+    GENERATE_SUBTASK_RATIO,
+    IMAGE_FACTOR,
+    MAX_PIXELS,
+    MIN_PIXELS,
+    MODEL_TYPE,
+    PRIORITY_ORDER,
+    RESOLUTION,
+    TOKENIZER_MAX_LENGTH,
+)
+from lerobot.policies.wall_x.qwen_model.configuration_qwen2_5_vl import Qwen2_5_VLConfig
+from lerobot.policies.wall_x.qwen_model.qwen2_5_vl_moe import (
+    Qwen2_5_VisionTransformerPretrainedModel,
+    Qwen2_5_VLACausalLMOutputWithPast,
+    Qwen2_5_VLMoEModel,
+)
+from lerobot.policies.wall_x.utils import (
+    get_wallx_normal_text,
+    preprocesser_call,
+    process_grounding_points,
+    replace_action_token,
+)
+from lerobot.utils.constants import ACTION, OBS_STATE
+
+logger = logging.get_logger(__name__)
+
+
+class SinusoidalPosEmb(nn.Module):
+    """Sinusoidal positional embedding for diffusion timesteps."""
+
+    def __init__(self, dim):
+        super().__init__()
+        self.dim = dim
+
+    def forward(self, x):
+        device = x.device
+        half_dim = self.dim // 2
+        emb = math.log(10000) / (half_dim - 1)
+        emb = torch.exp(torch.arange(half_dim, device=device) * -emb)
+        emb = x[:, None] * emb[None, :]
+        emb = torch.cat((emb.sin(), emb.cos()), dim=-1)
+        return emb
+
+
+class ActionHead(nn.Module):
+    """
+    Action prediction head with flow matching.
+
+    Implements Beta-distributed noise scheduling and temporal embeddings
+    for action sequence prediction.
+    """
+
+    def __init__(self, config):
+        super().__init__()
+
+        self.config = config
+        self.action_dim = sum(config.dof_config.values())
+        self.propri_dim = sum(config.agent_pos_config.values())
+        self.hidden_size = config.hidden_size
+
+        # Beta distribution for noise scheduling
+        self.beta_alpha = 1.5
+        self.beta_beta = 1.0
+        self.s = 0.999
+
+        # Sinusoidal timestep embedding
+        self.time_embed = SinusoidalPosEmb(config.hidden_size)
+
+        # Action embedding network
+        # *2 for action + DOF mask concatenation
+        self.w1 = nn.Linear(self.action_dim * 2, self.hidden_size, bias=False)
+        self.w2 = nn.Linear(self.hidden_size * 2, self.hidden_size, bias=False)  # *2 for action + time
+        self.w3 = nn.Linear(self.hidden_size, self.hidden_size, bias=False)
+        self.act_fn = nn.SiLU()
+
+        # Project back to action space
+        self.action_proj_back = nn.Linear(self.hidden_size, self.action_dim, bias=False)
+
+        # Proprioception projection
+        self.propri_proj = nn.Linear(self.propri_dim * 2, self.hidden_size, bias=False)
+
+    def sample_time(self, batch_size, device):
+        """Sample timesteps using Beta distribution (always in float32 for numerical stability)."""
+        beta_dist = Beta(
+            torch.tensor(self.beta_alpha, dtype=torch.float32, device=device),
+            torch.tensor(self.beta_beta, dtype=torch.float32, device=device),
+        )
+        sample = beta_dist.sample([batch_size])
+        time = (1 - sample) * self.s
+        return time
+
+    def forward(self, action_chunk, dof_mask=None):
+        """
+        Process action sequences with noise injection for training.
+
+        Args:
+            action_chunk: Action sequences [batch, seq_len, action_dim]
+            dof_mask: DOF mask [batch, seq_len, action_dim]
+
+        Returns:
+            tuple: (action_embeddings, flow_target)
+        """
+        batch_size = action_chunk.shape[0]
+        device = action_chunk.device
+        weight_dtype = self.w1.weight.dtype
+
+        # Sample time outside of autocast (Beta distribution needs float32)
+        time = self.sample_time(batch_size, device)
+        t = time.unsqueeze(-1).unsqueeze(-1)
+
+        # Noise and flow computation in float32
+        noise = torch.randn_like(action_chunk, dtype=torch.float32)
+        action_chunk_f32 = action_chunk.to(torch.float32)
+        noisy_action = (1 - t) * noise + t * action_chunk_f32
+        flow = action_chunk_f32 - noise
+
+        # Project noisy actions
+        if dof_mask is not None:
+            noisy_action = torch.cat([noisy_action, dof_mask.to(torch.float32)], dim=-1)
+
+        # Convert to weight dtype for linear layers
+        noisy_action = noisy_action.to(dtype=weight_dtype)
+        action_embed = self.w1(noisy_action)
+
+        # Generate time embeddings and combine
+        time_embed = self.time_embed(time)
+        time_embed = time_embed.unsqueeze(1).repeat(1, action_embed.shape[1], 1)
+        time_embed = time_embed.to(dtype=weight_dtype)
+
+        concat_embed = torch.cat([action_embed, time_embed], dim=-1)
+        concat_embed = self.w2(concat_embed)
+        embed = self.w3(self.act_fn(concat_embed))
+
+        return embed, flow
+
+    def step(self, timestep, noisy_action, dof_mask=None):
+        """Single denoising step for inference."""
+        weight_dtype = self.w1.weight.dtype
+
+        if dof_mask is not None:
+            noisy_action = torch.cat([noisy_action, dof_mask], dim=-1)
+        noisy_action = noisy_action.to(dtype=weight_dtype)
+
+        time_embed = self.time_embed(timestep)
+        action_embed = self.w1(noisy_action)
+
+        time_embed = time_embed.unsqueeze(1).repeat(1, action_embed.shape[1], 1)
+        time_embed = time_embed.to(device=noisy_action.device, dtype=weight_dtype)
+
+        concat_embed = torch.cat([action_embed, time_embed], dim=-1)
+        concat_embed = self.w2(concat_embed)
+        embed = self.w3(self.act_fn(concat_embed))
+
+        return embed
+
+    def flow_loss(self, action_hidden_states, flow, dof_mask=None):
+        """Compute flow matching loss (all computations in float32 for stability)."""
+        # Ensure all inputs are float32
+        action_hidden_states = action_hidden_states.to(torch.float32)
+        flow = flow.to(torch.float32)
+
+        action_pred = self.action_proj_back(action_hidden_states)
+        loss = F.mse_loss(action_pred, flow, reduction="none")
+
+        if dof_mask is not None:
+            dof_mask = dof_mask.reshape(-1, dof_mask.shape[-1]).to(torch.float32)
+            loss = loss * dof_mask
+
+        return loss
+
+    def proprioception_proj(self, proprioception, dof_mask=None, use_history=False):
+        """Project proprioceptive data to hidden space."""
+        # Ensure proper device and dtype alignment
+        proprioception = proprioception.to(device=self.propri_proj.weight.device).to(
+            dtype=self.propri_proj.weight.dtype
+        )
+
+        if dof_mask is not None:
+            # Concatenate proprioception with DOF mask
+            # TODO: Use variable-based dimension checking for better flexibility
+            if use_history:
+                proprioception = torch.cat([proprioception, dof_mask], dim=-1)
+            else:
+                proprioception = torch.cat([proprioception, dof_mask], dim=-1)
+
+        proprioception = proprioception.to(device=self.propri_proj.weight.device).to(
+            dtype=self.propri_proj.weight.dtype
+        )
+        return self.propri_proj(proprioception)
+
+
+class Qwen2_5_VLMoEForAction(Qwen2_5_VLForConditionalGeneration):
+    """
+    Qwen2.5 Vision-Language Mixture of Experts model for action processing.
+
+    This model extends the base Qwen2.5 VL model with action token processing capabilities
+    and optional LoRA fine-tuning support.
+    """
+
+    _tied_weights_keys = {"lm_head.weight": "model.embed_tokens.weight"}
+    config_class = Qwen2_5_VLConfig
+    _no_split_modules = ["Qwen2_5_VLDecoderLayer_with_MoE", "Qwen2_5_VLVisionBlock"]
+
+    def init_weights(self):
+        if getattr(self.model, "language_model", None) is not None:
+            return
+        super().init_weights()
+
+    @classmethod
+    def from_pretrained(
+        cls,
+        pretrained_name_or_path,
+        config=None,
+        action_tokenizer_path=None,
+        attn_implementation: str = "eager",
+        cache_dir: str | PathLike | None = None,
+        force_download: bool = False,
+        local_files_only: bool = False,
+        token: str | bool | None = None,
+        revision: str = "main",
+        strict: bool = False,
+        **kwargs: Any,
+    ):
+        """
+        Load model from pretrained model path.
+
+        Args:
+            pretrained_model_path (str): Model directory path containing model.safetensors file
+            config_path (str, optional): Configuration file path, if None will look for qwen25_config.json in pretrained_model_path
+            action_tokenizer_path (str, optional): Action tokenizer path, if None will load from default config
+            attn_implementation (str, optional): Attention implementation, if None will load from default config
+            **kwargs: Additional arguments
+
+        Returns:
+            Qwen2_5_VLMoEForAction: Loaded model instance
+        """
+        if config is None:
+            config = cls.config_class.from_pretrained(
+                pretrained_name_or_path,
+                cache_dir=cache_dir,
+                force_download=force_download,
+                local_files_only=local_files_only,
+                token=token,
+                revision=revision,
+                strict=strict,
+                **kwargs,
+            )
+        if attn_implementation is not None:
+            config._attn_implementation = attn_implementation
+        processor = AutoProcessor.from_pretrained(pretrained_name_or_path, use_fast=True)
+        if action_tokenizer_path is not None:
+            action_tokenizer = AutoProcessor.from_pretrained(action_tokenizer_path, trust_remote_code=True)
+            processor.action_processor = action_tokenizer
+        else:
+            action_tokenizer = None
+
+        # add pad_token_id to config
+        config.pad_token_id = processor.tokenizer.pad_token_id
+        config.text_config.pad_token_id = processor.tokenizer.pad_token_id
+
+        # Initialize model with configuration and processor
+        model = cls(config, processor=processor, action_tokenizer=action_tokenizer, **kwargs)
+
+        # Resize token embeddings to match processor tokenizer vocabulary size
+        model.resize_token_embeddings(len(processor.tokenizer))
+
+        # Try to load the model.safetensors file
+        print(f"Loading model from: {pretrained_name_or_path}")
+        try:
+            from transformers.utils import cached_file
+
+            # Try safetensors first
+            resolved_file = cached_file(
+                pretrained_name_or_path,
+                "model.safetensors",
+                cache_dir=kwargs.get("cache_dir"),
+                force_download=kwargs.get("force_download", False),
+                resume_download=kwargs.get("resume_download"),
+                proxies=kwargs.get("proxies"),
+                token=kwargs.get("token"),
+                revision=kwargs.get("revision"),
+                local_files_only=kwargs.get("local_files_only", False),
+            )
+            from safetensors.torch import load_file
+
+            sd = load_file(resolved_file)
+            print("✓ Loaded state dict from model.safetensors")
+        except Exception as e:
+            print(f"Could not load state dict from remote files: {e}")
+            print("Returning model without loading pretrained weights")
+            return model
+
+        state_dict = {}
+        # filter normalizer statistic params
+        del_keys = []
+        for key in sd.keys():
+            if "action_preprocessor.normalizer" in key:
+                del_keys.append(key)
+        for key in del_keys:
+            del sd[key]
+        state_dict.update(sd)
+
+        model.load_state_dict(state_dict, strict=False)
+
+        return model
+
+    def __init__(
+        self,
+        config,
+        use_fast_tokenizer=False,
+        processor=None,
+        action_tokenizer=None,
+        action_mapper=None,
+        flow_loss_weight=1.0,
+    ):
+        """
+        Initialize the Qwen2.5 VLMoE model for action processing.
+
+        Args:
+            config: Model configuration
+            use_fast_tokenizer (bool): Whether to use fast tokenizer
+            processor: Text and image processor
+            action_tokenizer: Action-specific tokenizer
+            action_mapper: Action mapping utility
+            flow_loss_weight (float): Weight for flow loss computation
+        """
+        super().__init__(config)
+
+        # Initialize vision transformer and language model components
+        self.visual = Qwen2_5_VisionTransformerPretrainedModel._from_config(config.vision_config)
+        self.model = Qwen2_5_VLMoEModel(config)
+        self.vocab_size = config.vocab_size
+        self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
+
+        # Initialize loss function without reduction for channel-wise loss computation
+        self.loss_fct = CrossEntropyLoss(reduction="none")
+        self.flow_loss_weight = flow_loss_weight
+        self.use_fast_tokenizer = use_fast_tokenizer
+        self.processor = processor
+        self.action_tokenizer = action_tokenizer
+
+        # Define action token IDs
+        self.define_action_token_id()
+
+        # Cache for rope deltas
+        self.rope_deltas = None
+
+        # Initialize action preprocessor
+        self.action_preprocessor = ActionHead(config)
+
+        # Apply LoRA if specified in configuration
+        if hasattr(config, "use_lora") and config.use_lora:
+            self.add_lora(
+                r=config.lora_r,
+                lora_alpha=config.lora_alpha,
+                target_modules=config.lora_target_modules,
+                lora_dropout=config.lora_dropout,
+            )
+
+        # Initialize weights and apply final processing
+        self.post_init()
+
+    def to_bfloat16_for_selected_params(self):
+        self.to(dtype=torch.bfloat16)
+
+        params_to_keep_float32 = []
+
+        for name, param in self.named_parameters():
+            if "input_layernorm" in name or "post_attention_layernorm" in name or "model.norm" in name:
+                params_to_keep_float32.append(name)
+            if "action_preprocessor" in name:
+                params_to_keep_float32.append(name)
+
+        for name, param in self.named_parameters():
+            if name in params_to_keep_float32:
+                param.data = param.data.to(torch.float32)
+
+    def define_action_token_id(self):
+        """
+        Define action token IDs based on tokenizer configuration.
+
+        Creates mappings for fast action tokens, proprioception tokens, and general action tokens.
+        """
+        # Create list of fast action token IDs
+        fast_action_token_list = []
+        if self.use_fast_tokenizer:
+            for i in range(self.processor.tokenizer.init_kwargs["action_token_vocab_size"]):
+                action_token_id = self.processor.tokenizer.convert_tokens_to_ids(f"<|action_token_{i}|>")
+                fast_action_token_list.append(action_token_id)
+
+        # Get special action token IDs
+        action_token_id = self.processor.tokenizer.convert_tokens_to_ids("<|action|>")
+        propri_token_id = self.processor.tokenizer.convert_tokens_to_ids("<|propri|>")
+
+        # Store action token ID mappings
+        self.action_token_id_set = {
+            "fast_action_token_list": fast_action_token_list,
+            "propri_token_id": propri_token_id,
+            "action_token_id": action_token_id,
+        }
+
+    def add_lora(self, r=8, lora_alpha=32, target_modules=["q_proj", "v_proj"], lora_dropout=0.1):
+        """
+        Add LoRA (Low-Rank Adaptation) adapters to the model.
+
+        Args:
+            r (int): Rank of adaptation
+            lora_alpha (int): LoRA scaling parameter
+            target_modules (list): List of module names to apply LoRA to
+            lora_dropout (float): Dropout probability for LoRA layers
+        """
+        config = LoraConfig(
+            r=r,
+            lora_alpha=lora_alpha,
+            target_modules=target_modules,
+            lora_dropout=lora_dropout,
+            bias="none",
+            task_type="CAUSAL_LM",
+        )
+        self.model = get_peft_model(self.model, config)
+
+        # Print information about trainable parameters
+        self.model.print_trainable_parameters()
+
+    def get_input_embeddings(self):
+        """Get input embeddings layer."""
+        return self.model.embed_tokens
+
+    def set_input_embeddings(self, value):
+        """Set input embeddings layer."""
+        self.model.embed_tokens = value
+
+    def get_output_embeddings(self):
+        """Get output embeddings layer."""
+        return self.lm_head
+
+    def set_output_embeddings(self, new_embeddings):
+        """Set output embeddings layer."""
+        self.lm_head = new_embeddings
+
+    def set_decoder(self, decoder):
+        """Set the decoder model."""
+        self.model = decoder
+
+    def get_decoder(self):
+        """Get the decoder model."""
+        return self.model
+
+    def get_rope_index(
+        self,
+        input_ids: torch.LongTensor | None = None,
+        image_grid_thw: torch.LongTensor | None = None,
+        video_grid_thw: torch.LongTensor | None = None,
+        second_per_grid_ts: torch.Tensor | None = None,
+        attention_mask: torch.Tensor | None = None,
+    ) -> tuple[torch.Tensor, torch.Tensor]:
+        """
+        Calculate 3D RoPE (Rotary Position Embedding) indices for vision and text tokens.
+
+        This method computes position embeddings that account for the temporal, height, and width
+        dimensions of vision tokens (images/videos) while maintaining standard 1D position embeddings
+        for text tokens.
+
+        For vision tokens, 3D position embeddings are calculated based on:
+        - Temporal dimension: Time patches in videos
+        - Height dimension: Vertical patches in images/video frames
+        - Width dimension: Horizontal patches in images/video frames
+
+        For text tokens, standard 1D position embeddings are used, continuing from the maximum
+        vision position ID plus 1.
+
+        Args:
+            input_ids (torch.LongTensor, optional): Input token IDs of shape (batch_size, sequence_length)
+            image_grid_thw (torch.LongTensor, optional): Image grid dimensions (num_images, 3) for [temporal, height, width]
+            video_grid_thw (torch.LongTensor, optional): Video grid dimensions (num_videos, 3) for [temporal, height, width]
+            second_per_grid_ts (torch.Tensor, optional): Time interval per temporal grid (num_videos,)
+            attention_mask (torch.Tensor, optional): Attention mask (batch_size, sequence_length)
+
+        Returns:
+            tuple:
+                - position_ids (torch.LongTensor): 3D position IDs of shape (3, batch_size, sequence_length)
+                - mrope_position_deltas (torch.Tensor): Position deltas for mRoPE of shape (batch_size, 1)
+        """
+        spatial_merge_size = self.config.vision_config.spatial_merge_size
+        image_token_id = self.config.image_token_id
+        video_token_id = self.config.video_token_id
+        vision_start_token_id = self.config.vision_start_token_id
+        mrope_position_deltas = []
+
+        if input_ids is not None and (image_grid_thw is not None or video_grid_thw is not None):
+            total_input_ids = input_ids
+            if attention_mask is None:
+                attention_mask = torch.ones_like(total_input_ids)
+
+            # Initialize 3D position IDs tensor
+            position_ids = torch.ones(
+                3,
+                input_ids.shape[0],
+                input_ids.shape[1],
+                dtype=input_ids.dtype,
+                device=input_ids.device,
+            )
+
+            image_index, video_index = 0, 0
+            attention_mask = attention_mask.to(total_input_ids.device)
+
+            # Process each sequence in the batch
+            for i, input_ids in enumerate(total_input_ids):
+                input_ids = input_ids[attention_mask[i] == 1]
+                image_nums, video_nums = 0, 0
+
+                # Find vision tokens and count images/videos
+                vision_start_indices = torch.argwhere(input_ids == vision_start_token_id).squeeze(1)
+                vision_tokens = input_ids[vision_start_indices + 1]
+                image_nums = (vision_tokens == image_token_id).sum()
+                video_nums = (vision_tokens == video_token_id).sum()
+
+                input_tokens = input_ids.tolist()
+                llm_pos_ids_list: list = []
+                st = 0
+                remain_images, remain_videos = image_nums, video_nums
+
+                # Process each vision token (image or video)
+                for _ in range(image_nums + video_nums):
+                    # Find next image or video token
+                    if image_token_id in input_tokens and remain_images > 0:
+                        ed_image = input_tokens.index(image_token_id, st)
+                    else:
+                        ed_image = len(input_tokens) + 1
+
+                    if video_token_id in input_tokens and remain_videos > 0:
+                        ed_video = input_tokens.index(video_token_id, st)
+                    else:
+                        ed_video = len(input_tokens) + 1
+
+                    # Determine if processing image or video token
+                    if ed_image < ed_video:
+                        # Process image token
+                        t, h, w = (
+                            image_grid_thw[image_index][0],
+                            image_grid_thw[image_index][1],
+                            image_grid_thw[image_index][2],
+                        )
+                        second_per_grid_t = 0
+                        image_index += 1
+                        remain_images -= 1
+                        ed = ed_image
+                    else:
+                        # Process video token
+                        t, h, w = (
+                            video_grid_thw[video_index][0],
+                            video_grid_thw[video_index][1],
+                            video_grid_thw[video_index][2],
+                        )
+                        if second_per_grid_ts is not None:
+                            second_per_grid_t = second_per_grid_ts[video_index]
+                        else:
+                            second_per_grid_t = 1.0
+                        video_index += 1
+                        remain_videos -= 1
+                        ed = ed_video
+
+                    # Calculate grid dimensions after spatial merging
+                    llm_grid_t, llm_grid_h, llm_grid_w = (
+                        t.item(),
+                        h.item() // spatial_merge_size,
+                        w.item() // spatial_merge_size,
+                    )
+                    text_len = ed - st
+
+                    # Add position IDs for text tokens before vision token
+                    st_idx = llm_pos_ids_list[-1].max() + 1 if len(llm_pos_ids_list) > 0 else 0
+                    llm_pos_ids_list.append(torch.arange(text_len).view(1, -1).expand(3, -1) + st_idx)
+
+                    # Calculate 3D position embeddings for vision tokens
+                    range_tensor = torch.arange(llm_grid_t).view(-1, 1)
+                    expanded_range = range_tensor.expand(-1, llm_grid_h * llm_grid_w)
+
+                    # Calculate temporal position IDs with time scaling
+                    time_tensor = (
+                        expanded_range * second_per_grid_t * self.config.vision_config.tokens_per_second
+                    )
+                    time_tensor_long = time_tensor.long()
+                    t_index = time_tensor_long.flatten()
+
+                    # Calculate spatial position IDs
+                    h_index = (
+                        torch.arange(llm_grid_h).view(1, -1, 1).expand(llm_grid_t, -1, llm_grid_w).flatten()
+                    )
+                    w_index = (
+                        torch.arange(llm_grid_w).view(1, 1, -1).expand(llm_grid_t, llm_grid_h, -1).flatten()
+                    )
+
+                    # Add 3D position IDs for vision tokens
+                    llm_pos_ids_list.append(torch.stack([t_index, h_index, w_index]) + text_len + st_idx)
+                    st = ed + llm_grid_t * llm_grid_h * llm_grid_w
+
+                # Add position IDs for remaining text tokens
+                if st < len(input_tokens):
+                    st_idx = llm_pos_ids_list[-1].max() + 1 if len(llm_pos_ids_list) > 0 else 0
+                    text_len = len(input_tokens) - st
+                    llm_pos_ids_list.append(torch.arange(text_len).view(1, -1).expand(3, -1) + st_idx)
+
+                # Concatenate all position IDs for this sequence
+                llm_positions = torch.cat(llm_pos_ids_list, dim=1).reshape(3, -1)
+                position_ids[..., i, attention_mask[i] == 1] = llm_positions.to(position_ids.device)
+                mrope_position_deltas.append(llm_positions.max() + 1 - len(total_input_ids[i]))
+
+            mrope_position_deltas = torch.tensor(mrope_position_deltas, device=input_ids.device).unsqueeze(1)
+            return position_ids, mrope_position_deltas
+        else:
+            # Handle case without vision tokens - use standard 1D position embeddings
+            if attention_mask is not None:
+                position_ids = attention_mask.long().cumsum(-1) - 1
+                position_ids.masked_fill_(attention_mask == 0, 1)
+                position_ids = position_ids.unsqueeze(0).expand(3, -1, -1).to(attention_mask.device)
+                max_position_ids = position_ids.max(0, keepdim=False)[0].max(-1, keepdim=True)[0]
+                mrope_position_deltas = max_position_ids + 1 - attention_mask.shape[-1]
+            else:
+                position_ids = (
+                    torch.arange(input_ids.shape[1], device=input_ids.device)
+                    .view(1, 1, -1)
+                    .expand(3, input_ids.shape[0], -1)
+                )
+                mrope_position_deltas = torch.zeros(
+                    [input_ids.shape[0], 1],
+                    device=input_ids.device,
+                    dtype=input_ids.dtype,
+                )
+
+            return position_ids, mrope_position_deltas
+
+    def train_step_forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        moe_token_types: torch.LongTensor | None = None,  # MoE token type assignments
+        labels: torch.LongTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+        pixel_values: torch.Tensor | None = None,
+        pixel_values_videos: torch.FloatTensor | None = None,
+        image_grid_thw: torch.LongTensor | None = None,
+        video_grid_thw: torch.LongTensor | None = None,
+        action_chunk: torch.FloatTensor | None = None,  # Action trajectory chunks
+        proprioception: torch.FloatTensor | None = None,  # Joint position/orientation data
+        rope_deltas: torch.LongTensor | None = None,
+        cache_position: torch.LongTensor | None = None,
+        second_per_grid_ts: torch.Tensor | None = None,
+        dof_mask: torch.FloatTensor | None = None,
+        agent_pos_mask: torch.FloatTensor | None = None,
+        **kwargs,
+    ) -> tuple | Qwen2_5_VLACausalLMOutputWithPast:
+        """
+        Forward pass for training with multi-modal inputs including vision, text, and action data.
+
+        This method handles the complete forward pass during training, processing various input modalities
+        including images, videos, text, proprioceptive data, and action sequences. It computes losses
+        for both language modeling and action prediction using flow matching.
+
+        Args:
+            input_ids (torch.LongTensor, optional): Input token IDs
+            attention_mask (torch.Tensor, optional): Attention mask for input tokens
+            position_ids (torch.LongTensor, optional): Position IDs for tokens
+            past_key_values (List[torch.FloatTensor], optional): Cached key-value pairs for generation
+            inputs_embeds (torch.FloatTensor, optional): Pre-computed input embeddings
+            moe_token_types (torch.LongTensor, optional): Token type assignments for MoE routing
+            labels (torch.LongTensor, optional): Target labels for loss computation
+            use_cache (bool, optional): Whether to use key-value caching
+            output_attentions (bool, optional): Whether to return attention weights
+            output_hidden_states (bool, optional): Whether to return hidden states
+            return_dict (bool, optional): Whether to return structured output
+            pixel_values (torch.Tensor, optional): Image pixel values
+            pixel_values_videos (torch.FloatTensor, optional): Video pixel values
+            image_grid_thw (torch.LongTensor, optional): Image grid dimensions (temporal, height, width)
+            video_grid_thw (torch.LongTensor, optional): Video grid dimensions (temporal, height, width)
+            action_chunk (torch.FloatTensor, optional): Action trajectory data chunks
+            proprioception (torch.FloatTensor, optional): Proprioceptive sensor data (joint positions, etc.)
+            rope_deltas (torch.LongTensor, optional): RoPE position deltas
+            cache_position (torch.LongTensor, optional): Cache position indices
+            second_per_grid_ts (torch.Tensor, optional): Time interval per temporal grid
+            dof_mask (torch.FloatTensor, optional): Degrees of freedom mask for action tokens
+            agent_pos_mask (torch.FloatTensor, optional): Agent position mask for proprioceptive data
+            **kwargs: Additional keyword arguments
+
+        Returns:
+            Union[Tuple, Qwen2_5_VLACausalLMOutputWithPast]: Model outputs including losses, logits,
+                and auxiliary information, or tuple if return_dict=False
+        """
+        batch_size, seq_length = input_ids.shape
+
+        # Set output configuration from model config if not specified
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        # Calculate RoPE position IDs if not provided
+        # Note: Cannot calculate rope deltas with 4D attention mask. TODO: Fix this limitation
+        if position_ids is None and (attention_mask is None or attention_mask.ndim == 2):
+            # Calculate RoPE index once per generation in the pre-fill stage only
+            if (
+                (cache_position is not None and cache_position[0] == 0)
+                or self.rope_deltas is None
+                or (past_key_values is None or past_key_values.get_seq_length() == 0)
+            ):
+                position_ids, rope_deltas = self.get_rope_index(
+                    input_ids,
+                    image_grid_thw,
+                    video_grid_thw,
+                    second_per_grid_ts,
+                    attention_mask,
+                )
+                self.rope_deltas = rope_deltas
+            # Use previously calculated rope deltas to get correct position IDs
+            else:
+                delta = (
+                    (cache_position[0] + self.rope_deltas).to(self.device)
+                    if cache_position is not None
+                    else 0
+                )
+                position_ids = torch.arange(seq_length, device=self.device)
+                position_ids = position_ids.view(1, -1).expand(batch_size, -1)
+                if cache_position is not None:  # otherwise `deltas` is an int `0`
+                    delta = delta.repeat_interleave(batch_size // delta.shape[0], dim=0)
+                position_ids = position_ids.add(delta)
+                position_ids = position_ids.unsqueeze(0).expand(3, -1, -1)
+
+        # Process input embeddings with multi-modal data
+        if inputs_embeds is None:
+            inputs_embeds = self.model.embed_tokens(input_ids)
+
+            # Process image embeddings
+            if pixel_values is not None:
+                pixel_values = pixel_values.type(self.visual.dtype)
+                image_embeds = self.visual(pixel_values, grid_thw=image_grid_thw)
+                mask = input_ids == self.config.image_token_id
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                image_mask = mask_expanded.to(inputs_embeds.device)
+
+                image_embeds = image_embeds.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(image_mask, image_embeds)
+
+            # Process video embeddings
+            if pixel_values_videos is not None:
+                pixel_values_videos = pixel_values_videos.type(self.visual.dtype)
+                video_embeds = self.visual(pixel_values_videos, grid_thw=video_grid_thw)
+                n_video_tokens = (input_ids == self.config.video_token_id).sum().item()
+                n_video_features = video_embeds.shape[0]
+
+                # Validate video token and feature count match
+                if n_video_tokens != n_video_features:
+                    raise ValueError(
+                        f"Video features and video tokens do not match: tokens: {n_video_tokens}, features {n_video_features}"
+                    )
+                mask = input_ids == self.config.video_token_id
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                video_mask = mask_expanded.to(inputs_embeds.device)
+
+                video_embeds = video_embeds.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(video_mask, video_embeds)
+
+            # Process proprioceptive data (joint positions, orientations, etc.)
+            if proprioception is not None:
+                proprioception = proprioception.to(inputs_embeds.device).to(inputs_embeds.dtype)
+                agent_pos_mask = agent_pos_mask.to(inputs_embeds.device).to(inputs_embeds.dtype)
+                proprioception = self.action_preprocessor.proprioception_proj(
+                    proprioception,
+                    agent_pos_mask,
+                    use_history=proprioception.shape[1] > 1,
+                )
+                mask = input_ids == self.action_token_id_set["propri_token_id"]
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                proprioception_mask = mask_expanded.to(inputs_embeds.device)
+
+                proprioception = proprioception.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(proprioception_mask, proprioception)
+            elif self.training:
+                # Dummy forward pass to ensure gradient registration in DDP
+                # This handles cases where one process has proprioception data while another doesn't
+                # Without this, DDP would hang waiting for a gradient that will never be computed
+                dummy_input = torch.randn(
+                    2,
+                    self.action_preprocessor.propri_dim * 2,
+                    device=inputs_embeds.device,
+                )
+                dummy_forward = self.action_preprocessor.proprioception_proj(dummy_input)
+                dummy_loss = sum(p.sum() for p in dummy_forward)
+                inputs_embeds = inputs_embeds + 0 * dummy_loss
+
+            # Process action chunk data
+            if action_chunk is not None:
+                action_chunk = action_chunk.to(inputs_embeds.device).to(inputs_embeds.dtype)
+                dof_mask = dof_mask.to(inputs_embeds.device).to(inputs_embeds.dtype)
+                noisy_action_emb, flow = self.action_preprocessor(action_chunk, dof_mask)
+                mask = input_ids == self.action_token_id_set["action_token_id"]
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                action_mask = mask_expanded.to(inputs_embeds.device)
+
+                noisy_action_emb = noisy_action_emb.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(action_mask, noisy_action_emb)
+
+            if attention_mask is not None:
+                attention_mask = attention_mask.to(inputs_embeds.device)
+
+        # Forward pass through the main model
+        outputs = self.model(
+            input_ids=None,
+            position_ids=position_ids,
+            attention_mask=attention_mask,
+            past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds,
+            moe_token_types=moe_token_types,  # Pass token types for MoE routing
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+        )
+
+        hidden_states = outputs[0]
+        hidden_states = hidden_states.to(self.lm_head.weight.dtype)
+        logits = self.lm_head(hidden_states)
+
+        # Initialize loss computation variables
+        loss = None
+        cross_entropy_loss, flow_loss = None, None
+        channel_loss_dict = None
+        channel_loss_count_dict = None
+
+        # Compute losses if labels are provided
+        if labels is not None:
+            loss = torch.tensor(0.0, device=hidden_states.device, dtype=torch.float32)
+
+            # Compute standard cross-entropy loss for language modeling
+            shift_logits = logits[..., :-1, :].contiguous().to(torch.float32)
+            shift_labels = labels[..., 1:].contiguous()
+            shift_logits = shift_logits.view(-1, self.config.vocab_size)
+            shift_labels = shift_labels.view(-1)
+
+            # Enable model parallelism by moving labels to correct device
+            shift_labels = shift_labels.to(shift_logits.device)
+            non_ignored_mask = shift_labels != -100
+            _cross_entropy_loss = self.loss_fct(shift_logits, shift_labels)
+            cross_entropy_loss = (
+                _cross_entropy_loss[non_ignored_mask].mean()
+                if non_ignored_mask.any()
+                else torch.tensor(0.0, device=shift_logits.device, dtype=torch.float32)
+            )
+
+            # Add cross-entropy loss to total loss if valid
+            if not torch.isnan(cross_entropy_loss):
+                loss = loss + cross_entropy_loss.to(torch.float32)
+            else:
+                with torch.no_grad():
+                    cross_entropy_loss.detach()
+
+        if action_chunk is not None:
+            action_mask = input_ids == self.action_token_id_set["action_token_id"]
+            if action_mask.any():
+                action_hidden_states = hidden_states[action_mask].to(torch.float32)
+                flow = flow.reshape(-1, flow.shape[-1]).to(torch.float32)
+                _flow_loss = self.action_preprocessor.flow_loss(action_hidden_states, flow, dof_mask)
+                if isinstance(_flow_loss, torch.Tensor):
+                    flow_loss = _flow_loss.mean()
+                if loss is not None:
+                    loss = loss + self.flow_loss_weight * flow_loss.to(torch.float32)
+                else:
+                    loss = self.flow_loss_weight * flow_loss.to(torch.float32)
+                _flow_loss = _flow_loss.view(dof_mask.shape[0], dof_mask.shape[1], dof_mask.shape[2])
+
+        # Return outputs based on return_dict setting
+        if not return_dict:
+            output = (logits,) + outputs[1:]
+            return (loss,) + output if loss is not None else output
+
+        return Qwen2_5_VLACausalLMOutputWithPast(
+            loss=loss,
+            cross_entropy_loss=(cross_entropy_loss.clone() if cross_entropy_loss is not None else None),
+            flow_loss=flow_loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+            rope_deltas=self.rope_deltas,
+            channel_loss_dict=channel_loss_dict,
+            channel_loss_count_dict=channel_loss_count_dict,
+        )
+
+    def predict_action(self, predict_mode: str, **kwargs):
+        """
+        Predict actions using specified prediction mode.
+
+        Args:
+            predict_mode (str): Prediction mode, either "fast" or "diffusion"
+            **kwargs: Additional arguments passed to the predict method
+
+        Returns:
+            tuple: (predicted_action, ground_truth_action) where ground_truth_action may be None
+        """
+        assert predict_mode in ["fast", "diffusion"]
+
+        output = self.predict(predict_mode=predict_mode, **kwargs)
+
+        return output["predict_action"], output.get("gt_action", None)
+
+    @torch.no_grad()
+    def predict(
+        self,
+        predict_mode: str,
+        pred_horizon: int | None = None,
+        action_dim: int | None = None,
+        input_ids: torch.LongTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        moe_token_types: torch.LongTensor | None = None,
+        labels: torch.LongTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+        pixel_values: torch.Tensor | None = None,
+        pixel_values_videos: torch.FloatTensor | None = None,
+        image_grid_thw: torch.LongTensor | None = None,
+        video_grid_thw: torch.LongTensor | None = None,
+        action_chunk: torch.FloatTensor | None = None,
+        proprioception: torch.FloatTensor | None = None,
+        rope_deltas: torch.LongTensor | None = None,
+        cache_position: torch.LongTensor | None = None,
+        second_per_grid_ts: torch.Tensor | None = None,
+        num_inference_timesteps: int | None = 10,
+        dof_mask: torch.FloatTensor | None = None,
+        agent_pos_mask: torch.FloatTensor | None = None,
+        re_generate: bool = False,
+        **kwargs,
+    ):
+        """
+        Multi-modal prediction method supporting text generation, fast action prediction, and diffusion-based action prediction.
+
+        This method handles three prediction modes:
+        1. "text": Pure text generation using autoregressive decoding
+        2. "fast": Fast action prediction using discrete action tokens
+        3. "diffusion": Continuous action prediction using diffusion/flow matching
+
+        Args:
+            predict_mode (str): Prediction mode ("text", "fast", or "diffusion")
+            pred_horizon (int, optional): Prediction horizon for action sequences
+            action_dim (int, optional): Dimensionality of action space
+            input_ids (torch.LongTensor, optional): Input token IDs
+            attention_mask (torch.Tensor, optional): Attention mask for input tokens
+            position_ids (torch.LongTensor, optional): Position IDs for tokens
+            past_key_values (List[torch.FloatTensor], optional): Cached key-value pairs
+            inputs_embeds (torch.FloatTensor, optional): Pre-computed input embeddings
+            moe_token_types (torch.LongTensor, optional): Token type assignments for MoE routing
+            labels (torch.LongTensor, optional): Target labels for evaluation
+            use_cache (bool, optional): Whether to use key-value caching
+            output_attentions (bool, optional): Whether to return attention weights
+            output_hidden_states (bool, optional): Whether to return hidden states
+            return_dict (bool, optional): Whether to return structured output
+            pixel_values (torch.Tensor, optional): Image pixel values
+            pixel_values_videos (torch.FloatTensor, optional): Video pixel values
+            image_grid_thw (torch.LongTensor, optional): Image grid dimensions
+            video_grid_thw (torch.LongTensor, optional): Video grid dimensions
+            action_chunk (torch.FloatTensor, optional): Ground truth action sequences
+            proprioception (torch.FloatTensor, optional): Proprioceptive sensor data
+            rope_deltas (torch.LongTensor, optional): RoPE position deltas
+            cache_position (torch.LongTensor, optional): Cache position indices
+            second_per_grid_ts (torch.Tensor, optional): Time interval per temporal grid
+            num_inference_timesteps (int, optional): Number of diffusion inference steps
+            dof_mask (torch.FloatTensor, optional): Degrees of freedom mask
+            agent_pos_mask (torch.FloatTensor, optional): Agent position mask
+            re_generate (bool, optional): Whether to use sampling for regeneration
+            **kwargs: Additional keyword arguments
+
+        Returns:
+            dict: Dictionary containing prediction results with keys like:
+                - 'predict_action': Predicted action sequences
+                - 'gt_action': Ground truth actions (if available)
+                - 'input_text': Input text (for text/fast modes)
+                - 'predict_output_text': Generated text (for text/fast modes)
+                - 'gt_output_text': Ground truth text (for text/fast modes)
+        """
+        batch_size = input_ids.shape[0] if input_ids is not None else inputs_embeds.shape[0]
+
+        # Text and fast modes require batch size 1 for autoregressive generation
+        if predict_mode in ["text", "fast"]:
+            assert batch_size == 1, "predict only support batch size 1 for ar generation"
+
+        # Set output configuration from model config if not specified
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        # Process input embeddings with multi-modal data
+        if inputs_embeds is None:
+            inputs_embeds = self.model.embed_tokens(input_ids)
+
+            # Process image embeddings
+            if pixel_values is not None:
+                pixel_values = pixel_values.type(self.visual.dtype)
+                image_embeds = self.visual(pixel_values, grid_thw=image_grid_thw)
+                n_image_tokens = (input_ids == self.config.image_token_id).sum().item()
+                n_image_features = image_embeds.shape[0]
+
+                # Validate image token and feature count match
+                if n_image_tokens != n_image_features:
+                    raise ValueError(
+                        f"Image features and image tokens do not match: tokens: {n_image_tokens}, features {n_image_features}"
+                    )
+
+                mask = input_ids == self.config.image_token_id
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                image_mask = mask_expanded.to(inputs_embeds.device)
+
+                image_embeds = image_embeds.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(image_mask, image_embeds)
+
+            # Process video embeddings
+            if pixel_values_videos is not None:
+                pixel_values_videos = pixel_values_videos.type(self.visual.dtype)
+                video_embeds = self.visual(pixel_values_videos, grid_thw=video_grid_thw)
+                n_video_tokens = (input_ids == self.config.video_token_id).sum().item()
+                n_video_features = video_embeds.shape[0]
+
+                # Validate video token and feature count match
+                if n_video_tokens != n_video_features:
+                    raise ValueError(
+                        f"Video features and video tokens do not match: tokens: {n_video_tokens}, features {n_video_features}"
+                    )
+
+                mask = input_ids == self.config.video_token_id
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                video_mask = mask_expanded.to(inputs_embeds.device)
+
+                video_embeds = video_embeds.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(video_mask, video_embeds)
+
+            # Process proprioceptive data
+            if proprioception is not None:
+                proprioception = proprioception.to(inputs_embeds.device).to(inputs_embeds.dtype)
+                agent_pos_mask = agent_pos_mask.to(inputs_embeds.device).to(inputs_embeds.dtype)
+                proprio_embed = self.action_preprocessor.proprioception_proj(
+                    proprioception,
+                    agent_pos_mask,
+                    use_history=proprioception.shape[1] > 1,
+                )
+                proprioception_mask = input_ids == self.action_token_id_set["propri_token_id"]
+                proprio_embed = proprio_embed.to(torch.bfloat16)
+                inputs_embeds[proprioception_mask] = proprio_embed.reshape(-1, inputs_embeds.shape[-1])
+
+            if attention_mask is not None:
+                attention_mask = attention_mask.to(inputs_embeds.device)
+
+        # Calculate RoPE position IDs if not provided
+        # Note: Cannot calculate rope deltas with 4D attention mask. TODO: Fix this limitation
+        if position_ids is None and (attention_mask is None or attention_mask.ndim == 2):
+            # Calculate RoPE index once per generation in the pre-fill stage only
+            if (
+                (cache_position is not None and cache_position[0] == 0)
+                or self.rope_deltas is None
+                or (past_key_values is None or past_key_values.get_seq_length() == 0)
+            ):
+                position_ids, rope_deltas = self.get_rope_index(
+                    input_ids,
+                    image_grid_thw,
+                    video_grid_thw,
+                    second_per_grid_ts,
+                    attention_mask,
+                )
+                self.rope_deltas = rope_deltas
+            # Use previously calculated rope deltas to get correct position IDs
+            else:
+                batch_size, seq_length, _ = inputs_embeds.shape
+                delta = (
+                    (cache_position[0] + self.rope_deltas).to(inputs_embeds.device)
+                    if cache_position is not None
+                    else 0
+                )
+                position_ids = torch.arange(seq_length, device=inputs_embeds.device)
+                position_ids = position_ids.view(1, -1).expand(batch_size, -1)
+                if cache_position is not None:  # otherwise `deltas` is an int `0`
+                    delta = delta.repeat_interleave(batch_size // delta.shape[0], dim=0)
+                position_ids = position_ids.add(delta)
+                position_ids = position_ids.unsqueeze(0).expand(3, -1, -1)
+
+        # Prepare action chunk data if provided
+        if action_chunk is not None:
+            action_chunk = action_chunk.to(inputs_embeds.device).to(torch.float32)
+
+        output = {}
+
+        # Split input sequence for text and fast modes (not needed for diffusion)
+        if predict_mode == "text" or predict_mode == "fast":
+            # Look for generation prompt tokens: <|im_start|>assistant
+            generation_prompt_ids = torch.tensor(
+                [151644, 77091], device=input_ids.device, dtype=input_ids.dtype
+            )
+            matches = (input_ids[0, :-1] == generation_prompt_ids[0]) & (
+                input_ids[0, 1:] == generation_prompt_ids[1]
+            )
+
+            if matches.any():
+                split_pos = torch.nonzero(matches, as_tuple=True)[0][0].item()
+                # Extract ground truth output tokens (including newline)
+                gt_output_ids = input_ids[:, split_pos + 3 :]
+                # Remove output part from input, keeping prompt
+                input_ids = input_ids[:, : split_pos + 3]
+                inputs_embeds = inputs_embeds[:, : split_pos + 3, :]
+                if attention_mask is not None:
+                    attention_mask = attention_mask[:, : split_pos + 3]
+                if labels is not None:
+                    labels = labels[:, split_pos + 3 :]
+            else:
+                raise ValueError(
+                    "input_ids does not contain the generation prompt tokens <|im_start|>assistant"
+                )
+
+            # Decode input text for output
+            input_text = self.processor.batch_decode(
+                input_ids, skip_special_tokens=False, clean_up_tokenization_spaces=True
+            )
+            output["input_text"] = input_text
+
+        # Handle text and fast prediction modes using autoregressive generation
+        if predict_mode == "text" or predict_mode == "fast":
+            # Initialize MoE token types for generation
+            moe_token_types = torch.zeros_like(input_ids)
+            batch = {
+                "input_ids": input_ids,
+                "attention_mask": attention_mask,
+                "pixel_values": pixel_values,
+                "moe_token_types": moe_token_types,
+                "image_grid_thw": image_grid_thw,
+                "dof_mask": dof_mask,
+                "agent_pos_mask": agent_pos_mask,
+                "proprioception": proprioception,
+            }
+
+            # Generate output tokens
+            predict_output_ids = self.generate(
+                **batch,
+                max_new_tokens=100,
+                eos_token_id=[self.processor.tokenizer.eos_token_id],
+                use_cache=True,
+                pad_token_id=self.processor.tokenizer.pad_token_id,
+                temperature=(1.0 if not re_generate else 0.7),  # Higher temperature for regeneration
+                do_sample=(False if not re_generate else True),  # Enable sampling for regeneration
+            )
+
+            # Decode generated and ground truth text
+            gt_output_text = self.processor.batch_decode(
+                gt_output_ids,
+                skip_special_tokens=False,
+                clean_up_tokenization_spaces=True,
+            )
+            predict_output_text = self.processor.batch_decode(
+                predict_output_ids,
+                skip_special_tokens=False,
+                clean_up_tokenization_spaces=True,
+            )
+            output["gt_output_text"] = gt_output_text
+            output["predict_output_text"] = predict_output_text
+
+        # Convert tokens to actions for fast prediction mode
+        if predict_mode == "fast":
+            action_id = []
+            # Extract action tokens from generated sequence
+            for token_id_i in predict_output_ids[0]:
+                if token_id_i.item() >= self.processor.tokenizer.init_kwargs["action_token_start_index"]:
+                    action_id.append(
+                        token_id_i.item() - self.processor.tokenizer.init_kwargs["action_token_start_index"]
+                    )
+
+            predict_action = self.processor.action_processor.decode(
+                [action_id], time_horizon=pred_horizon, action_dim=action_dim
+            )
+            # Handle action decoding errors
+            if np.sum(predict_action) == 0:
+                print("Error in decoding action, predict_action is None")
+                output["predict_action"] = None
+            else:
+                # Convert discrete tokens to continuous actions
+                predict_action = torch.tensor(predict_action, device=self.device)
+                dof_mask = dof_mask.to(self.device).to(pixel_values.dtype)
+                # removed unnormalization step for now
+                predict_action = predict_action[:, :, dof_mask[0, 0, :].bool()]
+                output["predict_action"] = predict_action
+
+            # Process ground truth actions if available
+            if action_chunk is not None:
+                # Apply DOF mask to get ground truth actions
+                # removed unnormalization step for now
+                action_chunk = action_chunk[:, :, dof_mask[0, 0, :].bool()]
+                output["gt_action"] = action_chunk
+            else:
+                output["gt_action"] = None
+
+        # Handle diffusion-based action prediction
+        if predict_mode == "diffusion":
+            # Initialize with random noise
+            noisy_action = torch.randn(
+                size=(batch_size, pred_horizon, action_dim),
+                dtype=torch.float32,
+                device=inputs_embeds.device,
+            )
+            dof_mask = dof_mask.to(inputs_embeds.device).to(torch.float32)
+
+            def step(timestep, noisy_action):
+                """
+                Single denoising step for diffusion process.
+
+                Args:
+                    timestep: Current diffusion timestep
+                    noisy_action: Current noisy action estimate
+
+                Returns:
+                    torch.Tensor: Predicted clean action
+                """
+                action_mask = input_ids == self.action_token_id_set["action_token_id"]
+                assert action_mask.any(), "No action token found in input_ids"
+
+                # Prepare timestep for batch processing
+                timestep = timestep.unsqueeze(0).repeat(noisy_action.shape[0])
+                action_embed = self.action_preprocessor.step(
+                    timestep=timestep, noisy_action=noisy_action, dof_mask=dof_mask
+                )
+                action_embed = action_embed.reshape(-1, inputs_embeds.shape[-1])
+
+                # Ensure action_embed has the correct dtype and device before assignment
+                action_embed = action_embed.to(dtype=inputs_embeds.dtype, device=inputs_embeds.device)
+
+                # Create temporary copy of embeddings (clone preserves dtype)
+                temp_inputs_embeds = inputs_embeds.clone()
+                temp_inputs_embeds[action_mask] = action_embed
+
+                # Forward pass through transformer
+                transformer_outputs = self.model(
+                    input_ids=None,
+                    attention_mask=attention_mask,
+                    position_ids=position_ids,
+                    past_key_values=past_key_values,
+                    inputs_embeds=temp_inputs_embeds,
+                    moe_token_types=moe_token_types,
+                    use_cache=True,
+                    output_attentions=False,
+                    output_hidden_states=False,
+                    return_dict=True,
+                )
+
+                # Extract action predictions from hidden states
+                hidden_states = transformer_outputs.last_hidden_state
+                action_mask = input_ids == self.action_token_id_set["action_token_id"]
+                action_hidden_states = hidden_states[action_mask].to(torch.float32)
+                pred = self.action_preprocessor.action_proj_back(action_hidden_states)
+                return pred.reshape(batch_size, pred_horizon, action_dim)
+
+            # Perform ODE integration for diffusion sampling
+            times = torch.linspace(
+                0,
+                1,
+                num_inference_timesteps + 1,
+                device=inputs_embeds.device,
+                dtype=torch.float32,
+            )
+            action_trajectory = odeint(step, noisy_action, times, method="euler")
+
+            # Extract final predicted action
+            # Removed unnormalization step for now
+            predict_action = action_trajectory[-1]
+            output["predict_action"] = predict_action
+
+            # Process ground truth actions if available
+            # removed unnormalization step for now
+            if action_chunk is not None:
+                output["gt_action"] = action_chunk[:, :, dof_mask[0, 0, :].bool()]
+
+        return output
+
+    def forward(self, mode: str | None = None, predict_mode: str | None = "text", **kwargs):
+        """
+        Main forward pass dispatcher for different execution modes.
+
+        This method routes execution to appropriate forward functions based on the specified mode:
+        - No mode (None): Training step with gradient disabled
+        - 'predict': Prediction/inference mode
+        - 'train': Training mode with gradients enabled
+        - 'validate': Validation mode with gradients disabled
+
+        Args:
+            mode (str, optional): Execution mode. If None, defaults to training step without gradients
+            predict_mode (str, optional): Prediction mode for 'predict' mode ("text", "fast", or "diffusion")
+            **kwargs: Additional arguments passed to the selected forward function
+
+        Returns:
+            Model outputs appropriate for the selected mode
+
+        Todo:
+            - Add support for distinguishing multi-modal data types in prediction mode
+        """
+        if not mode:
+            with torch.no_grad():
+                return self.train_step_forward(**kwargs)
+        elif mode == "predict":
+            return self.predict(predict_mode=predict_mode, **kwargs)
+        elif mode == "train":
+            return self.train_step_forward(use_cache=False, **kwargs)
+        elif mode == "validate":
+            with torch.no_grad():
+                return self.train_step_forward(use_cache=False, **kwargs)
+        else:
+            raise NotImplementedError("invalid key")
+
+    def prepare_inputs_for_generation(
+        self,
+        input_ids,
+        past_key_values=None,
+        attention_mask=None,
+        inputs_embeds=None,
+        moe_token_types=None,
+        cache_position=None,
+        position_ids=None,
+        use_cache=True,
+        pixel_values=None,
+        pixel_values_videos=None,
+        image_grid_thw=None,
+        video_grid_thw=None,
+        second_per_grid_ts=None,
+        proprioception=None,
+        dof_mask=None,
+        agent_pos_mask=None,
+        **kwargs,
+    ):
+        """
+        Prepare inputs for autoregressive generation with multi-modal support.
+
+        This method handles input preparation for generation, including proper slicing of inputs
+        based on cache position, MoE token type management, and multi-modal data handling.
+        Vision inputs are selectively forwarded only when needed during generation.
+
+        Args:
+            input_ids: Input token IDs
+            past_key_values: Cached key-value pairs from previous generation steps
+            attention_mask: Attention mask for input tokens
+            inputs_embeds: Pre-computed input embeddings
+            moe_token_types: Token type assignments for MoE routing
+            cache_position: Current cache position for generation
+            position_ids: Position IDs for tokens
+            use_cache: Whether to use key-value caching
+            pixel_values: Image pixel values
+            pixel_values_videos: Video pixel values
+            image_grid_thw: Image grid dimensions
+            video_grid_thw: Video grid dimensions
+            second_per_grid_ts: Time interval per temporal grid
+            proprioception: Proprioceptive sensor data
+            dof_mask: Degrees of freedom mask
+            agent_pos_mask: Agent position mask
+            **kwargs: Additional arguments
+
+        Returns:
+            dict: Prepared model inputs for generation step
+
+        Todo:
+            - Test this function thoroughly with various input configurations
+
+        Note:
+            This is an overridden method that handles specific cases for multi-modal generation:
+            - Slices input_ids through cache_position to keep only unprocessed tokens
+            - Handles special cases for input_embeds, generation methods, and GPU synchronization
+            - Manages vision inputs to avoid unnecessary forward passes
+        """
+        # Initialize MoE token types if not provided
+        if moe_token_types is None:
+            moe_token_types = torch.zeros_like(
+                input_ids
+            )  # FIXME: Handle case when input_embeds is used instead
+        else:
+            # Ensure moe_token_types length matches input_ids
+            if moe_token_types.shape[1] < input_ids.shape[1]:
+                # Calculate required padding length
+                pad_length = input_ids.shape[1] - moe_token_types.shape[1]
+                # Create padding tensor with default token type (0)
+                pad_tensor = torch.zeros(
+                    (moe_token_types.shape[0], pad_length),
+                    dtype=moe_token_types.dtype,
+                    device=moe_token_types.device,
+                )
+                # Concatenate padding to existing moe_token_types
+                moe_token_types = torch.cat([moe_token_types, pad_tensor], dim=1)
+
+        # Handle input slicing based on cache state and special cases
+        if past_key_values is not None:
+            if inputs_embeds is not None and input_ids.shape[1] == 0:  # Exception 4: input_embeds case
+                inputs_embeds = inputs_embeds[:, -cache_position.shape[0] :]
+                moe_token_types = moe_token_types[:, -cache_position.shape[0] :]
+            elif inputs_embeds is not None or (  # Exception 1: input_embeds provided
+                is_torchdynamo_compiling() or cache_position[-1] >= input_ids.shape[1]
+            ):  # Exception 3: GPU sync edge case
+                input_ids = input_ids[:, -cache_position.shape[0] :]
+                moe_token_types = moe_token_types[:, -cache_position.shape[0] :]
+            elif input_ids.shape[1] != cache_position.shape[0]:  # Default case (Exception 2 is no-op)
+                cache_pos = cache_position.clone()
+                input_ids = input_ids[:, cache_pos]
+                moe_token_types = moe_token_types[:, cache_pos]
+
+        # Skip vision inputs for continuation steps (not initial generation)
+        if cache_position[0] != 0:
+            pixel_values = None
+            pixel_values_videos = None
+
+        # Determine whether to use inputs_embeds or input_ids for this generation step
+        if inputs_embeds is not None and len(cache_position) == inputs_embeds.shape[1]:
+            model_inputs = {"inputs_embeds": inputs_embeds, "input_ids": None}
+        else:
+            model_inputs = {"input_ids": input_ids, "inputs_embeds": None}
+
+        # Prepare 4D causal attention mask for static cache
+        if isinstance(past_key_values, StaticCache) and attention_mask.ndim == 2:
+            if model_inputs["inputs_embeds"] is not None:
+                batch_size, sequence_length, _ = inputs_embeds.shape
+                device = inputs_embeds.device
+            else:
+                batch_size, sequence_length = input_ids.shape
+                device = input_ids.device
+
+            attention_mask = self.model._prepare_4d_causal_attention_mask_with_cache_position(
+                attention_mask,
+                sequence_length=sequence_length,
+                target_length=past_key_values.get_max_cache_shape(),
+                dtype=self.lm_head.weight.dtype,
+                device=device,
+                cache_position=cache_position,
+                batch_size=batch_size,
+                config=self.config,
+                past_key_values=past_key_values,
+            )
+
+        # Assemble all model inputs for generation
+        model_inputs.update(
+            {
+                "position_ids": position_ids,
+                "past_key_values": past_key_values,
+                "moe_token_types": moe_token_types,
+                "use_cache": use_cache,
+                "attention_mask": attention_mask,
+                "pixel_values": pixel_values,
+                "pixel_values_videos": pixel_values_videos,
+                "image_grid_thw": image_grid_thw,
+                "video_grid_thw": video_grid_thw,
+                "cache_position": cache_position,
+                "second_per_grid_ts": second_per_grid_ts,
+                "proprioception": proprioception,
+                "dof_mask": dof_mask,
+                "agent_pos_mask": agent_pos_mask,
+            }
+        )
+        return model_inputs
+
+    def _get_image_nums_and_video_nums(
+        self,
+        input_ids: torch.LongTensor | None,
+    ) -> tuple[torch.Tensor, torch.Tensor]:
+        """
+        Get the number of images and videos for each sample to calculate tensor separation lengths.
+
+        These parameters are computed directly from input_ids rather than being passed through
+        the processor to avoid unpredictable impacts from interface modifications.
+
+        Args:
+            input_ids (torch.LongTensor): Input token IDs of shape (batch_size, sequence_length)
+
+        Returns:
+            tuple:
+                - image_nums (torch.LongTensor): Number of images per sample
+                - video_nums (torch.LongTensor): Number of videos per sample
+        """
+        image_token_id = self.config.image_token_id
+        video_token_id = self.config.video_token_id
+        vision_start_token_id = self.config.vision_start_token_id
+
+        # Find vision start tokens and their following tokens
+        vision_start_mask = input_ids == vision_start_token_id
+        vision_first_mask = torch.roll(vision_start_mask, shifts=1, dims=1)
+        image_mask = input_ids == image_token_id
+        video_mask = input_ids == video_token_id
+
+        # Count images and videos following vision start tokens
+        image_nums = torch.sum(vision_first_mask & image_mask, dim=1)
+        video_nums = torch.sum(vision_first_mask & video_mask, dim=1)
+
+        return image_nums, video_nums
+
+    def _expand_inputs_for_generation(
+        self,
+        expand_size: int = 1,
+        is_encoder_decoder: bool = False,
+        input_ids: torch.LongTensor | None = None,
+        **model_kwargs,
+    ) -> tuple[torch.LongTensor, dict[str, Any]]:
+        """
+        Expand inputs for generation with support for multi-modal tensors.
+
+        This is an overridden method that supports expanding tensors without a standard batch
+        size dimension, specifically for vision-related tensors:
+        - pixel_values.shape[0] = sum(sequence_lengths for all image samples)
+        - image_grid_thw.shape[0] = sum(num_images for all samples)
+        - Similar patterns for video tensors
+
+        Args:
+            expand_size (int): Factor by which to expand inputs (for beam search, etc.)
+            is_encoder_decoder (bool): Whether using encoder-decoder architecture
+            input_ids (torch.LongTensor, optional): Input token IDs
+            **model_kwargs: Additional model arguments to expand
+
+        Returns:
+            tuple: (expanded_input_ids, expanded_model_kwargs)
+        """
+        if expand_size == 1:
+            return input_ids, model_kwargs
+
+        # Define keys for vision-related tensors that need special handling
+        visual_keys = [
+            "pixel_values",
+            "image_grid_thw",
+            "pixel_values_videos",
+            "video_grid_thw",
+            "second_per_grid_ts",
+        ]
+
+        def _expand_dict_for_generation_visual(dict_to_expand):
+            """Expand vision-related tensors based on image/video counts per sample."""
+            image_grid_thw = model_kwargs.get("image_grid_thw", None)
+            video_grid_thw = model_kwargs.get("video_grid_thw", None)
+            image_nums, video_nums = self._get_image_nums_and_video_nums(input_ids)
+
+            def _repeat_interleave_samples(x, lengths, repeat_times):
+                """Split tensor by lengths and repeat each sample."""
+                samples = torch.split(x, lengths)
+                repeat_args = [repeat_times] + [1] * (x.dim() - 1)
+                result = torch.cat([sample.repeat(*repeat_args) for sample in samples], dim=0)
+                return result
+
+            for key in dict_to_expand:
+                if key == "pixel_values":
+                    # Split images into samples and compute sequence lengths
+                    samples = torch.split(image_grid_thw, list(image_nums))
+                    lengths = [torch.prod(sample, dim=1).sum() for sample in samples]
+                    dict_to_expand[key] = _repeat_interleave_samples(
+                        dict_to_expand[key], lengths=lengths, repeat_times=expand_size
+                    )
+                elif key == "image_grid_thw":
+                    # Expand based on number of images per sample
+                    lengths = list(image_nums)
+                    dict_to_expand[key] = _repeat_interleave_samples(
+                        dict_to_expand[key], lengths=lengths, repeat_times=expand_size
+                    )
+                elif key == "pixel_values_videos":
+                    # Split videos into samples and compute sequence lengths
+                    samples = torch.split(video_grid_thw, list(video_nums))
+                    lengths = [torch.prod(sample, dim=1).sum() for sample in samples]
+                    dict_to_expand[key] = _repeat_interleave_samples(
+                        dict_to_expand[key], lengths=lengths, repeat_times=expand_size
+                    )
+                elif key == "video_grid_thw":
+                    # Expand based on number of videos per sample
+                    lengths = list(video_nums)
+                    dict_to_expand[key] = _repeat_interleave_samples(
+                        dict_to_expand[key], lengths=lengths, repeat_times=expand_size
+                    )
+                elif key == "second_per_grid_ts":
+                    # Handle list-type temporal grid data
+                    if not isinstance(dict_to_expand[key], list):
+                        raise TypeError(
+                            f"Expected value for key '{key}' to be a list, but got {type(dict_to_expand[key])} instead."
+                        )
+                    tensor = torch.tensor(dict_to_expand[key])
+                    lengths = list(video_nums)
+                    tensor = _repeat_interleave_samples(tensor, lengths=lengths, repeat_times=expand_size)
+                    dict_to_expand[key] = tensor.tolist()
+            return dict_to_expand
+
+        def _expand_dict_for_generation(dict_to_expand):
+            """Expand standard tensors using repeat_interleave."""
+            for key in dict_to_expand:
+                if (
+                    key != "cache_position"
+                    and dict_to_expand[key] is not None
+                    and isinstance(dict_to_expand[key], torch.Tensor)
+                    and key not in visual_keys
+                ):
+                    dict_to_expand[key] = dict_to_expand[key].repeat_interleave(expand_size, dim=0)
+            return dict_to_expand
+
+        # Expand visual inputs only if input_ids is available for counting images/videos
+        # If input_ids is unavailable, visual inputs won't be used, so no expansion needed
+        if input_ids is not None and input_ids.numel() != 0:
+            model_kwargs = _expand_dict_for_generation_visual(model_kwargs)
+
+        # Expand input_ids using standard repeat_interleave
+        if input_ids is not None:
+            input_ids = input_ids.repeat_interleave(expand_size, dim=0)
+
+        # Expand all other model arguments
+        model_kwargs = _expand_dict_for_generation(model_kwargs)
+
+        # Handle encoder-decoder specific expansion
+        if is_encoder_decoder:
+            if model_kwargs.get("encoder_outputs") is None:
+                raise ValueError(
+                    "If `is_encoder_decoder` is True, make sure that `encoder_outputs` is defined."
+                )
+            model_kwargs["encoder_outputs"] = _expand_dict_for_generation(model_kwargs["encoder_outputs"])
+
+        return input_ids, model_kwargs
+
+
+class WallXPolicy(PreTrainedPolicy):
+    """
+    Wall-X policy for cross-embodiment robotic control.
+
+    Integrates Qwen2.5-VL vision-language model with action prediction
+    using flow matching for continuous action spaces.
+    """
+
+    config_class = WallXConfig
+    name = "wall_x"
+
+    def __init__(self, config: WallXConfig, **kwargs):
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        # Initialize the wall-x model
+        self.model = Qwen2_5_VLMoEForAction.from_pretrained(
+            pretrained_name_or_path=config.pretrained_name_or_path,
+            action_tokenizer_path=config.action_tokenizer_path,
+            attn_implementation=config.attn_implementation,
+        )
+        self.model.to(config.device)
+        self.model.to_bfloat16_for_selected_params()
+
+        self.reset()
+
+    def reset(self):
+        """Reset action queue."""
+        self._queues = {
+            ACTION: deque(maxlen=self.config.n_action_steps),
+        }
+
+    def get_optim_params(self):
+        """Get parameters for optimization."""
+        return self.parameters()
+
+    def preprocess_inputs(
+        self,
+        batch: dict[str, Any],
+    ) -> BatchFeature:
+        """
+        Convert a batch of LeRobot dataset items to Wall-X model input format.
+
+        This processes a batched dictionary where tensors have batch dimension first.
+
+        Args:
+            batch: Dictionary with batched tensors:
+                - "observation.state": (batch_size, state_dim) or (batch_size, n_obs_steps, state_dim)
+                - "action": (batch_size, chunk_size, action_dim)
+                - "observation.images.<key>": (batch_size, C, H, W)
+                - "task": List[str] of length batch_size
+
+        Returns:
+            BatchFeature containing batched model inputs
+        """
+        use_fast_tokenizer = self.config.use_fast_tokenizer
+
+        # Get batch size from state tensor
+        batch_size = batch[OBS_STATE].shape[0]
+
+        # ==================== PROCESS ALL SAMPLES ====================
+        all_image_inputs = []
+        all_texts = []
+
+        # Find image keys in batch
+        img_keys = [key for key in self.config.image_features if key in batch]
+
+        for i in range(batch_size):
+            # Vision preprocessing per sample
+            processed_frames = []
+            orig_height, orig_width = None, None
+            resized_height, resized_width = None, None
+
+            for key in img_keys:
+                current_obs = batch[key][i].clone()  # (C, H, W)
+                if current_obs.dim() == 3:
+                    current_obs = current_obs.permute(1, 2, 0)  # (H, W, C)
+
+                img_pil = Image.fromarray((current_obs * 255).to(torch.uint8).cpu().numpy())
+                orig_width, orig_height = img_pil.size
+
+                target_size = RESOLUTION
+                if target_size != -1:
+                    if orig_width > orig_height:
+                        new_width = target_size
+                        new_height = int(target_size * orig_height / orig_width)
+                    else:
+                        new_height = target_size
+                        new_width = int(target_size * orig_width / orig_height)
+                    img_pil = img_pil.resize((new_width, new_height))
+
+                current_width, current_height = img_pil.size
+                resized_height, resized_width = smart_resize(
+                    current_height,
+                    current_width,
+                    factor=IMAGE_FACTOR,
+                    min_pixels=MIN_PIXELS,
+                    max_pixels=MAX_PIXELS,
+                )
+                resized_img = img_pil.resize((resized_width, resized_height))
+                processed_frames.append(resized_img)
+
+            all_image_inputs.append(processed_frames)
+
+            # Text preprocessing
+            task_text = batch["task"][i] if isinstance(batch["task"], list) else batch["task"]
+            instruction_info = {"instruction": task_text}
+
+            frame_index = batch["frame_index"][i] if "frame_index" in batch else 0
+            complete_text, _ = get_wallx_normal_text(
+                instruction_info,
+                self.config.chunk_size,
+                frame_index,
+                PRIORITY_ORDER,
+                img_keys,
+                generate_subtask_ratio=GENERATE_SUBTASK_RATIO,
+            )
+
+            text = process_grounding_points(
+                complete_text, orig_height, orig_width, resized_height, resized_width, MODEL_TYPE
+            )
+            all_texts.append(text)
+
+        # ==================== PROCESS AGENT POS ====================
+        agent_pos = batch[OBS_STATE]  # (batch_size, state_dim)
+        if agent_pos.dim() == 2:
+            agent_pos = agent_pos.unsqueeze(1)  # (batch_size, 1, state_dim)
+        agent_pos_mask = (~torch.isnan(agent_pos)).float()
+        agent_pos = agent_pos.nan_to_num(nan=0.0)
+
+        if agent_pos.shape[-1] != 20:
+            pad_size = 20 - agent_pos.shape[-1]
+            agent_pos = torch.cat(
+                [
+                    agent_pos,
+                    torch.zeros(agent_pos.shape[0], agent_pos.shape[1], pad_size, device=agent_pos.device),
+                ],
+                dim=-1,
+            )
+            agent_pos_mask = torch.cat(
+                [
+                    agent_pos_mask,
+                    torch.zeros(
+                        agent_pos_mask.shape[0],
+                        agent_pos_mask.shape[1],
+                        pad_size,
+                        device=agent_pos_mask.device,
+                    ),
+                ],
+                dim=-1,
+            )
+
+        # ==================== PROCESS ACTIONS ====================
+        action = batch.get(ACTION)  # (batch_size, chunk_size, action_dim)
+        if action is not None:
+            if action.dim() == 2:
+                action = action.unsqueeze(1)
+            dof_mask = (~torch.isnan(action)).float()
+            action = action.nan_to_num(nan=0.0)
+
+            if action.shape[-1] != 20:
+                pad_size = 20 - action.shape[-1]
+                action = torch.cat(
+                    [action, torch.zeros(action.shape[0], action.shape[1], pad_size, device=action.device)],
+                    dim=-1,
+                )
+                dof_mask = torch.cat(
+                    [
+                        dof_mask,
+                        torch.zeros(dof_mask.shape[0], dof_mask.shape[1], pad_size, device=dof_mask.device),
+                    ],
+                    dim=-1,
+                )
+        else:
+            action_dim = self.config.output_features[ACTION].shape[0]
+            dof_mask = torch.cat(
+                [
+                    torch.ones(
+                        batch_size, self.config.chunk_size, action_dim, device=batch[OBS_STATE].device
+                    ),
+                    torch.zeros(
+                        batch_size, self.config.chunk_size, 20 - action_dim, device=batch[OBS_STATE].device
+                    ),
+                ],
+                dim=-1,
+            )
+
+        # ==================== ACTION TOKEN REPLACEMENT ====================
+        all_texts = replace_action_token(
+            all_texts,
+            action,
+            self.model.action_tokenizer if use_fast_tokenizer else None,
+            dof_mask,
+        )
+
+        # ==================== TOKENIZATION ====================
+        inputs = preprocesser_call(
+            processor=self.model.processor,
+            text=all_texts,
+            images=all_image_inputs,
+            videos=None,
+            padding=True,
+            truncation=True,
+            return_tensors="pt",
+            max_length=TOKENIZER_MAX_LENGTH,
+        )
+
+        # ==================== ADDITIONAL INPUTS ====================
+        action_token_id = self.model.processor.tokenizer.convert_tokens_to_ids("<|action|>")
+        moe_token_types = inputs.input_ids == action_token_id
+
+        inputs["proprioception"] = agent_pos
+        inputs["agent_pos_mask"] = agent_pos_mask
+        inputs["action_chunk"] = action
+        inputs["dof_mask"] = dof_mask
+        inputs["moe_token_types"] = moe_token_types
+        inputs["frame_index"] = (
+            batch["frame_index"]
+            if "frame_index" in batch
+            else torch.zeros(batch_size, device=batch[OBS_STATE].device)
+        )
+
+        # Move all tensors to the correct device
+        device = self.config.device
+        for key, value in inputs.items():
+            if isinstance(value, torch.Tensor):
+                inputs[key] = value.to(device)
+
+        return inputs
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, dict]:
+        """
+        Training forward pass using Qwen2_5_VLMoEForAction.
+
+        Args:
+            batch: Dictionary containing preprocessed inputs from preprocess_inputs()
+                   Expected keys: input_ids, attention_mask, pixel_values, image_grid_thw,
+                   proprioception, agent_pos_mask, action_chunk, dof_mask, moe_token_types,
+                   etc.
+
+        Returns:
+            tuple: (loss, loss_dict)
+        """
+        batch = self.preprocess_inputs(
+            batch,
+        )
+
+        # Call the underlying model's forward with mode="train"
+        outputs = self.model(**batch, mode="train")
+
+        # Extract losses from output
+        loss = outputs.loss
+        loss_dict = {
+            "loss": loss.item() if loss is not None else 0.0,
+        }
+
+        if outputs.flow_loss is not None:
+            loss_dict["flow_loss"] = outputs.flow_loss.item()
+        if outputs.cross_entropy_loss is not None:
+            loss_dict["cross_entropy_loss"] = outputs.cross_entropy_loss.item()
+
+        # Add channel losses if available
+        if outputs.channel_loss_dict is not None:
+            for key, value in outputs.channel_loss_dict.items():
+                if isinstance(value, torch.Tensor):
+                    loss_dict[f"channel_{key}"] = value.item()
+
+        return loss, loss_dict
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor]) -> Tensor:
+        """Predict action chunk for evaluation."""
+        self.eval()
+        self._queues = populate_queues(self._queues, batch, exclude_keys=[ACTION])
+
+        batch = self.preprocess_inputs(
+            batch,
+        )
+
+        if self.config.prediction_mode == "diffusion":
+            output = self.model(
+                **batch,
+                action_dim=self.config.max_action_dim,
+                pred_horizon=self.config.chunk_size,
+                mode="predict",
+                predict_mode="diffusion",
+            )
+        elif self.config.prediction_mode == "fast":
+            output = self.model(
+                **batch,
+                action_dim=self.config.output_features[ACTION].shape[0],
+                pred_horizon=self.config.chunk_size,
+                mode="predict",
+                predict_mode="fast",
+            )
+        else:
+            raise NotImplementedError(f"Prediction mode {self.config.prediction_mode} not implemented")
+
+        # Extract action tensor from output dictionary
+        actions = output["predict_action"]
+
+        # Unpad actions to actual action dimension
+        action_dim = self.config.output_features[ACTION].shape[0]
+        actions = actions[:, :, :action_dim]
+
+        return actions
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select single action for environment execution."""
+        self.eval()
+        self._queues = populate_queues(self._queues, batch, exclude_keys=[ACTION])
+
+        # Use action queue
+        if len(self._queues[ACTION]) == 0:
+            actions = self.predict_action_chunk(batch)
+            self._queues[ACTION].extend(actions.transpose(0, 1)[: self.config.n_action_steps])
+
+        return self._queues[ACTION].popleft()
diff --git a/lerobot/src/lerobot/policies/wall_x/processor_wall_x.py b/lerobot/src/lerobot/policies/wall_x/processor_wall_x.py
new file mode 100644
index 0000000000000000000000000000000000000000..e4e281541bd65db38fc176b72adc9cf7b0955a43
--- /dev/null
+++ b/lerobot/src/lerobot/policies/wall_x/processor_wall_x.py
@@ -0,0 +1,133 @@
+#!/usr/bin/env python
+
+# Copyright 2025 HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from typing import Any
+
+import torch
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.policies.wall_x.configuration_wall_x import WallXConfig
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    ComplementaryDataProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    ProcessorStepRegistry,
+    RenameObservationsProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.utils.constants import POLICY_POSTPROCESSOR_DEFAULT_NAME, POLICY_PREPROCESSOR_DEFAULT_NAME
+
+
+def make_wall_x_pre_post_processors(
+    config: WallXConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Constructs pre-processor and post-processor pipelines for the Wall-X policy.
+
+    The pre-processing pipeline prepares input data for the model by:
+    1. Renaming features to match pretrained configurations
+    2. Adding a batch dimension
+    4. Normalizing input and output features based on dataset statistics
+    5. Moving all data to the specified device
+
+    The post-processing pipeline handles the model's output by:
+    1. Unnormalizing the output actions to their original scale
+    2. Moving data to the CPU
+
+    Args:
+        config: The configuration object for the Wall-X policy
+        dataset_stats: A dictionary of statistics for normalization
+
+    Returns:
+        A tuple containing the configured pre-processor and post-processor pipelines
+    """
+
+    input_steps = [
+        RenameObservationsProcessorStep(rename_map={}),
+        AddBatchDimensionProcessorStep(),
+        WallXTaskProcessor(),  # Process task description
+        NormalizerProcessorStep(
+            features={**config.input_features, **config.output_features},
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+        DeviceProcessorStep(device=config.device),
+    ]
+
+    output_steps = [
+        UnnormalizerProcessorStep(
+            features=config.output_features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
+
+
+@ProcessorStepRegistry.register(name="wall_x_task_processor")
+class WallXTaskProcessor(ComplementaryDataProcessorStep):
+    """
+    A processor step that ensures the task description is properly formatted for Wall-X.
+
+    This step handles task preprocessing similar to Qwen-VL requirements.
+    """
+
+    def complementary_data(self, complementary_data):
+        if "task" not in complementary_data:
+            return complementary_data
+
+        task = complementary_data["task"]
+        if task is None:
+            # Provide default task if none specified
+            complementary_data["task"] = "Execute the robot action."
+            return complementary_data
+
+        new_complementary_data = dict(complementary_data)
+
+        # Handle both string and list of strings
+        if isinstance(task, str):
+            # Single string: ensure proper formatting
+            if not task.endswith("."):
+                new_complementary_data["task"] = f"{task}."
+        elif isinstance(task, list) and all(isinstance(t, str) for t in task):
+            # List of strings: format each
+            new_complementary_data["task"] = [t if t.endswith(".") else f"{t}." for t in task]
+
+        return new_complementary_data
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
diff --git a/lerobot/src/lerobot/policies/wall_x/qwen_model/configuration_qwen2_5_vl.py b/lerobot/src/lerobot/policies/wall_x/qwen_model/configuration_qwen2_5_vl.py
new file mode 100644
index 0000000000000000000000000000000000000000..19874b6ff1ad73d8429e48cf61530080db571b44
--- /dev/null
+++ b/lerobot/src/lerobot/policies/wall_x/qwen_model/configuration_qwen2_5_vl.py
@@ -0,0 +1,250 @@
+from transformers.configuration_utils import PretrainedConfig
+from transformers.modeling_rope_utils import rope_config_validation
+
+
+class Qwen2_5_VLVisionConfig(PretrainedConfig):
+    model_type = "qwen2_5_vl"
+    base_config_key = "vision_config"
+
+    def __init__(
+        self,
+        depth=32,
+        hidden_size=3584,
+        hidden_act="silu",
+        intermediate_size=3420,
+        num_heads=16,
+        in_channels=3,
+        patch_size=14,
+        spatial_merge_size=2,
+        temporal_patch_size=2,
+        tokens_per_second=4,
+        window_size=112,
+        out_hidden_size=3584,
+        fullatt_block_indexes=[7, 15, 23, 31],
+        initializer_range=0.02,
+        **kwargs,
+    ):
+        super().__init__(**kwargs)
+
+        self.depth = depth
+        self.hidden_size = hidden_size
+        self.hidden_act = hidden_act
+        self.intermediate_size = intermediate_size
+        self.num_heads = num_heads
+        self.in_channels = in_channels
+        self.patch_size = patch_size
+        self.spatial_merge_size = spatial_merge_size
+        self.temporal_patch_size = temporal_patch_size
+        self.tokens_per_second = tokens_per_second
+        self.window_size = window_size
+        self.fullatt_block_indexes = fullatt_block_indexes
+        self.out_hidden_size = out_hidden_size
+        self.initializer_range = initializer_range
+
+
+class Qwen2_5_VLConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`Qwen2_5_VLModel`]. It is used to instantiate a
+    Qwen2-VL model according to the specified arguments, defining the model architecture. Instantiating a configuration
+    with the defaults will yield a similar configuration to that of
+    Qwen2-VL-7B-Instruct [Qwen/Qwen2-VL-7B-Instruct](https://huggingface.co/Qwen/Qwen2-VL-7B-Instruct).
+
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+
+
+    Args:
+        vocab_size (`int`, *optional*, defaults to 152064):
+            Vocabulary size of the Qwen2_5_VL model. Defines the number of different tokens that can be represented by the
+            `inputs_ids` passed when calling [`Qwen2_5_VLModel`]
+        hidden_size (`int`, *optional*, defaults to 8192):
+            Dimension of the hidden representations.
+        intermediate_size (`int`, *optional*, defaults to 29568):
+            Dimension of the MLP representations.
+        num_hidden_layers (`int`, *optional*, defaults to 80):
+            Number of hidden layers in the Transformer encoder.
+        num_attention_heads (`int`, *optional*, defaults to 64):
+            Number of attention heads for each attention layer in the Transformer encoder.
+        num_key_value_heads (`int`, *optional*, defaults to 8):
+            This is the number of key_value heads that should be used to implement Grouped Query Attention. If
+            `num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
+            `num_key_value_heads=1` the model will use Multi Query Attention (MQA) otherwise GQA is used. When
+            converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
+            by meanpooling all the original heads within that group. For more details checkout [this
+            paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to `32`.
+        hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
+            The non-linear activation function (function or string) in the decoder.
+        max_position_embeddings (`int`, *optional*, defaults to 32768):
+            The maximum sequence length that this model might ever be used with.
+        initializer_range (`float`, *optional*, defaults to 0.02):
+            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
+        rms_norm_eps (`float`, *optional*, defaults to 1e-05):
+            The epsilon used by the rms normalization layers.
+        use_cache (`bool`, *optional*, defaults to `True`):
+            Whether or not the model should return the last key/values attentions (not used by all models). Only
+            relevant if `config.is_decoder=True`.
+        tie_word_embeddings (`bool`, *optional*, defaults to `False`):
+            Whether the model's input and output word embeddings should be tied.
+        rope_theta (`float`, *optional*, defaults to 1000000.0):
+            The base period of the RoPE embeddings.
+        use_sliding_window (`bool`, *optional*, defaults to `False`):
+            Whether to use sliding window attention.
+        sliding_window (`int`, *optional*, defaults to 4096):
+            Sliding window attention (SWA) window size. If not specified, will default to `4096`.
+        max_window_layers (`int`, *optional*, defaults to 80):
+            The number of layers that use SWA (Sliding Window Attention). The bottom layers use SWA while the top use full attention.
+        attention_dropout (`float`, *optional*, defaults to 0.0):
+            The dropout ratio for the attention probabilities.
+        vision_config (`Dict`, *optional*):
+            The config for the visual encoder initialization.
+        rope_scaling (`Dict`, *optional*):
+            Dictionary containing the scaling configuration for the RoPE embeddings. NOTE: if you apply new rope type
+            and you expect the model to work on longer `max_position_embeddings`, we recommend you to update this value
+            accordingly.
+            Expected contents:
+                `rope_type` (`str`):
+                    The sub-variant of RoPE to use. Can be one of ['default', 'linear', 'dynamic', 'yarn', 'longrope',
+                    'llama3'], with 'default' being the original RoPE implementation.
+                `factor` (`float`, *optional*):
+                    Used with all rope types except 'default'. The scaling factor to apply to the RoPE embeddings. In
+                    most scaling types, a `factor` of x will enable the model to handle sequences of length x *
+                    original maximum pre-trained length.
+                `original_max_position_embeddings` (`int`, *optional*):
+                    Used with 'dynamic', 'longrope' and 'llama3'. The original max position embeddings used during
+                    pretraining.
+                `attention_factor` (`float`, *optional*):
+                    Used with 'yarn' and 'longrope'. The scaling factor to be applied on the attention
+                    computation. If unspecified, it defaults to value recommended by the implementation, using the
+                    `factor` field to infer the suggested value.
+                `beta_fast` (`float`, *optional*):
+                    Only used with 'yarn'. Parameter to set the boundary for extrapolation (only) in the linear
+                    ramp function. If unspecified, it defaults to 32.
+                `beta_slow` (`float`, *optional*):
+                    Only used with 'yarn'. Parameter to set the boundary for interpolation (only) in the linear
+                    ramp function. If unspecified, it defaults to 1.
+                `short_factor` (`List[float]`, *optional*):
+                    Only used with 'longrope'. The scaling factor to be applied to short contexts (<
+                    `original_max_position_embeddings`). Must be a list of numbers with the same length as the hidden
+                    size divided by the number of attention heads divided by 2
+                `long_factor` (`List[float]`, *optional*):
+                    Only used with 'longrope'. The scaling factor to be applied to long contexts (<
+                    `original_max_position_embeddings`). Must be a list of numbers with the same length as the hidden
+                    size divided by the number of attention heads divided by 2
+                `low_freq_factor` (`float`, *optional*):
+                    Only used with 'llama3'. Scaling factor applied to low frequency components of the RoPE
+                `high_freq_factor` (`float`, *optional*):
+                    Only used with 'llama3'. Scaling factor applied to high frequency components of the RoPE
+
+    ```python
+    >>> from transformers import Qwen2_5_VLForConditionalGeneration, Qwen2_5_VLConfig
+
+    >>> # Initializing a Qwen2_5_VL style configuration
+    >>> configuration = Qwen2_5_VLConfig()
+
+    >>> # Initializing a model from the Qwen2-VL-7B style configuration
+    >>> model = Qwen2_5_VLForConditionalGeneration(configuration)
+
+    >>> # Accessing the model configuration
+    >>> configuration = model.config
+    ```"""
+
+    model_type = "qwen2_5_vl"
+    sub_configs = {"vision_config": Qwen2_5_VLVisionConfig}
+    keys_to_ignore_at_inference = ["past_key_values"]
+    # Default tensor parallel plan for base model `Qwen2_5_VL`
+    base_model_tp_plan = {
+        "layers.*.self_attn.q_proj": "colwise",
+        "layers.*.self_attn.k_proj": "colwise",
+        "layers.*.self_attn.v_proj": "colwise",
+        "layers.*.self_attn.o_proj": "rowwise",
+        "layers.*.mlp.gate_proj": "colwise",
+        "layers.*.mlp.up_proj": "colwise",
+        "layers.*.mlp.down_proj": "rowwise",
+    }
+    base_model_pp_plan = {
+        "embed_tokens": (["input_ids"], ["inputs_embeds"]),
+        "layers": (["hidden_states", "attention_mask"], ["hidden_states"]),
+        "norm": (["hidden_states"], ["hidden_states"]),
+    }
+
+    def __init__(
+        self,
+        vocab_size=152064,
+        hidden_size=8192,
+        intermediate_size=29568,
+        num_hidden_layers=80,
+        num_attention_heads=64,
+        num_key_value_heads=8,
+        hidden_act="silu",
+        max_position_embeddings=32768,
+        initializer_range=0.02,
+        rms_norm_eps=1e-05,
+        use_cache=True,
+        tie_word_embeddings=False,
+        rope_theta=1000000.0,
+        use_sliding_window=False,
+        sliding_window=4096,
+        max_window_layers=80,
+        attention_dropout=0.0,
+        vision_config=None,
+        rope_scaling=None,
+        num_experts=4,
+        experts=None,
+        dof_config=None,
+        noise_scheduler=None,
+        dim_inputs=(1536, 1536),
+        attention_moe=False,
+        mlp_moe=False,
+        **kwargs,
+    ):
+        if isinstance(vision_config, dict):
+            self.vision_config = self.sub_configs["vision_config"](**vision_config)
+        elif vision_config is None:
+            self.vision_config = self.sub_configs["vision_config"]()
+
+        self.vocab_size = vocab_size
+        self.max_position_embeddings = max_position_embeddings
+        self.hidden_size = hidden_size
+        self.intermediate_size = intermediate_size
+        self.num_hidden_layers = num_hidden_layers
+        self.num_attention_heads = num_attention_heads
+        self.use_sliding_window = use_sliding_window
+        self.sliding_window = sliding_window
+        self.max_window_layers = max_window_layers
+        self.layer_types = ["dense"] * num_hidden_layers
+
+        # for backward compatibility
+        if num_key_value_heads is None:
+            num_key_value_heads = num_attention_heads
+
+        self.num_key_value_heads = num_key_value_heads
+        self.hidden_act = hidden_act
+        self.initializer_range = initializer_range
+        self.rms_norm_eps = rms_norm_eps
+        self.use_cache = use_cache
+        self.rope_theta = rope_theta
+        self.attention_dropout = attention_dropout
+        self.rope_scaling = rope_scaling
+
+        self.num_experts = num_experts
+        self.experts = experts
+        self.dof_config = dof_config
+        self.noise_scheduler = noise_scheduler
+        self.dim_inputs = tuple(dim_inputs)
+        self.attention_moe = attention_moe
+        self.mlp_moe = mlp_moe
+
+        if self.rope_scaling is not None and "type" in self.rope_scaling:
+            if self.rope_scaling["type"] == "mrope":
+                self.rope_scaling["type"] = "default"
+            self.rope_scaling["rope_type"] = self.rope_scaling["type"]
+        rope_config_validation(self, ignore_keys={"mrope_section"})
+
+        super().__init__(tie_word_embeddings=tie_word_embeddings, **kwargs)
+
+    @property
+    def text_config(self):
+        return self
+
+
+__all__ = ["Qwen2_5_VLConfig"]
diff --git a/lerobot/src/lerobot/policies/wall_x/qwen_model/qwen2_5_vl_moe.py b/lerobot/src/lerobot/policies/wall_x/qwen_model/qwen2_5_vl_moe.py
new file mode 100644
index 0000000000000000000000000000000000000000..ecf3eb37171082468406d33f9b63ccce676f4014
--- /dev/null
+++ b/lerobot/src/lerobot/policies/wall_x/qwen_model/qwen2_5_vl_moe.py
@@ -0,0 +1,2817 @@
+import math
+from dataclasses import dataclass
+from typing import Any
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from torch.nn import CrossEntropyLoss
+from transformers import AutoConfig
+from transformers.activations import ACT2FN
+from transformers.cache_utils import (
+    Cache,
+    DynamicCache,
+    StaticCache,
+)
+from transformers.generation import GenerationMixin
+from transformers.modeling_attn_mask_utils import AttentionMaskConverter
+from transformers.modeling_outputs import BaseModelOutputWithPast, ModelOutput
+from transformers.modeling_rope_utils import ROPE_INIT_FUNCTIONS
+from transformers.modeling_utils import PreTrainedModel
+from transformers.utils import (
+    add_start_docstrings,
+    add_start_docstrings_to_model_forward,
+    is_flash_attn_2_available,
+    is_flash_attn_greater_or_equal_2_10,
+    is_torchdynamo_compiling,
+    logging,
+    replace_return_docstrings,
+)
+
+from .configuration_qwen2_5_vl import Qwen2_5_VLConfig, Qwen2_5_VLVisionConfig
+
+
+# TODO(Steven): SlidingWindowCache was removed in transformers v5. Define a placeholder so isinstance checks
+# always return False (which is the correct behavior when no sliding window cache is in use).
+class _SlidingWindowCachePlaceholder:
+    pass
+
+
+SlidingWindowCache = _SlidingWindowCachePlaceholder
+
+if is_flash_attn_2_available():
+    from flash_attn import flash_attn_func, flash_attn_varlen_func
+    from flash_attn.layers.rotary import apply_rotary_emb
+else:
+    flash_attn_varlen_func = None
+    apply_rotary_emb = None
+    flash_attn_func = None
+
+
+if is_flash_attn_2_available():
+    pass
+else:
+    flash_attn_varlen_func = None
+
+
+logger = logging.get_logger(__name__)
+
+_CONFIG_FOR_DOC = "Qwen2_5_VLConfig"
+
+
+class Qwen2_5_VLMLP(nn.Module):
+    def __init__(self, config, bias: bool = False):
+        super().__init__()
+        self.hidden_size = config.hidden_size
+        self.intermediate_size = config.intermediate_size
+        self.gate_proj = nn.Linear(self.hidden_size, self.intermediate_size, bias=bias)
+        self.up_proj = nn.Linear(self.hidden_size, self.intermediate_size, bias=bias)
+        self.down_proj = nn.Linear(self.intermediate_size, self.hidden_size, bias=bias)
+        self.act_fn = ACT2FN[config.hidden_act]
+
+    def forward(self, hidden_state):
+        return self.down_proj(self.act_fn(self.gate_proj(hidden_state)) * self.up_proj(hidden_state))
+
+
+class Qwen2_5_VisionPatchEmbed(nn.Module):
+    def __init__(
+        self,
+        patch_size: int = 14,
+        temporal_patch_size: int = 2,
+        in_channels: int = 3,
+        embed_dim: int = 1152,
+    ) -> None:
+        super().__init__()
+        self.patch_size = patch_size
+        self.temporal_patch_size = temporal_patch_size
+        self.in_channels = in_channels
+        self.embed_dim = embed_dim
+
+        kernel_size = [temporal_patch_size, patch_size, patch_size]
+        self.proj = nn.Conv3d(
+            in_channels,
+            embed_dim,
+            kernel_size=kernel_size,
+            stride=kernel_size,
+            bias=False,
+        )
+
+    def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
+        target_dtype = self.proj.weight.dtype
+        hidden_states = hidden_states.view(
+            -1,
+            self.in_channels,
+            self.temporal_patch_size,
+            self.patch_size,
+            self.patch_size,
+        )
+        hidden_states = self.proj(hidden_states.to(dtype=target_dtype)).view(-1, self.embed_dim)
+        return hidden_states
+
+
+class Qwen2_5_VisionRotaryEmbedding(nn.Module):
+    def __init__(self, dim: int, theta: float = 10000.0) -> None:
+        super().__init__()
+        inv_freq = 1.0 / (theta ** (torch.arange(0, dim, 2, dtype=torch.float) / dim))
+        self.register_buffer("inv_freq", inv_freq, persistent=False)
+
+    def forward(self, seqlen: int) -> torch.Tensor:
+        seq = torch.arange(seqlen, device=self.inv_freq.device, dtype=self.inv_freq.dtype)
+        freqs = torch.outer(seq, self.inv_freq)
+        return freqs
+
+
+class Qwen2RMSNorm(nn.Module):
+    def __init__(self, hidden_size, eps=1e-6):
+        """
+        Qwen2RMSNorm is equivalent to T5LayerNorm
+        """
+        super().__init__()
+        self.weight = nn.Parameter(torch.ones(hidden_size))
+        self.variance_epsilon = eps
+
+    def forward(self, hidden_states):
+        input_dtype = hidden_states.dtype
+        hidden_states = hidden_states.to(torch.float32)
+        variance = hidden_states.pow(2).mean(-1, keepdim=True)
+        hidden_states = hidden_states * torch.rsqrt(variance + self.variance_epsilon)
+        return self.weight * hidden_states.to(input_dtype)
+
+    def extra_repr(self):
+        return f"{tuple(self.weight.shape)}, eps={self.variance_epsilon}"
+
+
+class Qwen2_5_VLPatchMerger(nn.Module):
+    def __init__(self, dim: int, context_dim: int, spatial_merge_size: int = 2) -> None:
+        super().__init__()
+        self.hidden_size = context_dim * (spatial_merge_size**2)
+        self.ln_q = Qwen2RMSNorm(context_dim, eps=1e-6)
+        self.mlp = nn.Sequential(
+            nn.Linear(self.hidden_size, self.hidden_size),
+            nn.GELU(),
+            nn.Linear(self.hidden_size, dim),
+        )
+
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        x = self.mlp(self.ln_q(x).view(-1, self.hidden_size))
+        return x
+
+
+def apply_rotary_pos_emb_flashatt(
+    q: torch.Tensor, k: torch.Tensor, cos: torch.Tensor, sin: torch.Tensor
+) -> tuple[torch.Tensor, torch.Tensor]:
+    cos = cos.chunk(2, dim=-1)[0].contiguous()
+    sin = sin.chunk(2, dim=-1)[0].contiguous()
+    q_embed = apply_rotary_emb(q.float(), cos.float(), sin.float()).type_as(q)
+    k_embed = apply_rotary_emb(k.float(), cos.float(), sin.float()).type_as(k)
+    return q_embed, k_embed
+
+
+class Qwen2_5_VLVisionFlashAttention2(nn.Module):
+    def __init__(self, dim: int, num_heads: int = 16) -> None:
+        super().__init__()
+        self.num_heads = num_heads
+        self.qkv = nn.Linear(dim, dim * 3, bias=True)
+        self.proj = nn.Linear(dim, dim)
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        cu_seqlens: torch.Tensor,
+        max_seqlen: int | None = None,
+        rotary_pos_emb: torch.Tensor | None = None,
+        position_embeddings: tuple[torch.Tensor, torch.Tensor] | None = None,
+    ) -> torch.Tensor:
+        seq_length = hidden_states.shape[0]
+        q, k, v = (
+            self.qkv(hidden_states).reshape(seq_length, 3, self.num_heads, -1).permute(1, 0, 2, 3).unbind(0)
+        )
+        if position_embeddings is None:
+            logger.warning_once(
+                "The attention layers in this model are transitioning from computing the RoPE embeddings internally "
+                "through `rotary_pos_emb` (2D tensor of RoPE theta values), to using externally computed "
+                "`position_embeddings` (Tuple of tensors, containing cos and sin). In v4.54 `rotary_pos_emb` will be "
+                "removed and `position_embeddings` will be mandatory."
+            )
+            emb = torch.cat((rotary_pos_emb, rotary_pos_emb), dim=-1)
+            cos = emb.cos().float()
+            sin = emb.sin().float()
+        else:
+            cos, sin = position_embeddings
+        q, k = apply_rotary_pos_emb_flashatt(q.unsqueeze(0), k.unsqueeze(0), cos, sin)
+        q = q.squeeze(0)
+        k = k.squeeze(0)
+
+        if max_seqlen is None:
+            max_seqlen = (cu_seqlens[1:] - cu_seqlens[:-1]).max().item()
+        attn_output = flash_attn_varlen_func(q, k, v, cu_seqlens, cu_seqlens, max_seqlen, max_seqlen).reshape(
+            seq_length, -1
+        )
+        attn_output = self.proj(attn_output)
+        return attn_output
+
+
+def rotate_half(x):
+    """Rotates half the hidden dims of the input."""
+    x1 = x[..., : x.shape[-1] // 2]
+    x2 = x[..., x.shape[-1] // 2 :]
+    return torch.cat((-x2, x1), dim=-1)
+
+
+def apply_rotary_pos_emb_vision(
+    q: torch.Tensor, k: torch.Tensor, cos: torch.Tensor, sin: torch.Tensor
+) -> tuple[torch.Tensor, torch.Tensor]:
+    orig_q_dtype = q.dtype
+    orig_k_dtype = k.dtype
+    q, k = q.float(), k.float()
+    cos, sin = cos.unsqueeze(-2), sin.unsqueeze(-2)
+    q_embed = (q * cos) + (rotate_half(q) * sin)
+    k_embed = (k * cos) + (rotate_half(k) * sin)
+    q_embed = q_embed.to(orig_q_dtype)
+    k_embed = k_embed.to(orig_k_dtype)
+    return q_embed, k_embed
+
+
+class Qwen2_5_VLVisionAttention(nn.Module):
+    def __init__(self, dim: int, num_heads: int = 16) -> None:
+        super().__init__()
+        self.num_heads = num_heads
+        self.head_dim = dim // num_heads
+        self.qkv = nn.Linear(dim, dim * 3, bias=True)
+        self.proj = nn.Linear(dim, dim)
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        cu_seqlens: torch.Tensor,
+        max_seqlen: int | None = None,
+        rotary_pos_emb: torch.Tensor | None = None,
+        position_embeddings: tuple[torch.Tensor, torch.Tensor] | None = None,
+    ) -> torch.Tensor:
+        seq_length = hidden_states.shape[0]
+        q, k, v = (
+            self.qkv(hidden_states).reshape(seq_length, 3, self.num_heads, -1).permute(1, 0, 2, 3).unbind(0)
+        )
+        if position_embeddings is None:
+            logger.warning_once(
+                "The attention layers in this model are transitioning from computing the RoPE embeddings internally "
+                "through `rotary_pos_emb` (2D tensor of RoPE theta values), to using externally computed "
+                "`position_embeddings` (Tuple of tensors, containing cos and sin). In v4.54 `rotary_pos_emb` will be "
+                "removed and `position_embeddings` will be mandatory."
+            )
+            emb = torch.cat((rotary_pos_emb, rotary_pos_emb), dim=-1)
+            cos = emb.cos().float()
+            sin = emb.sin().float()
+        else:
+            cos, sin = position_embeddings
+        q, k = apply_rotary_pos_emb_vision(q, k, cos, sin)
+
+        attention_mask = torch.full(
+            [1, seq_length, seq_length],
+            torch.finfo(q.dtype).min,
+            device=q.device,
+            dtype=q.dtype,
+        )
+        for i in range(1, len(cu_seqlens)):
+            attention_mask[
+                ...,
+                cu_seqlens[i - 1] : cu_seqlens[i],
+                cu_seqlens[i - 1] : cu_seqlens[i],
+            ] = 0
+
+        q = q.transpose(0, 1)
+        k = k.transpose(0, 1)
+        v = v.transpose(0, 1)
+        attn_weights = torch.matmul(q, k.transpose(1, 2)) / math.sqrt(self.head_dim)
+        attn_weights = attn_weights + attention_mask
+        attn_weights = nn.functional.softmax(attn_weights, dim=-1, dtype=torch.float32).to(q.dtype)
+        attn_output = torch.matmul(attn_weights, v)
+        attn_output = attn_output.transpose(0, 1)
+        attn_output = attn_output.reshape(seq_length, -1)
+        attn_output = self.proj(attn_output)
+        return attn_output
+
+
+class Qwen2_5_VLVisionSdpaAttention(nn.Module):
+    def __init__(self, dim: int, num_heads: int = 16) -> None:
+        super().__init__()
+        self.num_heads = num_heads
+        self.qkv = nn.Linear(dim, dim * 3, bias=True)
+        self.proj = nn.Linear(dim, dim)
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        cu_seqlens: torch.Tensor,
+        max_seqlen: int | None = None,
+        rotary_pos_emb: torch.Tensor | None = None,
+        position_embeddings: tuple[torch.Tensor, torch.Tensor] | None = None,
+    ) -> torch.Tensor:
+        seq_length = hidden_states.shape[0]
+        q, k, v = (
+            self.qkv(hidden_states).reshape(seq_length, 3, self.num_heads, -1).permute(1, 0, 2, 3).unbind(0)
+        )
+        if position_embeddings is None:
+            logger.warning_once(
+                "The attention layers in this model are transitioning from computing the RoPE embeddings internally "
+                "through `rotary_pos_emb` (2D tensor of RoPE theta values), to using externally computed "
+                "`position_embeddings` (Tuple of tensors, containing cos and sin). In v4.54 `rotary_pos_emb` will be "
+                "removed and `position_embeddings` will be mandatory."
+            )
+            emb = torch.cat((rotary_pos_emb, rotary_pos_emb), dim=-1)
+            cos = emb.cos().float()
+            sin = emb.sin().float()
+        else:
+            cos, sin = position_embeddings
+        q, k = apply_rotary_pos_emb_vision(q, k, cos, sin)
+
+        attention_mask = torch.zeros([1, seq_length, seq_length], device=q.device, dtype=torch.bool)
+        for i in range(1, len(cu_seqlens)):
+            attention_mask[
+                ...,
+                cu_seqlens[i - 1] : cu_seqlens[i],
+                cu_seqlens[i - 1] : cu_seqlens[i],
+            ] = True
+        q = q.transpose(0, 1)
+        k = k.transpose(0, 1)
+        v = v.transpose(0, 1)
+        attn_output = F.scaled_dot_product_attention(q, k, v, attention_mask, dropout_p=0.0)
+        attn_output = attn_output.transpose(0, 1)
+        attn_output = attn_output.reshape(seq_length, -1)
+        attn_output = self.proj(attn_output)
+        return attn_output
+
+
+QWEN2_5_VL_VISION_ATTENTION_CLASSES = {
+    "eager": Qwen2_5_VLVisionAttention,
+    "flash_attention_2": Qwen2_5_VLVisionFlashAttention2,
+    "sdpa": Qwen2_5_VLVisionSdpaAttention,
+}
+
+
+class Qwen2_5_VLVisionBlock(nn.Module):
+    def __init__(self, config, attn_implementation: str = "sdpa") -> None:
+        super().__init__()
+        self.norm1 = Qwen2RMSNorm(config.hidden_size, eps=1e-6)
+        self.norm2 = Qwen2RMSNorm(config.hidden_size, eps=1e-6)
+        self.attn = QWEN2_5_VL_VISION_ATTENTION_CLASSES[attn_implementation](
+            config.hidden_size, num_heads=config.num_heads
+        )
+        self.mlp = Qwen2_5_VLMLP(config, bias=True)
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        cu_seqlens: torch.Tensor,
+        max_seqlen: int | None = None,
+        rotary_pos_emb: torch.Tensor | None = None,
+        position_embeddings: tuple[torch.Tensor, torch.Tensor] | None = None,
+    ) -> torch.Tensor:
+        hidden_states = hidden_states + self.attn(
+            self.norm1(hidden_states),
+            cu_seqlens=cu_seqlens,
+            max_seqlen=max_seqlen,
+            rotary_pos_emb=rotary_pos_emb,
+            position_embeddings=position_embeddings,
+        )
+        hidden_states = hidden_states + self.mlp(self.norm2(hidden_states))
+        return hidden_states
+
+
+Qwen2_5_VL_START_DOCSTRING = r"""
+    This model inherits from [`PreTrainedModel`]. Check the superclass documentation for the generic methods the
+    library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads
+    etc.)
+
+    This model is also a PyTorch [torch.nn.Module](https://pytorch.org/docs/stable/nn.html#torch.nn.Module) subclass.
+    Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage
+    and behavior.
+
+    Parameters:
+        config ([`Qwen2_5_VLConfig`]):
+            Model configuration class with all the parameters of the model. Initializing with a config file does not
+            load the weights associated with the model, only the configuration. Check out the
+            [`~PreTrainedModel.from_pretrained`] method to load the model weights.
+"""
+
+
+@add_start_docstrings(
+    "The bare Qwen2_5_VL Model outputting raw hidden-states without any specific head on top.",
+    Qwen2_5_VL_START_DOCSTRING,
+)
+class Qwen2_5_VLPreTrainedModel(PreTrainedModel):
+    config_class = Qwen2_5_VLConfig
+    base_model_prefix = "model"
+    supports_gradient_checkpointing = True
+    _no_split_modules = ["Qwen2_5_VLDecoderLayer", "Qwen2_5_VLVisionBlock"]
+    _skip_keys_device_placement = "past_key_values"
+    _supports_flash_attn_2 = True
+    _supports_sdpa = True
+    _supports_cache_class = True
+    _supports_static_cache = (
+        False  # TODO (joao): fix. torch.compile failing probably due to `cache_positions`
+    )
+
+    def _init_weights(self, module):
+        std = self.config.initializer_range
+        if isinstance(module, (nn.Linear, nn.Conv3d)):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.bias is not None:
+                module.bias.data.zero_()
+        elif isinstance(module, nn.Embedding):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.padding_idx is not None:
+                module.weight.data[module.padding_idx].zero_()
+
+
+class Qwen2_5_VisionTransformerPretrainedModel(Qwen2_5_VLPreTrainedModel):
+    config_class = Qwen2_5_VLVisionConfig
+    _no_split_modules = ["Qwen2_5_VLVisionBlock"]
+
+    def __init__(self, config, *inputs, **kwargs) -> None:
+        super().__init__(config, *inputs, **kwargs)
+        self.spatial_merge_size = config.spatial_merge_size
+        self.patch_size = config.patch_size
+        self.fullatt_block_indexes = config.fullatt_block_indexes
+        self.window_size = config.window_size
+        self.spatial_merge_unit = self.spatial_merge_size * self.spatial_merge_size
+
+        self.patch_embed = Qwen2_5_VisionPatchEmbed(
+            patch_size=config.patch_size,
+            temporal_patch_size=config.temporal_patch_size,
+            in_channels=config.in_channels,
+            embed_dim=config.hidden_size,
+        )
+
+        head_dim = config.hidden_size // config.num_heads
+        self.rotary_pos_emb = Qwen2_5_VisionRotaryEmbedding(head_dim // 2)
+
+        self.blocks = nn.ModuleList(
+            [Qwen2_5_VLVisionBlock(config, config._attn_implementation) for _ in range(config.depth)]
+        )
+        self.merger = Qwen2_5_VLPatchMerger(
+            dim=config.out_hidden_size,
+            context_dim=config.hidden_size,
+            spatial_merge_size=config.spatial_merge_size,
+        )
+        self.gradient_checkpointing = False
+
+    def rot_pos_emb(self, grid_thw):
+        pos_ids = []
+        for t, h, w in grid_thw:
+            hpos_ids = torch.arange(h).unsqueeze(1).expand(-1, w)
+            hpos_ids = hpos_ids.reshape(
+                h // self.spatial_merge_size,
+                self.spatial_merge_size,
+                w // self.spatial_merge_size,
+                self.spatial_merge_size,
+            )
+            hpos_ids = hpos_ids.permute(0, 2, 1, 3)
+            hpos_ids = hpos_ids.flatten()
+
+            wpos_ids = torch.arange(w).unsqueeze(0).expand(h, -1)
+            wpos_ids = wpos_ids.reshape(
+                h // self.spatial_merge_size,
+                self.spatial_merge_size,
+                w // self.spatial_merge_size,
+                self.spatial_merge_size,
+            )
+            wpos_ids = wpos_ids.permute(0, 2, 1, 3)
+            wpos_ids = wpos_ids.flatten()
+            pos_ids.append(torch.stack([hpos_ids, wpos_ids], dim=-1).repeat(t, 1))
+        pos_ids = torch.cat(pos_ids, dim=0)
+        max_grid_size = grid_thw[:, 1:].max()
+        rotary_pos_emb_full = self.rotary_pos_emb(max_grid_size)
+        rotary_pos_emb = rotary_pos_emb_full[pos_ids].flatten(1)
+        return rotary_pos_emb
+
+    def get_window_index(self, grid_thw):
+        window_index: list = []
+        cu_window_seqlens: list = [0]
+        window_index_id = 0
+        vit_merger_window_size = self.window_size // self.spatial_merge_size // self.patch_size
+
+        for grid_t, grid_h, grid_w in grid_thw:
+            llm_grid_h, llm_grid_w = (
+                grid_h // self.spatial_merge_size,
+                grid_w // self.spatial_merge_size,
+            )
+            index = torch.arange(grid_t * llm_grid_h * llm_grid_w).reshape(grid_t, llm_grid_h, llm_grid_w)
+            pad_h = vit_merger_window_size - llm_grid_h % vit_merger_window_size
+            pad_w = vit_merger_window_size - llm_grid_w % vit_merger_window_size
+            num_windows_h = (llm_grid_h + pad_h) // vit_merger_window_size
+            num_windows_w = (llm_grid_w + pad_w) // vit_merger_window_size
+            index_padded = F.pad(index, (0, pad_w, 0, pad_h), "constant", -100)
+            index_padded = index_padded.reshape(
+                grid_t,
+                num_windows_h,
+                vit_merger_window_size,
+                num_windows_w,
+                vit_merger_window_size,
+            )
+            index_padded = index_padded.permute(0, 1, 3, 2, 4).reshape(
+                grid_t,
+                num_windows_h * num_windows_w,
+                vit_merger_window_size,
+                vit_merger_window_size,
+            )
+            seqlens = (index_padded != -100).sum([2, 3]).reshape(-1)
+            index_padded = index_padded.reshape(-1)
+            index_new = index_padded[index_padded != -100]
+            window_index.append(index_new + window_index_id)
+            cu_seqlens_tmp = seqlens.cumsum(0) * self.spatial_merge_unit + cu_window_seqlens[-1]
+            cu_window_seqlens.extend(cu_seqlens_tmp.tolist())
+            window_index_id += (grid_t * llm_grid_h * llm_grid_w).item()
+        window_index = torch.cat(window_index, dim=0)
+
+        return window_index, cu_window_seqlens
+
+    def forward(self, hidden_states: torch.Tensor, grid_thw: torch.Tensor) -> torch.Tensor:
+        """
+        Args:
+            hidden_states (`torch.Tensor` of shape `(seq_len, hidden_size)`):
+                The final hidden states of the model.
+            grid_thw (`torch.Tensor` of shape `(num_images_or_videos, 3)`):
+                The temporal, height and width of feature shape of each image in LLM.
+
+        Returns:
+            `torch.Tensor`: hidden_states.
+        """
+        hidden_states = self.patch_embed(hidden_states)
+        rotary_pos_emb = self.rot_pos_emb(grid_thw)
+        window_index, cu_window_seqlens = self.get_window_index(grid_thw)
+        window_index = window_index.to(hidden_states.device)
+        cu_window_seqlens = torch.tensor(
+            cu_window_seqlens,
+            device=hidden_states.device,
+            dtype=grid_thw.dtype if torch.jit.is_tracing() else torch.int32,
+        )
+        cu_window_seqlens = torch.unique_consecutive(cu_window_seqlens)
+
+        seq_len, _ = hidden_states.size()
+        hidden_states = hidden_states.reshape(seq_len // self.spatial_merge_unit, self.spatial_merge_unit, -1)
+        hidden_states = hidden_states[window_index, :, :]
+        hidden_states = hidden_states.reshape(seq_len, -1)
+        rotary_pos_emb = rotary_pos_emb.reshape(
+            seq_len // self.spatial_merge_unit, self.spatial_merge_unit, -1
+        )
+        rotary_pos_emb = rotary_pos_emb[window_index, :, :]
+        rotary_pos_emb = rotary_pos_emb.reshape(seq_len, -1)
+        emb = torch.cat((rotary_pos_emb, rotary_pos_emb), dim=-1)
+        position_embeddings = (emb.cos(), emb.sin())
+
+        cu_seqlens = torch.repeat_interleave(grid_thw[:, 1] * grid_thw[:, 2], grid_thw[:, 0]).cumsum(
+            dim=0,
+            # Select dtype based on the following factors:
+            #  - FA2 requires that cu_seqlens_q must have dtype int32
+            #  - torch.onnx.export requires that cu_seqlens_q must have same dtype as grid_thw
+            # See https://github.com/huggingface/transformers/pull/34852 for more information
+            dtype=grid_thw.dtype if torch.jit.is_tracing() else torch.int32,
+        )
+        cu_seqlens = F.pad(cu_seqlens, (1, 0), value=0)
+        max_seqlen_full = (cu_seqlens[1:] - cu_seqlens[:-1]).max().item()
+        max_seqlen_window = (cu_window_seqlens[1:] - cu_window_seqlens[:-1]).max().item()
+
+        for layer_num, blk in enumerate(self.blocks):
+            if layer_num in self.fullatt_block_indexes:
+                cu_seqlens_now = cu_seqlens
+                max_seqlen_now = max_seqlen_full
+            else:
+                cu_seqlens_now = cu_window_seqlens
+                max_seqlen_now = max_seqlen_window
+            if self.gradient_checkpointing and self.training:
+                hidden_states = self._gradient_checkpointing_func(
+                    blk.__call__,
+                    hidden_states,
+                    cu_seqlens_now,
+                    None,
+                    position_embeddings,
+                )
+            else:
+                hidden_states = blk(
+                    hidden_states,
+                    cu_seqlens=cu_seqlens_now,
+                    max_seqlen=max_seqlen_now,
+                    position_embeddings=position_embeddings,
+                )
+
+        hidden_states = self.merger(hidden_states)
+        reverse_indices = torch.argsort(window_index)
+        hidden_states = hidden_states[reverse_indices, :]
+
+        return hidden_states
+
+
+def _compute_default_rope_parameters_qwen2_5_vl(config, device=None):
+    """
+    compute default rope parameters for Qwen2_5_VL
+    """
+    base = config.text_config.rope_parameters["rope_theta"]
+    dim = config.hidden_size // config.num_attention_heads
+    inv_freq = 1.0 / (
+        base ** (torch.arange(0, dim, 2, dtype=torch.int64).to(device=device, dtype=torch.float) / dim)
+    )
+    return inv_freq, 1.0
+
+
+class Qwen2_5_VLRotaryEmbedding(nn.Module):
+    def __init__(self, config: Qwen2_5_VLConfig, device=None):
+        super().__init__()
+        # BC: "rope_type" was originally "type"
+        if hasattr(config, "rope_scaling") and config.rope_scaling is not None:
+            self.rope_type = config.rope_scaling.get("rope_type", config.rope_scaling.get("type"))
+        elif hasattr(config, "rope_parameters") and config.rope_parameters is not None:
+            self.rope_type = config.rope_parameters.get("rope_type", "default")
+        else:
+            self.rope_type = "default"
+        self.max_seq_len_cached = config.max_position_embeddings
+        self.original_max_seq_len = config.max_position_embeddings
+
+        self.config = config
+
+        if self.rope_type == "default":
+            self.rope_init_fn = _compute_default_rope_parameters_qwen2_5_vl
+            self.rope_kwargs = {}
+        else:
+            rope_type_key = "linear" if self.rope_type == "linear" else self.rope_type
+            self.rope_init_fn = ROPE_INIT_FUNCTIONS[rope_type_key]
+            self.rope_kwargs = {}
+
+        inv_freq, self.attention_scaling = self.rope_init_fn(self.config, device)
+        self.register_buffer("inv_freq", inv_freq, persistent=False)
+        self.original_inv_freq = self.inv_freq
+
+    def _dynamic_frequency_update(self, position_ids, device):
+        """
+        dynamic RoPE layers should recompute `inv_freq` in the following situations:
+        1 - growing beyond the cached sequence length (allow scaling)
+        2 - the current sequence length is in the original scale (avoid losing precision with small sequences)
+        """
+        seq_len = torch.max(position_ids) + 1
+        if seq_len > self.max_seq_len_cached:  # growth
+            inv_freq, self.attention_scaling = self.rope_init_fn(
+                self.config, device, seq_len=seq_len, **self.rope_kwargs
+            )
+            self.register_buffer(
+                "inv_freq", inv_freq, persistent=False
+            )  # TODO joao: may break with compilation
+            self.max_seq_len_cached = seq_len
+
+        if (
+            seq_len < self.original_max_seq_len and self.max_seq_len_cached > self.original_max_seq_len
+        ):  # reset
+            self.register_buffer("inv_freq", self.original_inv_freq, persistent=False)
+            self.max_seq_len_cached = self.original_max_seq_len
+
+    @torch.no_grad()
+    def forward(self, x, position_ids):
+        if "dynamic" in self.rope_type:
+            self._dynamic_frequency_update(position_ids, device=x.device)
+
+        # Core RoPE block. In contrast to other models, Qwen2_5_VL has different position ids for thw grids
+        # So we expand the inv_freq to shape (3, ...)
+        inv_freq_expanded = self.inv_freq[None, None, :, None].float().expand(3, position_ids.shape[1], -1, 1)
+        position_ids_expanded = position_ids[:, :, None, :].float()  # shape (3, bs, 1, positions)
+        # Force float32 (see https://github.com/huggingface/transformers/pull/29285)
+        device_type = x.device.type
+        device_type = device_type if isinstance(device_type, str) and device_type != "mps" else "cpu"
+        with torch.autocast(device_type=device_type, enabled=False):
+            freqs = (inv_freq_expanded.float() @ position_ids_expanded.float()).transpose(2, 3)
+            emb = torch.cat((freqs, freqs), dim=-1)
+            cos = emb.cos()
+            sin = emb.sin()
+
+        # Advanced RoPE types (e.g. yarn) apply a post-processing scaling factor, equivalent to scaling attention
+        cos = cos * self.attention_scaling
+        sin = sin * self.attention_scaling
+
+        return cos.to(dtype=x.dtype), sin.to(dtype=x.dtype)
+
+
+class Qwen2MLP(nn.Module):
+    def __init__(self, config):
+        super().__init__()
+        self.config = config
+        self.hidden_size = config.hidden_size
+        self.intermediate_size = config.intermediate_size
+        self.gate_proj = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
+        self.up_proj = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
+        self.down_proj = nn.Linear(self.intermediate_size, self.hidden_size, bias=False)
+        self.act_fn = ACT2FN[config.hidden_act]
+
+    def forward(self, x):
+        down_proj = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
+        return down_proj
+
+
+def apply_multimodal_rotary_pos_emb(q, k, cos, sin, mrope_section, unsqueeze_dim=1):
+    """Applies Rotary Position Embedding with Multimodal Sections to the query and key tensors (https://qwenlm.github.io/blog/qwen2-vl/).
+
+    Explanation:
+        Multimodal 3D rotary position embedding is an extension to 1D rotary position embedding. The input embedding
+        sequence contains vision (images / videos) embedding and text embedding or just contains text embedding. For
+        vision embedding part, we apply rotary position embedding on temporal, height and width dimension separately.
+        Here we split the channel dimension to 3 chunks for the temporal, height and width rotary position embedding.
+        For text embedding part, we just apply 1D rotary position embedding. The three rotary position index (temporal,
+        height and width) of text embedding is always the same, so the text embedding rotary position embedding has no
+        difference with modern LLMs.
+
+    Args:
+        q (`torch.Tensor`): The query tensor.
+        k (`torch.Tensor`): The key tensor.
+        cos (`torch.Tensor`): The cosine part of the rotary embedding.
+        sin (`torch.Tensor`): The sine part of the rotary embedding.
+        position_ids (`torch.Tensor`):
+            The position indices of the tokens corresponding to the query and key tensors. For example, this can be
+            used to pass offsetted position ids when working with a KV-cache.
+        mrope_section(`List(int)`):
+            Multimodal rope section is for channel dimension of temporal, height and width in rope calculation.
+        unsqueeze_dim (`int`, *optional*, defaults to 1):
+            The 'unsqueeze_dim' argument specifies the dimension along which to unsqueeze cos[position_ids] and
+            sin[position_ids] so that they can be properly broadcasted to the dimensions of q and k. For example, note
+            that cos[position_ids] and sin[position_ids] have the shape [batch_size, seq_len, head_dim]. Then, if q and
+            k have the shape [batch_size, heads, seq_len, head_dim], then setting unsqueeze_dim=1 makes
+            cos[position_ids] and sin[position_ids] broadcastable to the shapes of q and k. Similarly, if q and k have
+            the shape [batch_size, seq_len, heads, head_dim], then set unsqueeze_dim=2.
+    Returns:
+        `tuple(torch.Tensor)` comprising of the query and key tensors rotated using the Rotary Position Embedding.
+    """
+    mrope_section = mrope_section * 2
+    cos = torch.cat([m[i % 3] for i, m in enumerate(cos.split(mrope_section, dim=-1))], dim=-1).unsqueeze(
+        unsqueeze_dim
+    )
+    sin = torch.cat([m[i % 3] for i, m in enumerate(sin.split(mrope_section, dim=-1))], dim=-1).unsqueeze(
+        unsqueeze_dim
+    )
+
+    q_embed = (q * cos) + (rotate_half(q) * sin)
+    k_embed = (k * cos) + (rotate_half(k) * sin)
+    return q_embed, k_embed
+
+
+def repeat_kv(hidden_states: torch.Tensor, n_rep: int) -> torch.Tensor:
+    """
+    This is the equivalent of torch.repeat_interleave(x, dim=1, repeats=n_rep). The hidden states go from (batch,
+    num_key_value_heads, seqlen, head_dim) to (batch, num_attention_heads, seqlen, head_dim)
+    """
+    batch, num_key_value_heads, slen, head_dim = hidden_states.shape
+    if n_rep == 1:
+        return hidden_states
+    hidden_states = hidden_states[:, :, None, :, :].expand(batch, num_key_value_heads, n_rep, slen, head_dim)
+    return hidden_states.reshape(batch, num_key_value_heads * n_rep, slen, head_dim)
+
+
+class Qwen2_5_VLAttention(nn.Module):
+    """
+    Multi-headed attention from 'Attention Is All You Need' paper. Modified to use sliding window attention: Longformer
+    and "Generating Long Sequences with Sparse Transformers".
+    """
+
+    def __init__(self, config: Qwen2_5_VLConfig, layer_idx: int | None = None):
+        super().__init__()
+        self.config = config
+        self.layer_idx = layer_idx
+        if layer_idx is None:
+            logger.warning_once(
+                f"Instantiating {self.__class__.__name__} without passing `layer_idx` is not recommended and will "
+                "to errors during the forward call, if caching is used. Please make sure to provide a `layer_idx` "
+                "when creating this class."
+            )
+
+        self.hidden_size = config.hidden_size
+        self.num_heads = config.num_attention_heads
+        self.head_dim = self.hidden_size // self.num_heads
+        self.num_key_value_heads = config.num_key_value_heads
+        self.num_key_value_groups = self.num_heads // self.num_key_value_heads
+        self.is_causal = True
+        self.attention_dropout = config.attention_dropout
+        self.rope_scaling = config.rope_scaling
+
+        if (self.head_dim * self.num_heads) != self.hidden_size:
+            raise ValueError(
+                f"hidden_size must be divisible by num_heads (got `hidden_size`: {self.hidden_size}"
+                f" and `num_heads`: {self.num_heads})."
+            )
+        self.q_proj = nn.Linear(self.hidden_size, self.num_heads * self.head_dim, bias=True)
+        self.k_proj = nn.Linear(self.hidden_size, self.num_key_value_heads * self.head_dim, bias=True)
+        self.v_proj = nn.Linear(self.hidden_size, self.num_key_value_heads * self.head_dim, bias=True)
+        self.o_proj = nn.Linear(self.num_heads * self.head_dim, self.hidden_size, bias=False)
+
+        self.rotary_emb = Qwen2_5_VLRotaryEmbedding(config=config)
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_value: Cache | None = None,
+        output_attentions: bool = False,
+        use_cache: bool = False,
+        cache_position: torch.LongTensor | None = None,
+        position_embeddings: tuple[torch.Tensor, torch.Tensor]
+        | None = None,  # necessary, but kept here for BC
+    ) -> tuple[torch.Tensor, torch.Tensor | None, tuple[torch.Tensor] | None]:
+        bsz, q_len, _ = hidden_states.size()
+
+        query_states = self.q_proj(hidden_states)
+        key_states = self.k_proj(hidden_states)
+        value_states = self.v_proj(hidden_states)
+
+        query_states = query_states.view(bsz, q_len, -1, self.head_dim).transpose(1, 2)
+        key_states = key_states.view(bsz, q_len, -1, self.head_dim).transpose(1, 2)
+        value_states = value_states.view(bsz, q_len, -1, self.head_dim).transpose(1, 2)
+
+        cos, sin = position_embeddings
+        query_states, key_states = apply_multimodal_rotary_pos_emb(
+            query_states, key_states, cos, sin, self.rope_scaling["mrope_section"]
+        )
+
+        if past_key_value is not None:
+            cache_kwargs = {
+                "sin": sin,
+                "cos": cos,
+                "cache_position": cache_position,
+            }  # Specific to RoPE models
+            key_states, value_states = past_key_value.update(
+                key_states, value_states, self.layer_idx, cache_kwargs
+            )
+
+        # repeat k/v heads if n_kv_heads < n_heads
+        key_states = repeat_kv(key_states, self.num_key_value_groups)
+        value_states = repeat_kv(value_states, self.num_key_value_groups)
+
+        attn_weights = torch.matmul(query_states, key_states.transpose(2, 3)) / math.sqrt(self.head_dim)
+
+        if attention_mask is not None:  # no matter the length, we just slice it
+            causal_mask = attention_mask[:, :, :, : key_states.shape[-2]]
+            attn_weights = attn_weights + causal_mask
+
+        # Fix precision issues in Qwen2-VL float16 inference
+        # Replace inf values with zeros in attention weights to prevent NaN propagation
+        if query_states.dtype == torch.float16:
+            attn_weights = torch.where(
+                torch.isinf(attn_weights), torch.zeros_like(attn_weights), attn_weights
+            )
+
+        # upcast attention to fp32
+        attn_weights = nn.functional.softmax(attn_weights, dim=-1, dtype=torch.float32).to(query_states.dtype)
+        attn_weights = nn.functional.dropout(attn_weights, p=self.attention_dropout, training=self.training)
+        attn_output = torch.matmul(attn_weights, value_states)
+
+        if attn_output.size() != (bsz, self.num_heads, q_len, self.head_dim):
+            raise ValueError(
+                f"`attn_output` should be of size {(bsz, self.num_heads, q_len, self.head_dim)}, but is"
+                f" {attn_output.size()}"
+            )
+
+        attn_output = attn_output.transpose(1, 2).contiguous()
+        attn_output = attn_output.reshape(bsz, q_len, -1)
+
+        attn_output = self.o_proj(attn_output)
+
+        if not output_attentions:
+            attn_weights = None
+
+        return attn_output, attn_weights, past_key_value
+
+
+class Qwen2_5_VLFlashAttention2(Qwen2_5_VLAttention):
+    """
+    Qwen2_5_VL flash attention module, following Qwen2_5_VL attention module. This module inherits from `Qwen2_5_VLAttention`
+    as the weights of the module stays untouched. The only required change would be on the forward pass
+    where it needs to correctly call the public API of flash attention and deal with padding tokens
+    in case the input contains any of them. Additionally, for sliding window attention, we apply SWA only to the bottom
+    config.max_window_layers layers.
+    """
+
+    def __init__(self, *args, **kwargs):
+        super().__init__(*args, **kwargs)
+
+        # TODO: Should be removed once Flash Attention for RoCm is bumped to 2.1.
+        # flash_attn<2.1 generates top-left aligned causal mask, while what is needed here is bottom-right alignment, that was made default for flash_attn>=2.1. This attribute is used to handle this difference. Reference: https://github.com/Dao-AILab/flash-attention/releases/tag/v2.1.0.
+        # Beware that with flash_attn<2.1, using q_seqlen != k_seqlen (except for the case q_seqlen == 1) produces a wrong mask (top-left).
+        self._flash_attn_uses_top_left_mask = not is_flash_attn_greater_or_equal_2_10()
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_value: Cache | None = None,
+        output_attentions: bool = False,
+        use_cache: bool = False,
+        cache_position: torch.LongTensor | None = None,
+        position_embeddings: tuple[torch.Tensor, torch.Tensor]
+        | None = None,  # necessary, but kept here for BC
+    ):
+        bsz, q_len, _ = hidden_states.size()
+        query_states = self.q_proj(hidden_states)
+        key_states = self.k_proj(hidden_states)
+        value_states = self.v_proj(hidden_states)
+
+        query_states = query_states.view(bsz, q_len, -1, self.head_dim).transpose(1, 2)
+        key_states = key_states.view(bsz, q_len, -1, self.head_dim).transpose(1, 2)
+        value_states = value_states.view(bsz, q_len, -1, self.head_dim).transpose(1, 2)
+
+        # Because the input can be padded, the absolute sequence length depends on the max position id.
+        cos, sin = position_embeddings
+        query_states, key_states = apply_multimodal_rotary_pos_emb(
+            query_states, key_states, cos, sin, self.rope_scaling["mrope_section"]
+        )
+        if past_key_value is not None:
+            cache_kwargs = {
+                "sin": sin,
+                "cos": cos,
+                "cache_position": cache_position,
+            }  # Specific to RoPE models
+            key_states, value_states = past_key_value.update(
+                key_states, value_states, self.layer_idx, cache_kwargs
+            )
+
+        # repeat k/v heads if n_kv_heads < n_heads
+        # key_states = repeat_kv(key_states, self.num_key_value_groups)
+        # value_states = repeat_kv(value_states, self.num_key_value_groups)
+        dropout_rate = 0.0 if not self.training else self.attention_dropout
+
+        # In PEFT, usually we cast the layer norms in float32 for training stability reasons
+        # therefore the input hidden states gets silently casted in float32. Hence, we need
+        # cast them back in float16 just to be sure everything works as expected.
+        input_dtype = query_states.dtype
+        if input_dtype == torch.float32:
+            if torch.is_autocast_enabled():
+                target_dtype = torch.get_autocast_gpu_dtype()
+            # Handle the case where the model is quantized
+            elif hasattr(self.config, "_pre_quantization_dtype"):
+                target_dtype = self.config._pre_quantization_dtype
+            else:
+                target_dtype = self.q_proj.weight.dtype
+
+            logger.warning_once(
+                f"The input hidden states seems to be silently casted in float32, this might be related to"
+                f" the fact you have upcasted embedding or layer norm layers in float32. We will cast back the input in"
+                f" {target_dtype}."
+            )
+
+            query_states = query_states.to(target_dtype)
+            key_states = key_states.to(target_dtype)
+            value_states = value_states.to(target_dtype)
+
+        # Reashape to the expected shape for Flash Attention
+        query_states = query_states.transpose(1, 2)
+        key_states = key_states.transpose(1, 2)
+        value_states = value_states.transpose(1, 2)
+
+        attn_output = flash_attn_func(
+            query_states,
+            key_states,
+            value_states,
+            dropout_rate,
+            softmax_scale=None,
+            causal=self.is_causal,
+        )
+
+        attn_output = attn_output.reshape(bsz, q_len, self.hidden_size).contiguous()
+        attn_output = self.o_proj(attn_output)
+
+        if not output_attentions:
+            attn_weights = None
+
+        return attn_output, attn_weights, past_key_value
+
+
+class Qwen2_5_VLSdpaAttention(Qwen2_5_VLAttention):
+    """
+    Qwen2 attention module using torch.nn.functional.scaled_dot_product_attention. This module inherits from
+    `Qwen2Attention` as the weights of the module stays untouched. The only changes are on the forward pass to adapt to
+    SDPA API.
+    """
+
+    # Adapted from Qwen2Attention.forward
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_value: Cache | None = None,
+        output_attentions: bool = False,
+        use_cache: bool = False,
+        cache_position: torch.LongTensor | None = None,
+        position_embeddings: tuple[torch.Tensor, torch.Tensor]
+        | None = None,  # necessary, but kept here for BC
+    ) -> tuple[torch.Tensor, torch.Tensor | None, tuple[torch.Tensor] | None]:
+        if output_attentions:
+            # TODO: Improve this warning with e.g. `model.config.attn_implementation = "manual"` once this is implemented.
+            logger.warning_once(
+                "Qwen2_5_VLModel is using Qwen2_5_VLSdpaAttention, but `torch.nn.functional.scaled_dot_product_attention` does not support `output_attentions=True`. Falling back to the manual attention implementation, "
+                'but specifying the manual implementation will be required from Transformers version v5.0.0 onwards. This warning can be removed using the argument `attn_implementation="eager"` when loading the model.'
+            )
+            return super().forward(
+                hidden_states=hidden_states,
+                attention_mask=attention_mask,
+                position_ids=position_ids,
+                past_key_value=past_key_value,
+                output_attentions=output_attentions,
+                use_cache=use_cache,
+                cache_position=cache_position,
+                position_embeddings=position_embeddings,
+            )
+
+        bsz, q_len, _ = hidden_states.size()
+
+        query_states = self.q_proj(hidden_states)
+        key_states = self.k_proj(hidden_states)
+        value_states = self.v_proj(hidden_states)
+
+        query_states = query_states.view(bsz, q_len, -1, self.head_dim).transpose(1, 2)
+        key_states = key_states.view(bsz, q_len, -1, self.head_dim).transpose(1, 2)
+        value_states = value_states.view(bsz, q_len, -1, self.head_dim).transpose(1, 2)
+
+        cos, sin = position_embeddings
+        query_states, key_states = apply_multimodal_rotary_pos_emb(
+            query_states, key_states, cos, sin, self.rope_scaling["mrope_section"]
+        )
+
+        if past_key_value is not None:
+            cache_kwargs = {
+                "sin": sin,
+                "cos": cos,
+                "cache_position": cache_position,
+            }  # Specific to RoPE models
+            key_states, value_states = past_key_value.update(
+                key_states, value_states, self.layer_idx, cache_kwargs
+            )
+
+        key_states = repeat_kv(key_states, self.num_key_value_groups)
+        value_states = repeat_kv(value_states, self.num_key_value_groups)
+
+        causal_mask = attention_mask
+        if attention_mask is not None:  # no matter the length, we just slice it
+            causal_mask = attention_mask[:, :, :, : key_states.shape[-2]]
+
+        # SDPA with memory-efficient backend is currently (torch==2.1.2) bugged with non-contiguous inputs with custom attn_mask,
+        # Reference: https://github.com/pytorch/pytorch/issues/112577.
+        if query_states.device.type == "cuda" and attention_mask is not None:
+            query_states = query_states.contiguous()
+            key_states = key_states.contiguous()
+            value_states = value_states.contiguous()
+
+        # We dispatch to SDPA's Flash Attention or Efficient kernels via this `is_causal` if statement instead of an inline conditional assignment
+        # in SDPA to support both torch.compile's dynamic shapes and full graph options. An inline conditional prevents dynamic shapes from compiling.
+        # The q_len > 1 is necessary to match with AttentionMaskConverter.to_causal_4d that does not create a causal mask in case q_len == 1.
+        is_causal = True if causal_mask is None and q_len > 1 else False
+
+        attn_output = torch.nn.functional.scaled_dot_product_attention(
+            query_states,
+            key_states,
+            value_states,
+            attn_mask=causal_mask,
+            dropout_p=self.attention_dropout if self.training else 0.0,
+            is_causal=is_causal,
+        )
+
+        attn_output = attn_output.transpose(1, 2).contiguous()
+        attn_output = attn_output.view(bsz, q_len, self.hidden_size)
+
+        attn_output = self.o_proj(attn_output)
+
+        return attn_output, None, past_key_value
+
+
+QWEN2_5_VL_ATTENTION_CLASSES = {
+    "eager": Qwen2_5_VLAttention,
+    "flash_attention_2": Qwen2_5_VLFlashAttention2,
+    "sdpa": Qwen2_5_VLSdpaAttention,
+}
+
+
+class Qwen2_5_VLDecoderLayer(nn.Module):
+    def __init__(self, config: Qwen2_5_VLConfig, layer_idx: int):
+        super().__init__()
+        self.hidden_size = config.hidden_size
+
+        if config.use_sliding_window and config._attn_implementation != "flash_attention_2":
+            logger.warning_once(
+                f"Sliding Window Attention is enabled but not implemented for `{config._attn_implementation}`; "
+                "unexpected results may be encountered."
+            )
+        self.self_attn = QWEN2_5_VL_ATTENTION_CLASSES[config._attn_implementation](config, layer_idx)
+
+        self.mlp = Qwen2MLP(config)
+        self.input_layernorm = Qwen2RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+        self.post_attention_layernorm = Qwen2RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_value: tuple[torch.Tensor] | None = None,
+        output_attentions: bool | None = False,
+        use_cache: bool | None = False,
+        cache_position: torch.LongTensor | None = None,
+        position_embeddings: tuple[torch.Tensor, torch.Tensor]
+        | None = None,  # necessary, but kept here for BC
+        **kwargs,
+    ) -> tuple[torch.FloatTensor, tuple[torch.FloatTensor, torch.FloatTensor] | None]:
+        """
+        Args:
+            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
+            attention_mask (`torch.FloatTensor`, *optional*): attention mask of size
+                `(batch, sequence_length)` where padding elements are indicated by 0.
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+            use_cache (`bool`, *optional*):
+                If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding
+                (see `past_key_values`).
+            past_key_value (`Tuple(torch.FloatTensor)`, *optional*): cached past key and value projection states
+            cache_position (`torch.LongTensor` of shape `(sequence_length)`, *optional*):
+                Indices depicting the position of the input sequence tokens in the sequence.
+            position_embeddings (`Tuple[torch.FloatTensor, torch.FloatTensor]`, *optional*):
+                Tuple containing the cosine and sine positional embeddings of shape `(batch_size, seq_len, head_dim)`,
+                with `head_dim` being the embedding dimension of each attention head.
+            kwargs (`dict`, *optional*):
+                Arbitrary kwargs to be ignored, used for FSDP and other methods that injects code
+                into the model
+        """
+
+        residual = hidden_states
+
+        hidden_states = self.input_layernorm(hidden_states)
+
+        # Self Attention
+        hidden_states, self_attn_weights, present_key_value = self.self_attn(
+            hidden_states=hidden_states,
+            attention_mask=attention_mask,
+            position_ids=position_ids,
+            past_key_value=past_key_value,
+            output_attentions=output_attentions,
+            use_cache=use_cache,
+            cache_position=cache_position,
+            position_embeddings=position_embeddings,
+        )
+        hidden_states = residual + hidden_states
+
+        # Fully Connected
+        residual = hidden_states
+        hidden_states = self.post_attention_layernorm(hidden_states)
+        hidden_states = self.mlp(hidden_states)
+        hidden_states = residual + hidden_states
+
+        outputs = (hidden_states,)
+
+        if output_attentions:
+            outputs += (self_attn_weights,)
+
+        if use_cache:
+            outputs += (present_key_value,)
+
+        return outputs
+
+
+@add_start_docstrings(
+    "The bare Qwen2_5_VL Model outputting raw hidden-states without any specific head on top.",
+    Qwen2_5_VL_START_DOCSTRING,
+)
+class Qwen2_5_VLModel(Qwen2_5_VLPreTrainedModel):
+    def __init__(self, config: Qwen2_5_VLConfig):
+        super().__init__(config)
+        self.padding_idx = config.pad_token_id
+        self.vocab_size = config.vocab_size
+
+        self.embed_tokens = nn.Embedding(config.vocab_size, config.hidden_size, self.padding_idx)
+        self.layers = nn.ModuleList(
+            [Qwen2_5_VLDecoderLayer(config, layer_idx) for layer_idx in range(config.num_hidden_layers)]
+        )
+        self._attn_implementation = config._attn_implementation
+        self.norm = Qwen2RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+        self.rotary_emb = Qwen2_5_VLRotaryEmbedding(config=config)
+
+        self.gradient_checkpointing = False
+        # Initialize weights and apply final processing
+        self.post_init()
+
+    def get_input_embeddings(self):
+        return self.embed_tokens
+
+    def set_input_embeddings(self, value):
+        self.embed_tokens = value
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+        cache_position: torch.LongTensor | None = None,
+    ) -> tuple | BaseModelOutputWithPast:
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        if (input_ids is None) ^ (inputs_embeds is not None):
+            raise ValueError("You must specify exactly one of input_ids or inputs_embeds")
+
+        if self.gradient_checkpointing and self.training:
+            if use_cache:
+                logger.warning_once(
+                    "`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`..."
+                )
+                use_cache = False
+
+        # torch.jit.trace() doesn't support cache objects in the output
+        if use_cache and past_key_values is None and not torch.jit.is_tracing():
+            past_key_values = DynamicCache()
+
+        if inputs_embeds is None:
+            inputs_embeds = self.embed_tokens(input_ids)
+
+        if cache_position is None:
+            past_seen_tokens = past_key_values.get_seq_length() if past_key_values is not None else 0
+            cache_position = torch.arange(
+                past_seen_tokens,
+                past_seen_tokens + inputs_embeds.shape[1],
+                device=inputs_embeds.device,
+            )
+
+        # the hard coded `3` is for temporal, height and width.
+        if position_ids is None:
+            position_ids = cache_position.view(1, 1, -1).expand(3, inputs_embeds.shape[0], -1)
+        elif position_ids.dim() == 2:
+            position_ids = position_ids[None, ...].expand(3, position_ids.shape[0], -1)
+
+        causal_mask = self._update_causal_mask(
+            attention_mask,
+            inputs_embeds,
+            cache_position,
+            past_key_values,
+            output_attentions,
+        )
+
+        hidden_states = inputs_embeds
+
+        # create position embeddings to be shared across the decoder layers
+        position_embeddings = self.rotary_emb(hidden_states, position_ids)
+
+        # decoder layers
+        all_hidden_states = () if output_hidden_states else None
+        all_self_attns = () if output_attentions else None
+        next_decoder_cache = None
+
+        for decoder_layer in self.layers:
+            if output_hidden_states:
+                all_hidden_states += (hidden_states,)
+
+            if self.gradient_checkpointing and self.training:
+                layer_outputs = self._gradient_checkpointing_func(
+                    decoder_layer.__call__,
+                    hidden_states,
+                    causal_mask,
+                    position_ids,
+                    past_key_values,
+                    output_attentions,
+                    use_cache,
+                    cache_position,
+                    position_embeddings,
+                )
+            else:
+                layer_outputs = decoder_layer(
+                    hidden_states,
+                    attention_mask=causal_mask,
+                    position_ids=position_ids,
+                    past_key_value=past_key_values,
+                    output_attentions=output_attentions,
+                    use_cache=use_cache,
+                    cache_position=cache_position,
+                    position_embeddings=position_embeddings,
+                )
+
+            hidden_states = layer_outputs[0]
+
+            if use_cache:
+                next_decoder_cache = layer_outputs[2 if output_attentions else 1]
+
+            if output_attentions:
+                all_self_attns += (layer_outputs[1],)
+
+        hidden_states = self.norm(hidden_states)
+
+        # add hidden states from the last decoder layer
+        if output_hidden_states:
+            all_hidden_states += (hidden_states,)
+
+        next_cache = next_decoder_cache if use_cache else None
+
+        if not return_dict:
+            return tuple(
+                v for v in [hidden_states, next_cache, all_hidden_states, all_self_attns] if v is not None
+            )
+        return BaseModelOutputWithPast(
+            last_hidden_state=hidden_states,
+            past_key_values=next_cache,
+            hidden_states=all_hidden_states,
+            attentions=all_self_attns,
+        )
+
+    def _update_causal_mask(
+        self,
+        attention_mask: torch.Tensor,
+        input_tensor: torch.Tensor,
+        cache_position: torch.Tensor,
+        past_key_values: Cache,
+        output_attentions: bool,
+    ):
+        if self.config._attn_implementation == "flash_attention_2":
+            if attention_mask is not None and past_key_values is not None:
+                is_padding_right = attention_mask[:, -1].sum().item() != input_tensor.size()[0]
+                if is_padding_right:
+                    raise ValueError(
+                        "You are attempting to perform batched generation with padding_side='right'"
+                        " this may lead to unexpected behaviour for Flash Attention version of Qwen2_5_VL. Make sure to "
+                        " call `tokenizer.padding_side  = 'left'` before tokenizing the input. "
+                    )
+            if attention_mask is not None and 0.0 in attention_mask:
+                return attention_mask
+            return None
+
+        # For SDPA, when possible, we will rely on its `is_causal` argument instead of its `attn_mask` argument, in
+        # order to dispatch on Flash Attention 2. This feature is not compatible with static cache, as SDPA will fail
+        # to infer the attention mask.
+        past_seen_tokens = past_key_values.get_seq_length() if past_key_values is not None else 0
+        using_static_cache = isinstance(past_key_values, StaticCache)
+        using_sliding_window_cache = isinstance(past_key_values, SlidingWindowCache)
+
+        # When output attentions is True, sdpa implementation's forward method calls the eager implementation's forward
+        if (
+            self.config._attn_implementation == "sdpa"
+            and not (using_static_cache or using_sliding_window_cache)
+            and not output_attentions
+        ):
+            if AttentionMaskConverter._ignore_causal_mask_sdpa(
+                attention_mask,
+                inputs_embeds=input_tensor,
+                past_key_values_length=past_seen_tokens,
+                sliding_window=self.config.sliding_window,
+                is_training=self.training,
+            ):
+                return None
+
+        dtype, device = input_tensor.dtype, input_tensor.device
+        min_dtype = torch.finfo(dtype).min
+        sequence_length = input_tensor.shape[1]
+        # SlidingWindowCache or StaticCache
+        if using_sliding_window_cache or using_static_cache:
+            target_length = past_key_values.get_max_cache_shape()
+        # DynamicCache or no cache
+        else:
+            target_length = (
+                attention_mask.shape[-1]
+                if isinstance(attention_mask, torch.Tensor)
+                else past_seen_tokens + sequence_length + 1
+            )
+
+        # In case the provided `attention` mask is 2D, we generate a causal mask here (4D).
+        causal_mask = self._prepare_4d_causal_attention_mask_with_cache_position(
+            attention_mask,
+            sequence_length=sequence_length,
+            target_length=target_length,
+            dtype=dtype,
+            device=device,
+            cache_position=cache_position,
+            batch_size=input_tensor.shape[0],
+            config=self.config,
+            past_key_values=past_key_values,
+        )
+
+        if (
+            self.config._attn_implementation == "sdpa"
+            and attention_mask is not None
+            and attention_mask.device.type in ["cuda", "xpu"]
+            and not output_attentions
+        ):
+            # Attend to all tokens in fully masked rows in the causal_mask, for example the relevant first rows when
+            # using left padding. This is required by F.scaled_dot_product_attention memory-efficient attention path.
+            # Details: https://github.com/pytorch/pytorch/issues/110213
+            causal_mask = AttentionMaskConverter._unmask_unattended(causal_mask, min_dtype)
+
+        return causal_mask
+
+    @staticmethod
+    def _prepare_4d_causal_attention_mask_with_cache_position(
+        attention_mask: torch.Tensor,
+        sequence_length: int,
+        target_length: int,
+        dtype: torch.dtype,
+        device: torch.device,
+        cache_position: torch.Tensor,
+        batch_size: int,
+        config: Qwen2_5_VLConfig,
+        past_key_values: Cache,
+    ):
+        """
+        Creates a causal 4D mask of shape `(batch_size, 1, query_length, key_value_length)` from a 2D mask of shape
+        `(batch_size, key_value_length)`, or if the input `attention_mask` is already 4D, do nothing.
+
+        Args:
+            attention_mask (`torch.Tensor`):
+                A 2D attention mask of shape `(batch_size, key_value_length)` or a 4D attention mask of shape `(batch_size, 1, query_length, key_value_length)`.
+            sequence_length (`int`):
+                The sequence length being processed.
+            target_length (`int`):
+                The target length: when generating with static cache, the mask should be as long as the static cache, to account for the 0 padding, the part of the cache that is not filled yet.
+            dtype (`torch.dtype`):
+                The dtype to use for the 4D attention mask.
+            device (`torch.device`):
+                The device to place the 4D attention mask on.
+            cache_position (`torch.Tensor`):
+                Indices depicting the position of the input sequence tokens in the sequence.
+            batch_size (`torch.Tensor`):
+                Batch size.
+            config (`Qwen2_5_VLConfig`):
+                The model's configuration class
+            past_key_values (`Cache`):
+                The cache class that is being used currently to generate
+        """
+        if attention_mask is not None and attention_mask.dim() == 4:
+            # In this case we assume that the mask comes already in inverted form and requires no inversion or slicing.
+            causal_mask = attention_mask
+        else:
+            min_dtype = torch.finfo(dtype).min
+            causal_mask = torch.full(
+                (sequence_length, target_length),
+                fill_value=min_dtype,
+                dtype=dtype,
+                device=device,
+            )
+            diagonal_attend_mask = torch.arange(target_length, device=device) > cache_position.reshape(-1, 1)
+            if config.sliding_window is not None:
+                # if we have sliding window, we should not attend to tokens beyond sliding window length, so we mask them out also
+                # the check is needed to verify is current checkpoint was trained with sliding window or not
+                if not isinstance(past_key_values, SlidingWindowCache) or sequence_length > target_length:
+                    sliding_attend_mask = torch.arange(target_length, device=device) <= (
+                        cache_position.reshape(-1, 1) - config.sliding_window
+                    )
+                    diagonal_attend_mask.bitwise_or_(sliding_attend_mask)
+            causal_mask *= diagonal_attend_mask
+            causal_mask = causal_mask[None, None, :, :].expand(batch_size, 1, -1, -1)
+            if attention_mask is not None:
+                causal_mask = causal_mask.clone()  # copy to contiguous memory for in-place edit
+                if attention_mask.shape[-1] > target_length:
+                    attention_mask = attention_mask[:, :target_length]
+                mask_length = attention_mask.shape[-1]
+                padding_mask = causal_mask[:, :, :, :mask_length] + attention_mask[:, None, None, :].to(
+                    causal_mask.device
+                )
+                padding_mask = padding_mask == 0
+                causal_mask[:, :, :, :mask_length] = causal_mask[:, :, :, :mask_length].masked_fill(
+                    padding_mask, min_dtype
+                )
+        return causal_mask
+
+
+@dataclass
+class Qwen2_5_VLCausalLMOutputWithPast(ModelOutput):
+    """
+    Base class for Qwen2_5_VL causal language model (or autoregressive) outputs.
+
+    Args:
+        loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided):
+            Language modeling loss (for next-token prediction).
+        logits (`torch.FloatTensor` of shape `(batch_size, sequence_length, config.vocab_size)`):
+            Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).
+        past_key_values (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
+            Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of shape
+            `(batch_size, num_heads, sequence_length, embed_size_per_head)`)
+
+            Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see
+            `past_key_values` input) to speed up sequential decoding.
+        hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
+            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
+            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
+
+            Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.
+        attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
+            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
+            sequence_length)`.
+
+            Attentions weights after the attention softmax, used to compute the weighted average in the self-attention
+            heads.
+        rope_deltas (`torch.LongTensor` of shape `(batch_size, )`, *optional*):
+            The rope index difference between sequence length and multimodal rope.
+    """
+
+    loss: torch.FloatTensor | None = None
+    logits: torch.FloatTensor = None
+    past_key_values: list[torch.FloatTensor] | None = None
+    hidden_states: tuple[torch.FloatTensor] | None = None
+    attentions: tuple[torch.FloatTensor] | None = None
+    rope_deltas: torch.LongTensor | None = None
+
+
+QWEN2_5_VL_INPUTS_DOCSTRING = r"""
+    Args:
+        input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
+            Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide
+            it.
+
+            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+
+            [What are input IDs?](../glossary#input-ids)
+        attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:
+
+            - 1 for tokens that are **not masked**,
+            - 0 for tokens that are **masked**.
+
+            [What are attention masks?](../glossary#attention-mask)
+
+            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+
+            If `past_key_values` is used, optionally only the last `decoder_input_ids` have to be input (see
+            `past_key_values`).
+
+            If you want to change padding behavior, you should read [`modeling_opt._prepare_decoder_attention_mask`]
+            and modify to your needs. See diagram 1 in [the paper](https://arxiv.org/abs/1910.13461) for more
+            information on the default strategy.
+
+            - 1 indicates the head is **not masked**,
+            - 0 indicates the head is **masked**.
+        position_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
+            config.n_positions - 1]`. [What are position IDs?](../glossary#position-ids)
+        past_key_values (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
+            Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of shape
+            `(batch_size, num_heads, sequence_length, embed_size_per_head)`) and 2 additional tensors of shape
+            `(batch_size, num_heads, encoder_sequence_length, embed_size_per_head)`.
+
+            Contains pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention
+            blocks) that can be used (see `past_key_values` input) to speed up sequential decoding.
+
+            If `past_key_values` are used, the user can optionally input only the last `decoder_input_ids` (those that
+            don't have their past key value states given to this model) of shape `(batch_size, 1)` instead of all
+            `decoder_input_ids` of shape `(batch_size, sequence_length)`.
+        inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
+            Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
+            is useful if you want more control over how to convert `input_ids` indices into associated vectors than the
+            model's internal embedding lookup matrix.
+        use_cache (`bool`, *optional*):
+            If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding (see
+            `past_key_values`).
+        output_attentions (`bool`, *optional*):
+            Whether or not to return the attentions tensors of all attention layers. See `attentions` under returned
+            tensors for more detail.
+        output_hidden_states (`bool`, *optional*):
+            Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors for
+            more detail.
+        return_dict (`bool`, *optional*):
+            Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
+        pixel_values (`torch.FloatTensor` of shape `(seq_length, num_channels * image_size * image_size)):
+            The tensors corresponding to the input images. Pixel values can be obtained using
+            [`AutoImageProcessor`]. See [`Qwen2_5_VLImageProcessor.__call__`] for details. [`Qwen2_5_VLProcessor`] uses
+            [`Qwen2_5_VLImageProcessor`] for processing images.
+        pixel_values_videos (`torch.FloatTensor` of shape `(seq_length, num_channels * temporal_size * image_size * image_size)):
+            The tensors corresponding to the input videos. Pixel values can be obtained using
+            [`AutoImageProcessor`]. See [`Qwen2_5_VLImageProcessor.__call__`] for details. [`Qwen2_5_VLProcessor`] uses
+            [`Qwen2_5_VLImageProcessor`] for processing videos.
+        image_grid_thw (`torch.LongTensor` of shape `(num_images, 3)`, *optional*):
+            The temporal, height and width of feature shape of each image in LLM.
+        video_grid_thw (`torch.LongTensor` of shape `(num_videos, 3)`, *optional*):
+            The temporal, height and width of feature shape of each video in LLM.
+        rope_deltas (`torch.LongTensor` of shape `(batch_size, )`, *optional*):
+            The rope index difference between sequence length and multimodal rope.
+"""
+
+
+class Qwen2_5_VLForConditionalGeneration(Qwen2_5_VLPreTrainedModel, GenerationMixin):
+    _tied_weights_keys = {"lm_head.weight": "model.embed_tokens.weight"}
+    config_class = Qwen2_5_VLConfig
+    _no_split_modules = ["Qwen2_5_VLDecoderLayer", "Qwen2_5_VLVisionBlock"]
+
+    def __init__(self, config):
+        super().__init__(config)
+        self.visual = Qwen2_5_VisionTransformerPretrainedModel._from_config(config.vision_config)
+        self.model = Qwen2_5_VLModel(config)
+        self.vocab_size = config.vocab_size
+        self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
+        self.rope_deltas = None  # cache rope_deltas here
+
+        # Initialize weights and apply final processing
+        self.post_init()
+
+    def get_input_embeddings(self):
+        return self.model.embed_tokens
+
+    def set_input_embeddings(self, value):
+        self.model.embed_tokens = value
+
+    def get_output_embeddings(self):
+        return self.lm_head
+
+    def set_output_embeddings(self, new_embeddings):
+        self.lm_head = new_embeddings
+
+    def set_decoder(self, decoder):
+        self.model = decoder
+
+    def get_decoder(self):
+        return self.model
+
+    def get_rope_index(
+        self,
+        input_ids: torch.LongTensor | None = None,
+        image_grid_thw: torch.LongTensor | None = None,
+        video_grid_thw: torch.LongTensor | None = None,
+        second_per_grid_ts: torch.Tensor | None = None,
+        attention_mask: torch.Tensor | None = None,
+    ) -> tuple[torch.Tensor, torch.Tensor]:
+        """
+        Calculate the 3D rope index based on image and video's temporal, height and width in LLM.
+
+        Explanation:
+            Each embedding sequence contains vision embedding and text embedding or just contains text embedding.
+
+            For pure text embedding sequence, the rotary position embedding has no difference with modern LLMs.
+            Examples:
+                input_ids: [T T T T T], here T is for text.
+                temporal position_ids: [0, 1, 2, 3, 4]
+                height position_ids: [0, 1, 2, 3, 4]
+                width position_ids: [0, 1, 2, 3, 4]
+
+            For vision and text embedding sequence, we calculate 3D rotary position embedding for vision part
+            and 1D rotary position embedding for text part.
+            Examples:
+                Temporal (Time): 3 patches, representing different segments of the video in time.
+                Height: 2 patches, dividing each frame vertically.
+                Width: 2 patches, dividing each frame horizontally.
+                We also have some important parameters:
+                fps (Frames Per Second): The video's frame rate, set to 1. This means one frame is processed each second.
+                tokens_per_second: This is a crucial parameter. It dictates how many "time-steps" or "temporal tokens" are conceptually packed into a one-second interval of the video. In this case, we have 25 tokens per second. So each second of the video will be represented with 25 separate time points. It essentially defines the temporal granularity.
+                temporal_patch_size: The number of frames that compose one temporal patch. Here, it's 2 frames.
+                interval: The step size for the temporal position IDs, calculated as tokens_per_second * temporal_patch_size / fps. In this case, 25 * 2 / 1 = 50. This means that each temporal patch will be have a difference of 50 in the temporal position IDs.
+                input_ids: [V V V V V V V V V V V V T T T T T], here V is for vision.
+                vision temporal position_ids: [0, 0, 0, 0, 50, 50, 50, 50, 100, 100, 100, 100]
+                vision height position_ids: [0, 0, 1, 1, 0, 0, 1, 1, 0, 0, 1, 1]
+                vision width position_ids: [0, 1, 0, 1, 0, 1, 0, 1, 0, 1, 0, 1]
+                text temporal position_ids: [101, 102, 103, 104, 105]
+                text height position_ids: [101, 102, 103, 104, 105]
+                text width position_ids: [101, 102, 103, 104, 105]
+                Here we calculate the text start position_ids as the max vision position_ids plus 1.
+
+        Args:
+            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
+                Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide
+                it.
+            image_grid_thw (`torch.LongTensor` of shape `(num_images, 3)`, *optional*):
+                The temporal, height and width of feature shape of each image in LLM.
+            video_grid_thw (`torch.LongTensor` of shape `(num_videos, 3)`, *optional*):
+                The temporal, height and width of feature shape of each video in LLM.
+            second_per_grid_ts (`torch.Tensor` of shape `(num_videos)`, *optional*):
+                The time interval (in seconds) for each grid along the temporal dimension in the 3D position IDs.
+            attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
+                Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:
+
+                - 1 for tokens that are **not masked**,
+                - 0 for tokens that are **masked**.
+
+        Returns:
+            position_ids (`torch.LongTensor` of shape `(3, batch_size, sequence_length)`)
+            mrope_position_deltas (`torch.Tensor` of shape `(batch_size)`)
+        """
+        spatial_merge_size = self.config.vision_config.spatial_merge_size
+        image_token_id = self.config.image_token_id
+        video_token_id = self.config.video_token_id
+        vision_start_token_id = self.config.vision_start_token_id
+        mrope_position_deltas = []
+        if input_ids is not None and (image_grid_thw is not None or video_grid_thw is not None):
+            total_input_ids = input_ids
+            if attention_mask is None:
+                attention_mask = torch.ones_like(total_input_ids)
+            position_ids = torch.ones(
+                3,
+                input_ids.shape[0],
+                input_ids.shape[1],
+                dtype=input_ids.dtype,
+                device=input_ids.device,
+            )
+            image_index, video_index = 0, 0
+            attention_mask = attention_mask.to(total_input_ids.device)
+            for i, input_ids in enumerate(total_input_ids):
+                input_ids = input_ids[attention_mask[i] == 1]
+                image_nums, video_nums = 0, 0
+                vision_start_indices = torch.argwhere(input_ids == vision_start_token_id).squeeze(1)
+                vision_tokens = input_ids[vision_start_indices + 1]
+                image_nums = (vision_tokens == image_token_id).sum()
+                video_nums = (vision_tokens == video_token_id).sum()
+                input_tokens = input_ids.tolist()
+                llm_pos_ids_list: list = []
+                st = 0
+                remain_images, remain_videos = image_nums, video_nums
+                for _ in range(image_nums + video_nums):
+                    if image_token_id in input_tokens and remain_images > 0:
+                        ed_image = input_tokens.index(image_token_id, st)
+                    else:
+                        ed_image = len(input_tokens) + 1
+                    if video_token_id in input_tokens and remain_videos > 0:
+                        ed_video = input_tokens.index(video_token_id, st)
+                    else:
+                        ed_video = len(input_tokens) + 1
+                    if ed_image < ed_video:
+                        t, h, w = (
+                            image_grid_thw[image_index][0],
+                            image_grid_thw[image_index][1],
+                            image_grid_thw[image_index][2],
+                        )
+                        second_per_grid_t = 0
+                        image_index += 1
+                        remain_images -= 1
+                        ed = ed_image
+
+                    else:
+                        t, h, w = (
+                            video_grid_thw[video_index][0],
+                            video_grid_thw[video_index][1],
+                            video_grid_thw[video_index][2],
+                        )
+                        if second_per_grid_ts is not None:
+                            second_per_grid_t = second_per_grid_ts[video_index]
+                        else:
+                            second_per_grid_t = 1.0
+                        video_index += 1
+                        remain_videos -= 1
+                        ed = ed_video
+                    llm_grid_t, llm_grid_h, llm_grid_w = (
+                        t.item(),
+                        h.item() // spatial_merge_size,
+                        w.item() // spatial_merge_size,
+                    )
+                    text_len = ed - st
+
+                    st_idx = llm_pos_ids_list[-1].max() + 1 if len(llm_pos_ids_list) > 0 else 0
+                    llm_pos_ids_list.append(torch.arange(text_len).view(1, -1).expand(3, -1) + st_idx)
+
+                    range_tensor = torch.arange(llm_grid_t).view(-1, 1)
+                    expanded_range = range_tensor.expand(-1, llm_grid_h * llm_grid_w)
+
+                    time_tensor = (
+                        expanded_range * second_per_grid_t * self.config.vision_config.tokens_per_second
+                    )
+
+                    time_tensor_long = time_tensor.long()
+                    t_index = time_tensor_long.flatten()
+
+                    h_index = (
+                        torch.arange(llm_grid_h).view(1, -1, 1).expand(llm_grid_t, -1, llm_grid_w).flatten()
+                    )
+                    w_index = (
+                        torch.arange(llm_grid_w).view(1, 1, -1).expand(llm_grid_t, llm_grid_h, -1).flatten()
+                    )
+                    llm_pos_ids_list.append(torch.stack([t_index, h_index, w_index]) + text_len + st_idx)
+                    st = ed + llm_grid_t * llm_grid_h * llm_grid_w
+
+                if st < len(input_tokens):
+                    st_idx = llm_pos_ids_list[-1].max() + 1 if len(llm_pos_ids_list) > 0 else 0
+                    text_len = len(input_tokens) - st
+                    llm_pos_ids_list.append(torch.arange(text_len).view(1, -1).expand(3, -1) + st_idx)
+
+                llm_positions = torch.cat(llm_pos_ids_list, dim=1).reshape(3, -1)
+                position_ids[..., i, attention_mask[i] == 1] = llm_positions.to(position_ids.device)
+                mrope_position_deltas.append(llm_positions.max() + 1 - len(total_input_ids[i]))
+            mrope_position_deltas = torch.tensor(mrope_position_deltas, device=input_ids.device).unsqueeze(1)
+            return position_ids, mrope_position_deltas
+        else:
+            if attention_mask is not None:
+                position_ids = attention_mask.long().cumsum(-1) - 1
+                position_ids.masked_fill_(attention_mask == 0, 1)
+                position_ids = position_ids.unsqueeze(0).expand(3, -1, -1).to(attention_mask.device)
+                max_position_ids = position_ids.max(0, keepdim=False)[0].max(-1, keepdim=True)[0]
+                mrope_position_deltas = max_position_ids + 1 - attention_mask.shape[-1]
+            else:
+                position_ids = (
+                    torch.arange(input_ids.shape[1], device=input_ids.device)
+                    .view(1, 1, -1)
+                    .expand(3, input_ids.shape[0], -1)
+                )
+                mrope_position_deltas = torch.zeros(
+                    [input_ids.shape[0], 1],
+                    device=input_ids.device,
+                    dtype=input_ids.dtype,
+                )
+
+            return position_ids, mrope_position_deltas
+
+    @add_start_docstrings_to_model_forward(QWEN2_5_VL_INPUTS_DOCSTRING)
+    @replace_return_docstrings(output_type=Qwen2_5_VLCausalLMOutputWithPast, config_class=_CONFIG_FOR_DOC)
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        labels: torch.LongTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+        pixel_values: torch.Tensor | None = None,
+        pixel_values_videos: torch.FloatTensor | None = None,
+        image_grid_thw: torch.LongTensor | None = None,
+        video_grid_thw: torch.LongTensor | None = None,
+        rope_deltas: torch.LongTensor | None = None,
+        cache_position: torch.LongTensor | None = None,
+        second_per_grid_ts: torch.Tensor | None = None,
+    ) -> tuple | Qwen2_5_VLCausalLMOutputWithPast:
+        r"""
+        Args:
+            labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+                Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
+                config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
+                (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.
+
+        Returns:
+
+        Example:
+
+        ```python
+        >>> from PIL import Image
+        >>> import requests
+        >>> from transformers import AutoProcessor, Qwen2_5_VLForConditionalGeneration
+
+        >>> model = Qwen2_5_VLForConditionalGeneration.from_pretrained("Qwen/Qwen2.5-VL-7B-Instruct")
+        >>> processor = AutoProcessor.from_pretrained("Qwen/Qwen2.5-VL-7B-Instruct")
+
+        >>> messages = [
+            {
+                "role": "user",
+                "content": [
+                    {"type": "image"},
+                    {"type": "text", "text": "What is shown in this image?"},
+                ],
+            },
+        ]
+        >>> url = "https://www.ilankelman.org/stopsigns/australia.jpg"
+        >>> image = Image.open(requests.get(url, stream=True).raw)
+
+        >>> text = processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
+        >>> inputs = processor(text=[text], images=[image], vision_infos=[vision_infos])
+
+        >>> # Generate
+        >>> generate_ids = model.generate(inputs.input_ids, max_length=30)
+        >>> tokenizer.batch_decode(generate_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
+        "The image shows a street scene with a red stop sign in the foreground. In the background, there is a large red gate with Chinese characters ..."
+        ```"""
+
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        if inputs_embeds is None:
+            inputs_embeds = self.model.embed_tokens(input_ids)
+            if pixel_values is not None:
+                pixel_values = pixel_values.type(self.visual.dtype)
+                image_embeds = self.visual(pixel_values, grid_thw=image_grid_thw)
+                n_image_tokens = (input_ids == self.config.image_token_id).sum().item()
+                n_image_features = image_embeds.shape[0]
+                if n_image_tokens != n_image_features:
+                    raise ValueError(
+                        f"Image features and image tokens do not match: tokens: {n_image_tokens}, features {n_image_features}"
+                    )
+
+                mask = input_ids == self.config.image_token_id
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                image_mask = mask_expanded.to(inputs_embeds.device)
+
+                image_embeds = image_embeds.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(image_mask, image_embeds)
+
+            if pixel_values_videos is not None:
+                pixel_values_videos = pixel_values_videos.type(self.visual.dtype)
+                video_embeds = self.visual(pixel_values_videos, grid_thw=video_grid_thw)
+                n_video_tokens = (input_ids == self.config.video_token_id).sum().item()
+                n_video_features = video_embeds.shape[0]
+                if n_video_tokens != n_video_features:
+                    raise ValueError(
+                        f"Video features and video tokens do not match: tokens: {n_video_tokens}, features {n_video_features}"
+                    )
+
+                mask = input_ids == self.config.video_token_id
+                mask_unsqueezed = mask.unsqueeze(-1)
+                mask_expanded = mask_unsqueezed.expand_as(inputs_embeds)
+                video_mask = mask_expanded.to(inputs_embeds.device)
+
+                video_embeds = video_embeds.to(inputs_embeds.device, inputs_embeds.dtype)
+                inputs_embeds = inputs_embeds.masked_scatter(video_mask, video_embeds)
+
+            if attention_mask is not None:
+                attention_mask = attention_mask.to(inputs_embeds.device)
+
+        # if we get 4D attention mask we cannot calculate rope deltas anymore. TODO @raushan fixme
+        if position_ids is None and (attention_mask is None or attention_mask.ndim == 2):
+            # calculate RoPE index once per generation in the pre-fill stage only
+            if (
+                (cache_position is not None and cache_position[0] == 0)
+                or self.rope_deltas is None
+                or (past_key_values is None or past_key_values.get_seq_length() == 0)
+            ):
+                position_ids, rope_deltas = self.get_rope_index(
+                    input_ids,
+                    image_grid_thw,
+                    video_grid_thw,
+                    second_per_grid_ts,
+                    attention_mask,
+                )
+                self.rope_deltas = rope_deltas
+            # then use the prev pre-calculated rope-deltas to get the correct position ids
+            else:
+                batch_size, seq_length, _ = inputs_embeds.shape
+                delta = (
+                    (cache_position[0] + self.rope_deltas).to(inputs_embeds.device)
+                    if cache_position is not None
+                    else 0
+                )
+                position_ids = torch.arange(seq_length, device=inputs_embeds.device)
+                position_ids = position_ids.view(1, -1).expand(batch_size, -1)
+                if cache_position is not None:  # otherwise `deltas` is an int `0`
+                    delta = delta.repeat_interleave(batch_size // delta.shape[0], dim=0)
+                position_ids = position_ids.add(delta)
+                position_ids = position_ids.unsqueeze(0).expand(3, -1, -1)
+
+        outputs = self.model(
+            input_ids=None,
+            position_ids=position_ids,
+            attention_mask=attention_mask,
+            past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            cache_position=cache_position,
+        )
+
+        hidden_states = outputs[0]
+        logits = self.lm_head(hidden_states)
+
+        loss = None
+        if labels is not None:
+            # Upcast to float if we need to compute the loss to avoid potential precision issues
+            logits = logits.float()
+            # Shift so that tokens < n predict n
+            shift_logits = logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            # Flatten the tokens
+            loss_fct = CrossEntropyLoss()
+            shift_logits = shift_logits.view(-1, self.config.vocab_size)
+            shift_labels = shift_labels.view(-1)
+            # Enable model parallelism
+            shift_labels = shift_labels.to(shift_logits.device)
+            loss = loss_fct(shift_logits, shift_labels)
+
+        if not return_dict:
+            output = (logits,) + outputs[1:]
+            return (loss,) + output if loss is not None else output
+
+        return Qwen2_5_VLCausalLMOutputWithPast(
+            loss=loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+            rope_deltas=self.rope_deltas,
+        )
+
+    def prepare_inputs_for_generation(
+        self,
+        input_ids,
+        past_key_values=None,
+        attention_mask=None,
+        inputs_embeds=None,
+        cache_position=None,
+        position_ids=None,
+        use_cache=True,
+        pixel_values=None,
+        pixel_values_videos=None,
+        image_grid_thw=None,
+        video_grid_thw=None,
+        second_per_grid_ts=None,
+        **kwargs,
+    ):
+        # Overwritten -- in specific circumstances we don't want to forward image inputs to the model
+
+        # If we have cache: let's slice `input_ids` through `cache_position`, to keep only the unprocessed tokens
+        # Exception 1: when passing input_embeds, input_ids may be missing entries
+        # Exception 2: some generation methods do special slicing of input_ids, so we don't need to do it here
+        # Exception 3: with synced GPUs cache_position may go out of bounds, but we only want dummy token in that case.
+        #              (we can't check exception 3 while compiling)
+        # Exception 4: If input_embeds are passed then slice it through `cache_position`, to keep only the unprocessed tokens and
+        # generate the first token for each sequence. Later use the generated Input ids for continuation.
+        if past_key_values is not None:
+            if inputs_embeds is not None and input_ids.shape[1] == 0:  # Exception 4
+                inputs_embeds = inputs_embeds[:, -cache_position.shape[0] :]
+            elif inputs_embeds is not None or (  # Exception 1
+                is_torchdynamo_compiling() or cache_position[-1] >= input_ids.shape[1]
+            ):  # Exception 3
+                input_ids = input_ids[:, -cache_position.shape[0] :]
+            elif (
+                input_ids.shape[1] != cache_position.shape[0]
+            ):  # Default case (the "else", a no op, is Exception 2)
+                input_ids = input_ids[:, cache_position]
+
+        if cache_position[0] != 0:
+            pixel_values = None
+            pixel_values_videos = None
+
+        # if `inputs_embeds` are passed, we only want to use them in the 1st generation step
+        if inputs_embeds is not None and len(cache_position) == inputs_embeds.shape[1]:
+            model_inputs = {"inputs_embeds": inputs_embeds, "input_ids": None}
+        else:
+            model_inputs = {"input_ids": input_ids, "inputs_embeds": None}
+
+        if isinstance(past_key_values, StaticCache) and attention_mask.ndim == 2:
+            if model_inputs["inputs_embeds"] is not None:
+                batch_size, sequence_length, _ = inputs_embeds.shape
+                device = inputs_embeds.device
+            else:
+                batch_size, sequence_length = input_ids.shape
+                device = input_ids.device
+
+            attention_mask = self.model._prepare_4d_causal_attention_mask_with_cache_position(
+                attention_mask,
+                sequence_length=sequence_length,
+                target_length=past_key_values.get_max_cache_shape(),
+                dtype=self.lm_head.weight.dtype,
+                device=device,
+                cache_position=cache_position,
+                batch_size=batch_size,
+                config=self.config,
+                past_key_values=past_key_values,
+            )
+
+        model_inputs.update(
+            {
+                "position_ids": position_ids,
+                "past_key_values": past_key_values,
+                "use_cache": use_cache,
+                "attention_mask": attention_mask,
+                "pixel_values": pixel_values,
+                "pixel_values_videos": pixel_values_videos,
+                "image_grid_thw": image_grid_thw,
+                "video_grid_thw": video_grid_thw,
+                "cache_position": cache_position,
+                "second_per_grid_ts": second_per_grid_ts,
+            }
+        )
+        return model_inputs
+
+    def _get_image_nums_and_video_nums(
+        self,
+        input_ids: torch.LongTensor | None,
+    ) -> tuple[torch.Tensor, torch.Tensor]:
+        """
+        Get the number of images and videos for each sample to calculate the separation length of the sample tensor.
+        These parameters are not passed through the processor to avoid unpredictable impacts from interface modifications.
+
+        Args:
+            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
+                Indices of input sequence tokens in the vocabulary.
+
+        Returns:
+            image_nums (`torch.LongTensor` of shape `(batch_size, num_images_sample)`)
+            video_nums (`torch.LongTensor` of shape `(batch_size, num_videos_sample)`)
+        """
+        image_token_id = self.config.image_token_id
+        video_token_id = self.config.video_token_id
+        vision_start_token_id = self.config.vision_start_token_id
+
+        vision_start_mask = input_ids == vision_start_token_id
+        vision_first_mask = torch.roll(vision_start_mask, shifts=1, dims=1)
+        image_mask = input_ids == image_token_id
+        video_mask = input_ids == video_token_id
+        image_nums = torch.sum(vision_first_mask & image_mask, dim=1)
+        video_nums = torch.sum(vision_first_mask & video_mask, dim=1)
+
+        return image_nums, video_nums
+
+    def _expand_inputs_for_generation(
+        self,
+        expand_size: int = 1,
+        is_encoder_decoder: bool = False,
+        input_ids: torch.LongTensor | None = None,
+        **model_kwargs,
+    ) -> tuple[torch.LongTensor, dict[str, Any]]:
+        # Overwritten -- Support for expanding tensors without a batch size dimension
+        # e.g., pixel_values, image_grid_thw, pixel_values_videos, video_grid_thw, second_per_grid_t
+        # pixel_values.shape[0] is sum(seqlen_images for samples)
+        # image_grid_thw.shape[0] is sum(num_images for samples)
+
+        if expand_size == 1:
+            return input_ids, model_kwargs
+
+        visual_keys = [
+            "pixel_values",
+            "image_grid_thw",
+            "pixel_values_videos",
+            "video_grid_thw",
+            "second_per_grid_ts",
+        ]
+
+        def _expand_dict_for_generation_visual(dict_to_expand):
+            image_grid_thw = model_kwargs.get("image_grid_thw", None)
+            video_grid_thw = model_kwargs.get("video_grid_thw", None)
+            image_nums, video_nums = self._get_image_nums_and_video_nums(input_ids)
+
+            def _repeat_interleave_samples(x, lengths, repeat_times):
+                samples = torch.split(x, lengths)
+                repeat_args = [repeat_times] + [1] * (x.dim() - 1)
+                result = torch.cat([sample.repeat(*repeat_args) for sample in samples], dim=0)
+                return result
+
+            for key in dict_to_expand:
+                if key == "pixel_values":
+                    # split images into samples
+                    samples = torch.split(image_grid_thw, list(image_nums))
+                    # compute the sequence length of images for each sample
+                    lengths = [torch.prod(sample, dim=1).sum() for sample in samples]
+                    dict_to_expand[key] = _repeat_interleave_samples(
+                        dict_to_expand[key], lengths=lengths, repeat_times=expand_size
+                    )
+                elif key == "image_grid_thw":
+                    # get the num of images for each sample
+                    lengths = list(image_nums)
+                    dict_to_expand[key] = _repeat_interleave_samples(
+                        dict_to_expand[key], lengths=lengths, repeat_times=expand_size
+                    )
+                elif key == "pixel_values_videos":
+                    samples = torch.split(video_grid_thw, list(video_nums))
+                    lengths = [torch.prod(sample, dim=1).sum() for sample in samples]
+                    dict_to_expand[key] = _repeat_interleave_samples(
+                        dict_to_expand[key], lengths=lengths, repeat_times=expand_size
+                    )
+                elif key == "video_grid_thw":
+                    lengths = list(video_nums)
+                    dict_to_expand[key] = _repeat_interleave_samples(
+                        dict_to_expand[key], lengths=lengths, repeat_times=expand_size
+                    )
+                elif key == "second_per_grid_ts":
+                    if not isinstance(dict_to_expand[key], list):
+                        raise TypeError(
+                            f"Expected value for key '{key}' to be a list, but got {type(dict_to_expand[key])} instead."
+                        )
+                    tensor = torch.tensor(dict_to_expand[key])
+                    lengths = list(video_nums)
+                    tensor = _repeat_interleave_samples(tensor, lengths=lengths, repeat_times=expand_size)
+                    dict_to_expand[key] = tensor.tolist()
+            return dict_to_expand
+
+        def _expand_dict_for_generation(dict_to_expand):
+            for key in dict_to_expand:
+                if (
+                    key != "cache_position"
+                    and dict_to_expand[key] is not None
+                    and isinstance(dict_to_expand[key], torch.Tensor)
+                    and key not in visual_keys
+                ):
+                    dict_to_expand[key] = dict_to_expand[key].repeat_interleave(expand_size, dim=0)
+            return dict_to_expand
+
+        # input_ids is required for expanding visual inputs
+        # If input_ids is unavailable, visual inputs will not be used; therefore, there is no need to expand visual inputs.
+        if input_ids is not None and input_ids.numel() != 0:
+            model_kwargs = _expand_dict_for_generation_visual(model_kwargs)
+
+        if input_ids is not None:
+            input_ids = input_ids.repeat_interleave(expand_size, dim=0)
+
+        model_kwargs = _expand_dict_for_generation(model_kwargs)
+
+        if is_encoder_decoder:
+            if model_kwargs.get("encoder_outputs") is None:
+                raise ValueError(
+                    "If `is_encoder_decoder` is True, make sure that `encoder_outputs` is defined."
+                )
+            model_kwargs["encoder_outputs"] = _expand_dict_for_generation(model_kwargs["encoder_outputs"])
+
+        return input_ids, model_kwargs
+
+
+@dataclass
+class Qwen2_5_VLACausalLMOutputWithPast(ModelOutput):
+    loss: torch.FloatTensor | None = None
+    flow_loss: torch.FloatTensor | None = None
+    cross_entropy_loss: torch.FloatTensor | None = None
+    logits: torch.FloatTensor | None = None
+    past_key_values: list[torch.FloatTensor] | None = None
+    hidden_states: tuple[torch.FloatTensor] | None = None
+    attentions: tuple[torch.FloatTensor] | None = None
+    rope_deltas: torch.LongTensor | None = None
+
+    channel_loss_dict: dict[torch.FloatTensor] | None = None
+    channel_loss_count_dict: dict[torch.FloatTensor] | None = None
+
+
+class BlockSparseMLP(nn.Module):
+    def __init__(self, config):
+        super().__init__()
+
+        self.hidden_size = config["hidden_size"]
+        self.intermediate_size = config["intermediate_size"]
+        self.hidden_act = config["hidden_act"]
+        self.gate_proj = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
+        self.up_proj = nn.Linear(self.hidden_size, self.intermediate_size, bias=False)
+        self.down_proj = nn.Linear(self.intermediate_size, self.hidden_size, bias=False)
+        self.act_fn = ACT2FN[self.hidden_act]
+
+    def forward(self, hidden_state):
+        return self.down_proj(self.act_fn(self.gate_proj(hidden_state)) * self.up_proj(hidden_state))
+
+
+class SparseMoeBlock(nn.Module):
+    def __init__(self, config, num_experts: int):
+        super().__init__()
+        self.num_experts = num_experts
+        self.experts = nn.ModuleList([BlockSparseMLP(config.experts[i]) for i in range(num_experts)])
+
+        if not hasattr(config, "dim_inputs") or not config.dim_inputs:
+            raise ValueError("Config must contain valid dim_inputs")
+
+        self.dim_inputs = config.dim_inputs
+
+    def forward(self, hidden_states: torch.Tensor, experts_indices: torch.Tensor) -> torch.Tensor:
+        """
+        Route different hidden_states to corresponding experts for processing.
+
+        Args:
+            hidden_states (torch.Tensor): Tensor of shape (batch_size, seq_length, hidden_dim).
+            experts_indices (torch.Tensor): Tensor of shape (batch_size, seq_length),
+                indicating the expert index assigned to each token.
+
+        Returns:
+            output (torch.Tensor): Tensor of shape (batch_size, seq_length, hidden_dim).
+        """
+        batch_size, seq_length, hidden_dim = hidden_states.size()
+        output = torch.zeros_like(hidden_states)
+
+        for expert_idx, expert in enumerate(self.experts):
+            mask = experts_indices == expert_idx
+            if mask.sum() == 0:
+                continue
+            dim_input = self.dim_inputs[expert_idx]
+
+            selected_hidden = hidden_states[mask]
+            processed_hidden = expert(selected_hidden[:, :dim_input])
+
+            batch_indices, seq_indices = torch.where(mask)
+            output[batch_indices, seq_indices, :dim_input] = processed_hidden
+
+        return output
+
+
+QWEN2_5_VL_ATTENTION_CLASSES = {
+    "eager": Qwen2_5_VLAttention,
+    "flash_attention_2": Qwen2_5_VLFlashAttention2,
+    "sdpa": Qwen2_5_VLSdpaAttention,
+}
+
+
+class Qwen2_5_VLDecoderLayer_with_MoE(nn.Module):
+    def __init__(self, config: Qwen2_5_VLConfig, layer_idx: int, num_experts: int):
+        super().__init__()
+        self.hidden_size = config.hidden_size
+
+        if config.use_sliding_window and config._attn_implementation != "flash_attention_2":
+            logger.warning_once(
+                f"Sliding Window Attention is enabled but not implemented for `{config._attn_implementation}`; "
+                "unexpected results may be encountered."
+            )
+
+        self.self_attn = QWEN2_5_VL_ATTENTION_CLASSES[config._attn_implementation](config, layer_idx)
+
+        self.input_layernorm = Qwen2RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+        self.post_attention_layernorm = Qwen2RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+
+        if config.mlp_moe:
+            self.moe = SparseMoeBlock(config, num_experts=num_experts)
+            self.mlp = None
+        else:
+            self.mlp = Qwen2_5_VLMLP(config)
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_value: tuple[torch.Tensor] | None = None,
+        token_types=None,
+        output_attentions: bool | None = False,
+        use_cache: bool | None = False,
+        cache_position: torch.LongTensor | None = None,
+        position_embeddings: tuple[torch.Tensor, torch.Tensor] | None = None,
+        **kwargs,
+    ) -> tuple[torch.FloatTensor, tuple[torch.FloatTensor, torch.FloatTensor] | None]:
+        """
+        Args:
+            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
+            attention_mask (`torch.FloatTensor`, *optional*): attention mask of size
+                `(batch, sequence_length)` where padding elements are indicated by 0.
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+            use_cache (`bool`, *optional*):
+                If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding
+                (see `past_key_values`).
+            past_key_value (`Tuple(torch.FloatTensor)`, *optional*): cached past key and value projection states
+            cache_position (`torch.LongTensor` of shape `(sequence_length)`, *optional*):
+                Indices depicting the position of the input sequence tokens in the sequence.
+            position_embeddings (`Tuple[torch.FloatTensor, torch.FloatTensor]`, *optional*):
+                Tuple containing the cosine and sine positional embeddings of shape `(batch_size, seq_len, head_dim)`,
+                with `head_dim` being the embedding dimension of each attention head.
+            kwargs (`dict`, *optional*):
+                Arbitrary kwargs to be ignored, used for FSDP and other methods that injects code
+                into the model
+        """
+        residual = hidden_states
+        hidden_states = hidden_states.to(self.input_layernorm.weight.dtype)
+        hidden_states = self.input_layernorm(hidden_states)
+        hidden_states = hidden_states.to(self.self_attn.q_proj.weight.dtype)
+        # Self Attention
+        hidden_states, self_attn_weights, present_key_value = self.self_attn(
+            hidden_states=hidden_states,
+            attention_mask=attention_mask,
+            position_ids=position_ids,
+            past_key_value=past_key_value,
+            output_attentions=output_attentions,
+            use_cache=use_cache,
+            cache_position=cache_position,
+            position_embeddings=position_embeddings,
+        )
+        hidden_states = residual + hidden_states
+
+        # Fully Connected
+        residual = hidden_states
+        hidden_states = hidden_states.to(self.post_attention_layernorm.weight.dtype)
+        hidden_states = self.post_attention_layernorm(hidden_states)
+        if self.mlp is None:  # using moe mlp
+            hidden_states = hidden_states.to(self.moe.experts[0].down_proj.weight.dtype)
+            hidden_states = self.moe(hidden_states, token_types)
+        else:
+            hidden_states = hidden_states.to(self.mlp.down_proj.weight.dtype)
+            hidden_states = self.mlp(hidden_states)
+
+        hidden_states = residual + hidden_states
+
+        outputs = (hidden_states,)
+
+        if output_attentions:
+            outputs += (self_attn_weights,)
+        if use_cache:
+            outputs += (present_key_value,)
+        return outputs
+
+
+class Qwen2_5_VLMoEModel(Qwen2_5_VLPreTrainedModel):
+    """Qwen2.5-VL model with Mixture of Experts (MoE) architecture.
+
+    This model extends the base Qwen2.5-VL model by incorporating MoE layers
+    for improved scalability and specialization across different token types.
+    """
+
+    @classmethod
+    def from_pretrained(
+        cls,
+        pretrained_model_name_or_path: str,
+        num_experts: int | None = None,
+        *args,
+        **kwargs,
+    ):
+        """Load a pretrained model with optional MoE configuration.
+
+        Args:
+            pretrained_model_name_or_path: Path or name of the pretrained model
+            num_experts: Number of experts for MoE layers (if not in config)
+            *args: Additional arguments passed to parent class
+            **kwargs: Additional keyword arguments passed to parent class
+
+        Returns:
+            Initialized model instance with MoE configuration
+        """
+        config = kwargs.get("config")
+        if config is None:
+            config = AutoConfig.from_pretrained(pretrained_model_name_or_path)
+
+        # Override number of experts if specified
+        if num_experts is not None:
+            config.num_experts = num_experts
+
+        kwargs["config"] = config
+        return super().from_pretrained(pretrained_model_name_or_path, *args, **kwargs)
+
+    def __init__(self, config: Qwen2_5_VLConfig):
+        """Initialize the Qwen2.5-VL MoE model.
+
+        Args:
+            config: Model configuration containing architecture parameters
+        """
+        super().__init__(config)
+
+        # Basic model parameters
+        self.padding_idx = config.pad_token_id
+        self.vocab_size = config.vocab_size
+
+        # Model components
+        self.embed_tokens = nn.Embedding(config.vocab_size, config.hidden_size, self.padding_idx)
+
+        # Decoder layers with MoE support
+        self.layers = nn.ModuleList(
+            [
+                Qwen2_5_VLDecoderLayer_with_MoE(config, layer_idx, config.num_experts)
+                for layer_idx in range(config.num_hidden_layers)
+            ]
+        )
+
+        # Model configuration
+        self._attn_implementation = config._attn_implementation
+        self.norm = Qwen2RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
+        self.rotary_emb = Qwen2_5_VLRotaryEmbedding(config=config)
+        self.gradient_checkpointing = False
+
+        # Initialize weights and apply final processing
+        self.post_init()
+
+    def get_input_embeddings(self) -> nn.Embedding:
+        """Get the input embedding layer.
+
+        Returns:
+            The token embedding layer
+        """
+        return self.embed_tokens
+
+    def set_input_embeddings(self, value: nn.Embedding) -> None:
+        """Set the input embedding layer.
+
+        Args:
+            value: New embedding layer to use
+        """
+        self.embed_tokens = value
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor | None = None,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        moe_token_types: torch.LongTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+        cache_position: torch.LongTensor | None = None,
+        **kwargs,
+    ) -> tuple | BaseModelOutputWithPast:
+        # Set default output options
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        # Validate inputs
+        if (input_ids is None) ^ (inputs_embeds is not None):
+            raise ValueError("You must specify exactly one of input_ids or inputs_embeds")
+
+        if moe_token_types is None:
+            raise ValueError("moe_token_types must be provided for MoE routing")
+
+        # Handle gradient checkpointing compatibility
+        if self.gradient_checkpointing and self.training:
+            if use_cache:
+                logger.warning_once(
+                    "`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`..."
+                )
+                use_cache = False
+
+        # Initialize cache if needed
+        if use_cache and past_key_values is None and not torch.jit.is_tracing():
+            past_key_values = DynamicCache()
+
+        # Get input embeddings
+        if inputs_embeds is None:
+            inputs_embeds = self.embed_tokens(input_ids)
+
+        # Set up cache position
+        if cache_position is None:
+            past_seen_tokens = past_key_values.get_seq_length() if past_key_values is not None else 0
+            cache_position = torch.arange(
+                past_seen_tokens,
+                past_seen_tokens + inputs_embeds.shape[1],
+                device=inputs_embeds.device,
+            )
+
+        # Set up position IDs (hardcoded 3 dimensions for temporal, height, width)
+        if position_ids is None:
+            position_ids = cache_position.view(1, 1, -1).expand(3, inputs_embeds.shape[0], -1)
+        elif position_ids.dim() == 2:
+            position_ids = position_ids[None, ...].expand(3, position_ids.shape[0], -1)
+
+        # Create causal attention mask
+        causal_mask = self._update_causal_mask(
+            attention_mask,
+            inputs_embeds,
+            cache_position,
+            past_key_values,
+            output_attentions,
+            moe_token_types,
+        )
+
+        hidden_states = inputs_embeds
+
+        # Create position embeddings to be shared across decoder layers
+        position_embeddings = self.rotary_emb(hidden_states, position_ids)
+
+        # Initialize output collections
+        all_hidden_states = () if output_hidden_states else None
+        all_self_attns = () if output_attentions else None
+        next_decoder_cache = None
+
+        # Process through decoder layers
+        for decoder_layer in self.layers:
+            if output_hidden_states:
+                all_hidden_states += (hidden_states,)
+
+            if self.gradient_checkpointing and self.training:
+                # Use gradient checkpointing during training
+                layer_outputs = self._gradient_checkpointing_func(
+                    decoder_layer.__call__,
+                    hidden_states,
+                    causal_mask,
+                    position_ids,
+                    past_key_values,
+                    moe_token_types,
+                    output_attentions,
+                    use_cache,
+                    cache_position,
+                    position_embeddings,
+                )
+            else:
+                # Regular forward pass
+                layer_outputs = decoder_layer(
+                    hidden_states,
+                    attention_mask=causal_mask,
+                    position_ids=position_ids,
+                    past_key_value=past_key_values,
+                    token_types=moe_token_types,
+                    output_attentions=output_attentions,
+                    use_cache=use_cache,
+                    cache_position=cache_position,
+                    position_embeddings=position_embeddings,
+                )
+
+            hidden_states = layer_outputs[0]
+
+            # Update cache if using it
+            if use_cache:
+                next_decoder_cache = layer_outputs[2 if output_attentions else 1]
+
+            # Collect attention weights if requested
+            if output_attentions:
+                all_self_attns += (layer_outputs[1],)
+
+        # Apply final layer normalization
+        hidden_states = self.norm(hidden_states)
+
+        # Add final hidden states if collecting all states
+        if output_hidden_states:
+            all_hidden_states += (hidden_states,)
+
+        next_cache = next_decoder_cache if use_cache else None
+
+        # Return outputs in requested format
+        if not return_dict:
+            return tuple(
+                v for v in [hidden_states, next_cache, all_hidden_states, all_self_attns] if v is not None
+            )
+
+        return BaseModelOutputWithPast(
+            last_hidden_state=hidden_states,
+            past_key_values=next_cache,
+            hidden_states=all_hidden_states,
+            attentions=all_self_attns,
+        )
+
+    def _update_causal_mask(
+        self,
+        attention_mask: torch.Tensor,
+        input_tensor: torch.Tensor,
+        cache_position: torch.Tensor,
+        past_key_values: Cache,
+        output_attentions: bool,
+        moe_token_types: torch.LongTensor | None = None,
+    ):
+        """Update causal attention mask with support for bidirectional attention for specific token types.
+
+        This method creates and modifies attention masks to support different attention patterns:
+        - Standard causal (unidirectional) attention for most tokens
+        - Bidirectional attention for specific token types (e.g., MoE routing tokens)
+
+        Args:
+            attention_mask: Input attention mask to avoid attending to padding tokens
+            input_tensor: Input embeddings tensor for shape and device information
+            cache_position: Position indices for caching mechanisms
+            past_key_values: Cached key-value pairs from previous forward passes
+            output_attentions: Whether attention weights will be returned
+            moe_token_types: Optional tensor indicating token types for MoE routing
+                            (type 1 tokens will use bidirectional attention)
+
+        Returns:
+            Updated causal attention mask, or None if using Flash Attention 2
+        """
+        # Flash Attention 2 handles masking internally
+        if self.config._attn_implementation == "flash_attention_2":
+            return None
+
+        # Calculate sequence lengths for cache management
+        past_seen_tokens = past_key_values.get_seq_length() if past_key_values is not None else 0
+        using_static_cache = isinstance(past_key_values, StaticCache)
+        using_sliding_window_cache = isinstance(past_key_values, SlidingWindowCache)
+
+        # For SDPA (Scaled Dot Product Attention), use `is_causal` argument when possible
+        # instead of explicit attention mask to enable Flash Attention 2 dispatch
+        # Note: This optimization is not compatible with static cache
+        if (
+            self.config._attn_implementation == "sdpa"
+            and not (using_static_cache or using_sliding_window_cache)
+            and not output_attentions
+        ):
+            # Check if we can ignore the causal mask and rely on SDPA's internal handling
+            if AttentionMaskConverter._ignore_causal_mask_sdpa(
+                attention_mask,
+                inputs_embeds=input_tensor,
+                past_key_values_length=past_seen_tokens,
+                sliding_window=self.config.sliding_window,
+                is_training=self.training,
+            ):
+                return None
+
+        # Extract tensor properties for mask creation
+        dtype, device = input_tensor.dtype, input_tensor.device
+        min_dtype = torch.finfo(dtype).min
+        sequence_length = input_tensor.shape[1]
+
+        # Determine target length based on cache type
+        if using_sliding_window_cache or using_static_cache:
+            # Use maximum cache shape for sliding window or static caches
+            target_length = past_key_values.get_max_cache_shape()
+        else:
+            # For dynamic cache or no cache, calculate based on attention mask or sequence length
+            target_length = (
+                attention_mask.shape[-1]
+                if isinstance(attention_mask, torch.Tensor)
+                else past_seen_tokens + sequence_length + 1
+            )
+
+        # Generate 4D causal attention mask from 2D input mask if provided
+        causal_mask = self._prepare_4d_causal_attention_mask_with_cache_position(
+            attention_mask,
+            sequence_length=sequence_length,
+            target_length=target_length,
+            dtype=dtype,
+            device=device,
+            cache_position=cache_position,
+            batch_size=input_tensor.shape[0],
+            config=self.config,
+            past_key_values=past_key_values,
+        )
+
+        # Modify mask to support bidirectional attention for specific token types
+        if moe_token_types is not None:
+            # Identify positions of type 1 tokens (MoE routing tokens)
+            type1_tokens = (moe_token_types == 1).unsqueeze(1).unsqueeze(2)  # Shape: [B, 1, 1, S]
+
+            # Create bidirectional attention region for type 1 tokens
+            # This allows type 1 tokens to attend to each other bidirectionally
+            type1_mask = torch.zeros_like(causal_mask)  # Shape: [B, num_heads, S, S]
+            type1_region = type1_tokens & type1_tokens.transpose(-1, -2)  # Shape: [B, 1, S, S]
+            type1_mask = type1_mask.masked_fill(type1_region, 1.0).to(torch.bool)
+
+            # Apply bidirectional attention: zero out causal constraints in type 1 regions
+            causal_mask = torch.where(
+                type1_mask,  # Where type 1 tokens interact with each other
+                torch.zeros_like(causal_mask),  # Remove causal masking (allow bidirectional)
+                causal_mask,  # Keep original causal masking for other regions
+            )
+
+        # Handle special case for SDPA with CUDA/XPU devices
+        if (
+            self.config._attn_implementation == "sdpa"
+            and attention_mask is not None
+            and attention_mask.device.type in ["cuda", "xpu"]
+            and not output_attentions
+        ):
+            # Ensure attention to all tokens in fully masked rows for memory-efficient attention
+            # This is required for F.scaled_dot_product_attention's memory-efficient path
+            # when using left padding. See: https://github.com/pytorch/pytorch/issues/110213
+            causal_mask = AttentionMaskConverter._unmask_unattended(causal_mask, min_dtype)
+
+        return causal_mask
+
+    @staticmethod
+    def _prepare_4d_causal_attention_mask_with_cache_position(
+        attention_mask: torch.Tensor,
+        sequence_length: int,
+        target_length: int,
+        dtype: torch.dtype,
+        device: torch.device,
+        cache_position: torch.Tensor,
+        batch_size: int,
+        config: Qwen2_5_VLConfig,
+        past_key_values: Cache,
+    ):
+        """
+        Creates a causal 4D mask of shape `(batch_size, 1, query_length, key_value_length)` from a 2D mask of shape
+        `(batch_size, key_value_length)`, or if the input `attention_mask` is already 4D, do nothing.
+
+        Args:
+            attention_mask (`torch.Tensor`):
+                A 2D attention mask of shape `(batch_size, key_value_length)` or a 4D attention mask of shape `(batch_size, 1, query_length, key_value_length)`.
+            sequence_length (`int`):
+                The sequence length being processed.
+            target_length (`int`):
+                The target length: when generating with static cache, the mask should be as long as the static cache, to account for the 0 padding, the part of the cache that is not filled yet.
+            dtype (`torch.dtype`):
+                The dtype to use for the 4D attention mask.
+            device (`torch.device`):
+                The device to place the 4D attention mask on.
+            cache_position (`torch.Tensor`):
+                Indices depicting the position of the input sequence tokens in the sequence.
+            batch_size (`torch.Tensor`):
+                Batch size.
+            config (`Qwen2_5_VLConfig`):
+                The model's configuration class
+            past_key_values (`Cache`):
+                The cache class that is being used currently to generate
+        """
+        if attention_mask is not None and attention_mask.dim() == 4:
+            # In this case we assume that the mask comes already in inverted form and requires no inversion or slicing.
+            causal_mask = attention_mask
+        else:
+            min_dtype = torch.finfo(dtype).min
+            causal_mask = torch.full(
+                (sequence_length, target_length),
+                fill_value=min_dtype,
+                dtype=dtype,
+                device=device,
+            )
+            diagonal_attend_mask = torch.arange(target_length, device=device) > cache_position.reshape(-1, 1)
+            if config.sliding_window is not None:
+                # if we have sliding window, we should not attend to tokens beyond sliding window length, so we mask them out also
+                # the check is needed to verify is current checkpoint was trained with sliding window or not
+                if not isinstance(past_key_values, SlidingWindowCache) or sequence_length > target_length:
+                    sliding_attend_mask = torch.arange(target_length, device=device) <= (
+                        cache_position.reshape(-1, 1) - config.sliding_window
+                    )
+                    diagonal_attend_mask.bitwise_or_(sliding_attend_mask)
+            causal_mask *= diagonal_attend_mask
+            causal_mask = causal_mask[None, None, :, :].expand(batch_size, 1, -1, -1)
+            if attention_mask is not None:
+                causal_mask = causal_mask.clone()  # copy to contiguous memory for in-place edit
+                if attention_mask.shape[-1] > target_length:
+                    attention_mask = attention_mask[:, :target_length]
+                mask_length = attention_mask.shape[-1]
+                padding_mask = causal_mask[:, :, :, :mask_length] + attention_mask[:, None, None, :].to(
+                    causal_mask.device
+                )
+                padding_mask = padding_mask == 0
+                causal_mask[:, :, :, :mask_length] = causal_mask[:, :, :, :mask_length].masked_fill(
+                    padding_mask, min_dtype
+                )
+        return causal_mask
+
+
+__all__ = [
+    "Qwen2_5_VLForConditionalGeneration",
+    "Qwen2_5_VLModel",
+    "Qwen2_5_VLPreTrainedModel",
+    "Qwen2_5_VLDecoderLayer_with_MoE",
+    "Qwen2_5_VLMoEModel",
+]
diff --git a/lerobot/src/lerobot/policies/wall_x/utils.py b/lerobot/src/lerobot/policies/wall_x/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..e08ef69d51bb41eec60ca388c0b4d9cbe3ea853d
--- /dev/null
+++ b/lerobot/src/lerobot/policies/wall_x/utils.py
@@ -0,0 +1,631 @@
+#!/usr/bin/env python
+
+# Copyright 2025 HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Wall-X Utility Functions.
+
+Contains data processing utilities, text formatting functions, and helper classes
+for the Wall-X cross-embodiment robotic control model.
+"""
+
+import random
+import re
+from collections import OrderedDict
+from dataclasses import dataclass, field
+from typing import Any
+
+import torch
+from transformers import BatchFeature
+
+from lerobot.policies.wall_x.constant import (
+    CAMERA_NAME_MAPPING,
+)
+from lerobot.utils.constants import OBS_IMAGES
+
+
+@dataclass
+class X2RDataProcessingConfig:
+    """Configuration class for X2R data processing pipeline.
+
+    This class contains all the necessary parameters for processing robotic data
+    including camera mappings, tactile sensor configurations, action predictions,
+    and various processing options.
+    """
+
+    # Action prediction configuration
+    predict_action_keys: list[str] = field(default_factory=list)
+    obs_action_keys: list[str] = field(default_factory=list)
+
+    # Image resolution settings for different views
+    resolution: dict[str, int] = field(
+        default_factory=lambda: {
+            "face_view": -1,
+            "left_wrist_view": 128,
+            "right_wrist_view": 128,
+        }
+    )
+
+    # Dataset splitting
+    train_test_split: float = 0.9
+    split_seed: int = 42
+
+    # Instruction handling
+    priority_order: dict[str, float] | None = None
+
+    # Vision model parameters
+    model_type: str = "qwen2_5"
+    max_pixels: int = 16384 * 28 * 28
+    min_pixels: int = 4 * 28 * 28
+    image_factor: int = 28
+
+    generate_subtask_ratio: float = 0.0
+
+    def __post_init__(self):
+        """Post-initialization validation and setup."""
+        # Validate train/test split
+        if not 0 < self.train_test_split < 1:
+            raise ValueError(f"train_test_split must be between 0 and 1, got {self.train_test_split}")
+
+    def as_dict(self) -> dict:
+        """Convert configuration to dictionary format.
+
+        Returns:
+            Dict: Configuration as dictionary
+        """
+        return self.__dict__
+
+    def update(self, **kwargs) -> "X2RDataProcessingConfig":
+        """Update configuration parameters.
+
+        Args:
+            **kwargs: Key-value pairs to update
+
+        Returns:
+            X2RDataProcessingConfig: Updated configuration instance
+        """
+        for key, value in kwargs.items():
+            if hasattr(self, key):
+                setattr(self, key, value)
+            else:
+                raise ValueError(f"Unknown configuration parameter: {key}")
+        return self
+
+
+def preprocesser_call(
+    processor,
+    images: list | Any | None = None,
+    text: str | list[str] | None = None,
+    videos: list | Any | None = None,
+    padding: bool | str = False,
+    truncation: bool | None = None,
+    max_length: int | None = None,
+    return_tensors: str = "pt",
+) -> BatchFeature:
+    """Unified preprocessing function for Wall-X model handling text, image and video inputs.
+
+    Processes inputs into format suitable for multimodal transformer models, including:
+    - Text tokenization and special token handling
+    - Image/video processing through image processor
+    - Attention mask and label generation
+    - Padding and truncation handling
+
+    Args:
+        processor: Multimodal processor containing tokenizer and image processor
+        images: Input images (PIL, numpy arrays, or torch tensors)
+        text: Text or list of texts to tokenize
+        videos: Input videos (numpy arrays or torch tensors)
+        padding: Whether to pad sequences to same length
+        truncation: Whether to truncate sequences longer than max_length
+        max_length: Maximum length for truncation/padding
+        return_tensors: Format for returned tensors ('pt', 'np', etc.)
+
+    Returns:
+        BatchFeature containing processed inputs with keys:
+        - input_ids: Tokenized text
+        - attention_mask: Attention mask for text
+        - pixel_values: Processed image pixels
+        - pixel_values_videos: Processed video frames
+        - image_grid_thw: Image grid dimensions for LLM
+        - video_grid_thw: Video grid dimensions for LLM
+        - labels: Training labels with masking
+    """
+    # Process image inputs
+    if images is not None and len(images) > 0:
+        image_inputs = processor.image_processor(images=images, return_tensors=return_tensors)
+        image_grid_thw = image_inputs["image_grid_thw"]
+    else:
+        image_inputs = {}
+        image_grid_thw = None
+
+    # Process video inputs
+    if videos is not None:
+        videos_inputs = processor.image_processor(videos=videos, return_tensors=return_tensors)
+        video_grid_thw = videos_inputs["video_grid_thw"]
+    else:
+        videos_inputs = {}
+        video_grid_thw = None
+
+    # Ensure text input is in list format
+    if not isinstance(text, list):
+        text = [text]
+
+    # Process image placeholder tokens in text
+    if image_grid_thw is not None:
+        merge_length = processor.image_processor.merge_size**2
+        index = 0
+        for i in range(len(text)):
+            while "<|image_pad|>" in text[i]:
+                # Add bounds checking to avoid index overflow
+                if index >= len(image_grid_thw):
+                    print(
+                        f"Warning: Number of image placeholders ({index + 1}) "
+                        f"exceeds actual images ({len(image_grid_thw)}), "
+                        f"skipping remaining placeholder processing"
+                    )
+                    break
+                # Replace image placeholder with actual token count
+                token_count = image_grid_thw[index].prod() // merge_length
+                text[i] = text[i].replace("<|image_pad|>", "<|placeholder|>" * token_count, 1)
+                index += 1
+            text[i] = text[i].replace("<|placeholder|>", "<|image_pad|>")
+
+    # Process video placeholder tokens in text
+    if video_grid_thw is not None:
+        merge_length = processor.image_processor.merge_size**2
+        index = 0
+        for i in range(len(text)):
+            while "<|video_pad|>" in text[i]:
+                # Replace video placeholder with actual token count
+                token_count = video_grid_thw[index].prod() // merge_length
+                text[i] = text[i].replace("<|video_pad|>", "<|placeholder|>" * token_count, 1)
+                index += 1
+            text[i] = text[i].replace("<|placeholder|>", "<|video_pad|>")
+
+    # Tokenize complete input text
+    text_inputs = processor.tokenizer(
+        text,
+        return_tensors=return_tensors,
+        padding=padding,
+        truncation=truncation,
+        max_length=max_length,
+    )
+
+    # Get pad token ID for label generation
+    pad_token_id = processor.tokenizer.pad_token_id
+    if pad_token_id is None:
+        pad_token_id = processor.tokenizer.eos_token_id
+
+    # Generate labels for multi-turn dialogue, keeping only assistant response loss
+    labels = torch.full_like(text_inputs.input_ids, -100)
+    assistant_marker = "<|im_start|>assistant\n"
+    im_end_token_id = processor.tokenizer.convert_tokens_to_ids("<|im_end|>")
+    assistant_tokens = processor.tokenizer("<|im_start|>assistant\n", add_special_tokens=False).input_ids
+
+    for i in range(len(text)):
+        assistant_regions = []
+        parts = text[i].split(assistant_marker)
+
+        # Process each part to determine which tokens belong to assistant responses
+        # Count left padding tokens
+        num_left_pads = 0
+        for token_id in text_inputs.input_ids[i]:
+            if token_id == pad_token_id:
+                num_left_pads += 1
+            else:
+                break
+        current_pos = num_left_pads
+
+        for j, part in enumerate(parts):
+            part_tokens = processor.tokenizer(part, add_special_tokens=False).input_ids
+            if j == 0:
+                # First part is system prompt or user question, all labels are -100
+                current_pos += len(part_tokens)
+                continue
+
+            # From second part onwards, each part starts with assistant response
+            for k in range(current_pos + 1, len(text_inputs.input_ids[i])):
+                if text_inputs.input_ids[i][k] == im_end_token_id:
+                    assistant_regions.append((current_pos + len(assistant_tokens), k + 2))
+                    break
+            current_pos += len(part_tokens) + 3
+
+        # Set labels for assistant response regions
+        for start, end in assistant_regions:
+            labels[i][start:end] = text_inputs.input_ids[i][start:end]
+
+    # Mask special action tokens in labels
+    action_token_id = processor.tokenizer.encode("<|action|>")[0]
+    propri_token_id = processor.tokenizer.encode("<|propri|>")[0]
+    labels[labels == action_token_id] = -100
+    labels[labels == propri_token_id] = -100
+    labels[labels == processor.tokenizer.pad_token_id] = -100
+
+    # Set labels to None if all are invalid to skip cross entropy loss
+    if (labels != -100).any().item():
+        text_inputs["labels"] = labels
+    else:
+        text_inputs["labels"] = None
+
+    return BatchFeature(data={**text_inputs, **image_inputs, **videos_inputs})
+
+
+def process_grounding_points(
+    text: str,
+    orig_height: int,
+    orig_width: int,
+    resized_height: int,
+    resized_width: int,
+    model_type: str,
+) -> str:
+    """Process grounding point coordinates in text based on image resizing.
+
+    Adjusts coordinate values in <point> tags to match resized image dimensions
+    for different model types (qwen2, qwen2_5).
+
+    Args:
+        text: Input text containing <point> tags with coordinates
+        orig_height: Original image height
+        orig_width: Original image width
+        resized_height: Resized image height
+        resized_width: Resized image width
+        model_type: Model type for coordinate processing ('qwen2' or 'qwen2_5')
+
+    Returns:
+        Text with adjusted coordinate values
+    """
+    # Regex pattern to match <point> tags and their contents
+    point_pattern = re.compile(r"<point>(.*?)</point>")
+
+    def process_match(match):
+        """Process a single point match and adjust coordinates."""
+        coords_str = match.group(1)
+        try:
+            # Extract coordinates from string
+            coords = list(map(int, re.findall(r"\d+", coords_str)))
+
+            # Calculate resize scale factors
+            scale_w = resized_width / orig_width
+            scale_h = resized_height / orig_height
+
+            if len(coords) == 2:
+                x, y = coords
+                if model_type == "qwen2_5":
+                    # Qwen2.5 uses pixel coordinates
+                    new_x = max(0, min(round(x * scale_w), resized_width - 1))
+                    new_y = max(0, min(round(y * scale_h), resized_height - 1))
+                elif model_type == "qwen2":
+                    # Qwen2 normalizes to [0, 1000) range
+                    new_x = max(0, min(999.999, (x / orig_width) * 1000))
+                    new_y = max(0, min(999.999, (y / orig_height) * 1000))
+                else:
+                    raise ValueError(f"Unsupported model type: {model_type}")
+                coords = [new_x, new_y]
+
+            elif len(coords) == 4:
+                x1, y1, x2, y2 = coords
+                if model_type == "qwen2_5":
+                    new_x1 = max(0, min(round(x1 * scale_w), resized_width - 1))
+                    new_y1 = max(0, min(round(y1 * scale_h), resized_height - 1))
+                    new_x2 = max(0, min(round(x2 * scale_w), resized_width - 1))
+                    new_y2 = max(0, min(round(y2 * scale_h), resized_height - 1))
+                elif model_type == "qwen2":
+                    new_x1 = max(0, min(999.999, (x1 / orig_width) * 1000))
+                    new_y1 = max(0, min(999.999, (y1 / orig_height) * 1000))
+                    new_x2 = max(0, min(999.999, (x2 / orig_width) * 1000))
+                    new_y2 = max(0, min(999.999, (y2 / orig_height) * 1000))
+                else:
+                    raise ValueError(f"Unsupported model type: {model_type}")
+                coords = [new_x1, new_y1, new_x2, new_y2]
+
+            # Return processed point tag
+            return f"<point>[{', '.join(map(str, coords))}]</point>"
+
+        except (ValueError, TypeError):
+            # Return original content if processing fails
+            return match.group(0)
+
+    # Replace all matching point tags
+    processed_text = point_pattern.sub(process_match, text)
+    return processed_text
+
+
+def get_frame_instruction(
+    instruction_info: dict[str, Any],
+    frame_idx: int | None = None,
+    truncate_keys: list[str] | None = None,
+) -> tuple[dict[str, Any], int | None]:
+    """Extract frame-specific instruction from instruction dictionary.
+
+    Args:
+        instruction_info: Dictionary containing instruction components
+        frame_idx: Current frame index
+        truncate_keys: Keys that trigger truncation when found
+
+    Returns:
+        Tuple of (frame_instruction_dict, split_end_frame)
+    """
+    if truncate_keys is None:
+        truncate_keys = [
+            "subtask_generation",
+            "distribute",
+            "subtask_generation_zh",
+            "distribute_zh",
+        ]
+
+    instruction_for_frame = {}
+    split_end = None
+
+    for key, value in instruction_info.items():
+        if isinstance(value, dict):
+            # Handle frame-range specific instructions
+            for frame_range, frame_instruction in value.items():
+                start_frame, end_frame = map(int, frame_range.split(" "))
+                if start_frame <= frame_idx < end_frame or (start_frame == frame_idx):
+                    instruction_for_frame[key] = frame_instruction
+                    if truncate_keys is not None and split_end is None and key in truncate_keys:
+                        split_end = end_frame + 1
+                    break
+        else:
+            instruction_for_frame[key] = value
+
+    return instruction_for_frame, split_end
+
+
+def get_task_instruction(
+    frame_instruction_info: dict[str, Any], priority_order: OrderedDict | None = None
+) -> str:
+    """Construct task instruction from available instruction fields using priority sampling.
+
+    Args:
+        frame_instruction_info: Dictionary containing instruction fields
+        priority_order: OrderedDict specifying sampling probability for each field
+
+    Returns:
+        Combined instruction string with priority components
+    """
+    # Default priority settings
+    default_priority_order = OrderedDict(
+        {
+            "subtask_generation": 0.25,
+            "subtask_generation_zh": 0.25,
+            "distribute": 0.25,
+            "distribute_zh": 0.25,
+        }
+    )
+
+    if priority_order is not None:
+        priority_order = OrderedDict(priority_order)
+    else:
+        priority_order = default_priority_order
+
+    got_instruction = False
+    task_instruction = ""
+
+    # Sample instruction components based on priority probabilities
+    for key, prob in priority_order.items():
+        if key in frame_instruction_info and frame_instruction_info[key] != "":
+            if got_instruction:
+                if random.random() >= prob:
+                    continue
+
+            task_instruction += f"\n{frame_instruction_info[key]}"
+            got_instruction = True
+            break
+
+    # Fall back to base instruction if no priority components found
+    if not got_instruction:
+        task_instruction = frame_instruction_info.get("instruction", "")
+
+    return task_instruction
+
+
+def get_wallx_normal_text(
+    instruction_info: dict[str, Any],
+    action_chunk_size: int,
+    frame_idx: int,
+    priority_order: OrderedDict | None = None,
+    img_keys: list[str] | None = None,
+    generate_subtask_ratio: float = 0.0,
+) -> tuple[str, bool]:
+    """Construct complete multimodal prompt text for Wall-X model.
+
+    Formats input using special tokens including:
+    - System message
+    - User observations (with image placeholders)
+    - Task instructions
+    - Proprioception prompts
+    - Assistant responses (with action tokens)
+
+    Args:
+        instruction_info: Dictionary containing instruction components
+        action_chunk_size: Number of action tokens to generate
+        frame_idx: Current frame index
+        priority_order: Priority order for instruction sampling
+        img_keys: List of image keys
+        generate_subtask_ratio: Probability of generating subtask instead of actions
+
+    Returns:
+        Tuple of (formatted_prompt_text, is_subtask_generation)
+    """
+    # Special tokens for formatting
+    role_start_symbol = "<|im_start|>"
+    role_end_symbol = "<|im_end|>"
+    vision_start_symbol = "<|vision_start|>"
+    vision_end_symbol = "<|vision_end|>"
+    image_pad_symbol = "<|image_pad|>"
+    propri_symbol = "<|propri|>"
+    action_symbol = "<|action|>"
+    action_fast_symbol = "<|action_fast|>"
+
+    # System prologue
+    prologue = f"{role_start_symbol}system\nYou are a helpful assistant.{role_end_symbol}\n"
+
+    # User request with observation
+    user_request = f"{role_start_symbol}user\nObservation:"
+    if img_keys:
+        img_keys = img_key_mapping(img_keys)
+        for key in img_keys:
+            user_request += f" {key}: {vision_start_symbol}{image_pad_symbol}{vision_end_symbol}"
+    user_request += "\nInstruction:"
+
+    # Get frame-specific instruction
+    frame_instruction_info, _ = get_frame_instruction(instruction_info, frame_idx=frame_idx)
+
+    generate_subtask = False
+    priority_keys = ["subtask_generation", "distribute"]
+
+    # Decide whether to generate subtask or actions
+    if (
+        bool(set(frame_instruction_info.keys()) & set(priority_keys))
+        and random.random() < generate_subtask_ratio
+    ):
+        # Generate subtask (equivalent to VQA task)
+        instruction = frame_instruction_info.get("instruction", "")
+        text_prompt = "\nPredict the next action in language.\n"
+        user_message = f"{user_request} {instruction}{text_prompt}{role_end_symbol}\n"
+
+        # Find output instruction from priority keys
+        for key in priority_keys:
+            if key in frame_instruction_info:
+                output_instruction = frame_instruction_info[key]
+                break
+
+        assistant_output = f"{role_start_symbol}assistant\n{output_instruction}\n{role_end_symbol}"
+        generate_subtask = True
+    else:
+        # Generate actions
+        instruction = get_task_instruction(frame_instruction_info, priority_order=priority_order)
+        text_prompt = f"\nPredict the next action in robot action.\nProprioception: {propri_symbol}\n"
+        user_message = f"{user_request} {instruction}{text_prompt}{role_end_symbol}\n"
+        assistant_output = f"{role_start_symbol}assistant\n{action_fast_symbol}{role_end_symbol}\n{action_symbol * action_chunk_size}"
+
+    complete_text = prologue + user_message + assistant_output
+    return complete_text, generate_subtask
+
+
+def img_key_mapping(img_keys: list[str]) -> list[str]:
+    """Map image keys to camera names.
+
+    Args:
+        img_keys: List of image keys
+
+    Returns:
+        List of camera names
+    """
+    processed_img_keys = []
+    for key in img_keys:
+        key = key.replace(OBS_IMAGES + ".", "")
+        if key in CAMERA_NAME_MAPPING:
+            key = CAMERA_NAME_MAPPING[key]
+        else:
+            if "view" in key:
+                key = key.replace("_", " ")
+            else:
+                key = key + " view"
+        processed_img_keys.append(key)
+    return processed_img_keys
+
+
+def get_action_tokens(normalized_actions: torch.Tensor | list, action_tokenizer) -> list[list[str]]:
+    """Convert normalized actions to action token strings.
+
+    Args:
+        normalized_actions: Normalized action arrays/tensors
+        action_tokenizer: Tokenizer for converting actions to tokens
+
+    Returns:
+        List of action token string lists for each sample
+    """
+    if isinstance(normalized_actions, torch.Tensor):
+        normalized_actions = normalized_actions.cpu().numpy()
+
+    all_action_tokens = []
+    for i in range(len(normalized_actions)):
+        if isinstance(normalized_actions[i], torch.Tensor):
+            normalized_actions[i] = normalized_actions[i].cpu().numpy()
+
+        token_id = action_tokenizer(normalized_actions[i])
+        action_tokens = [f"<|action_token_{j}|>" for j in token_id[0]]
+        all_action_tokens.append(action_tokens)
+
+    return all_action_tokens
+
+
+def pad_action_token_strs(
+    actions_token_lists: list[list[str]],
+    pad_token: str = "<|endoftext|>",  # nosec B107
+) -> list[str]:
+    """Pad action token lists to same length and join as strings.
+
+    Args:
+        actions_token_lists: List of action token lists for each sample
+        pad_token: Token used for padding
+
+    Returns:
+        List of padded action token strings
+    """
+    max_len = max(len(tokens) for tokens in actions_token_lists)
+    padded_action_strs = []
+
+    for tokens in actions_token_lists:
+        padded_tokens = tokens + ["<|im_end|>\n"] + [pad_token] * (max_len - len(tokens))
+        padded_action_strs.append("".join(padded_tokens))
+
+    return padded_action_strs
+
+
+def replace_action_token(
+    text: list[str],
+    norm_action: torch.Tensor | None,
+    action_tokenizer,
+    dof_masks: torch.Tensor | None = None,
+) -> list[str]:
+    """Replace action placeholders in text with actual action tokens.
+
+    Args:
+        text: List of text strings with action placeholders
+        norm_action: Normalized action tensors
+        action_tokenizer: Tokenizer for converting actions to tokens
+        dof_masks: Masks for degrees of freedom
+
+    Returns:
+        List of text strings with action tokens replaced
+    """
+    if action_tokenizer is not None and norm_action is not None:
+        # Extract actions based on chunk sizes and DOF masks
+        norm_action = [action[:32, dof_masks[i, 0].bool()] for i, action in enumerate(norm_action)]
+
+        # Convert to action tokens and pad
+        actions_fast_tokens = get_action_tokens(norm_action, action_tokenizer)
+        actions_fast_token_strs = pad_action_token_strs(actions_fast_tokens)
+
+        # Replace action placeholders with actual tokens
+        actions_fast_token_idx = 0
+        for i in range(len(text)):
+            if "<|action_fast|>" in text[i]:
+                text[i] = text[i].replace(
+                    "<|action_fast|><|im_end|>\n",
+                    actions_fast_token_strs[actions_fast_token_idx],
+                )
+                actions_fast_token_idx += 1
+
+        # Remove remaining action placeholders
+        text = [t.replace("<|action|>", "") for t in text]
+    else:
+        # Remove action placeholders when no tokenizer available
+        text = [t.replace("<|action_fast|><|im_end|>\n", "") for t in text]
+
+    return text
diff --git a/lerobot/src/lerobot/policies/xvla/__init__.py b/lerobot/src/lerobot/policies/xvla/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..71b04e76fc18e32ca3abe7b67ea4e0edff71f14e
--- /dev/null
+++ b/lerobot/src/lerobot/policies/xvla/__init__.py
@@ -0,0 +1,6 @@
+# register the processor steps
+from lerobot.policies.xvla.processor_xvla import (
+    XVLAAddDomainIdProcessorStep,
+    XVLAImageNetNormalizeProcessorStep,
+    XVLAImageToFloatProcessorStep,
+)
diff --git a/lerobot/src/lerobot/policies/xvla/action_hub.py b/lerobot/src/lerobot/policies/xvla/action_hub.py
new file mode 100644
index 0000000000000000000000000000000000000000..e8411de9dec17ef754617a0b4d79ed504255361e
--- /dev/null
+++ b/lerobot/src/lerobot/policies/xvla/action_hub.py
@@ -0,0 +1,588 @@
+# ------------------------------------------------------------------------------
+# Copyright 2025 2toINF and HuggingFace Inc. (https://github.com/2toINF)
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# ------------------------------------------------------------------------------
+
+from __future__ import annotations
+
+from collections.abc import Iterable
+
+import torch
+import torch.nn as nn
+
+# =============================================================================
+# Registry
+# =============================================================================
+ACTION_REGISTRY: dict[str, type[BaseActionSpace]] = {}
+
+
+def register_action(name: str):
+    """Decorator for registering a new action space."""
+
+    def _wrap(cls):
+        key = name.lower()
+        if key in ACTION_REGISTRY:
+            raise KeyError(f"ActionSpace '{key}' already registered -> {ACTION_REGISTRY[key]}")
+        ACTION_REGISTRY[key] = cls
+        cls.name = key
+        return cls
+
+    return _wrap
+
+
+def build_action_space(name: str, **kwargs) -> BaseActionSpace:
+    """Instantiate a registered action space by name."""
+    key = name.lower()
+    if key not in ACTION_REGISTRY:
+        raise KeyError(f"Unknown action space '{name}'. Available: {list(ACTION_REGISTRY.keys())}")
+    return ACTION_REGISTRY[key](**kwargs)
+
+
+# =============================================================================
+# Base class
+# =============================================================================
+class BaseActionSpace(nn.Module):
+    """
+    Abstract base class for all action-space definitions.
+
+    Each subclass defines:
+      - `dim_action`: dimension of the action vector.
+      - `gripper_idx`: indices of gripper channels.
+      - `compute_loss(pred, target)`: supervised loss for this space.
+      - `preprocess(proprio, action, mode)`: pre-step modifications.
+      - `postprocess(action)`: post-step corrections (e.g. apply sigmoid).
+    """
+
+    name: str = "base"
+    dim_action: int = 0
+    gripper_idx: tuple[int, ...] = ()
+
+    def __init__(self):
+        super().__init__()
+
+    # ---------------------------------------------------------------------
+    # Core supervised loss
+    # ---------------------------------------------------------------------
+    def compute_loss(self, pred: torch.Tensor, target: torch.Tensor) -> dict[str, torch.Tensor]:
+        raise NotImplementedError
+
+    def forward(self, pred: torch.Tensor, target: torch.Tensor) -> dict[str, torch.Tensor]:
+        """Alias for compute_loss."""
+        return self.compute_loss(pred, target)
+
+    # ---------------------------------------------------------------------
+    # Space-level hooks
+    # ---------------------------------------------------------------------
+    def preprocess(
+        self,
+        proprio: torch.Tensor,
+        action: torch.Tensor,
+        mode: str = "train",
+    ) -> tuple[torch.Tensor, torch.Tensor]:
+        """Default: return unchanged."""
+        return proprio, action
+
+    def postprocess(self, action: torch.Tensor) -> torch.Tensor:
+        """Default: return unchanged."""
+        return action
+
+
+# =============================================================================
+# Utilities
+# =============================================================================
+def _ensure_indices_valid(dim_action: int, idx: Iterable[int], name: str) -> None:
+    bad = [i for i in idx if i < 0 or i >= dim_action]
+    if bad:
+        raise IndexError(f"{name} contains out-of-range indices {bad} for action dim dim_action={dim_action}")
+
+
+# =============================================================================
+# Implementations
+# =============================================================================
+@register_action("ee6d")
+class EE6DActionSpace(BaseActionSpace):
+    """End-effector layout with xyz, 6D rotation, and gripper channels."""
+
+    dim_action = 20
+    gripper_idx = (9, 19)
+    GRIPPER_SCALE = 1.0
+    XYZ_SCALE = 500.0
+    ROT_SCALE = 10.0
+
+    POS_IDX_1 = (0, 1, 2)
+    POS_IDX_2 = (10, 11, 12)
+    ROT_IDX_1 = (3, 4, 5, 6, 7, 8)
+    ROT_IDX_2 = (13, 14, 15, 16, 17, 18)
+
+    def __init__(self):
+        super().__init__()
+        self.mse = nn.MSELoss()
+        self.bce = nn.BCEWithLogitsLoss()
+
+    def compute_loss(self, pred, target):
+        assert pred.shape == target.shape, "pred/target shapes must match"
+        batch_size, seq_len, action_dim = pred.shape
+        _ensure_indices_valid(action_dim, self.gripper_idx, "gripper_idx")
+
+        # Gripper BCE
+        g_losses = [self.bce(pred[:, :, gi], target[:, :, gi]) for gi in self.gripper_idx]
+        gripper_loss = sum(g_losses) / len(self.gripper_idx) * self.GRIPPER_SCALE
+
+        # XYZ position
+        pos_loss = (
+            self.mse(pred[:, :, self.POS_IDX_1], target[:, :, self.POS_IDX_1])
+            + self.mse(pred[:, :, self.POS_IDX_2], target[:, :, self.POS_IDX_2])
+        ) * self.XYZ_SCALE
+
+        # Rotation 6D
+        rot_loss = (
+            self.mse(pred[:, :, self.ROT_IDX_1], target[:, :, self.ROT_IDX_1])
+            + self.mse(pred[:, :, self.ROT_IDX_2], target[:, :, self.ROT_IDX_2])
+        ) * self.ROT_SCALE
+
+        return {
+            "position_loss": pos_loss,
+            "rotate6D_loss": rot_loss,
+            "gripper_loss": gripper_loss,
+        }
+
+    def preprocess(self, proprio, action, mode="train"):
+        """Zero-out gripper channels in proprio/action."""
+        proprio_m = proprio.clone()
+        action_m = action.clone()
+        proprio_m[..., self.gripper_idx] = 0.0
+        action_m[..., self.gripper_idx] = 0.0
+        return proprio_m, action_m
+
+    def postprocess(self, action: torch.Tensor) -> torch.Tensor:
+        """Apply sigmoid to gripper logits."""
+        if action.size(-1) > max(self.gripper_idx):
+            action[..., self.gripper_idx] = torch.sigmoid(action[..., self.gripper_idx])
+        return action
+
+
+@register_action("joint")
+class JointActionSpace(BaseActionSpace):
+    """Joint-space layout with joints + gripper only."""
+
+    dim_action = 14
+    gripper_idx = (6, 13)
+    GRIPPER_SCALE = 0.1
+    JOINTS_SCALE = 1.0
+
+    def __init__(self):
+        super().__init__()
+        self.mse = nn.MSELoss()
+        self.bce = nn.BCEWithLogitsLoss()
+
+    def compute_loss(self, pred, target):
+        assert pred.shape == target.shape
+        batch_size, seq_len, action_dim = pred.shape
+        _ensure_indices_valid(action_dim, self.gripper_idx, "gripper_idx")
+
+        g_losses = [self.bce(pred[:, :, gi], target[:, :, gi]) for gi in self.gripper_idx]
+        gripper_loss = sum(g_losses) / len(self.gripper_idx) * self.GRIPPER_SCALE
+
+        joints_idx = tuple(i for i in range(action_dim) if i not in set(self.gripper_idx))
+        joints_loss = self.mse(pred[:, :, joints_idx], target[:, :, joints_idx]) * self.JOINTS_SCALE
+
+        return {
+            "joints_loss": joints_loss,
+            "gripper_loss": gripper_loss,
+        }
+
+    def preprocess(self, proprio, action, mode="train"):
+        """Zero-out gripper channels in proprio/action."""
+        proprio_m = proprio.clone()
+        action_m = action.clone()
+        proprio_m[..., self.gripper_idx] = 0.0
+        action_m[..., self.gripper_idx] = 0.0
+        return proprio_m, action_m
+
+    def postprocess(self, action: torch.Tensor) -> torch.Tensor:
+        """Apply sigmoid to gripper logits."""
+        if action.size(-1) > max(self.gripper_idx):
+            action[..., self.gripper_idx] = torch.sigmoid(action[..., self.gripper_idx])
+        return action
+
+
+@register_action("agibot_ee6d")
+class AGIBOTEE6DActionSpace(BaseActionSpace):
+    """AGI-bot variant of EE6DActionSpace using MSE for all components."""
+
+    dim_action = 20
+    gripper_idx = (9, 19)
+    GRIPPER_SCALE = 10.0
+    XYZ_SCALE = 500.0
+    ROT_SCALE = 10.0
+    POS_IDX_1 = (0, 1, 2)
+    POS_IDX_2 = (10, 11, 12)
+    ROT_IDX_1 = (3, 4, 5, 6, 7, 8)
+    ROT_IDX_2 = (13, 14, 15, 16, 17, 18)
+
+    def __init__(self):
+        super().__init__()
+        self.mse = nn.MSELoss()
+
+    def compute_loss(self, pred, target):
+        assert pred.shape == target.shape
+        batch_size, seq_len, action_dim = pred.shape
+        _ensure_indices_valid(action_dim, self.gripper_idx, "gripper_idx")
+
+        gripper_loss = (
+            self.mse(pred[:, :, self.gripper_idx], target[:, :, self.gripper_idx]) * self.GRIPPER_SCALE
+        )
+        pos_loss = (
+            self.mse(pred[:, :, self.POS_IDX_1], target[:, :, self.POS_IDX_1])
+            + self.mse(pred[:, :, self.POS_IDX_2], target[:, :, self.POS_IDX_2])
+        ) * self.XYZ_SCALE
+        rot_loss = (
+            self.mse(pred[:, :, self.ROT_IDX_1], target[:, :, self.ROT_IDX_1])
+            + self.mse(pred[:, :, self.ROT_IDX_2], target[:, :, self.ROT_IDX_2])
+        ) * self.ROT_SCALE
+
+        return {
+            "position_loss": pos_loss,
+            "rotate6D_loss": rot_loss,
+            "gripper_loss": gripper_loss,
+        }
+
+    def preprocess(self, proprio, action, mode="train"):
+        """No preprocessing applied in AGIBOT variant."""
+        return proprio, action
+
+    def postprocess(self, action: torch.Tensor) -> torch.Tensor:
+        """AGIBOT does not postprocess."""
+        return action
+
+
+@register_action("franka_joint7")
+class FrankaJoint7ActionSpace(BaseActionSpace):
+    """
+    Franka Panda joint-space: 7 joints, with gripper.
+
+    - Real robot action dim: 7
+    - Model-facing dim: 20 (padded with zeros)
+      compatible with pretrained VLA models expecting 20D.
+    """
+
+    dim_action = 20  # model dimension
+    REAL_DIM = 7  # actual Franka joints
+
+    JOINTS_SCALE = 1.0
+
+    def __init__(self):
+        super().__init__()
+        self.mse = nn.MSELoss()
+
+    def _pad_to_model_dim(self, x: torch.Tensor) -> torch.Tensor:
+        """Pad 7 → 20 dims (zeros for the dummy channels)."""
+        if x is None:
+            return None
+        if x.size(-1) == self.dim_action:
+            return x
+        if x.size(-1) != self.REAL_DIM:
+            raise ValueError(
+                f"Expected last dim to be {self.REAL_DIM} or {self.dim_action}, got {x.size(-1)}"
+            )
+
+        pad_shape = list(x.shape[:-1]) + [self.dim_action - self.REAL_DIM]  # 13 zeros
+        pad = x.new_zeros(pad_shape)
+        return torch.cat([x, pad], dim=-1)
+
+    def _trim_to_real_dim(self, x: torch.Tensor) -> torch.Tensor:
+        """Trim model output 20 → 7 dims."""
+        return x[..., : self.REAL_DIM]
+
+    def compute_loss(self, pred, target):
+        """
+        pred :  [B, T, 20]
+        target : [B, T, 7] or [B, T, 20]
+
+        Only compute MSE on the first 7 dims.
+        """
+        pred = self._pad_to_model_dim(pred)
+        target = self._pad_to_model_dim(target)
+
+        assert pred.shape == target.shape
+
+        joints_loss = (
+            self.mse(
+                pred[:, :, : self.REAL_DIM],  # use only the first 7 joints
+                target[:, :, : self.REAL_DIM],
+            )
+            * self.JOINTS_SCALE
+        )
+
+        return {"joints_loss": joints_loss}
+
+    def preprocess(self, proprio, action, mode="train"):
+        """
+        During training:
+        - Pad [7] → [20]
+        """
+        return proprio, self._pad_to_model_dim(action)
+
+    def postprocess(self, action: torch.Tensor) -> torch.Tensor:
+        """
+        After model prediction:
+        - Trim [20] → [7] for real robot control.
+        """
+        return self._trim_to_real_dim(action)
+
+
+@register_action("auto")
+class AutoActionSpace(BaseActionSpace):
+    """
+    Auto-detecting action space that adapts to any action dimension.
+
+    - Auto-detects the real action dimension from the policy feature
+    - Model outputs max_dim for compatibility with pretrained models
+    - Loss is computed only on the first real_dim dimensions
+    - Postprocess trims output back to real_dim
+
+    Args:
+        real_dim: The actual action dimension from the dataset/policy feature
+        max_dim: The model's output dimension for pretrained VLA compatibility
+    """
+
+    JOINTS_SCALE = 1.0
+
+    def __init__(self, real_dim: int, max_dim: int):
+        super().__init__()
+        self.real_dim = real_dim
+        self.dim_action = max_dim  # Model-facing dimension
+        self.mse = nn.MSELoss()
+
+    def _pad_to_model_dim(self, x: torch.Tensor) -> torch.Tensor:
+        """Pad real_dim → max_dim (zeros for the dummy channels)."""
+        if x is None:
+            return None
+        if x.size(-1) == self.dim_action:
+            return x
+        if x.size(-1) != self.real_dim:
+            # If dimension doesn't match either, pad/trim to real_dim first
+            if x.size(-1) < self.real_dim:
+                pad_shape = list(x.shape[:-1]) + [self.real_dim - x.size(-1)]
+                pad = x.new_zeros(pad_shape)
+                x = torch.cat([x, pad], dim=-1)
+            else:
+                x = x[..., : self.real_dim]
+
+        pad_shape = list(x.shape[:-1]) + [self.dim_action - self.real_dim]
+        pad = x.new_zeros(pad_shape)
+        return torch.cat([x, pad], dim=-1)
+
+    def _trim_to_real_dim(self, x: torch.Tensor) -> torch.Tensor:
+        """Trim model output max_dim → real_dim."""
+        return x[..., : self.real_dim]
+
+    def compute_loss(self, pred: torch.Tensor, target: torch.Tensor) -> dict[str, torch.Tensor]:
+        """
+        Compute loss only on the first real_dim dimensions.
+
+        pred:   [B, T, max_dim] from the model
+        target: [B, T, real_dim] or [B, T, max_dim]
+
+        Loss = MSE(pred[:,:,:real_dim], target[:,:,:real_dim])
+        """
+        pred = self._pad_to_model_dim(pred)
+        target = self._pad_to_model_dim(target)
+        assert pred.shape == target.shape, f"Shape mismatch: pred {pred.shape} vs target {target.shape}"
+
+        # only compute loss on the real dimensions
+        joints_loss = (
+            self.mse(
+                pred[:, :, : self.real_dim],
+                target[:, :, : self.real_dim],
+            )
+            * self.JOINTS_SCALE
+        )
+
+        return {"joints_loss": joints_loss}
+
+    def preprocess(self, proprio: torch.Tensor, action: torch.Tensor, mode: str = "train"):
+        """
+        Pad action from real_dim to max_dim for the model.
+        """
+        return proprio, self._pad_to_model_dim(action)
+
+    def postprocess(self, action: torch.Tensor) -> torch.Tensor:
+        """
+        Trim model output from max_dim to real_dim for real robot control.
+        """
+        return self._trim_to_real_dim(action)
+
+
+@register_action("so101_bimanual")
+class BimanualSO101ActionSpace(BaseActionSpace):
+    """
+    Bimanual SO101 robot: 2 arms with 5 joints each + gripper.
+
+    Layout (real robot):
+    [left_arm (5 joints + gripper), right_arm (5 joints + gripper)]
+    - Left arm:  shoulder_pan, shoulder_lift, elbow_flex, wrist_flex, wrist_roll, gripper
+    - Right arm: shoulder_pan, shoulder_lift, elbow_flex, wrist_flex, wrist_roll, gripper
+
+    Real action dim: 12
+    Model-facing dim: 20 (extra 8 dummy dims at the end)
+    """
+
+    # Model output / training dimension (to match pretrained policy)
+    dim_action = 20
+
+    # Real robot action dimension
+    REAL_DIM = 12
+
+    # Indices of real vs dummy channels
+    REAL_IDXS = tuple(range(REAL_DIM))  # 0..11
+    DUMMY_IDXS = tuple(range(REAL_DIM, dim_action))  # 12..19
+
+    # Grippers live in the real part
+    gripper_idx = (5, 11)  # left_gripper at idx 5, right_gripper at idx 11
+    GRIPPER_SCALE = 1.0
+    JOINTS_SCALE = 1.0
+
+    # Indices for left and right arm joints (excluding grippers)
+    LEFT_ARM_JOINTS = (0, 1, 2, 3, 4)
+    RIGHT_ARM_JOINTS = (6, 7, 8, 9, 10)
+
+    def __init__(self):
+        super().__init__()
+        self.mse = nn.MSELoss()
+        self.bce = nn.BCEWithLogitsLoss()
+
+    # ---------- helpers ----------
+
+    def _pad_to_model_dim(self, x: torch.Tensor) -> torch.Tensor:
+        """If last dim is REAL_DIM (12), pad zeros to reach dim_action (20)."""
+        if x is None:
+            return None
+        if x.size(-1) == self.dim_action:
+            return x
+        if x.size(-1) != self.REAL_DIM:
+            raise ValueError(
+                f"Expected last dim to be {self.REAL_DIM} or {self.dim_action}, got {x.size(-1)}"
+            )
+        pad_shape = list(x.shape[:-1]) + [self.dim_action - self.REAL_DIM]
+        pad = x.new_zeros(pad_shape)
+        return torch.cat([x, pad], dim=-1)
+
+    def _trim_to_real_dim(self, x: torch.Tensor) -> torch.Tensor:
+        """Keep only the first REAL_DIM (12) dims for the real robot."""
+        return x[..., : self.REAL_DIM]
+
+    # ---------- loss ----------
+
+    def compute_loss(self, pred, target):
+        """
+        pred:  [B, T, 20] from the model
+        target: [B, T, 12] or [B, T, 20]
+        We pad target → 20 and compute loss only on the real dims.
+        """
+        # Ensure both are [B, T, 20]
+        pred = self._pad_to_model_dim(pred)
+        target = self._pad_to_model_dim(target)
+        assert pred.shape == target.shape
+
+        # ---- MSE for all real dims (0–11) ----
+        real_dims = 12
+
+        joints_loss = (
+            self.mse(
+                pred[:, :, :real_dims],
+                target[:, :, :real_dims],
+            )
+            * self.JOINTS_SCALE
+        )
+
+        left_arm_loss = self.mse(pred[:, :, :6], target[:, :, :6])
+        right_arm_loss = self.mse(pred[:, :, 6:12], target[:, :, 6:12])
+
+        gripper_loss = (
+            self.mse(
+                pred[:, :, [5, 11]],
+                target[:, :, [5, 11]],
+            )
+            * self.GRIPPER_SCALE
+        )
+
+        return {
+            "joints_loss": joints_loss,
+            "gripper_loss": gripper_loss,
+            "left_arm_loss": left_arm_loss,
+            "right_arm_loss": right_arm_loss,
+        }
+
+    # ---------- preprocess / postprocess ----------
+
+    def preprocess(self, proprio, action, mode="train"):
+        """
+        - If proprio/action are 12-dim, pad them to 20 for the model.
+        - Zero-out gripper channels in proprio/action to focus learning on joints.
+        """
+        proprio_m = self._pad_to_model_dim(proprio.clone())
+        action_m = self._pad_to_model_dim(action.clone()) if action is not None else None
+
+        proprio_m[..., self.gripper_idx] = 0.0
+        if action_m is not None:
+            action_m[..., self.gripper_idx] = 0.0
+
+        return proprio_m, action_m
+
+    def postprocess(self, action: torch.Tensor) -> torch.Tensor:
+        """
+        - Model outputs [*, 20]
+        - Apply sigmoid to gripper logits
+        - Return only the first 12 dims for the real robot:
+          ["left_shoulder_pan.pos",
+           "left_shoulder_lift.pos",
+           "left_elbow_flex.pos",
+           "left_wrist_flex.pos",
+           "left_wrist_roll.pos",
+           "left_gripper.pos",
+           "right_shoulder_pan.pos",
+           "right_shoulder_lift.pos",
+           "right_elbow_flex.pos",
+           "right_wrist_flex.pos",
+           "right_wrist_roll.pos",
+           "right_gripper.pos"]
+        """
+        # Ensure we at least have the real dims + grippers
+        if action.size(-1) < self.REAL_DIM:
+            raise ValueError(f"Expected at least {self.REAL_DIM} dims in action, got {action.size(-1)}")
+
+        # Apply sigmoid on gripper channels in model space (indices 5 and 11)
+        if action.size(-1) > max(self.gripper_idx):
+            action[..., self.gripper_idx] = torch.sigmoid(action[..., self.gripper_idx])
+
+        # Return only the real 12-dim control vector for the env
+        return self._trim_to_real_dim(action)
+
+
+# =============================================================================
+# Exports
+# =============================================================================
+__all__ = [
+    "BaseActionSpace",
+    "build_action_space",
+    "register_action",
+    "EE6DActionSpace",
+    "JointActionSpace",
+    "AGIBOTEE6DActionSpace",
+    "FrankaJoint7ActionSpace",
+    "AutoActionSpace",
+    "BimanualSO101ActionSpace",
+    "ACTION_REGISTRY",
+]
diff --git a/lerobot/src/lerobot/policies/xvla/configuration_florence2.py b/lerobot/src/lerobot/policies/xvla/configuration_florence2.py
new file mode 100644
index 0000000000000000000000000000000000000000..77f1b3a1d5bfb7508f4b53cc203aab7d535896b2
--- /dev/null
+++ b/lerobot/src/lerobot/policies/xvla/configuration_florence2.py
@@ -0,0 +1,355 @@
+# Copyright 2024 Microsoft and the HuggingFace Inc. team. All rights reserved.
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import warnings
+
+from transformers.configuration_utils import PretrainedConfig
+from transformers.utils import logging
+
+""" Florence-2 configuration"""
+
+logger = logging.get_logger(__name__)
+
+
+class Florence2VisionConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`Florence2VisionModel`]. It is used to instantiate a Florence2VisionModel
+    according to the specified arguments, defining the model architecture. Instantiating a configuration with the
+    defaults will yield a similar configuration to that of the Florence2VisionModel architecture.
+
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+
+    Args:
+        drop_path_rate (`float`, *optional*, defaults to 0.1):
+            The dropout rate of the drop path layer.
+        patch_size (`List[int]`, *optional*, defaults to [7, 3, 3, 3]):
+            The patch size of the image.
+        patch_stride (`List[int]`, *optional*, defaults to [4, 2, 2, 2]):
+            The patch stride of the image.
+        patch_padding (`List[int]`, *optional*, defaults to [3, 1, 1, 1]):
+            The patch padding of the image.
+        patch_prenorm (`List[bool]`, *optional*, defaults to [false, true, true, true]):
+            Whether to apply layer normalization before the patch embedding layer.
+        enable_checkpoint (`bool`, *optional*, defaults to False):
+            Whether to enable checkpointing.
+        dim_embed (`List[int]`, *optional*, defaults to [256, 512, 1024, 2048]):
+            The dimension of the embedding layer.
+        num_heads (`List[int]`, *optional*, defaults to [8, 16, 32, 64]):
+            The number of attention heads.
+        num_groups (`List[int]`, *optional*, defaults to [8, 16, 32, 64]):
+            The number of groups.
+        depths (`List[int]`, *optional*, defaults to [1, 1, 9, 1]):
+            The depth of the model.
+        window_size (`int`, *optional*, defaults to 12):
+            The window size of the model.
+        projection_dim (`int`, *optional*, defaults to 1024):
+            The dimension of the projection layer.
+        visual_temporal_embedding (`dict`, *optional*):
+            The configuration of the visual temporal embedding.
+        image_pos_embed (`dict`, *optional*):
+            The configuration of the image position embedding.
+        image_feature_source (`List[str]`, *optional*, defaults to ["spatial_avg_pool", "temporal_avg_pool"]):
+            The source of the image feature.
+    Example:
+
+    ```python
+    >>> from transformers import Florence2VisionConfig, Florence2VisionModel
+
+    >>> # Initializing a Florence2 Vision style configuration
+    >>> configuration = Florence2VisionConfig()
+
+    >>> # Initializing a model (with random weights)
+    >>> model = Florence2VisionModel(configuration)
+
+    >>> # Accessing the model configuration
+    >>> configuration = model.config
+    ```"""
+
+    model_type = "davit"
+    keys_to_ignore_at_inference = ["past_key_values"]
+
+    def __init__(
+        self,
+        drop_path_rate=0.1,
+        patch_size=None,
+        patch_stride=None,
+        patch_padding=None,
+        patch_prenorm=None,
+        enable_checkpoint=False,
+        dim_embed=None,
+        num_heads=None,
+        num_groups=None,
+        depths=None,
+        window_size=12,
+        projection_dim=1024,
+        visual_temporal_embedding=None,
+        image_pos_embed=None,
+        image_feature_source=None,
+        **kwargs,
+    ):
+        self.drop_path_rate = drop_path_rate
+        self.patch_size = patch_size if patch_size is not None else [7, 3, 3, 3]
+        self.patch_stride = patch_stride if patch_stride is not None else [4, 2, 2, 2]
+        self.patch_padding = patch_padding if patch_padding is not None else [3, 1, 1, 1]
+        self.patch_prenorm = patch_prenorm if patch_prenorm is not None else [False, True, True, True]
+        self.enable_checkpoint = enable_checkpoint
+        self.dim_embed = dim_embed if dim_embed is not None else [256, 512, 1024, 2048]
+        self.num_heads = num_heads if num_heads is not None else [8, 16, 32, 64]
+        self.num_groups = num_groups if num_groups is not None else [8, 16, 32, 64]
+        self.depths = depths if depths is not None else [1, 1, 9, 1]
+        self.window_size = window_size
+        self.projection_dim = projection_dim
+
+        if visual_temporal_embedding is None:
+            visual_temporal_embedding = {
+                "type": "COSINE",
+                "max_temporal_embeddings": 100,
+            }
+        self.visual_temporal_embedding = visual_temporal_embedding
+
+        if image_pos_embed is None:
+            image_pos_embed = {
+                "type": "learned_abs_2d",
+                "max_pos_embeddings": 1000,
+            }
+        self.image_pos_embed = image_pos_embed
+
+        self.image_feature_source = (
+            image_feature_source
+            if image_feature_source is not None
+            else ["spatial_avg_pool", "temporal_avg_pool"]
+        )
+
+        super().__init__(**kwargs)
+
+
+class Florence2LanguageConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`Florence2LanguagePreTrainedModel`]. It is used to instantiate a BART
+    model according to the specified arguments, defining the model architecture. Instantiating a configuration with the
+    defaults will yield a similar configuration to that of the BART
+    [facebook/bart-large](https://huggingface.co/facebook/bart-large) architecture.
+
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+
+
+    Args:
+        vocab_size (`int`, *optional*, defaults to 51289):
+            Vocabulary size of the Florence2Language model. Defines the number of different tokens that can be represented by the
+            `inputs_ids` passed when calling [`Florence2LanguageModel`].
+        d_model (`int`, *optional*, defaults to 1024):
+            Dimensionality of the layers and the pooler layer.
+        encoder_layers (`int`, *optional*, defaults to 12):
+            Number of encoder layers.
+        decoder_layers (`int`, *optional*, defaults to 12):
+            Number of decoder layers.
+        encoder_attention_heads (`int`, *optional*, defaults to 16):
+            Number of attention heads for each attention layer in the Transformer encoder.
+        decoder_attention_heads (`int`, *optional*, defaults to 16):
+            Number of attention heads for each attention layer in the Transformer decoder.
+        decoder_ffn_dim (`int`, *optional*, defaults to 4096):
+            Dimensionality of the "intermediate" (often named feed-forward) layer in decoder.
+        encoder_ffn_dim (`int`, *optional*, defaults to 4096):
+            Dimensionality of the "intermediate" (often named feed-forward) layer in decoder.
+        activation_function (`str` or `function`, *optional*, defaults to `"gelu"`):
+            The non-linear activation function (function or string) in the encoder and pooler. If string, `"gelu"`,
+            `"relu"`, `"silu"` and `"gelu_new"` are supported.
+        dropout (`float`, *optional*, defaults to 0.1):
+            The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.
+        attention_dropout (`float`, *optional*, defaults to 0.0):
+            The dropout ratio for the attention probabilities.
+        activation_dropout (`float`, *optional*, defaults to 0.0):
+            The dropout ratio for activations inside the fully connected layer.
+        classifier_dropout (`float`, *optional*, defaults to 0.0):
+            The dropout ratio for classifier.
+        max_position_embeddings (`int`, *optional*, defaults to 1024):
+            The maximum sequence length that this model might ever be used with. Typically set this to something large
+            just in case (e.g., 512 or 1024 or 2048).
+        init_std (`float`, *optional*, defaults to 0.02):
+            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
+        encoder_layerdrop (`float`, *optional*, defaults to 0.0):
+            The LayerDrop probability for the encoder. See the [LayerDrop paper](see https://arxiv.org/abs/1909.11556)
+            for more details.
+        decoder_layerdrop (`float`, *optional*, defaults to 0.0):
+            The LayerDrop probability for the decoder. See the [LayerDrop paper](see https://arxiv.org/abs/1909.11556)
+            for more details.
+        scale_embedding (`bool`, *optional*, defaults to `False`):
+            Scale embeddings by diving by sqrt(d_model).
+        use_cache (`bool`, *optional*, defaults to `True`):
+            Whether or not the model should return the last key/values attentions (not used by all models).
+        num_labels (`int`, *optional*, defaults to 3):
+            The number of labels to use in [`Florence2LanguageForSequenceClassification`].
+        forced_eos_token_id (`int`, *optional*, defaults to 2):
+            The id of the token to force as the last generated token when `max_length` is reached. Usually set to
+            `eos_token_id`.
+
+    Example:
+
+    ```python
+    >>> from transformers import Florence2LanguageConfig, Florence2LanguageModel
+
+    >>> # Initializing a Florence2 Language style configuration
+    >>> configuration = Florence2LanguageConfig()
+
+    >>> # Initializing a model (with random weights)
+    >>> model = Florence2LanguageModel(configuration)
+
+    >>> # Accessing the model configuration
+    >>> configuration = model.config
+    ```"""
+
+    model_type = "florence2_language"
+    keys_to_ignore_at_inference = ["past_key_values"]
+    attribute_map = {"num_attention_heads": "encoder_attention_heads", "hidden_size": "d_model"}
+
+    def __init__(
+        self,
+        vocab_size=51289,
+        max_position_embeddings=1024,
+        encoder_layers=12,
+        encoder_ffn_dim=4096,
+        encoder_attention_heads=16,
+        decoder_layers=12,
+        decoder_ffn_dim=4096,
+        decoder_attention_heads=16,
+        encoder_layerdrop=0.0,
+        decoder_layerdrop=0.0,
+        activation_function="gelu",
+        d_model=1024,
+        dropout=0.1,
+        attention_dropout=0.0,
+        activation_dropout=0.0,
+        init_std=0.02,
+        classifier_dropout=0.0,
+        scale_embedding=False,
+        use_cache=True,
+        num_labels=3,
+        pad_token_id=1,
+        bos_token_id=0,
+        eos_token_id=2,
+        is_encoder_decoder=True,
+        decoder_start_token_id=2,
+        forced_eos_token_id=2,
+        **kwargs,
+    ):
+        self.vocab_size = vocab_size
+        self.max_position_embeddings = max_position_embeddings
+        self.d_model = d_model
+        self.encoder_ffn_dim = encoder_ffn_dim
+        self.encoder_layers = encoder_layers
+        self.encoder_attention_heads = encoder_attention_heads
+        self.decoder_ffn_dim = decoder_ffn_dim
+        self.decoder_layers = decoder_layers
+        self.decoder_attention_heads = decoder_attention_heads
+        self.dropout = dropout
+        self.attention_dropout = attention_dropout
+        self.activation_dropout = activation_dropout
+        self.activation_function = activation_function
+        self.init_std = init_std
+        self.encoder_layerdrop = encoder_layerdrop
+        self.decoder_layerdrop = decoder_layerdrop
+        self.classifier_dropout = classifier_dropout
+        self.use_cache = use_cache
+        self.num_hidden_layers = encoder_layers
+        self.scale_embedding = scale_embedding  # scale factor will be sqrt(d_model) if True
+
+        super().__init__(
+            num_labels=num_labels,
+            pad_token_id=pad_token_id,
+            bos_token_id=bos_token_id,
+            eos_token_id=eos_token_id,
+            is_encoder_decoder=is_encoder_decoder,
+            decoder_start_token_id=decoder_start_token_id,
+            forced_eos_token_id=forced_eos_token_id,
+            **kwargs,
+        )
+
+        # ensure backward compatibility for BART CNN models
+        if not hasattr(self, "forced_bos_token_id"):
+            self.forced_bos_token_id = None
+        if self.forced_bos_token_id is None and kwargs.get("force_bos_token_to_be_generated", False):
+            self.forced_bos_token_id = self.bos_token_id
+            warnings.warn(
+                f"Please make sure the config includes `forced_bos_token_id={self.bos_token_id}` in future versions. "
+                "The config can simply be saved and uploaded again to be fixed.",
+                stacklevel=2,
+            )
+
+
+class Florence2Config(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`Florence2ForConditionalGeneration`]. It is used to instantiate an
+    Florence-2 model according to the specified arguments, defining the model architecture.
+
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+
+    Args:
+        vision_config (`Florence2VisionConfig`,  *optional*):
+            Custom vision config or dict
+        text_config (`Union[AutoConfig, dict]`, *optional*):
+            The config object of the text backbone.
+        ignore_index (`int`, *optional*, defaults to -100):
+            The ignore index for the loss function.
+        vocab_size (`int`, *optional*, defaults to 51289):
+            Vocabulary size of the Florence2model. Defines the number of different tokens that can be represented by the
+            `inputs_ids` passed when calling [`~Florence2ForConditionalGeneration`]
+        projection_dim (`int`, *optional*, defaults to 1024):
+            Dimension of the multimodal projection space.
+
+    Example:
+
+    ```python
+    >>> from transformers import Florence2ForConditionalGeneration, Florence2Config, CLIPVisionConfig, BartConfig
+
+    >>> # Initializing a clip-like vision config
+    >>> vision_config = CLIPVisionConfig()
+
+    >>> # Initializing a Bart config
+    >>> text_config = BartConfig()
+
+    >>> # Initializing a Florence-2 configuration
+    >>> configuration = Florence2Config(vision_config, text_config)
+
+    >>> # Initializing a model from the florence-2 configuration
+    >>> model = Florence2ForConditionalGeneration(configuration)
+
+    >>> # Accessing the model configuration
+    >>> configuration = model.config
+    ```"""
+
+    model_type = "florence2"
+    is_composition = False
+
+    def __init__(
+        self,
+        vision_config=None,
+        text_config=None,
+        ignore_index=-100,
+        vocab_size=51289,
+        projection_dim=1024,
+        **kwargs,
+    ):
+        self.ignore_index = ignore_index
+        self.vocab_size = vocab_size
+        self.projection_dim = projection_dim
+        if vision_config is not None:
+            vision_config = Florence2VisionConfig(**vision_config)
+        self.vision_config = vision_config
+
+        self.text_config = text_config
+        if text_config is not None:
+            self.text_config = Florence2LanguageConfig(**text_config)
+
+        super().__init__(**kwargs)
diff --git a/lerobot/src/lerobot/policies/xvla/configuration_xvla.py b/lerobot/src/lerobot/policies/xvla/configuration_xvla.py
new file mode 100644
index 0000000000000000000000000000000000000000..30700b0427c5e9884577cc502069ed400e444aa0
--- /dev/null
+++ b/lerobot/src/lerobot/policies/xvla/configuration_xvla.py
@@ -0,0 +1,203 @@
+#!/usr/bin/env python
+
+# ------------------------------------------------------------------------------
+# Copyright 2025 The HuggingFace Inc. team and 2toINF (https://github.com/2toINF)
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# ------------------------------------------------------------------------------
+
+from __future__ import annotations
+
+from dataclasses import dataclass, field
+from typing import TYPE_CHECKING, Any
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.optim.optimizers import XVLAAdamWConfig
+from lerobot.optim.schedulers import CosineDecayWithWarmupSchedulerConfig
+from lerobot.utils.constants import OBS_IMAGES
+
+# Conditional import for type checking and lazy loading
+from lerobot.utils.import_utils import _transformers_available
+
+if TYPE_CHECKING or _transformers_available:
+    from .configuration_florence2 import Florence2Config
+else:
+    Florence2Config = None
+
+
+@PreTrainedConfig.register_subclass("xvla")
+@dataclass
+class XVLAConfig(PreTrainedConfig):
+    """
+    Configuration class for the XVLA (Extended Vision-Language-Action) policy so it can
+    plug into the LeRobot training stack.
+
+    The config mirrors the knobs exposed in the original XVLA repository but also
+    declares the input/output feature contract required by LeRobot.
+    """
+
+    # Input / output structure
+    n_obs_steps: int = 1
+    chunk_size: int = 32
+    n_action_steps: int = 32
+    dtype: str = "float32"  # Options: "bfloat16", "float32"
+
+    normalization_mapping: dict[str, NormalizationMode] = field(
+        default_factory=lambda: {
+            "VISUAL": NormalizationMode.IDENTITY,
+            "STATE": NormalizationMode.IDENTITY,
+            "ACTION": NormalizationMode.IDENTITY,
+        }
+    )
+
+    # Florence2 backbone and tokenizer configuration
+    florence_config: dict[str, Any] = field(default_factory=dict)
+    tokenizer_name: str = "facebook/bart-large"
+    tokenizer_max_length: int = 64
+    tokenizer_padding_side: str = "right"
+    pad_language_to: str = "max_length"
+
+    # Transformer head
+    hidden_size: int = 1024
+    depth: int = 24
+    num_heads: int = 16
+    mlp_ratio: float = 4.0
+    num_domains: int = 30
+    len_soft_prompts: int = 32
+    dim_time: int = 32
+    max_len_seq: int = 512
+    use_hetero_proj: bool = False
+
+    # Action & proprioception
+    action_mode: str = "ee6d"
+    num_denoising_steps: int = 10
+    use_proprio: bool = True
+    max_state_dim: int = 32
+    max_action_dim: int = 20  # Maximum action dimension for padding (used by "auto" action mode)
+    domain_feature_key: str | None = None
+
+    # Vision preprocessing
+    resize_imgs_with_padding: tuple[int, int] | None = None
+    num_image_views: int | None = None
+    empty_cameras: int = 0
+
+    # Freezing options for VLM components
+    # By default, VLM encoders are frozen and only policy transformer + soft prompts train
+    freeze_vision_encoder: bool = False  # Freeze VLM vision encoder weights
+    freeze_language_encoder: bool = False  # Freeze VLM language encoder weights
+    train_policy_transformer: bool = True  # Allow policy transformer to train
+    train_soft_prompts: bool = True  # Allow soft prompts to train
+
+    # Training presets
+    optimizer_lr: float = 1e-4
+    optimizer_betas: tuple[float, float] = (0.9, 0.99)
+    optimizer_eps: float = 1e-8
+    optimizer_weight_decay: float = 0.0
+    optimizer_grad_clip_norm: float = 10.0
+    # Soft-prompt LR settings (for optional warm-up)
+    optimizer_soft_prompt_lr_scale: float = 1.0  # Scale factor for soft-prompt LR
+    optimizer_soft_prompt_warmup_lr_scale: float | None = None  # Start scale for warmup (e.g., 0.01)
+
+    scheduler_warmup_steps: int = 1_000
+    scheduler_decay_steps: int = 30_000
+    scheduler_decay_lr: float = 2.5e-6
+
+    def __post_init__(self) -> None:
+        super().__post_init__()
+
+        if self.chunk_size <= 0:
+            raise ValueError("`chunk_size` must be strictly positive.")
+        if self.n_action_steps > self.chunk_size:
+            raise ValueError(
+                f"`n_action_steps` ({self.n_action_steps}) must be <= `chunk_size` ({self.chunk_size})."
+            )
+        if self.num_image_views is not None and self.num_image_views <= 0:
+            raise ValueError("`num_image_views` must be > 0 when specified.")
+        if self.dtype not in ["bfloat16", "float32"]:
+            raise ValueError(f"Invalid dtype: {self.dtype}")
+        self._florence_config_obj: Florence2Config | None = None
+
+    def get_florence_config(self) -> Florence2Config:
+        """
+        Build (and cache) the Florence2 transformer config that should back the VLM.
+        """
+        if self._florence_config_obj is None:
+            config_dict = dict(self.florence_config)
+            if "vision_config" not in config_dict or config_dict["vision_config"] is None:
+                raise ValueError("vision_config is required")
+
+            if "text_config" not in config_dict or config_dict["text_config"] is None:
+                raise ValueError("text_config is required")
+            self._florence_config_obj = Florence2Config(**config_dict)
+        return self._florence_config_obj
+
+    def validate_features(self) -> None:
+        if not self.image_features:
+            raise ValueError("XVLA requires at least one visual feature in the inputs.")
+        if self.use_proprio and self.robot_state_feature is None:
+            raise ValueError("`use_proprio=True` requires a proprioceptive state feature.")
+        if self.num_image_views is None:
+            self.num_image_views = len(self.image_features) + self.empty_cameras
+        else:
+            self.num_image_views = max(self.num_image_views, len(self.image_features) + self.empty_cameras)
+
+        if self.empty_cameras > 0:
+            height, width = (480, 640)
+            if self.resize_imgs_with_padding is not None:
+                height, width = self.resize_imgs_with_padding
+            for idx in range(self.empty_cameras):
+                key = f"{OBS_IMAGES}.empty_camera_{idx}"
+                if key not in self.input_features:
+                    self.input_features[key] = PolicyFeature(
+                        type=FeatureType.VISUAL,
+                        shape=(3, height, width),
+                    )
+
+    def get_optimizer_preset(self) -> XVLAAdamWConfig:
+        """Return the XVLA-specific optimizer with differential learning rates.
+
+        This optimizer applies:
+        - 1/10 LR for VLM parameters (stable optimization)
+        - Full LR for transformer/action head
+        - Configurable LR for soft-prompts (with optional warm-up)
+        """
+        return XVLAAdamWConfig(
+            lr=self.optimizer_lr,
+            betas=self.optimizer_betas,
+            eps=self.optimizer_eps,
+            weight_decay=self.optimizer_weight_decay,
+            grad_clip_norm=self.optimizer_grad_clip_norm,
+            soft_prompt_lr_scale=self.optimizer_soft_prompt_lr_scale,
+            soft_prompt_warmup_lr_scale=self.optimizer_soft_prompt_warmup_lr_scale,
+        )
+
+    def get_scheduler_preset(self) -> CosineDecayWithWarmupSchedulerConfig:
+        return CosineDecayWithWarmupSchedulerConfig(
+            peak_lr=self.optimizer_lr,
+            decay_lr=self.scheduler_decay_lr,
+            num_warmup_steps=self.scheduler_warmup_steps,
+            num_decay_steps=self.scheduler_decay_steps,
+        )
+
+    @property
+    def observation_delta_indices(self) -> list[int] | None:
+        return None
+
+    @property
+    def action_delta_indices(self) -> list[int]:
+        return list(range(self.chunk_size))
+
+    @property
+    def reward_delta_indices(self) -> list[int] | None:
+        return None
diff --git a/lerobot/src/lerobot/policies/xvla/modeling_florence2.py b/lerobot/src/lerobot/policies/xvla/modeling_florence2.py
new file mode 100644
index 0000000000000000000000000000000000000000..e33efe5c30c8da5620722599acb2a32255be544e
--- /dev/null
+++ b/lerobot/src/lerobot/policies/xvla/modeling_florence2.py
@@ -0,0 +1,2762 @@
+# Copyright 2024 Microsoft and the HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""PyTorch Florence-2 model."""
+
+import math
+from collections import OrderedDict
+from dataclasses import dataclass
+
+import torch
+import torch.nn.functional as functional
+import torch.utils.checkpoint
+import torch.utils.checkpoint as checkpoint
+from einops import rearrange
+from torch import nn
+from torch.nn import CrossEntropyLoss
+from transformers.activations import ACT2FN
+from transformers.generation.utils import GenerationMixin
+from transformers.modeling_attn_mask_utils import (
+    _prepare_4d_attention_mask,
+    _prepare_4d_attention_mask_for_sdpa,
+    _prepare_4d_causal_attention_mask,
+    _prepare_4d_causal_attention_mask_for_sdpa,
+)
+from transformers.modeling_outputs import (
+    BaseModelOutput,
+    BaseModelOutputWithPastAndCrossAttentions,
+    Seq2SeqLMOutput,
+    Seq2SeqModelOutput,
+)
+from transformers.modeling_utils import PreTrainedModel
+from transformers.utils import (
+    ModelOutput,
+    add_start_docstrings,
+    add_start_docstrings_to_model_forward,
+    is_flash_attn_2_available,
+    is_flash_attn_greater_or_equal_2_10,
+    logging,
+    replace_return_docstrings,
+)
+
+from .configuration_florence2 import Florence2Config, Florence2LanguageConfig
+from .utils import drop_path
+
+if is_flash_attn_2_available():
+    from flash_attn import flash_attn_func, flash_attn_varlen_func
+    from flash_attn.bert_padding import index_first_axis, pad_input, unpad_input  # noqa
+
+logger = logging.get_logger(__name__)
+
+_CONFIG_FOR_DOC = "Florence2Config"
+
+
+class DropPath(nn.Module):
+    """Drop paths (Stochastic Depth) per sample  (when applied in main path of residual blocks)."""
+
+    def __init__(self, drop_prob: float = 0.0, scale_by_keep: bool = True):
+        super().__init__()
+        self.drop_prob = drop_prob
+        self.scale_by_keep = scale_by_keep
+
+    def forward(self, x):
+        return drop_path(x, self.drop_prob, self.training, self.scale_by_keep)
+
+    def extra_repr(self):
+        return f"drop_prob={round(self.drop_prob, 3):0.3f}"
+
+
+class LearnedAbsolutePositionEmbedding2D(nn.Module):
+    """
+    This module learns positional embeddings up to a fixed maximum size.
+    """
+
+    def __init__(self, embedding_dim=256, num_pos=50):
+        super().__init__()
+        self.row_embeddings = nn.Embedding(num_pos, embedding_dim // 2)
+        self.column_embeddings = nn.Embedding(num_pos, embedding_dim - (embedding_dim // 2))
+
+    def forward(self, pixel_values):
+        """
+        pixel_values: (batch_size, height, width, num_channels)
+        returns: (batch_size, height, width, embedding_dim * 2)
+        """
+        if len(pixel_values.shape) != 4:
+            raise ValueError("pixel_values must be a 4D tensor")
+        height, width = pixel_values.shape[1:3]
+        width_values = torch.arange(width, device=pixel_values.device)
+        height_values = torch.arange(height, device=pixel_values.device)
+        x_emb = self.column_embeddings(width_values)
+        y_emb = self.row_embeddings(height_values)
+        # (height, width, embedding_dim * 2)
+        pos = torch.cat(
+            [x_emb.unsqueeze(0).repeat(height, 1, 1), y_emb.unsqueeze(1).repeat(1, width, 1)], dim=-1
+        )
+        # (embedding_dim * 2, height, width)
+        pos = pos.permute(2, 0, 1)
+        pos = pos.unsqueeze(0)
+        # (batch_size, embedding_dim * 2, height, width)
+        pos = pos.repeat(pixel_values.shape[0], 1, 1, 1)
+        # (batch_size, height, width, embedding_dim * 2)
+        pos = pos.permute(0, 2, 3, 1)
+        return pos
+
+
+class PositionalEmbeddingCosine1D(nn.Module):
+    """
+    This class implements a very simple positional encoding. It follows closely
+    the encoder from the link below:
+    https://pytorch.org/tutorials/beginner/translation_transformer.html
+
+    Args:
+        embed_dim: The dimension of the embeddings.
+        dropout_prob: The dropout probability.
+        max_seq_len: The maximum length to precompute the positional encodings.
+    """
+
+    def __init__(self, embed_dim: int = 512, max_seq_len: int = 1024) -> None:
+        super().__init__()
+        self.embed_dim = embed_dim
+        self.max_seq_len = max_seq_len
+        # Generate the sinusoidal arrays.
+        factor = math.log(10000)
+        denominator = torch.exp(-factor * torch.arange(0, self.embed_dim, 2) / self.embed_dim)
+        # Matrix where rows correspond to a positional embedding as a function
+        # of the position index (i.e., the row index).
+        frequencies = torch.arange(0, self.max_seq_len).reshape(self.max_seq_len, 1) * denominator
+        pos_idx_to_embed = torch.zeros((self.max_seq_len, self.embed_dim))
+        # Populate uneven entries.
+        pos_idx_to_embed[:, 0::2] = torch.sin(frequencies)
+        pos_idx_to_embed[:, 1::2] = torch.cos(frequencies)
+        # Save the positional embeddings in a constant buffer.
+        self.register_buffer("pos_idx_to_embed", pos_idx_to_embed)
+
+    def forward(self, seq_embeds: torch.Tensor) -> torch.Tensor:
+        """
+        Args:
+            seq_embeds: The sequence embeddings in order. Allowed size:
+                1. [T, D], where T is the length of the sequence, and D is the
+                frame embedding dimension.
+                2. [B, T, D], where B is the batch size and T and D are the
+                same as above.
+
+        Returns a tensor of with the same dimensions as the input: i.e.,
+        [1, T, D] or [T, D].
+        """
+        shape_len = len(seq_embeds.shape)
+        assert 2 <= shape_len <= 3
+        len_seq = seq_embeds.size(-2)
+        assert len_seq <= self.max_seq_len
+        pos_embeds = self.pos_idx_to_embed[0 : seq_embeds.size(-2), :]
+        # Adapt pre-computed positional embeddings to the input.
+        if shape_len == 3:
+            pos_embeds = pos_embeds.view((1, pos_embeds.size(0), pos_embeds.size(1)))
+        return pos_embeds
+
+
+class LearnedAbsolutePositionEmbedding1D(nn.Module):
+    """
+    Learnable absolute positional embeddings for 1D sequences.
+
+    Args:
+        embed_dim: The dimension of the embeddings.
+        max_seq_len: The maximum length to precompute the positional encodings.
+    """
+
+    def __init__(self, embedding_dim: int = 512, num_pos: int = 1024) -> None:
+        super().__init__()
+        self.embeddings = nn.Embedding(num_pos, embedding_dim)
+        self.num_pos = num_pos
+
+    def forward(self, seq_embeds: torch.Tensor) -> torch.Tensor:
+        """
+        Args:
+            seq_embeds: The sequence embeddings in order. Allowed size:
+                1. [T, D], where T is the length of the sequence, and D is the
+                frame embedding dimension.
+                2. [B, T, D], where B is the batch size and T and D are the
+                same as above.
+
+        Returns a tensor of with the same dimensions as the input: i.e.,
+        [1, T, D] or [T, D].
+        """
+        shape_len = len(seq_embeds.shape)
+        assert 2 <= shape_len <= 3
+        len_seq = seq_embeds.size(-2)
+        assert len_seq <= self.num_pos
+        # [T, D]
+        pos_embeds = self.embeddings(torch.arange(len_seq).to(seq_embeds.device))
+        # Adapt pre-computed positional embeddings to the input.
+        if shape_len == 3:
+            pos_embeds = pos_embeds.view((1, pos_embeds.size(0), pos_embeds.size(1)))
+        return pos_embeds
+
+
+class MySequential(nn.Sequential):
+    def forward(self, *inputs):
+        for module in self._modules.values():
+            inputs = module(*inputs) if isinstance(inputs, tuple) else module(inputs)
+        return inputs
+
+
+class PreNorm(nn.Module):
+    def __init__(self, norm, fn, drop_path=None):
+        super().__init__()
+        self.norm = norm
+        self.fn = fn
+        self.drop_path = drop_path
+
+    def forward(self, x, *args, **kwargs):
+        shortcut = x
+        if self.norm is not None:
+            x, size = self.fn(self.norm(x), *args, **kwargs)
+        else:
+            x, size = self.fn(x, *args, **kwargs)
+
+        if self.drop_path:
+            x = self.drop_path(x)
+
+        x = shortcut + x
+
+        return x, size
+
+
+class Mlp(nn.Module):
+    def __init__(
+        self,
+        in_features,
+        hidden_features=None,
+        out_features=None,
+        act_layer=nn.GELU,
+    ):
+        super().__init__()
+        out_features = out_features or in_features
+        hidden_features = hidden_features or in_features
+        self.net = nn.Sequential(
+            OrderedDict(
+                [
+                    ("fc1", nn.Linear(in_features, hidden_features)),
+                    ("act", act_layer()),
+                    ("fc2", nn.Linear(hidden_features, out_features)),
+                ]
+            )
+        )
+
+    def forward(self, x, size):
+        return self.net(x), size
+
+
+class DepthWiseConv2d(nn.Module):
+    def __init__(
+        self,
+        dim_in,
+        kernel_size,
+        padding,
+        stride,
+        bias=True,
+    ):
+        super().__init__()
+        self.dw = nn.Conv2d(
+            dim_in, dim_in, kernel_size=kernel_size, padding=padding, groups=dim_in, stride=stride, bias=bias
+        )
+
+    def forward(self, x, size):
+        batch_size, num_tokens, channels = x.shape
+        height, width = size
+        assert num_tokens == height * width
+
+        x = self.dw(x.transpose(1, 2).view(batch_size, channels, height, width))
+        size = (x.size(-2), x.size(-1))
+        x = x.flatten(2).transpose(1, 2)
+        return x, size
+
+
+class ConvEmbed(nn.Module):
+    """Image to Patch Embedding"""
+
+    def __init__(
+        self, patch_size=7, in_chans=3, embed_dim=64, stride=4, padding=2, norm_layer=None, pre_norm=True
+    ):
+        super().__init__()
+        self.patch_size = patch_size
+
+        self.proj = nn.Conv2d(in_chans, embed_dim, kernel_size=patch_size, stride=stride, padding=padding)
+
+        dim_norm = in_chans if pre_norm else embed_dim
+        self.norm = norm_layer(dim_norm) if norm_layer else None
+
+        self.pre_norm = pre_norm
+
+    def forward(self, x, size):
+        height, width = size
+        if len(x.size()) == 3:
+            if self.norm and self.pre_norm:
+                x = self.norm(x)
+            x = rearrange(x, "b (h w) c -> b c h w", h=height, w=width)
+
+        x = self.proj(x)
+
+        _, _, height, width = x.shape
+        x = rearrange(x, "b c h w -> b (h w) c")
+        if self.norm and not self.pre_norm:
+            x = self.norm(x)
+
+        return x, (height, width)
+
+
+class ChannelAttention(nn.Module):
+    def __init__(self, dim, groups=8, qkv_bias=True):
+        super().__init__()
+
+        self.groups = groups
+        self.qkv = nn.Linear(dim, dim * 3, bias=qkv_bias)
+        self.proj = nn.Linear(dim, dim)
+
+    def forward(self, x, size):
+        batch_size, num_tokens, channels = x.shape
+
+        qkv = (
+            self.qkv(x)
+            .reshape(batch_size, num_tokens, 3, self.groups, channels // self.groups)
+            .permute(2, 0, 3, 1, 4)
+        )
+        q, k, v = qkv[0], qkv[1], qkv[2]
+
+        q = q * (float(num_tokens) ** -0.5)
+        attention = q.transpose(-1, -2) @ k
+        attention = attention.softmax(dim=-1)
+        x = (attention @ v.transpose(-1, -2)).transpose(-1, -2)
+        x = x.transpose(1, 2).reshape(batch_size, num_tokens, channels)
+        x = self.proj(x)
+        return x, size
+
+
+class ChannelBlock(nn.Module):
+    def __init__(
+        self,
+        dim,
+        groups,
+        mlp_ratio=4.0,
+        qkv_bias=True,
+        drop_path_rate=0.0,
+        act_layer=nn.GELU,
+        norm_layer=nn.LayerNorm,
+        conv_at_attn=True,
+        conv_at_ffn=True,
+    ):
+        super().__init__()
+
+        drop_path = DropPath(drop_path_rate) if drop_path_rate > 0.0 else nn.Identity()
+
+        self.conv1 = PreNorm(None, DepthWiseConv2d(dim, 3, 1, 1)) if conv_at_attn else None
+        self.channel_attn = PreNorm(
+            norm_layer(dim), ChannelAttention(dim, groups=groups, qkv_bias=qkv_bias), drop_path
+        )
+        self.conv2 = PreNorm(None, DepthWiseConv2d(dim, 3, 1, 1)) if conv_at_ffn else None
+        self.ffn = PreNorm(
+            norm_layer(dim),
+            Mlp(in_features=dim, hidden_features=int(dim * mlp_ratio), act_layer=act_layer),
+            drop_path,
+        )
+
+    def forward(self, x, size):
+        if self.conv1:
+            x, size = self.conv1(x, size)
+        x, size = self.channel_attn(x, size)
+
+        if self.conv2:
+            x, size = self.conv2(x, size)
+        x, size = self.ffn(x, size)
+
+        return x, size
+
+
+def window_partition(x, window_size: int):
+    batch_size, height, width, channels = x.shape
+    x = x.view(batch_size, height // window_size, window_size, width // window_size, window_size, channels)
+    windows = x.permute(0, 1, 3, 2, 4, 5).contiguous().view(-1, window_size, window_size, channels)
+    return windows
+
+
+def window_reverse(windows, batch_size: int, window_size: int, height: int, width: int):
+    # this will cause onnx conversion failed for dynamic axis, because treated as constant
+    # int(windows.shape[0] / (height * width / window_size / window_size))
+    x = windows.view(batch_size, height // window_size, width // window_size, window_size, window_size, -1)
+    x = x.permute(0, 1, 3, 2, 4, 5).contiguous().view(batch_size, height, width, -1)
+    return x
+
+
+class WindowAttention(nn.Module):
+    def __init__(self, dim, num_heads, window_size, qkv_bias=True):
+        super().__init__()
+        self.dim = dim
+        self.window_size = window_size
+        self.num_heads = num_heads
+        head_dim = dim // num_heads
+        self.scale = float(head_dim) ** -0.5
+
+        self.qkv = nn.Linear(dim, dim * 3, bias=qkv_bias)
+        self.proj = nn.Linear(dim, dim)
+
+        self.softmax = nn.Softmax(dim=-1)
+
+    def forward(self, x, size):
+        height, width = size
+        batch_size, seq_len, channels = x.shape
+        assert seq_len == height * width, "input feature has wrong size"
+
+        x = x.view(batch_size, height, width, channels)
+
+        pad_l = pad_t = 0
+        pad_r = (self.window_size - width % self.window_size) % self.window_size
+        pad_b = (self.window_size - height % self.window_size) % self.window_size
+        x = functional.pad(x, (0, 0, pad_l, pad_r, pad_t, pad_b))
+        _, height_padded, width_padded, _ = x.shape
+
+        x = window_partition(x, self.window_size)
+        x = x.view(-1, self.window_size * self.window_size, channels)
+
+        # W-MSA/SW-MSA
+        # attn_windows = self.attn(x_windows)
+
+        batch_windows, num_tokens, channels = x.shape
+        qkv = (
+            self.qkv(x)
+            .reshape(batch_windows, num_tokens, 3, self.num_heads, channels // self.num_heads)
+            .permute(2, 0, 3, 1, 4)
+        )
+        q, k, v = qkv[0], qkv[1], qkv[2]
+
+        q = q * self.scale
+        attn = q @ k.transpose(-2, -1)
+        attn = self.softmax(attn)
+
+        x = (attn @ v).transpose(1, 2).reshape(batch_windows, num_tokens, channels)
+        x = self.proj(x)
+
+        # merge windows
+        x = x.view(-1, self.window_size, self.window_size, channels)
+        x = window_reverse(x, batch_size, self.window_size, height_padded, width_padded)
+
+        if pad_r > 0 or pad_b > 0:
+            x = x[:, :height, :width, :].contiguous()
+
+        x = x.view(batch_size, height * width, channels)
+
+        return x, size
+
+
+class SpatialBlock(nn.Module):
+    def __init__(
+        self,
+        dim,
+        num_heads,
+        window_size,
+        mlp_ratio=4.0,
+        qkv_bias=True,
+        drop_path_rate=0.0,
+        act_layer=nn.GELU,
+        norm_layer=nn.LayerNorm,
+        conv_at_attn=True,
+        conv_at_ffn=True,
+    ):
+        super().__init__()
+
+        drop_path = DropPath(drop_path_rate) if drop_path_rate > 0.0 else nn.Identity()
+
+        self.conv1 = PreNorm(None, DepthWiseConv2d(dim, 3, 1, 1)) if conv_at_attn else None
+        self.window_attn = PreNorm(
+            norm_layer(dim), WindowAttention(dim, num_heads, window_size, qkv_bias=qkv_bias), drop_path
+        )
+        self.conv2 = PreNorm(None, DepthWiseConv2d(dim, 3, 1, 1)) if conv_at_ffn else None
+        self.ffn = PreNorm(
+            norm_layer(dim),
+            Mlp(in_features=dim, hidden_features=int(dim * mlp_ratio), act_layer=act_layer),
+            drop_path,
+        )
+
+    def forward(self, x, size):
+        if self.conv1:
+            x, size = self.conv1(x, size)
+        x, size = self.window_attn(x, size)
+
+        if self.conv2:
+            x, size = self.conv2(x, size)
+        x, size = self.ffn(x, size)
+        return x, size
+
+
+class DaViT(nn.Module):
+    """DaViT: Dual-Attention Transformer
+
+    Args:
+        in_chans (int): Number of input image channels. Default: 3.
+        num_classes (int): Number of classes for classification head. Default: 1000.
+        patch_size (tuple(int)): Patch size of convolution in different stages. Default: (7, 2, 2, 2).
+        patch_stride (tuple(int)): Patch stride of convolution in different stages. Default: (4, 2, 2, 2).
+        patch_padding (tuple(int)): Patch padding of convolution in different stages. Default: (3, 0, 0, 0).
+        patch_prenorm (tuple(bool)): If True, perform norm before convlution layer. Default: (True, False, False, False).
+        embed_dims (tuple(int)): Patch embedding dimension in different stages. Default: (64, 128, 192, 256).
+        num_heads (tuple(int)): Number of spatial attention heads in different stages. Default: (4, 8, 12, 16).
+        num_groups (tuple(int)): Number of channel groups in different stages. Default: (4, 8, 12, 16).
+        window_size (int): Window size. Default: 7.
+        mlp_ratio (float): Ratio of mlp hidden dim to embedding dim. Default: 4.
+        qkv_bias (bool): If True, add a learnable bias to query, key, value. Default: True.
+        drop_path_rate (float): Stochastic depth rate. Default: 0.1.
+        norm_layer (nn.Module): Normalization layer. Default: nn.LayerNorm.
+        enable_checkpoint (bool): If True, enable checkpointing. Default: False.
+        conv_at_attn (bool): If True, perform depthwise convolution before attention layer. Default: True.
+        conv_at_ffn (bool): If True, perform depthwise convolution before ffn layer. Default: True.
+    """
+
+    def __init__(
+        self,
+        in_chans=3,
+        num_classes=1000,
+        depths=(1, 1, 3, 1),
+        patch_size=(7, 2, 2, 2),
+        patch_stride=(4, 2, 2, 2),
+        patch_padding=(3, 0, 0, 0),
+        patch_prenorm=(False, False, False, False),
+        embed_dims=(64, 128, 192, 256),
+        num_heads=(3, 6, 12, 24),
+        num_groups=(3, 6, 12, 24),
+        window_size=7,
+        mlp_ratio=4.0,
+        qkv_bias=True,
+        drop_path_rate=0.1,
+        norm_layer=nn.LayerNorm,
+        enable_checkpoint=False,
+        conv_at_attn=True,
+        conv_at_ffn=True,
+    ):
+        super().__init__()
+
+        self.num_classes = num_classes
+        self.embed_dims = embed_dims
+        self.num_heads = num_heads
+        self.num_groups = num_groups
+        self.num_stages = len(self.embed_dims)
+        self.enable_checkpoint = enable_checkpoint
+        assert self.num_stages == len(self.num_heads) == len(self.num_groups)
+
+        num_stages = len(embed_dims)
+        dpr = [x.item() for x in torch.linspace(0, drop_path_rate, sum(depths) * 2)]
+
+        depth_offset = 0
+        convs = []
+        blocks = []
+        for i in range(num_stages):
+            conv_embed = ConvEmbed(
+                patch_size=patch_size[i],
+                stride=patch_stride[i],
+                padding=patch_padding[i],
+                in_chans=in_chans if i == 0 else self.embed_dims[i - 1],
+                embed_dim=self.embed_dims[i],
+                norm_layer=norm_layer,
+                pre_norm=patch_prenorm[i],
+            )
+            convs.append(conv_embed)
+
+            block = MySequential(
+                *[
+                    MySequential(
+                        OrderedDict(
+                            [
+                                (
+                                    "spatial_block",
+                                    SpatialBlock(
+                                        embed_dims[i],
+                                        num_heads[i],
+                                        window_size,
+                                        drop_path_rate=dpr[depth_offset + j * 2],
+                                        qkv_bias=qkv_bias,
+                                        mlp_ratio=mlp_ratio,
+                                        conv_at_attn=conv_at_attn,
+                                        conv_at_ffn=conv_at_ffn,
+                                    ),
+                                ),
+                                (
+                                    "channel_block",
+                                    ChannelBlock(
+                                        embed_dims[i],
+                                        num_groups[i],
+                                        drop_path_rate=dpr[depth_offset + j * 2 + 1],
+                                        qkv_bias=qkv_bias,
+                                        mlp_ratio=mlp_ratio,
+                                        conv_at_attn=conv_at_attn,
+                                        conv_at_ffn=conv_at_ffn,
+                                    ),
+                                ),
+                            ]
+                        )
+                    )
+                    for j in range(depths[i])
+                ]
+            )
+            blocks.append(block)
+            depth_offset += depths[i] * 2
+
+        self.convs = nn.ModuleList(convs)
+        self.blocks = nn.ModuleList(blocks)
+
+        self.norms = norm_layer(self.embed_dims[-1])
+        self.avgpool = nn.AdaptiveAvgPool1d(1)
+        self.head = nn.Linear(self.embed_dims[-1], num_classes) if num_classes > 0 else nn.Identity()
+
+    @property
+    def dim_out(self):
+        return self.embed_dims[-1]
+
+    def forward_features_unpool(self, x):
+        """
+        forward until avg pooling
+        Args:
+            x (_type_): input image tensor
+        """
+        input_size = (x.size(2), x.size(3))
+        for conv, block in zip(self.convs, self.blocks, strict=False):
+            x, input_size = conv(x, input_size)
+            if self.enable_checkpoint:
+                x, input_size = checkpoint.checkpoint(block, x, input_size)
+            else:
+                x, input_size = block(x, input_size)
+        return x
+
+    def forward_features(self, x):
+        x = self.forward_features_unpool(x)
+
+        # (batch_size, num_tokens, token_dim)
+        x = self.avgpool(x.transpose(1, 2))
+        # (batch_size, 1, num_tokens)
+        x = torch.flatten(x, 1)
+        x = self.norms(x)
+
+        return x
+
+    def forward(self, x):
+        x = self.forward_features(x)
+        x = self.head(x)
+        return x
+
+    @classmethod
+    def from_config(cls, config):
+        return cls(
+            depths=config.depths,
+            embed_dims=config.dim_embed,
+            num_heads=config.num_heads,
+            num_groups=config.num_groups,
+            patch_size=config.patch_size,
+            patch_stride=config.patch_stride,
+            patch_padding=config.patch_padding,
+            patch_prenorm=config.patch_prenorm,
+            drop_path_rate=config.drop_path_rate,
+            window_size=config.window_size,
+        )
+
+
+# Copied from transformers.models.llama.modeling_llama._get_unpad_data
+def _get_unpad_data(attention_mask):
+    seqlens_in_batch = attention_mask.sum(dim=-1, dtype=torch.int32)
+    indices = torch.nonzero(attention_mask.flatten(), as_tuple=False).flatten()
+    max_seqlen_in_batch = seqlens_in_batch.max().item()
+    cu_seqlens = functional.pad(torch.cumsum(seqlens_in_batch, dim=0, dtype=torch.int32), (1, 0))
+    return (
+        indices,
+        cu_seqlens,
+        max_seqlen_in_batch,
+    )
+
+
+def shift_tokens_right(input_ids: torch.Tensor, pad_token_id: int, decoder_start_token_id: int):
+    """
+    Shift input ids one token to the right.
+    """
+    shifted_input_ids = input_ids.new_zeros(input_ids.shape)
+    shifted_input_ids[:, 1:] = input_ids[:, :-1].clone()
+    shifted_input_ids[:, 0] = decoder_start_token_id
+
+    if pad_token_id is None:
+        raise ValueError("self.model.config.pad_token_id has to be defined.")
+    # replace possible -100 values in labels by `pad_token_id`
+    shifted_input_ids.masked_fill_(shifted_input_ids == -100, pad_token_id)
+
+    return shifted_input_ids
+
+
+class Florence2LearnedPositionalEmbedding(nn.Embedding):
+    """
+    This module learns positional embeddings up to a fixed maximum size.
+    """
+
+    def __init__(self, num_embeddings: int, embedding_dim: int):
+        # Florence2 is set up so that if padding_idx is specified then offset the embedding ids by 2
+        # and adjust num_embeddings appropriately. Other models don't have this hack
+        self.offset = 2
+        super().__init__(num_embeddings + self.offset, embedding_dim)
+
+    def forward(self, input_ids: torch.Tensor, past_key_values_length: int = 0):
+        """`input_ids' shape is expected to be [bsz x seqlen]."""
+
+        bsz, seq_len = input_ids.shape[:2]
+        positions = torch.arange(
+            past_key_values_length,
+            past_key_values_length + seq_len,
+            dtype=torch.long,
+            device=self.weight.device,
+        ).expand(bsz, -1)
+
+        return super().forward(positions + self.offset)
+
+
+class Florence2ScaledWordEmbedding(nn.Embedding):
+    """
+    This module overrides nn.Embeddings' forward by multiplying with embeddings scale.
+    """
+
+    def __init__(
+        self, num_embeddings: int, embedding_dim: int, padding_idx: int, embed_scale: float | None = 1.0
+    ):
+        super().__init__(num_embeddings, embedding_dim, padding_idx)
+        self.embed_scale = embed_scale
+
+    def forward(self, input_ids: torch.Tensor):
+        return super().forward(input_ids) * self.embed_scale
+
+
+class Florence2Attention(nn.Module):
+    """Multi-headed attention from 'Attention Is All You Need' paper"""
+
+    def __init__(
+        self,
+        embed_dim: int,
+        num_heads: int,
+        dropout: float = 0.0,
+        is_decoder: bool = False,
+        bias: bool = True,
+        is_causal: bool = False,
+        config: Florence2LanguageConfig | None = None,
+    ):
+        super().__init__()
+        self.embed_dim = embed_dim
+        self.num_heads = num_heads
+        self.dropout = dropout
+        self.head_dim = embed_dim // num_heads
+        self.config = config
+
+        if (self.head_dim * num_heads) != self.embed_dim:
+            raise ValueError(
+                f"embed_dim must be divisible by num_heads (got `embed_dim`: {self.embed_dim}"
+                f" and `num_heads`: {num_heads})."
+            )
+        self.scaling = self.head_dim**-0.5
+        self.is_decoder = is_decoder
+        self.is_causal = is_causal
+
+        self.k_proj = nn.Linear(embed_dim, embed_dim, bias=bias)
+        self.v_proj = nn.Linear(embed_dim, embed_dim, bias=bias)
+        self.q_proj = nn.Linear(embed_dim, embed_dim, bias=bias)
+        self.out_proj = nn.Linear(embed_dim, embed_dim, bias=bias)
+
+    def _shape(self, tensor: torch.Tensor, seq_len: int, bsz: int):
+        return tensor.view(bsz, seq_len, self.num_heads, self.head_dim).transpose(1, 2).contiguous()
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        key_value_states: torch.Tensor | None = None,
+        past_key_value: tuple[torch.Tensor] | None = None,
+        attention_mask: torch.Tensor | None = None,
+        layer_head_mask: torch.Tensor | None = None,
+        output_attentions: bool = False,
+    ) -> tuple[torch.Tensor, torch.Tensor | None, tuple[torch.Tensor] | None]:
+        """Input shape: Batch x Time x Channel"""
+
+        # if key_value_states are provided this layer is used as a cross-attention layer
+        # for the decoder
+        is_cross_attention = key_value_states is not None
+
+        bsz, tgt_len, _ = hidden_states.size()
+
+        # get query proj
+        query_states = self.q_proj(hidden_states) * self.scaling
+        # get key, value proj
+        # `past_key_value[0].shape[2] == key_value_states.shape[1]`
+        # is checking that the `sequence_length` of the `past_key_value` is the same as
+        # the provided `key_value_states` to support prefix tuning
+        if (
+            is_cross_attention
+            and past_key_value is not None
+            and past_key_value[0].shape[2] == key_value_states.shape[1]
+        ):
+            # reuse k,v, cross_attentions
+            key_states = past_key_value[0]
+            value_states = past_key_value[1]
+        elif is_cross_attention:
+            # cross_attentions
+            key_states = self._shape(self.k_proj(key_value_states), -1, bsz)
+            value_states = self._shape(self.v_proj(key_value_states), -1, bsz)
+        elif past_key_value is not None:
+            # reuse k, v, self_attention
+            key_states = self._shape(self.k_proj(hidden_states), -1, bsz)
+            value_states = self._shape(self.v_proj(hidden_states), -1, bsz)
+            key_states = torch.cat([past_key_value[0], key_states], dim=2)
+            value_states = torch.cat([past_key_value[1], value_states], dim=2)
+        else:
+            # self_attention
+            key_states = self._shape(self.k_proj(hidden_states), -1, bsz)
+            value_states = self._shape(self.v_proj(hidden_states), -1, bsz)
+
+        if self.is_decoder:
+            # if cross_attention save Tuple(torch.Tensor, torch.Tensor) of all cross attention key/value_states.
+            # Further calls to cross_attention layer can then reuse all cross-attention
+            # key/value_states (first "if" case)
+            # if uni-directional self-attention (decoder) save Tuple(torch.Tensor, torch.Tensor) of
+            # all previous decoder key/value_states. Further calls to uni-directional self-attention
+            # can concat previous decoder key/value_states to current projected key/value_states (third "elif" case)
+            # if encoder bi-directional self-attention `past_key_value` is always `None`
+            past_key_value = (key_states, value_states)
+
+        proj_shape = (bsz * self.num_heads, -1, self.head_dim)
+        query_states = self._shape(query_states, tgt_len, bsz).view(*proj_shape)
+        key_states = key_states.reshape(*proj_shape)
+        value_states = value_states.reshape(*proj_shape)
+
+        src_len = key_states.size(1)
+        attn_weights = torch.bmm(query_states, key_states.transpose(1, 2))
+
+        if attn_weights.size() != (bsz * self.num_heads, tgt_len, src_len):
+            raise ValueError(
+                f"Attention weights should be of size {(bsz * self.num_heads, tgt_len, src_len)}, but is"
+                f" {attn_weights.size()}"
+            )
+
+        if attention_mask is not None:
+            if attention_mask.size() != (bsz, 1, tgt_len, src_len):
+                raise ValueError(
+                    f"Attention mask should be of size {(bsz, 1, tgt_len, src_len)}, but is {attention_mask.size()}"
+                )
+            attn_weights = attn_weights.view(bsz, self.num_heads, tgt_len, src_len) + attention_mask
+            attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len)
+
+        attn_weights = nn.functional.softmax(attn_weights, dim=-1)
+
+        if layer_head_mask is not None:
+            if layer_head_mask.size() != (self.num_heads,):
+                raise ValueError(
+                    f"Head mask for a single layer should be of size {(self.num_heads,)}, but is"
+                    f" {layer_head_mask.size()}"
+                )
+            attn_weights = layer_head_mask.view(1, -1, 1, 1) * attn_weights.view(
+                bsz, self.num_heads, tgt_len, src_len
+            )
+            attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len)
+
+        if output_attentions:
+            # this operation is a bit awkward, but it's required to
+            # make sure that attn_weights keeps its gradient.
+            # In order to do so, attn_weights have to be reshaped
+            # twice and have to be reused in the following
+            attn_weights_reshaped = attn_weights.view(bsz, self.num_heads, tgt_len, src_len)
+            attn_weights = attn_weights_reshaped.view(bsz * self.num_heads, tgt_len, src_len)
+        else:
+            attn_weights_reshaped = None
+
+        attn_probs = nn.functional.dropout(attn_weights, p=self.dropout, training=self.training)
+
+        attn_output = torch.bmm(attn_probs, value_states)
+
+        if attn_output.size() != (bsz * self.num_heads, tgt_len, self.head_dim):
+            raise ValueError(
+                f"`attn_output` should be of size {(bsz * self.num_heads, tgt_len, self.head_dim)}, but is"
+                f" {attn_output.size()}"
+            )
+
+        attn_output = attn_output.view(bsz, self.num_heads, tgt_len, self.head_dim)
+        attn_output = attn_output.transpose(1, 2)
+
+        # Use the `embed_dim` from the config (stored in the class) rather than `hidden_state` because `attn_output` can be
+        # partitioned across GPUs when using tensor-parallelism.
+        attn_output = attn_output.reshape(bsz, tgt_len, self.embed_dim)
+
+        attn_output = self.out_proj(attn_output)
+
+        return attn_output, attn_weights_reshaped, past_key_value
+
+
+class Florence2FlashAttention2(Florence2Attention):
+    """
+    Florence2 flash attention module. This module inherits from `Florence2Attention` as the weights of the module stays
+    untouched. The only required change would be on the forward pass where it needs to correctly call the public API of
+    flash attention and deal with padding tokens in case the input contains any of them.
+    """
+
+    # Copied from transformers.models.llama.modeling_llama.LlamaFlashAttention2.__init__
+    def __init__(self, *args, **kwargs):
+        super().__init__(*args, **kwargs)
+
+        # TODO: Should be removed once Flash Attention for RoCm is bumped to 2.1.
+        # flash_attn<2.1 generates top-left aligned causal mask, while what is needed here is bottom-right alignment, that was made default for flash_attn>=2.1. This attribute is used to handle this difference. Reference: https://github.com/Dao-AILab/flash-attention/releases/tag/v2.1.0.
+        # Beware that with flash_attn<2.1, using q_seqlen != k_seqlen (except for the case q_seqlen == 1) produces a wrong mask (top-left).
+        self._flash_attn_uses_top_left_mask = not is_flash_attn_greater_or_equal_2_10()
+
+    def _reshape(self, tensor: torch.Tensor, seq_len: int, bsz: int):
+        return tensor.view(bsz, seq_len, self.num_heads, self.head_dim)
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        key_value_states: torch.Tensor | None = None,
+        past_key_value: tuple[torch.Tensor] | None = None,
+        attention_mask: torch.Tensor | None = None,
+        layer_head_mask: torch.Tensor | None = None,
+        output_attentions: bool = False,
+    ) -> tuple[torch.Tensor, torch.Tensor | None, tuple[torch.Tensor] | None]:
+        # Florence2FlashAttention2 attention does not support output_attentions
+        if output_attentions:
+            raise ValueError("Florence2FlashAttention2 attention does not support output_attentions")
+
+        # if key_value_states are provided this layer is used as a cross-attention layer
+        # for the decoder
+        is_cross_attention = key_value_states is not None
+
+        bsz, q_len, _ = hidden_states.size()
+
+        # get query proj
+        query_states = self._reshape(self.q_proj(hidden_states), -1, bsz)
+        # get key, value proj
+        # `past_key_value[0].shape[2] == key_value_states.shape[1]`
+        # is checking that the `sequence_length` of the `past_key_value` is the same as
+        # the provided `key_value_states` to support prefix tuning
+        if (
+            is_cross_attention
+            and past_key_value is not None
+            and past_key_value[0].shape[2] == key_value_states.shape[1]
+        ):
+            # reuse k,v, cross_attentions
+            key_states = past_key_value[0].transpose(1, 2)
+            value_states = past_key_value[1].transpose(1, 2)
+        elif is_cross_attention:
+            # cross_attentions
+            key_states = self._reshape(self.k_proj(key_value_states), -1, bsz)
+            value_states = self._reshape(self.v_proj(key_value_states), -1, bsz)
+        elif past_key_value is not None:
+            # reuse k, v, self_attention
+            key_states = self._reshape(self.k_proj(hidden_states), -1, bsz)
+            value_states = self._reshape(self.v_proj(hidden_states), -1, bsz)
+            key_states = torch.cat([past_key_value[0].transpose(1, 2), key_states], dim=1)
+            value_states = torch.cat([past_key_value[1].transpose(1, 2), value_states], dim=1)
+        else:
+            # self_attention
+            key_states = self._reshape(self.k_proj(hidden_states), -1, bsz)
+            value_states = self._reshape(self.v_proj(hidden_states), -1, bsz)
+
+        if self.is_decoder:
+            # if cross_attention save Tuple(torch.Tensor, torch.Tensor) of all cross attention key/value_states.
+            # Further calls to cross_attention layer can then reuse all cross-attention
+            # key/value_states (first "if" case)
+            # if uni-directional self-attention (decoder) save Tuple(torch.Tensor, torch.Tensor) of
+            # all previous decoder key/value_states. Further calls to uni-directional self-attention
+            # can concat previous decoder key/value_states to current projected key/value_states (third "elif" case)
+            # if encoder bi-directional self-attention `past_key_value` is always `None`
+            past_key_value = (key_states.transpose(1, 2), value_states.transpose(1, 2))
+
+        kv_seq_len = key_states.shape[-2]
+        if past_key_value is not None:
+            kv_seq_len += past_key_value[0].shape[-2]
+
+        # In PEFT, usually we cast the layer norms in float32 for training stability reasons
+        # therefore the input hidden states gets silently casted in float32. Hence, we need
+        # cast them back in the correct dtype just to be sure everything works as expected.
+        # This might slowdown training & inference so it is recommended to not cast the LayerNorms
+        # in fp32. (LlamaRMSNorm handles it correctly)
+
+        input_dtype = query_states.dtype
+        if input_dtype == torch.float32:
+            if torch.is_autocast_enabled():
+                target_dtype = torch.get_autocast_gpu_dtype()
+            # Handle the case where the model is quantized
+            elif hasattr(self.config, "_pre_quantization_dtype"):
+                target_dtype = self.config._pre_quantization_dtype
+            else:
+                target_dtype = self.q_proj.weight.dtype
+
+            logger.warning_once(
+                f"The input hidden states seems to be silently casted in float32, this might be related to"
+                f" the fact you have upcasted embedding or layer norm layers in float32. We will cast back the input in"
+                f" {target_dtype}."
+            )
+
+            query_states = query_states.to(target_dtype)
+            key_states = key_states.to(target_dtype)
+            value_states = value_states.to(target_dtype)
+
+        attn_output = self._flash_attention_forward(
+            query_states, key_states, value_states, attention_mask, q_len, dropout=self.dropout
+        )
+
+        attn_output = attn_output.reshape(bsz, q_len, -1)
+        attn_output = self.out_proj(attn_output)
+
+        if not output_attentions:
+            attn_weights = None
+
+        return attn_output, attn_weights, past_key_value
+
+    # Copied from transformers.models.llama.modeling_llama.LlamaFlashAttention2._flash_attention_forward
+    def _flash_attention_forward(
+        self,
+        query_states,
+        key_states,
+        value_states,
+        attention_mask,
+        query_length,
+        dropout=0.0,
+        softmax_scale=None,
+    ):
+        """
+        Calls the forward method of Flash Attention - if the input hidden states contain at least one padding token
+        first unpad the input, then computes the attention scores and pad the final attention scores.
+
+        Args:
+            query_states (`torch.Tensor`):
+                Input query states to be passed to Flash Attention API
+            key_states (`torch.Tensor`):
+                Input key states to be passed to Flash Attention API
+            value_states (`torch.Tensor`):
+                Input value states to be passed to Flash Attention API
+            attention_mask (`torch.Tensor`):
+                The padding mask - corresponds to a tensor of size `(batch_size, seq_len)` where 0 stands for the
+                position of padding tokens and 1 for the position of non-padding tokens.
+            dropout (`float`):
+                Attention dropout
+            softmax_scale (`float`, *optional*):
+                The scaling of QK^T before applying softmax. Default to 1 / sqrt(head_dim)
+        """
+        if not self._flash_attn_uses_top_left_mask:
+            causal = self.is_causal
+        else:
+            # TODO: Remove the `query_length != 1` check once Flash Attention for RoCm is bumped to 2.1. For details, please see the comment in LlamaFlashAttention2 __init__.
+            causal = self.is_causal and query_length != 1
+
+        # Contains at least one padding token in the sequence
+        if attention_mask is not None:
+            batch_size = query_states.shape[0]
+            query_states, key_states, value_states, indices_q, cu_seq_lens, max_seq_lens = self._upad_input(
+                query_states, key_states, value_states, attention_mask, query_length
+            )
+
+            cu_seqlens_q, cu_seqlens_k = cu_seq_lens
+            max_seqlen_in_batch_q, max_seqlen_in_batch_k = max_seq_lens
+
+            attn_output_unpad = flash_attn_varlen_func(
+                query_states,
+                key_states,
+                value_states,
+                cu_seqlens_q=cu_seqlens_q,
+                cu_seqlens_k=cu_seqlens_k,
+                max_seqlen_q=max_seqlen_in_batch_q,
+                max_seqlen_k=max_seqlen_in_batch_k,
+                dropout_p=dropout,
+                softmax_scale=softmax_scale,
+                causal=causal,
+            )
+
+            attn_output = pad_input(attn_output_unpad, indices_q, batch_size, query_length)
+        else:
+            attn_output = flash_attn_func(
+                query_states, key_states, value_states, dropout, softmax_scale=softmax_scale, causal=causal
+            )
+
+        return attn_output
+
+    # Copied from transformers.models.llama.modeling_llama.LlamaFlashAttention2._upad_input
+    def _upad_input(self, query_layer, key_layer, value_layer, attention_mask, query_length):
+        indices_k, cu_seqlens_k, max_seqlen_in_batch_k = _get_unpad_data(attention_mask)
+        batch_size, kv_seq_len, num_key_value_heads, head_dim = key_layer.shape
+
+        key_layer = index_first_axis(
+            key_layer.reshape(batch_size * kv_seq_len, num_key_value_heads, head_dim), indices_k
+        )
+        value_layer = index_first_axis(
+            value_layer.reshape(batch_size * kv_seq_len, num_key_value_heads, head_dim), indices_k
+        )
+        if query_length == kv_seq_len:
+            query_layer = index_first_axis(
+                query_layer.reshape(batch_size * kv_seq_len, self.num_heads, head_dim), indices_k
+            )
+            cu_seqlens_q = cu_seqlens_k
+            max_seqlen_in_batch_q = max_seqlen_in_batch_k
+            indices_q = indices_k
+        elif query_length == 1:
+            max_seqlen_in_batch_q = 1
+            cu_seqlens_q = torch.arange(
+                batch_size + 1, dtype=torch.int32, device=query_layer.device
+            )  # There is a memcpy here, that is very bad.
+            indices_q = cu_seqlens_q[:-1]
+            query_layer = query_layer.squeeze(1)
+        else:
+            # The -q_len: slice assumes left padding.
+            attention_mask = attention_mask[:, -query_length:]
+            query_layer, indices_q, cu_seqlens_q, max_seqlen_in_batch_q = unpad_input(
+                query_layer, attention_mask
+            )
+
+        return (
+            query_layer,
+            key_layer,
+            value_layer,
+            indices_q,
+            (cu_seqlens_q, cu_seqlens_k),
+            (max_seqlen_in_batch_q, max_seqlen_in_batch_k),
+        )
+
+
+class Florence2SdpaAttention(Florence2Attention):
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        key_value_states: torch.Tensor | None = None,
+        past_key_value: tuple[torch.Tensor] | None = None,
+        attention_mask: torch.Tensor | None = None,
+        layer_head_mask: torch.Tensor | None = None,
+        output_attentions: bool = False,
+    ) -> tuple[torch.Tensor, torch.Tensor | None, tuple[torch.Tensor] | None]:
+        """Input shape: Batch x Time x Channel"""
+        if output_attentions or layer_head_mask is not None:
+            # TODO: Improve this warning with e.g. `model.config._attn_implementation = "manual"` once this is implemented.
+            logger.warning_once(
+                "Florence2Model is using Florence2SdpaAttention, but `torch.nn.functional.scaled_dot_product_attention` does not support `output_attentions=True` or `layer_head_mask` not None. Falling back to the manual attention"
+                ' implementation, but specifying the manual implementation will be required from Transformers version v5.0.0 onwards. This warning can be removed using the argument `attn_implementation="eager"` when loading the model.'
+            )
+            return super().forward(
+                hidden_states,
+                key_value_states=key_value_states,
+                past_key_value=past_key_value,
+                attention_mask=attention_mask,
+                layer_head_mask=layer_head_mask,
+                output_attentions=output_attentions,
+            )
+
+        # if key_value_states are provided this layer is used as a cross-attention layer
+        # for the decoder
+        is_cross_attention = key_value_states is not None
+
+        bsz, tgt_len, _ = hidden_states.size()
+
+        # get query proj
+        query_states = self.q_proj(hidden_states)
+        # get key, value proj
+        # `past_key_value[0].shape[2] == key_value_states.shape[1]`
+        # is checking that the `sequence_length` of the `past_key_value` is the same as
+        # the provided `key_value_states` to support prefix tuning
+        if (
+            is_cross_attention
+            and past_key_value is not None
+            and past_key_value[0].shape[2] == key_value_states.shape[1]
+        ):
+            # reuse k,v, cross_attentions
+            key_states = past_key_value[0]
+            value_states = past_key_value[1]
+        elif is_cross_attention:
+            # cross_attentions
+            key_states = self._shape(self.k_proj(key_value_states), -1, bsz)
+            value_states = self._shape(self.v_proj(key_value_states), -1, bsz)
+        elif past_key_value is not None:
+            # reuse k, v, self_attention
+            key_states = self._shape(self.k_proj(hidden_states), -1, bsz)
+            value_states = self._shape(self.v_proj(hidden_states), -1, bsz)
+            key_states = torch.cat([past_key_value[0], key_states], dim=2)
+            value_states = torch.cat([past_key_value[1], value_states], dim=2)
+        else:
+            # self_attention
+            key_states = self._shape(self.k_proj(hidden_states), -1, bsz)
+            value_states = self._shape(self.v_proj(hidden_states), -1, bsz)
+
+        if self.is_decoder:
+            # if cross_attention save Tuple(torch.Tensor, torch.Tensor) of all cross attention key/value_states.
+            # Further calls to cross_attention layer can then reuse all cross-attention
+            # key/value_states (first "if" case)
+            # if uni-directional self-attention (decoder) save Tuple(torch.Tensor, torch.Tensor) of
+            # all previous decoder key/value_states. Further calls to uni-directional self-attention
+            # can concat previous decoder key/value_states to current projected key/value_states (third "elif" case)
+            # if encoder bi-directional self-attention `past_key_value` is always `None`
+            past_key_value = (key_states, value_states)
+
+        query_states = self._shape(query_states, tgt_len, bsz)
+
+        # We dispatch to SDPA's Flash Attention or Efficient kernels via this `is_causal` if statement instead of an inline conditional assignment
+        # in SDPA to support both torch.compile's dynamic shapes and full graph options. An inline conditional prevents dynamic shapes from compiling.
+        # The tgt_len > 1 is necessary to match with AttentionMaskConverter.to_causal_4d that does not create a causal mask in case tgt_len == 1.
+        is_causal = bool(self.is_causal and attention_mask is None and tgt_len > 1)
+
+        # NOTE: SDPA with memory-efficient backend is currently (torch==2.1.2) bugged when using non-contiguous inputs and a custom attn_mask,
+        # but we are fine here as `_shape` do call `.contiguous()`. Reference: https://github.com/pytorch/pytorch/issues/112577
+        attn_output = torch.nn.functional.scaled_dot_product_attention(
+            query_states,
+            key_states,
+            value_states,
+            attn_mask=attention_mask,
+            dropout_p=self.dropout if self.training else 0.0,
+            is_causal=is_causal,
+        )
+
+        if attn_output.size() != (bsz, self.num_heads, tgt_len, self.head_dim):
+            raise ValueError(
+                f"`attn_output` should be of size {(bsz, self.num_heads, tgt_len, self.head_dim)}, but is"
+                f" {attn_output.size()}"
+            )
+
+        attn_output = attn_output.transpose(1, 2)
+
+        # Use the `embed_dim` from the config (stored in the class) rather than `hidden_state` because `attn_output` can be
+        # partitioned across GPUs when using tensor-parallelism.
+        attn_output = attn_output.reshape(bsz, tgt_len, self.embed_dim)
+
+        attn_output = self.out_proj(attn_output)
+
+        return attn_output, None, past_key_value
+
+
+FLORENCE2_ATTENTION_CLASSES = {
+    "eager": Florence2Attention,
+    "sdpa": Florence2SdpaAttention,
+    "flash_attention_2": Florence2FlashAttention2,
+}
+
+
+class Florence2EncoderLayer(nn.Module):
+    def __init__(self, config: Florence2LanguageConfig):
+        super().__init__()
+        self.embed_dim = config.d_model
+
+        self.self_attn = FLORENCE2_ATTENTION_CLASSES[config._attn_implementation](
+            embed_dim=self.embed_dim,
+            num_heads=config.encoder_attention_heads,
+            dropout=config.attention_dropout,
+            config=config,
+        )
+        self.self_attn_layer_norm = nn.LayerNorm(self.embed_dim)
+        self.dropout = config.dropout
+        self.activation_fn = ACT2FN[config.activation_function]
+        self.activation_dropout = config.activation_dropout
+        self.fc1 = nn.Linear(self.embed_dim, config.encoder_ffn_dim)
+        self.fc2 = nn.Linear(config.encoder_ffn_dim, self.embed_dim)
+        self.final_layer_norm = nn.LayerNorm(self.embed_dim)
+
+    def forward(
+        self,
+        hidden_states: torch.FloatTensor,
+        attention_mask: torch.FloatTensor,
+        layer_head_mask: torch.FloatTensor,
+        output_attentions: bool | None = False,
+    ) -> tuple[torch.FloatTensor, torch.FloatTensor | None]:
+        """
+        Args:
+            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
+            attention_mask (`torch.FloatTensor`): attention mask of size
+                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
+            layer_head_mask (`torch.FloatTensor`): mask for attention heads in a given layer of size
+                `(encoder_attention_heads,)`.
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+        """
+        residual = hidden_states
+        hidden_states, attn_weights, _ = self.self_attn(
+            hidden_states=hidden_states,
+            attention_mask=attention_mask,
+            layer_head_mask=layer_head_mask,
+            output_attentions=output_attentions,
+        )
+        hidden_states = nn.functional.dropout(hidden_states, p=self.dropout, training=self.training)
+        hidden_states = residual + hidden_states
+        hidden_states = self.self_attn_layer_norm(hidden_states)
+
+        residual = hidden_states
+        hidden_states = self.activation_fn(self.fc1(hidden_states))
+        hidden_states = nn.functional.dropout(
+            hidden_states, p=self.activation_dropout, training=self.training
+        )
+        hidden_states = self.fc2(hidden_states)
+        hidden_states = nn.functional.dropout(hidden_states, p=self.dropout, training=self.training)
+        hidden_states = residual + hidden_states
+        hidden_states = self.final_layer_norm(hidden_states)
+
+        if hidden_states.dtype == torch.float16 and (
+            torch.isinf(hidden_states).any() or torch.isnan(hidden_states).any()
+        ):
+            clamp_value = torch.finfo(hidden_states.dtype).max - 1000
+            hidden_states = torch.clamp(hidden_states, min=-clamp_value, max=clamp_value)
+
+        outputs = (hidden_states,)
+
+        if output_attentions:
+            outputs += (attn_weights,)
+
+        return outputs
+
+
+class Florence2DecoderLayer(nn.Module):
+    def __init__(self, config: Florence2LanguageConfig):
+        super().__init__()
+        self.embed_dim = config.d_model
+
+        self.self_attn = FLORENCE2_ATTENTION_CLASSES[config._attn_implementation](
+            embed_dim=self.embed_dim,
+            num_heads=config.decoder_attention_heads,
+            dropout=config.attention_dropout,
+            is_decoder=True,
+            is_causal=True,
+            config=config,
+        )
+        self.dropout = config.dropout
+        self.activation_fn = ACT2FN[config.activation_function]
+        self.activation_dropout = config.activation_dropout
+
+        self.self_attn_layer_norm = nn.LayerNorm(self.embed_dim)
+        self.encoder_attn = FLORENCE2_ATTENTION_CLASSES[config._attn_implementation](
+            self.embed_dim,
+            config.decoder_attention_heads,
+            dropout=config.attention_dropout,
+            is_decoder=True,
+            config=config,
+        )
+        self.encoder_attn_layer_norm = nn.LayerNorm(self.embed_dim)
+        self.fc1 = nn.Linear(self.embed_dim, config.decoder_ffn_dim)
+        self.fc2 = nn.Linear(config.decoder_ffn_dim, self.embed_dim)
+        self.final_layer_norm = nn.LayerNorm(self.embed_dim)
+
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: torch.Tensor | None = None,
+        encoder_hidden_states: torch.Tensor | None = None,
+        encoder_attention_mask: torch.Tensor | None = None,
+        layer_head_mask: torch.Tensor | None = None,
+        cross_attn_layer_head_mask: torch.Tensor | None = None,
+        past_key_value: tuple[torch.Tensor] | None = None,
+        output_attentions: bool | None = False,
+        use_cache: bool | None = True,
+    ) -> tuple[torch.FloatTensor, tuple[torch.FloatTensor, torch.FloatTensor] | None]:
+        """
+        Args:
+            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
+            attention_mask (`torch.FloatTensor`): attention mask of size
+                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
+            encoder_hidden_states (`torch.FloatTensor`):
+                cross attention input to the layer of shape `(batch, seq_len, embed_dim)`
+            encoder_attention_mask (`torch.FloatTensor`): encoder attention mask of size
+                `(batch, 1, tgt_len, src_len)` where padding elements are indicated by very large negative values.
+            layer_head_mask (`torch.FloatTensor`): mask for attention heads in a given layer of size
+                `(encoder_attention_heads,)`.
+            cross_attn_layer_head_mask (`torch.FloatTensor`): mask for cross-attention heads in a given layer of
+                size `(decoder_attention_heads,)`.
+            past_key_value (`Tuple(torch.FloatTensor)`): cached past key and value projection states
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+        """
+        residual = hidden_states
+
+        # Self Attention
+        # decoder uni-directional self-attention cached key/values tuple is at positions 1,2
+        self_attn_past_key_value = past_key_value[:2] if past_key_value is not None else None
+        # add present self-attn cache to positions 1,2 of present_key_value tuple
+        hidden_states, self_attn_weights, present_key_value = self.self_attn(
+            hidden_states=hidden_states,
+            past_key_value=self_attn_past_key_value,
+            attention_mask=attention_mask,
+            layer_head_mask=layer_head_mask,
+            output_attentions=output_attentions,
+        )
+        hidden_states = nn.functional.dropout(hidden_states, p=self.dropout, training=self.training)
+        hidden_states = residual + hidden_states
+        hidden_states = self.self_attn_layer_norm(hidden_states)
+
+        # Cross-Attention Block
+        cross_attn_present_key_value = None
+        cross_attn_weights = None
+        if encoder_hidden_states is not None:
+            residual = hidden_states
+
+            # cross_attn cached key/values tuple is at positions 3,4 of present_key_value tuple
+            cross_attn_past_key_value = past_key_value[-2:] if past_key_value is not None else None
+            hidden_states, cross_attn_weights, cross_attn_present_key_value = self.encoder_attn(
+                hidden_states=hidden_states,
+                key_value_states=encoder_hidden_states,
+                attention_mask=encoder_attention_mask,
+                layer_head_mask=cross_attn_layer_head_mask,
+                past_key_value=cross_attn_past_key_value,
+                output_attentions=output_attentions,
+            )
+            hidden_states = nn.functional.dropout(hidden_states, p=self.dropout, training=self.training)
+            hidden_states = residual + hidden_states
+            hidden_states = self.encoder_attn_layer_norm(hidden_states)
+
+            # add cross-attn to positions 3,4 of present_key_value tuple
+            present_key_value = present_key_value + cross_attn_present_key_value
+
+        # Fully Connected
+        residual = hidden_states
+        hidden_states = self.activation_fn(self.fc1(hidden_states))
+        hidden_states = nn.functional.dropout(
+            hidden_states, p=self.activation_dropout, training=self.training
+        )
+        hidden_states = self.fc2(hidden_states)
+        hidden_states = nn.functional.dropout(hidden_states, p=self.dropout, training=self.training)
+        hidden_states = residual + hidden_states
+        hidden_states = self.final_layer_norm(hidden_states)
+
+        outputs = (hidden_states,)
+
+        if output_attentions:
+            outputs += (self_attn_weights, cross_attn_weights)
+
+        if use_cache:
+            outputs += (present_key_value,)
+
+        return outputs
+
+
+class Florence2LanguagePreTrainedModel(PreTrainedModel):
+    config_class = Florence2LanguageConfig
+    base_model_prefix = "model"
+    supports_gradient_checkpointing = True
+    _keys_to_ignore_on_load_unexpected = ["encoder.version", "decoder.version"]
+    _no_split_modules = [r"Florence2EncoderLayer", r"Florence2DecoderLayer"]
+    _skip_keys_device_placement = "past_key_values"
+    _supports_flash_attn_2 = True
+    _supports_sdpa = True
+
+    def _init_weights(self, module):
+        std = self.config.init_std
+        if isinstance(module, nn.Linear):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.bias is not None:
+                module.bias.data.zero_()
+        elif isinstance(module, nn.Embedding):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.padding_idx is not None:
+                module.weight.data[module.padding_idx].zero_()
+        elif isinstance(module, nn.Conv2d):
+            nn.init.normal_(module.weight, std=0.02)
+            for name, _ in module.named_parameters():
+                if name == "bias":
+                    nn.init.constant_(module.bias, 0)
+        elif isinstance(module, (nn.LayerNorm, nn.BatchNorm2d)):
+            nn.init.constant_(module.weight, 1.0)
+            nn.init.constant_(module.bias, 0)
+
+    @property
+    def dummy_inputs(self):
+        pad_token = self.config.pad_token_id
+        input_ids = torch.tensor([[0, 6, 10, 4, 2], [0, 8, 12, 2, pad_token]], device=self.device)
+        dummy_inputs = {
+            "attention_mask": input_ids.ne(pad_token),
+            "input_ids": input_ids,
+        }
+        return dummy_inputs
+
+
+class Florence2Encoder(Florence2LanguagePreTrainedModel):
+    """
+    Transformer encoder consisting of *config.encoder_layers* self attention layers. Each layer is a
+    [`Florence2EncoderLayer`].
+
+    Args:
+        config: Florence2LanguageConfig
+        embed_tokens (nn.Embedding): output embedding
+    """
+
+    def __init__(self, config: Florence2LanguageConfig, embed_tokens: nn.Embedding | None = None):
+        super().__init__(config)
+
+        self.dropout = config.dropout
+        self.layerdrop = config.encoder_layerdrop
+
+        embed_dim = config.d_model
+        self.padding_idx = config.pad_token_id
+        self.max_source_positions = config.max_position_embeddings
+        embed_scale = math.sqrt(embed_dim) if config.scale_embedding else 1.0
+
+        self.embed_tokens = Florence2ScaledWordEmbedding(
+            config.vocab_size, embed_dim, self.padding_idx, embed_scale=embed_scale
+        )
+
+        if embed_tokens is not None:
+            self.embed_tokens.weight = embed_tokens.weight
+
+        self.embed_positions = Florence2LearnedPositionalEmbedding(
+            config.max_position_embeddings,
+            embed_dim,
+        )
+        self.layers = nn.ModuleList([Florence2EncoderLayer(config) for _ in range(config.encoder_layers)])
+        self._use_flash_attention_2 = config._attn_implementation == "flash_attention_2"
+        self._use_sdpa = config._attn_implementation == "sdpa"
+        self.layernorm_embedding = nn.LayerNorm(embed_dim)
+
+        self.gradient_checkpointing = False
+        # Initialize weights and apply final processing
+        self.post_init()
+
+    def get_input_embeddings(self):
+        return self.embed_tokens
+
+    def set_input_embeddings(self, value):
+        self.embed_tokens = value
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        head_mask: torch.Tensor | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+    ) -> tuple | BaseModelOutput:
+        r"""
+        Args:
+            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
+                Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you
+                provide it.
+
+                Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+                [`PreTrainedTokenizer.__call__`] for details.
+
+                [What are input IDs?](../glossary#input-ids)
+            attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
+                Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:
+
+                - 1 for tokens that are **not masked**,
+                - 0 for tokens that are **masked**.
+
+                [What are attention masks?](../glossary#attention-mask)
+            head_mask (`torch.Tensor` of shape `(encoder_layers, encoder_attention_heads)`, *optional*):
+                Mask to nullify selected heads of the attention modules. Mask values selected in `[0, 1]`:
+
+                - 1 indicates the head is **not masked**,
+                - 0 indicates the head is **masked**.
+
+            inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
+                Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation.
+                This is useful if you want more control over how to convert `input_ids` indices into associated vectors
+                than the model's internal embedding lookup matrix.
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+            output_hidden_states (`bool`, *optional*):
+                Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors
+                for more detail.
+            return_dict (`bool`, *optional*):
+                Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
+        """
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        # retrieve input_ids and inputs_embeds
+        if input_ids is not None and inputs_embeds is not None:
+            raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
+        elif input_ids is not None:
+            input = input_ids
+            input_ids = input_ids.view(-1, input_ids.shape[-1])
+        elif inputs_embeds is not None:
+            input = inputs_embeds[:, :, -1]
+        else:
+            raise ValueError("You have to specify either input_ids or inputs_embeds")
+
+        if inputs_embeds is None:
+            inputs_embeds = self.embed_tokens(input_ids)
+
+        embed_pos = self.embed_positions(input)
+        embed_pos = embed_pos.to(inputs_embeds.device)
+
+        hidden_states = inputs_embeds + embed_pos
+        hidden_states = self.layernorm_embedding(hidden_states)
+        hidden_states = nn.functional.dropout(hidden_states, p=self.dropout, training=self.training)
+
+        # expand attention_mask
+        if attention_mask is not None:
+            if self._use_flash_attention_2:
+                attention_mask = attention_mask if 0 in attention_mask else None
+            elif self._use_sdpa and head_mask is None and not output_attentions:
+                # output_attentions=True & head_mask can not be supported when using SDPA, fall back to
+                # the manual implementation that requires a 4D causal mask in all cases.
+                # [bsz, seq_len] -> [bsz, 1, tgt_seq_len, src_seq_len]
+                attention_mask = _prepare_4d_attention_mask_for_sdpa(attention_mask, inputs_embeds.dtype)
+            else:
+                # [bsz, seq_len] -> [bsz, 1, tgt_seq_len, src_seq_len]
+                attention_mask = _prepare_4d_attention_mask(attention_mask, inputs_embeds.dtype)
+
+        encoder_states = () if output_hidden_states else None
+        all_attentions = () if output_attentions else None
+
+        # check if head_mask has a correct number of layers specified if desired
+        if head_mask is not None and head_mask.size()[0] != (len(self.layers)):
+            raise ValueError(
+                f"The head_mask should be specified for {len(self.layers)} layers, but it is for"
+                f" {head_mask.size()[0]}."
+            )
+
+        for idx, encoder_layer in enumerate(self.layers):
+            if output_hidden_states:
+                encoder_states = encoder_states + (hidden_states,)
+            # add LayerDrop (see https://arxiv.org/abs/1909.11556 for description)
+            to_drop = False
+            if self.training:
+                dropout_probability = torch.rand([])
+                if dropout_probability < self.layerdrop:  # skip the layer
+                    to_drop = True
+
+            if to_drop:
+                layer_outputs = (None, None)
+            else:
+                if self.gradient_checkpointing and self.training:
+                    layer_outputs = self._gradient_checkpointing_func(
+                        encoder_layer.__call__,
+                        hidden_states,
+                        attention_mask,
+                        (head_mask[idx] if head_mask is not None else None),
+                        output_attentions,
+                    )
+                else:
+                    layer_outputs = encoder_layer(
+                        hidden_states,
+                        attention_mask,
+                        layer_head_mask=(head_mask[idx] if head_mask is not None else None),
+                        output_attentions=output_attentions,
+                    )
+
+                hidden_states = layer_outputs[0]
+
+            if output_attentions:
+                all_attentions = all_attentions + (layer_outputs[1],)
+
+        if output_hidden_states:
+            encoder_states = encoder_states + (hidden_states,)
+
+        if not return_dict:
+            return tuple(v for v in [hidden_states, encoder_states, all_attentions] if v is not None)
+        return BaseModelOutput(
+            last_hidden_state=hidden_states, hidden_states=encoder_states, attentions=all_attentions
+        )
+
+
+class Florence2Decoder(Florence2LanguagePreTrainedModel):
+    """
+    Transformer decoder consisting of *config.decoder_layers* layers. Each layer is a [`Florence2DecoderLayer`]
+
+    Args:
+        config: Florence2LanguageConfig
+        embed_tokens (nn.Embedding): output embedding
+    """
+
+    def __init__(self, config: Florence2LanguageConfig, embed_tokens: nn.Embedding | None = None):
+        super().__init__(config)
+        self.dropout = config.dropout
+        self.layerdrop = config.decoder_layerdrop
+        self.padding_idx = config.pad_token_id
+        self.max_target_positions = config.max_position_embeddings
+        embed_scale = math.sqrt(config.d_model) if config.scale_embedding else 1.0
+
+        self.embed_tokens = Florence2ScaledWordEmbedding(
+            config.vocab_size, config.d_model, self.padding_idx, embed_scale=embed_scale
+        )
+
+        if embed_tokens is not None:
+            self.embed_tokens.weight = embed_tokens.weight
+
+        self.embed_positions = Florence2LearnedPositionalEmbedding(
+            config.max_position_embeddings,
+            config.d_model,
+        )
+        self.layers = nn.ModuleList([Florence2DecoderLayer(config) for _ in range(config.decoder_layers)])
+        self._use_flash_attention_2 = config._attn_implementation == "flash_attention_2"
+        self._use_sdpa = config._attn_implementation == "sdpa"
+
+        self.layernorm_embedding = nn.LayerNorm(config.d_model)
+
+        self.gradient_checkpointing = False
+        # Initialize weights and apply final processing
+        self.post_init()
+
+    def get_input_embeddings(self):
+        return self.embed_tokens
+
+    def set_input_embeddings(self, value):
+        self.embed_tokens = value
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        encoder_hidden_states: torch.FloatTensor | None = None,
+        encoder_attention_mask: torch.LongTensor | None = None,
+        head_mask: torch.Tensor | None = None,
+        cross_attn_head_mask: torch.Tensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+    ) -> tuple | BaseModelOutputWithPastAndCrossAttentions:
+        r"""
+        Args:
+            input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
+                Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you
+                provide it.
+
+                Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+                [`PreTrainedTokenizer.__call__`] for details.
+
+                [What are input IDs?](../glossary#input-ids)
+            attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
+                Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:
+
+                - 1 for tokens that are **not masked**,
+                - 0 for tokens that are **masked**.
+
+                [What are attention masks?](../glossary#attention-mask)
+            encoder_hidden_states (`torch.FloatTensor` of shape `(batch_size, encoder_sequence_length, hidden_size)`, *optional*):
+                Sequence of hidden-states at the output of the last layer of the encoder. Used in the cross-attention
+                of the decoder.
+            encoder_attention_mask (`torch.LongTensor` of shape `(batch_size, encoder_sequence_length)`, *optional*):
+                Mask to avoid performing cross-attention on padding tokens indices of encoder input_ids. Mask values
+                selected in `[0, 1]`:
+
+                - 1 for tokens that are **not masked**,
+                - 0 for tokens that are **masked**.
+
+                [What are attention masks?](../glossary#attention-mask)
+            head_mask (`torch.Tensor` of shape `(decoder_layers, decoder_attention_heads)`, *optional*):
+                Mask to nullify selected heads of the attention modules. Mask values selected in `[0, 1]`:
+
+                - 1 indicates the head is **not masked**,
+                - 0 indicates the head is **masked**.
+
+            cross_attn_head_mask (`torch.Tensor` of shape `(decoder_layers, decoder_attention_heads)`, *optional*):
+                Mask to nullify selected heads of the cross-attention modules in the decoder to avoid performing
+                cross-attention on hidden heads. Mask values selected in `[0, 1]`:
+
+                - 1 indicates the head is **not masked**,
+                - 0 indicates the head is **masked**.
+
+            past_key_values (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
+                Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of
+                shape `(batch_size, num_heads, sequence_length, embed_size_per_head)`) and 2 additional tensors of
+                shape `(batch_size, num_heads, encoder_sequence_length, embed_size_per_head)`.
+
+                Contains pre-computed hidden-states (key and values in the self-attention blocks and in the
+                cross-attention blocks) that can be used (see `past_key_values` input) to speed up sequential decoding.
+
+                If `past_key_values` are used, the user can optionally input only the last `decoder_input_ids` (those
+                that don't have their past key value states given to this model) of shape `(batch_size, 1)` instead of
+                all `decoder_input_ids` of shape `(batch_size, sequence_length)`.
+            inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
+                Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation.
+                This is useful if you want more control over how to convert `input_ids` indices into associated vectors
+                than the model's internal embedding lookup matrix.
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+            output_hidden_states (`bool`, *optional*):
+                Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors
+                for more detail.
+            return_dict (`bool`, *optional*):
+                Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
+        """
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        # retrieve input_ids and inputs_embeds
+        if input_ids is not None and inputs_embeds is not None:
+            raise ValueError(
+                "You cannot specify both decoder_input_ids and decoder_inputs_embeds at the same time"
+            )
+        elif input_ids is not None:
+            input = input_ids
+            input_shape = input.shape
+            input_ids = input_ids.view(-1, input_shape[-1])
+        elif inputs_embeds is not None:
+            input_shape = inputs_embeds.size()[:-1]
+            input = inputs_embeds[:, :, -1]
+        else:
+            raise ValueError("You have to specify either decoder_input_ids or decoder_inputs_embeds")
+
+        # past_key_values_length
+        past_key_values_length = past_key_values[0][0].shape[2] if past_key_values is not None else 0
+
+        if inputs_embeds is None:
+            inputs_embeds = self.embed_tokens(input)
+
+        if self._use_flash_attention_2:
+            # 2d mask is passed through the layers
+            attention_mask = attention_mask if (attention_mask is not None and 0 in attention_mask) else None
+        elif self._use_sdpa and not output_attentions and cross_attn_head_mask is None:
+            # output_attentions=True & cross_attn_head_mask can not be supported when using SDPA, and we fall back on
+            # the manual implementation that requires a 4D causal mask in all cases.
+            attention_mask = _prepare_4d_causal_attention_mask_for_sdpa(
+                attention_mask,
+                input_shape,
+                inputs_embeds,
+                past_key_values_length,
+            )
+        else:
+            # 4d mask is passed through the layers
+            attention_mask = _prepare_4d_causal_attention_mask(
+                attention_mask, input_shape, inputs_embeds, past_key_values_length
+            )
+
+        # expand encoder attention mask
+        if encoder_hidden_states is not None and encoder_attention_mask is not None:
+            if self._use_flash_attention_2:
+                encoder_attention_mask = encoder_attention_mask if 0 in encoder_attention_mask else None
+            elif self._use_sdpa and cross_attn_head_mask is None and not output_attentions:
+                # output_attentions=True & cross_attn_head_mask can not be supported when using SDPA, and we fall back on
+                # the manual implementation that requires a 4D causal mask in all cases.
+                # [bsz, seq_len] -> [bsz, 1, tgt_seq_len, src_seq_len]
+                encoder_attention_mask = _prepare_4d_attention_mask_for_sdpa(
+                    encoder_attention_mask,
+                    inputs_embeds.dtype,
+                    tgt_len=input_shape[-1],
+                )
+            else:
+                # [bsz, seq_len] -> [bsz, 1, tgt_seq_len, src_seq_len]
+                encoder_attention_mask = _prepare_4d_attention_mask(
+                    encoder_attention_mask, inputs_embeds.dtype, tgt_len=input_shape[-1]
+                )
+
+        # embed positions
+        positions = self.embed_positions(input, past_key_values_length)
+        positions = positions.to(inputs_embeds.device)
+
+        hidden_states = inputs_embeds + positions
+        hidden_states = self.layernorm_embedding(hidden_states)
+
+        hidden_states = nn.functional.dropout(hidden_states, p=self.dropout, training=self.training)
+
+        if self.gradient_checkpointing and self.training and use_cache:
+            logger.warning_once(
+                "`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`..."
+            )
+            use_cache = False
+
+        # decoder layers
+        all_hidden_states = () if output_hidden_states else None
+        all_self_attns = () if output_attentions else None
+        all_cross_attentions = () if (output_attentions and encoder_hidden_states is not None) else None
+        next_decoder_cache = () if use_cache else None
+
+        # check if head_mask/cross_attn_head_mask has a correct number of layers specified if desired
+        for attn_mask, mask_name in zip(
+            [head_mask, cross_attn_head_mask], ["head_mask", "cross_attn_head_mask"], strict=False
+        ):
+            if attn_mask is not None and attn_mask.size()[0] != (len(self.layers)):
+                raise ValueError(
+                    f"The `{mask_name}` should be specified for {len(self.layers)} layers, but it is for"
+                    f" {head_mask.size()[0]}."
+                )
+
+        for idx, decoder_layer in enumerate(self.layers):
+            # add LayerDrop (see https://arxiv.org/abs/1909.11556 for description)
+            if output_hidden_states:
+                all_hidden_states += (hidden_states,)
+            if self.training:
+                dropout_probability = torch.rand([])
+                if dropout_probability < self.layerdrop:
+                    continue
+
+            past_key_value = past_key_values[idx] if past_key_values is not None else None
+
+            if self.gradient_checkpointing and self.training:
+                layer_outputs = self._gradient_checkpointing_func(
+                    decoder_layer.__call__,
+                    hidden_states,
+                    attention_mask,
+                    encoder_hidden_states,
+                    encoder_attention_mask,
+                    head_mask[idx] if head_mask is not None else None,
+                    cross_attn_head_mask[idx] if cross_attn_head_mask is not None else None,
+                    None,
+                    output_attentions,
+                    use_cache,
+                )
+            else:
+                layer_outputs = decoder_layer(
+                    hidden_states,
+                    attention_mask=attention_mask,
+                    encoder_hidden_states=encoder_hidden_states,
+                    encoder_attention_mask=encoder_attention_mask,
+                    layer_head_mask=(head_mask[idx] if head_mask is not None else None),
+                    cross_attn_layer_head_mask=(
+                        cross_attn_head_mask[idx] if cross_attn_head_mask is not None else None
+                    ),
+                    past_key_value=past_key_value,
+                    output_attentions=output_attentions,
+                    use_cache=use_cache,
+                )
+            hidden_states = layer_outputs[0]
+
+            if use_cache:
+                next_decoder_cache += (layer_outputs[3 if output_attentions else 1],)
+
+            if output_attentions:
+                all_self_attns += (layer_outputs[1],)
+
+                if encoder_hidden_states is not None:
+                    all_cross_attentions += (layer_outputs[2],)
+
+        # add hidden states from the last decoder layer
+        if output_hidden_states:
+            all_hidden_states += (hidden_states,)
+
+        next_cache = next_decoder_cache if use_cache else None
+        if not return_dict:
+            return tuple(
+                v
+                for v in [hidden_states, next_cache, all_hidden_states, all_self_attns, all_cross_attentions]
+                if v is not None
+            )
+        return BaseModelOutputWithPastAndCrossAttentions(
+            last_hidden_state=hidden_states,
+            past_key_values=next_cache,
+            hidden_states=all_hidden_states,
+            attentions=all_self_attns,
+            cross_attentions=all_cross_attentions,
+        )
+
+
+class Florence2LanguageModel(Florence2LanguagePreTrainedModel):
+    _tied_weights_keys = {
+        "encoder.embed_tokens.weight": "shared.weight",
+        "decoder.embed_tokens.weight": "shared.weight",
+    }
+
+    def __init__(self, config: Florence2LanguageConfig):
+        super().__init__(config)
+
+        padding_idx, vocab_size = config.pad_token_id, config.vocab_size
+        self.shared = nn.Embedding(vocab_size, config.d_model, padding_idx)
+
+        self.encoder = Florence2Encoder(config, self.shared)
+        self.decoder = Florence2Decoder(config, self.shared)
+
+        # Initialize weights and apply final processing
+        self.post_init()
+
+    def _tie_weights(self):
+        if self.config.tie_word_embeddings:
+            self._tie_or_clone_weights(self.encoder.embed_tokens, self.shared)
+            # self._tie_or_clone_weights(self.decoder.embed_tokens, self.shared)
+
+    def get_input_embeddings(self):
+        return self.shared
+
+    def set_input_embeddings(self, value):
+        self.shared = value
+        self.encoder.embed_tokens = self.shared
+        self.decoder.embed_tokens = self.shared
+
+    def get_encoder(self):
+        return self.encoder
+
+    def get_decoder(self):
+        return self.decoder
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        decoder_input_ids: torch.LongTensor | None = None,
+        decoder_attention_mask: torch.LongTensor | None = None,
+        head_mask: torch.Tensor | None = None,
+        decoder_head_mask: torch.Tensor | None = None,
+        cross_attn_head_mask: torch.Tensor | None = None,
+        encoder_outputs: list[torch.FloatTensor] | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        decoder_inputs_embeds: torch.FloatTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+    ) -> tuple | Seq2SeqModelOutput:
+        # different to other models, Florence2 automatically creates decoder_input_ids from
+        # input_ids if no decoder_input_ids are provided
+        if decoder_input_ids is None and decoder_inputs_embeds is None:
+            if input_ids is None:
+                raise ValueError(
+                    "If no `decoder_input_ids` or `decoder_inputs_embeds` are "
+                    "passed, `input_ids` cannot be `None`. Please pass either "
+                    "`input_ids` or `decoder_input_ids` or `decoder_inputs_embeds`."
+                )
+
+            decoder_input_ids = shift_tokens_right(
+                input_ids, self.config.pad_token_id, self.config.decoder_start_token_id
+            )
+
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        if encoder_outputs is None:
+            encoder_outputs = self.encoder(
+                input_ids=input_ids,
+                attention_mask=attention_mask,
+                head_mask=head_mask,
+                inputs_embeds=inputs_embeds,
+                output_attentions=output_attentions,
+                output_hidden_states=output_hidden_states,
+                return_dict=return_dict,
+            )
+        # If the user passed a tuple for encoder_outputs, we wrap it in a BaseModelOutput when return_dict=True
+        elif return_dict and not isinstance(encoder_outputs, BaseModelOutput):
+            encoder_outputs = BaseModelOutput(
+                last_hidden_state=encoder_outputs[0],
+                hidden_states=encoder_outputs[1] if len(encoder_outputs) > 1 else None,
+                attentions=encoder_outputs[2] if len(encoder_outputs) > 2 else None,
+            )
+
+        # decoder outputs consists of (dec_features, past_key_value, dec_hidden, dec_attn)
+        decoder_outputs = self.decoder(
+            input_ids=decoder_input_ids,
+            attention_mask=decoder_attention_mask,
+            encoder_hidden_states=encoder_outputs[0],
+            encoder_attention_mask=attention_mask,
+            head_mask=decoder_head_mask,
+            cross_attn_head_mask=cross_attn_head_mask,
+            past_key_values=past_key_values,
+            inputs_embeds=decoder_inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+        )
+
+        if not return_dict:
+            return decoder_outputs + encoder_outputs
+
+        return Seq2SeqModelOutput(
+            last_hidden_state=decoder_outputs.last_hidden_state,
+            past_key_values=decoder_outputs.past_key_values,
+            decoder_hidden_states=decoder_outputs.hidden_states,
+            decoder_attentions=decoder_outputs.attentions,
+            cross_attentions=decoder_outputs.cross_attentions,
+            encoder_last_hidden_state=encoder_outputs.last_hidden_state,
+            encoder_hidden_states=encoder_outputs.hidden_states,
+            encoder_attentions=encoder_outputs.attentions,
+        )
+
+
+class Florence2LanguageForConditionalGeneration(Florence2LanguagePreTrainedModel, GenerationMixin):
+    base_model_prefix = "model"
+    _tied_weights_keys = {
+        "model.encoder.embed_tokens.weight": "model.shared.weight",
+        "model.decoder.embed_tokens.weight": "model.shared.weight",
+    }
+    _keys_to_ignore_on_load_missing = ["final_logits_bias"]
+
+    def __init__(self, config: Florence2LanguageConfig):
+        super().__init__(config)
+        self.model = Florence2LanguageModel(config)
+        self.register_buffer("final_logits_bias", torch.zeros((1, self.model.shared.num_embeddings)))
+        self.lm_head = nn.Linear(config.d_model, self.model.shared.num_embeddings, bias=False)
+
+        # Initialize weights and apply final processing
+        self.post_init()
+
+    def _tie_weights(self):
+        if self.config.tie_word_embeddings:
+            self._tie_or_clone_weights(self.model.encoder.embed_tokens, self.model.shared)
+            # self._tie_or_clone_weights(self.model.decoder.embed_tokens, self.model.shared)
+            # self._tie_or_clone_weights(self.lm_head, self.model.shared)
+
+    def get_encoder(self):
+        return self.model.get_encoder()
+
+    def get_decoder(self):
+        return self.model.get_decoder()
+
+    def resize_token_embeddings(
+        self, new_num_tokens: int, pad_to_multiple_of: int | None = None, **kwargs
+    ) -> nn.Embedding:
+        new_embeddings = super().resize_token_embeddings(new_num_tokens, pad_to_multiple_of, **kwargs)
+        self._resize_final_logits_bias(new_embeddings.weight.shape[0])
+        return new_embeddings
+
+    def _resize_final_logits_bias(self, new_num_tokens: int) -> None:
+        old_num_tokens = self.final_logits_bias.shape[-1]
+        if new_num_tokens <= old_num_tokens:
+            new_bias = self.final_logits_bias[:, :new_num_tokens]
+        else:
+            extra_bias = torch.zeros(
+                (1, new_num_tokens - old_num_tokens), device=self.final_logits_bias.device
+            )
+            new_bias = torch.cat([self.final_logits_bias, extra_bias], dim=1)
+        self.register_buffer("final_logits_bias", new_bias)
+
+    def get_output_embeddings(self):
+        return self.lm_head
+
+    def set_output_embeddings(self, new_embeddings):
+        self.lm_head = new_embeddings
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        decoder_input_ids: torch.LongTensor | None = None,
+        decoder_attention_mask: torch.LongTensor | None = None,
+        head_mask: torch.Tensor | None = None,
+        decoder_head_mask: torch.Tensor | None = None,
+        cross_attn_head_mask: torch.Tensor | None = None,
+        encoder_outputs: list[torch.FloatTensor] | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        decoder_inputs_embeds: torch.FloatTensor | None = None,
+        labels: torch.LongTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+    ) -> tuple | Seq2SeqLMOutput:
+        r"""
+        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
+            config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
+            (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.
+
+        Returns:
+        """
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        if labels is not None:
+            if use_cache:
+                logger.warning("The `use_cache` argument is changed to `False` since `labels` is provided.")
+            use_cache = False
+            if decoder_input_ids is None and decoder_inputs_embeds is None:
+                decoder_input_ids = shift_tokens_right(
+                    labels, self.config.pad_token_id, self.config.decoder_start_token_id
+                )
+
+        outputs = self.model(
+            input_ids,
+            attention_mask=attention_mask,
+            decoder_input_ids=decoder_input_ids,
+            encoder_outputs=encoder_outputs,
+            decoder_attention_mask=decoder_attention_mask,
+            head_mask=head_mask,
+            decoder_head_mask=decoder_head_mask,
+            cross_attn_head_mask=cross_attn_head_mask,
+            past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds,
+            decoder_inputs_embeds=decoder_inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+        )
+
+        lm_logits = self.lm_head(outputs[0])
+        lm_logits = lm_logits + self.final_logits_bias.to(lm_logits.device)
+
+        masked_lm_loss = None
+        if labels is not None:
+            labels = labels.to(lm_logits.device)
+            loss_fct = CrossEntropyLoss()
+            masked_lm_loss = loss_fct(lm_logits.view(-1, self.config.vocab_size), labels.view(-1))
+
+        if not return_dict:
+            output = (lm_logits,) + outputs[1:]
+            return ((masked_lm_loss,) + output) if masked_lm_loss is not None else output
+
+        return Seq2SeqLMOutput(
+            loss=masked_lm_loss,
+            logits=lm_logits,
+            past_key_values=outputs.past_key_values,
+            decoder_hidden_states=outputs.decoder_hidden_states,
+            decoder_attentions=outputs.decoder_attentions,
+            cross_attentions=outputs.cross_attentions,
+            encoder_last_hidden_state=outputs.encoder_last_hidden_state,
+            encoder_hidden_states=outputs.encoder_hidden_states,
+            encoder_attentions=outputs.encoder_attentions,
+        )
+
+    def prepare_inputs_for_generation(
+        self,
+        decoder_input_ids,
+        past_key_values=None,
+        attention_mask=None,
+        decoder_attention_mask=None,
+        head_mask=None,
+        decoder_head_mask=None,
+        cross_attn_head_mask=None,
+        use_cache=None,
+        encoder_outputs=None,
+        **kwargs,
+    ):
+        # cut decoder_input_ids if past_key_values is used
+        if past_key_values is not None:
+            past_length = past_key_values[0][0].shape[2]
+
+            # Some generation methods already pass only the last input ID
+            if decoder_input_ids.shape[1] > past_length:
+                remove_prefix_length = past_length
+            else:
+                # Default to old behavior: keep only final ID
+                remove_prefix_length = decoder_input_ids.shape[1] - 1
+
+            decoder_input_ids = decoder_input_ids[:, remove_prefix_length:]
+
+        return {
+            "input_ids": None,  # encoder_outputs is defined. input_ids not needed
+            "encoder_outputs": encoder_outputs,
+            "past_key_values": past_key_values,
+            "decoder_input_ids": decoder_input_ids,
+            "attention_mask": attention_mask,
+            "decoder_attention_mask": decoder_attention_mask,
+            "head_mask": head_mask,
+            "decoder_head_mask": decoder_head_mask,
+            "cross_attn_head_mask": cross_attn_head_mask,
+            "use_cache": use_cache,  # change this to avoid caching (presumably for debugging)
+        }
+
+    def prepare_decoder_input_ids_from_labels(self, labels: torch.Tensor):
+        return shift_tokens_right(labels, self.config.pad_token_id, self.config.decoder_start_token_id)
+
+    @staticmethod
+    def _reorder_cache(past_key_values, beam_idx):
+        reordered_past = ()
+        for layer_past in past_key_values:
+            # cached cross_attention states don't have to be reordered -> they are always the same
+            reordered_past += (
+                tuple(
+                    past_state.index_select(0, beam_idx.to(past_state.device))
+                    for past_state in layer_past[:2]
+                )
+                + layer_past[2:],
+            )
+        return reordered_past
+
+
+@dataclass
+class Florence2Seq2SeqLMOutput(ModelOutput):
+    """
+    Base class for Florence-2 model's outputs that also contains : pre-computed hidden states that can speed up sequential
+    decoding.
+
+    Args:
+        loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided):
+            Language modeling loss.
+        logits (`torch.FloatTensor` of shape `(batch_size, sequence_length, config.vocab_size)`):
+            Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).
+        last_hidden_state (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
+            Sequence of hidden-states at the output of the last layer of the decoder of the model.
+
+            If `past_key_values` is used only the last hidden-state of the sequences of shape `(batch_size, 1,
+            hidden_size)` is output.
+        past_key_values (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
+            Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of shape
+            `(batch_size, num_heads, sequence_length, embed_size_per_head)`) and 2 additional tensors of shape
+            `(batch_size, num_heads, encoder_sequence_length, embed_size_per_head)`.
+
+            Contains pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention
+            blocks) that can be used (see `past_key_values` input) to speed up sequential decoding.
+        decoder_hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
+            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
+            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
+
+            Hidden-states of the decoder at the output of each layer plus the optional initial embedding outputs.
+        decoder_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
+            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
+            sequence_length)`.
+
+            Attentions weights of the decoder, after the attention softmax, used to compute the weighted average in the
+            self-attention heads.
+        cross_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
+            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
+            sequence_length)`.
+
+            Attentions weights of the decoder's cross-attention layer, after the attention softmax, used to compute the
+            weighted average in the cross-attention heads.
+        encoder_last_hidden_state (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
+            Sequence of hidden-states at the output of the last layer of the encoder of the model.
+        encoder_hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
+            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
+            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
+
+            Hidden-states of the encoder at the output of each layer plus the optional initial embedding outputs.
+        encoder_attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
+            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
+            sequence_length)`.
+
+            Attentions weights of the encoder, after the attention softmax, used to compute the weighted average in the
+            self-attention heads.
+        image_hidden_states (`tuple(torch.FloatTensor)`, *optional*):
+            Tuple of `torch.FloatTensor` (one for the output of the image embeddings, `(batch_size,
+            num_image_tokens, hidden_size)`.
+
+            image_hidden_states of the model produced by the vision encoder
+    """
+
+    loss: torch.FloatTensor | None = None
+    logits: torch.FloatTensor = None
+    last_hidden_state: torch.FloatTensor = None
+    past_key_values: tuple[tuple[torch.FloatTensor]] | None = None
+    decoder_hidden_states: tuple[torch.FloatTensor, ...] | None = None
+    decoder_attentions: tuple[torch.FloatTensor, ...] | None = None
+    cross_attentions: tuple[torch.FloatTensor, ...] | None = None
+    encoder_last_hidden_state: torch.FloatTensor | None = None
+    encoder_hidden_states: tuple[torch.FloatTensor, ...] | None = None
+    encoder_attentions: tuple[torch.FloatTensor, ...] | None = None
+    image_hidden_states: tuple[torch.FloatTensor, ...] | None = None
+
+
+FLORENCE2_START_DOCSTRING = r"""
+    This model inherits from [`PreTrainedModel`]. Check the superclass documentation for the generic methods the
+    library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads
+    etc.)
+
+    This model is also a PyTorch [torch.nn.Module](https://pytorch.org/docs/stable/nn.html#torch.nn.Module) subclass.
+    Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage
+    and behavior.
+
+    Parameters:
+        config ([`Florence2Config`] or [`Florence2VisionConfig`]):
+            Model configuration class with all the parameters of the model. Initializing with a config file does not
+            load the weights associated with the model, only the configuration. Check out the
+            [`~PreTrainedModel.from_pretrained`] method to load the model weights.
+"""
+
+
+@add_start_docstrings(
+    "The bare Florence-2 Model outputting raw hidden-states without any specific head on top.",
+    FLORENCE2_START_DOCSTRING,
+)
+class Florence2PreTrainedModel(PreTrainedModel):
+    config_class = Florence2Config
+    base_model_prefix = "model"
+    supports_gradient_checkpointing = True
+    _skip_keys_device_placement = "past_key_values"
+    _supports_flash_attn_2 = True
+    _supports_sdpa = True
+
+
+FLORENCE2_INPUTS_DOCSTRING = r"""
+    Args:
+        input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
+            Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide
+            it.
+
+            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+
+            [What are input IDs?](../glossary#input-ids)
+        pixel_values (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)):
+            The tensors corresponding to the input images. Pixel values can be obtained using
+            [`AutoImageProcessor`]. See [`CLIPImageProcessor.__call__`] for details ([]`Florence2Processor`] uses
+            [`CLIPImageProcessor`] for processing images).
+        attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:
+
+            - 1 for tokens that are **not masked**,
+            - 0 for tokens that are **masked**.
+
+            [What are attention masks?](../glossary#attention-mask)
+
+            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+
+            If `past_key_values` is used, optionally only the last `decoder_input_ids` have to be input (see
+            `past_key_values`).
+
+            If you want to change padding behavior, you should read [`modeling_opt._prepare_decoder_attention_mask`]
+            and modify to your needs. See diagram 1 in [the paper](https://arxiv.org/abs/1910.13461) for more
+            information on the default strategy.
+
+            - 1 indicates the head is **not masked**,
+            - 0 indicates the head is **masked**.
+        position_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
+            config.n_positions - 1]`. [What are position IDs?](../glossary#position-ids)
+        past_key_values (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
+            Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of shape
+            `(batch_size, num_heads, sequence_length, embed_size_per_head)`) and 2 additional tensors of shape
+            `(batch_size, num_heads, encoder_sequence_length, embed_size_per_head)`.
+
+            Contains pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention
+            blocks) that can be used (see `past_key_values` input) to speed up sequential decoding.
+
+            If `past_key_values` are used, the user can optionally input only the last `decoder_input_ids` (those that
+            don't have their past key value states given to this model) of shape `(batch_size, 1)` instead of all
+            `decoder_input_ids` of shape `(batch_size, sequence_length)`.
+        inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
+            Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
+            is useful if you want more control over how to convert `input_ids` indices into associated vectors than the
+            model's internal embedding lookup matrix.
+        use_cache (`bool`, *optional*):
+            If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding (see
+            `past_key_values`).
+        output_attentions (`bool`, *optional*):
+            Whether or not to return the attentions tensors of all attention layers. See `attentions` under returned
+            tensors for more detail.
+        output_hidden_states (`bool`, *optional*):
+            Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors for
+            more detail.
+        return_dict (`bool`, *optional*):
+            Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
+"""
+
+
+@add_start_docstrings(
+    """The FLORENCE2 model which consists of a vision backbone and a language model.""",
+    FLORENCE2_START_DOCSTRING,
+)
+class Florence2ForConditionalGeneration(Florence2PreTrainedModel):
+    _tied_weights_keys = {
+        "language_model.model.encoder.embed_tokens.weight": "language_model.model.shared.weight",
+        "language_model.model.decoder.embed_tokens.weight": "language_model.model.shared.weight",
+    }
+
+    def __init__(self, config: Florence2Config):
+        super().__init__(config)
+        assert config.vision_config.model_type == "davit", "only DaViT is supported for now"
+        self.vision_tower = DaViT.from_config(config=config.vision_config)
+        # remove unused layers
+        del self.vision_tower.head
+        del self.vision_tower.norms
+
+        self.vocab_size = config.vocab_size
+        self._attn_implementation = config._attn_implementation
+        self._build_image_projection_layers(config)
+
+        language_model = Florence2LanguageForConditionalGeneration(config=config.text_config)
+
+        self.language_model = language_model
+
+        self.pad_token_id = self.config.pad_token_id if self.config.pad_token_id is not None else -1
+        self.post_init()
+
+    def _build_image_projection_layers(self, config):
+        image_dim_out = config.vision_config.dim_embed[-1]
+        dim_projection = config.vision_config.projection_dim
+        self.image_projection = nn.Parameter(torch.empty(image_dim_out, dim_projection))
+        self.image_proj_norm = nn.LayerNorm(dim_projection)
+        image_pos_embed_config = config.vision_config.image_pos_embed
+        if image_pos_embed_config["type"] == "learned_abs_2d":
+            self.image_pos_embed = LearnedAbsolutePositionEmbedding2D(
+                embedding_dim=image_dim_out, num_pos=image_pos_embed_config["max_pos_embeddings"]
+            )
+        else:
+            raise NotImplementedError("Not implemented yet")
+
+        self.image_feature_source = config.vision_config.image_feature_source
+
+        # temporal embedding
+        visual_temporal_embedding_config = config.vision_config.visual_temporal_embedding
+        if visual_temporal_embedding_config["type"] == "COSINE":
+            self.visual_temporal_embed = PositionalEmbeddingCosine1D(
+                embed_dim=image_dim_out,
+                max_seq_len=visual_temporal_embedding_config["max_temporal_embeddings"],
+            )
+        else:
+            raise NotImplementedError("Not implemented yet")
+
+    def get_encoder(self):
+        return self.language_model.get_encoder()
+
+    def get_decoder(self):
+        return self.language_model.get_decoder()
+
+    def get_input_embeddings(self):
+        return self.language_model.get_input_embeddings()
+
+    def resize_token_embeddings(
+        self, new_num_tokens: int | None = None, pad_to_multiple_of=None, **kwargs
+    ) -> nn.Embedding:
+        model_embeds = self.language_model.resize_token_embeddings(
+            new_num_tokens, pad_to_multiple_of, **kwargs
+        )
+        # update vocab size
+        self.config.text_config.vocab_size = model_embeds.num_embeddings
+        self.config.vocab_size = model_embeds.num_embeddings
+        self.vocab_size = model_embeds.num_embeddings
+        return model_embeds
+
+    def _encode_image(self, pixel_values):
+        # Cast pixel_values to model's dtype
+        pixel_values = pixel_values.to(dtype=self.vision_tower.convs[0].proj.weight.dtype)
+
+        if len(pixel_values.shape) == 4:
+            batch_size, channels, height, width = pixel_values.shape
+            num_frames = 1
+            x = self.vision_tower.forward_features_unpool(pixel_values)
+        else:
+            raise ValueError(f"invalid image shape {pixel_values.shape}")
+
+        if self.image_pos_embed is not None:
+            x = x.view(batch_size * num_frames, -1, x.shape[-1])
+            num_tokens = x.shape[-2]
+            h, w = int(num_tokens**0.5), int(num_tokens**0.5)
+            assert h * w == num_tokens, "only support square feature maps for now"
+            x = x.view(batch_size * num_frames, h, w, x.shape[-1])
+            pos_embed = self.image_pos_embed(x)
+            x = x + pos_embed
+            x = x.view(batch_size, num_frames * h * w, x.shape[-1])
+
+        if self.visual_temporal_embed is not None:
+            visual_temporal_embed = self.visual_temporal_embed(
+                x.view(batch_size, num_frames, -1, x.shape[-1])[:, :, 0]
+            )
+            x = x.view(batch_size, num_frames, -1, x.shape[-1]) + visual_temporal_embed.view(
+                1, num_frames, 1, x.shape[-1]
+            )
+
+        x_feat_dict = {}
+
+        spatial_avg_pool_x = x.view(batch_size, num_frames, -1, x.shape[-1]).mean(dim=2)
+        x_feat_dict["spatial_avg_pool"] = spatial_avg_pool_x
+
+        temporal_avg_pool_x = x.view(batch_size, num_frames, -1, x.shape[-1]).mean(dim=1)
+        x_feat_dict["temporal_avg_pool"] = temporal_avg_pool_x
+
+        x = x.view(batch_size, num_frames, -1, x.shape[-1])[:, -1]
+        x_feat_dict["last_frame"] = x
+
+        new_x = []
+        for _image_feature_source in self.image_feature_source:
+            if _image_feature_source not in x_feat_dict:
+                raise ValueError(f"invalid image feature source: {_image_feature_source}")
+            new_x.append(x_feat_dict[_image_feature_source])
+
+        x = torch.cat(new_x, dim=1)
+
+        x = x @ self.image_projection
+        x = self.image_proj_norm(x)
+
+        return x
+
+    def _merge_input_ids_with_image_features(self, image_features, inputs_embeds):
+        batch_size, image_token_length = image_features.size()[:-1]
+        device = image_features.device
+        image_attention_mask = torch.ones(batch_size, image_token_length, device=device)
+
+        # task_prefix_embeds: [batch_size, padded_context_length, hidden_size]
+        # task_prefix_attention_mask: [batch_size, context_length]
+        if inputs_embeds is None:
+            return image_features, image_attention_mask
+
+        task_prefix_embeds = inputs_embeds
+        task_prefix_attention_mask = torch.ones(batch_size, task_prefix_embeds.size(1), device=device)
+
+        if len(task_prefix_attention_mask.shape) == 3:
+            task_prefix_attention_mask = task_prefix_attention_mask[:, 0]
+
+        # concat [image embeds, task prefix embeds]
+        inputs_embeds = torch.cat([image_features, task_prefix_embeds], dim=1)
+        attention_mask = torch.cat([image_attention_mask, task_prefix_attention_mask], dim=1)
+
+        return inputs_embeds, attention_mask
+
+    @add_start_docstrings_to_model_forward(FLORENCE2_INPUTS_DOCSTRING)
+    @replace_return_docstrings(output_type=Florence2Seq2SeqLMOutput, config_class=_CONFIG_FOR_DOC)
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        pixel_values: torch.FloatTensor = None,
+        attention_mask: torch.Tensor | None = None,
+        decoder_input_ids: torch.LongTensor | None = None,
+        decoder_attention_mask: torch.LongTensor | None = None,
+        head_mask: torch.Tensor | None = None,
+        decoder_head_mask: torch.Tensor | None = None,
+        cross_attn_head_mask: torch.Tensor | None = None,
+        encoder_outputs: list[torch.FloatTensor] | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: torch.FloatTensor | None = None,
+        decoder_inputs_embeds: torch.FloatTensor | None = None,
+        labels: torch.LongTensor | None = None,
+        use_cache: bool | None = None,
+        output_attentions: bool | None = None,
+        output_hidden_states: bool | None = None,
+        return_dict: bool | None = None,
+    ) -> tuple | Florence2Seq2SeqLMOutput:
+        r"""
+        Args:
+            labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+                Labels for computing the masked language modeling loss. Indices should either be in `[0, ...,
+                config.vocab_size]` or -100 (see `input_ids` docstring). Tokens with indices set to `-100` are ignored
+                (masked), the loss is only computed for the tokens with labels in `[0, ..., config.vocab_size]`.
+
+        Returns:
+
+        Example:
+
+        ```python
+        >>> from PIL import Image
+        >>> import requests
+        >>> from transformers import AutoProcessor, Florence2ForConditionalGeneration
+
+        >>> model = Florence2ForConditionalGeneration.from_pretrained("microsoft/Florence-2-large")
+        >>> processor = AutoProcessor.from_pretrained("microsoft/Florence-2-large")
+
+        >>> prompt = "<CAPTION>"
+        >>> url = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/transformers/tasks/car.jpg"
+        >>> image = Image.open(requests.get(url, stream=True).raw)
+
+        >>> inputs = processor(text=prompt, images=image, return_tensors="pt")
+
+        >>> # Generate
+        >>> generate_ids = model.generate(**inputs, max_length=100)
+        >>> processor.batch_decode(generate_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
+        "A green car parked in front of a yellow building."
+        ```"""
+        output_attentions = (
+            output_attentions if output_attentions is not None else self.config.output_attentions
+        )
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+
+        image_features = None
+        if inputs_embeds is None:
+            # 1. Extra the input embeddings
+            if input_ids is not None:
+                inputs_embeds = self.get_input_embeddings()(input_ids)
+            # 2. Merge text and images
+            if pixel_values is not None:
+                # (batch_size, num_image_tokens, hidden_size)
+                image_features = self._encode_image(pixel_values)
+                inputs_embeds, attention_mask = self._merge_input_ids_with_image_features(
+                    image_features, inputs_embeds
+                )
+
+        if inputs_embeds is not None:
+            attention_mask = attention_mask.to(inputs_embeds.dtype)
+        outputs = self.language_model(
+            attention_mask=attention_mask,
+            labels=labels,
+            inputs_embeds=inputs_embeds,
+            decoder_input_ids=decoder_input_ids,
+            encoder_outputs=encoder_outputs,
+            decoder_attention_mask=decoder_attention_mask,
+            head_mask=head_mask,
+            decoder_head_mask=decoder_head_mask,
+            cross_attn_head_mask=cross_attn_head_mask,
+            past_key_values=past_key_values,
+            decoder_inputs_embeds=decoder_inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+        )
+
+        logits = outputs.logits
+        logits = logits.float()
+        loss = outputs.loss
+        if not return_dict:
+            output = (logits,) + outputs[1:]
+            return (loss,) + output if loss is not None else output
+
+        return Florence2Seq2SeqLMOutput(
+            loss=loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            decoder_hidden_states=outputs.decoder_hidden_states,
+            decoder_attentions=outputs.decoder_attentions,
+            cross_attentions=outputs.cross_attentions,
+            encoder_last_hidden_state=outputs.encoder_last_hidden_state,
+            encoder_hidden_states=outputs.encoder_hidden_states,
+            encoder_attentions=outputs.encoder_attentions,
+            image_hidden_states=image_features,
+        )
+
+    def generate(self, input_ids, inputs_embeds=None, pixel_values=None, **kwargs):
+        if inputs_embeds is None:
+            # 1. Extra the input embeddings
+            if input_ids is not None:
+                inputs_embeds = self.get_input_embeddings()(input_ids)
+            # 2. Merge text and images
+            if pixel_values is not None:
+                image_features = self._encode_image(pixel_values)
+                inputs_embeds, attention_mask = self._merge_input_ids_with_image_features(
+                    image_features, inputs_embeds
+                )
+
+        return self.language_model.generate(input_ids=None, inputs_embeds=inputs_embeds, **kwargs)
+
+    def prepare_inputs_for_generation(
+        self,
+        decoder_input_ids,
+        past_key_values=None,
+        attention_mask=None,
+        pixel_values=None,
+        decoder_attention_mask=None,
+        head_mask=None,
+        decoder_head_mask=None,
+        cross_attn_head_mask=None,
+        use_cache=None,
+        encoder_outputs=None,
+        **kwargs,
+    ):
+        # cut decoder_input_ids if past_key_values is used
+        if past_key_values is not None:
+            past_length = past_key_values[0][0].shape[2]
+
+            # Some generation methods already pass only the last input ID
+            if decoder_input_ids.shape[1] > past_length:
+                remove_prefix_length = past_length
+            else:
+                # Default to old behavior: keep only final ID
+                remove_prefix_length = decoder_input_ids.shape[1] - 1
+
+            decoder_input_ids = decoder_input_ids[:, remove_prefix_length:]
+
+        return {
+            "input_ids": None,  # encoder_outputs is defined. input_ids not needed
+            "encoder_outputs": encoder_outputs,
+            "past_key_values": past_key_values,
+            "decoder_input_ids": decoder_input_ids,
+            "attention_mask": attention_mask,
+            "pixel_values": pixel_values,
+            "decoder_attention_mask": decoder_attention_mask,
+            "head_mask": head_mask,
+            "decoder_head_mask": decoder_head_mask,
+            "cross_attn_head_mask": cross_attn_head_mask,
+            "use_cache": use_cache,  # change this to avoid caching (presumably for debugging)
+        }
+
+    def prepare_decoder_input_ids_from_labels(self, labels: torch.Tensor):
+        return self.language_model.shift_tokens_right(labels)
+
+    def _reorder_cache(self, *args, **kwargs):
+        return self.language_model._reorder_cache(*args, **kwargs)
diff --git a/lerobot/src/lerobot/policies/xvla/modeling_xvla.py b/lerobot/src/lerobot/policies/xvla/modeling_xvla.py
new file mode 100644
index 0000000000000000000000000000000000000000..0436ae52772f1c725e6c04e54a8fcf9f21512583
--- /dev/null
+++ b/lerobot/src/lerobot/policies/xvla/modeling_xvla.py
@@ -0,0 +1,548 @@
+#!/usr/bin/env python
+
+# ------------------------------------------------------------------------------
+# Copyright 2025 The HuggingFace Inc. team and 2toINF (https://github.com/2toINF)
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# ------------------------------------------------------------------------------
+
+from __future__ import annotations
+
+import builtins
+import logging
+import os
+from collections import deque
+from pathlib import Path
+
+import torch
+import torch.nn.functional as F  # noqa: N812
+from torch import Tensor, nn
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.policies.pretrained import PreTrainedPolicy, T
+from lerobot.policies.utils import populate_queues
+from lerobot.utils.constants import ACTION, OBS_LANGUAGE_TOKENS, OBS_STATE
+
+from .action_hub import build_action_space
+from .configuration_florence2 import Florence2Config
+from .configuration_xvla import XVLAConfig
+from .modeling_florence2 import Florence2ForConditionalGeneration
+from .soft_transformer import SoftPromptedTransformer
+
+
+class XVLAModel(nn.Module):
+    """
+    XVLA backbone that stitches Florence-2 embeddings with the temporal/action transformer head.
+    """
+
+    def __init__(
+        self,
+        config: XVLAConfig,
+        florence_config: Florence2Config,
+        proprio_dim: int,
+    ) -> None:
+        super().__init__()
+        self.config = config
+        self.chunk_size: int = config.chunk_size
+        self.use_proprio: bool = config.use_proprio
+
+        # Build action space with auto-detection for "auto" mode
+        if config.action_mode.lower() == "auto":
+            # Auto-detect real action dim from config.action_feature
+            real_dim = (
+                config.action_feature.shape[-1]
+                if config.action_feature is not None
+                else config.max_action_dim
+            )
+            self.action_space = build_action_space(
+                config.action_mode.lower(),
+                real_dim=real_dim,
+                max_dim=config.max_action_dim,
+            )
+        else:
+            self.action_space = build_action_space(config.action_mode.lower())
+
+        self.dim_action = self.action_space.dim_action
+        self.dim_proprio = proprio_dim
+
+        self.vlm = Florence2ForConditionalGeneration(florence_config)
+        if hasattr(self.vlm, "language_model"):
+            lm = self.vlm.language_model
+            if hasattr(lm, "model") and hasattr(lm.model, "decoder"):
+                del lm.model.decoder
+            if hasattr(lm, "lm_head"):
+                del lm.lm_head
+
+        projection_dim = getattr(self.vlm.config, "projection_dim", None)
+        if projection_dim is None:
+            raise ValueError("Florence2 config must provide `projection_dim` for multimodal fusion.")
+
+        self.transformer = SoftPromptedTransformer(
+            hidden_size=config.hidden_size,
+            multi_modal_input_size=projection_dim,
+            depth=config.depth,
+            num_heads=config.num_heads,
+            mlp_ratio=config.mlp_ratio,
+            num_domains=config.num_domains,
+            dim_action=self.dim_action,
+            dim_propio=self.dim_proprio,
+            len_soft_prompts=config.len_soft_prompts,
+            dim_time=config.dim_time,
+            max_len_seq=config.max_len_seq,
+            use_hetero_proj=config.use_hetero_proj,
+        )
+
+        # Apply freezing based on config
+        self._apply_freezing()
+
+        # Apply dtype casting based on config
+        self._apply_dtype()
+
+    def _get_target_dtype(self) -> torch.dtype:
+        """Get the target dtype based on config."""
+        if self.config.dtype == "bfloat16":
+            return torch.bfloat16
+        return torch.float32
+
+    def _apply_dtype(self) -> None:
+        """
+        Apply dtype casting to model components based on config.
+        """
+        target_dtype = self._get_target_dtype()
+        self.to(dtype=target_dtype)
+
+    def _apply_freezing(self) -> None:
+        """
+        Freeze VLM vision and language encoders based on config options.
+        Keep only policy transformer and soft prompts trainable.
+        """
+        # Freeze vision encoder
+        if self.config.freeze_vision_encoder and hasattr(self.vlm, "vision_tower"):
+            for param in self.vlm.vision_tower.parameters():
+                param.requires_grad = False
+
+        # Freeze language encoder
+        if self.config.freeze_language_encoder and hasattr(self.vlm, "language_model"):
+            lm = self.vlm.language_model
+            # Freeze encoder
+            if hasattr(lm, "model") and hasattr(lm.model, "encoder"):
+                for param in lm.model.encoder.parameters():
+                    param.requires_grad = False
+            # Freeze shared embeddings
+            if hasattr(lm, "model") and hasattr(lm.model, "shared"):
+                for param in lm.model.shared.parameters():
+                    param.requires_grad = False
+
+        # Freeze or unfreeze policy transformer
+        if not self.config.train_policy_transformer:
+            for name, param in self.transformer.named_parameters():
+                if "soft_prompts" not in name:
+                    param.requires_grad = False
+
+        # Freeze or unfreeze soft prompts
+        if not self.config.train_soft_prompts and hasattr(self.transformer, "soft_prompt_hub"):
+            for param in self.transformer.soft_prompt_hub.parameters():
+                param.requires_grad = False
+
+    def forward_vlm(
+        self,
+        input_ids: torch.LongTensor,
+        pixel_values: torch.FloatTensor,
+        image_mask: torch.Tensor,
+    ) -> dict[str, torch.Tensor]:
+        """
+        Encode text and multi-view images via Florence2 encoder.
+        """
+        batch_size, num_views = pixel_values.shape[:2]
+        flat_mask = image_mask.view(-1).to(dtype=torch.bool)
+        flat_images = pixel_values.flatten(0, 1)
+        num_valid = int(flat_mask.sum().item())
+        if num_valid == 0:
+            raise ValueError("At least one image view must be valid per batch.")
+
+        valid_images = flat_images[flat_mask]
+        valid_feats = self.vlm._encode_image(valid_images)
+        tokens_per_view, hidden_dim = valid_feats.shape[1:]
+
+        image_features = valid_feats.new_zeros((batch_size * num_views, tokens_per_view, hidden_dim))
+        image_features[flat_mask] = valid_feats
+        image_features = image_features.view(batch_size, num_views, tokens_per_view, hidden_dim)
+        inputs_embeds = self.vlm.get_input_embeddings()(input_ids)
+        merged_embeds, attention_mask = self.vlm._merge_input_ids_with_image_features(
+            image_features[:, 0],
+            inputs_embeds,
+        )
+
+        enc_out = self.vlm.language_model.model.encoder(
+            attention_mask=attention_mask,
+            inputs_embeds=merged_embeds,
+        )[0]
+
+        aux_visual_inputs = image_features[:, 1:].reshape(batch_size, -1, hidden_dim)
+        return {"vlm_features": enc_out, "aux_visual_inputs": aux_visual_inputs}
+
+    def forward(
+        self,
+        input_ids: torch.LongTensor,
+        image_input: torch.FloatTensor,
+        image_mask: torch.Tensor,
+        domain_id: torch.LongTensor,
+        proprio: torch.Tensor,
+        action: torch.Tensor,
+    ) -> dict[str, torch.Tensor]:
+        """
+        Forward pass for the XVLA model.
+        """
+        target_dtype = self._get_target_dtype()
+        image_input = image_input.to(dtype=target_dtype)
+        proprio = proprio.to(dtype=target_dtype)
+        action = action.to(dtype=target_dtype)
+
+        enc = self.forward_vlm(input_ids, image_input, image_mask)
+
+        batch_size = input_ids.shape[0]
+        t = (
+            torch.rand(1, device=input_ids.device, dtype=target_dtype)
+            + torch.arange(batch_size, device=input_ids.device, dtype=target_dtype) / batch_size
+        ) % (1 - 1e-5)
+
+        action_noisy = torch.randn_like(action) * t.view(-1, 1, 1) + action * (1 - t).view(-1, 1, 1)
+        proprio_m, action_noisy_m = self.action_space.preprocess(proprio, action_noisy)
+
+        pred_action = self.transformer(
+            domain_id=domain_id,
+            action_with_noise=action_noisy_m,
+            t=t,
+            proprio=proprio_m,
+            **enc,
+        )
+        return self.action_space.compute_loss(pred_action, action)
+
+    @torch.no_grad()
+    def generate_actions(
+        self,
+        input_ids: torch.LongTensor,
+        image_input: torch.FloatTensor,
+        image_mask: torch.Tensor,
+        domain_id: torch.LongTensor,
+        proprio: torch.Tensor,
+        steps: int,
+    ) -> torch.Tensor:
+        self.eval()
+
+        target_dtype = self._get_target_dtype()
+        image_input = image_input.to(dtype=target_dtype)
+        proprio = proprio.to(dtype=target_dtype)
+
+        enc = self.forward_vlm(input_ids, image_input, image_mask)
+
+        batch_size = input_ids.shape[0]
+        action_dim = self.dim_action
+
+        x1 = torch.randn(batch_size, self.chunk_size, action_dim, device=proprio.device, dtype=target_dtype)
+        action = torch.zeros_like(x1)
+
+        steps = max(1, int(steps))
+        for i in range(steps, 0, -1):
+            t = torch.full((batch_size,), i / steps, device=proprio.device, dtype=target_dtype)
+            x_t = x1 * t.view(-1, 1, 1) + action * (1 - t).view(-1, 1, 1)
+            proprio_m, x_t_m = self.action_space.preprocess(proprio, x_t)
+            action = self.transformer(
+                domain_id=domain_id,
+                action_with_noise=x_t_m,
+                proprio=proprio_m,
+                t=t,
+                **enc,
+            )
+        return self.action_space.postprocess(action)
+
+
+class XVLAPolicy(PreTrainedPolicy):
+    """LeRobot-compliant wrapper built around the XVLA model."""
+
+    config_class = XVLAConfig
+    name = "xvla"
+
+    def __init__(self, config: XVLAConfig, **kwargs):
+        super().__init__(config)
+        config.validate_features()
+        florence_config = config.get_florence_config()
+        proprio_dim = config.max_state_dim if config.use_proprio else 0
+        self.model = XVLAModel(config=config, florence_config=florence_config, proprio_dim=proprio_dim)
+        self.reset()
+
+    def reset(self) -> None:
+        self._queues = {
+            ACTION: deque(maxlen=self.config.n_action_steps),
+        }
+
+    def get_optim_params(self) -> dict:
+        """Return trainable named parameters for optimization.
+
+        Returns a dict of name -> param for all trainable parameters.
+        This enables the xvla-adamw optimizer to apply differential learning rates
+        based on parameter names (e.g., 1/10 LR for VLM components).
+        """
+        return dict(filter(lambda kv: kv[1].requires_grad, self.named_parameters()))
+
+    def _prepare_state(self, batch: dict[str, Tensor], batch_size: int, device: torch.device) -> Tensor:
+        if not self.config.use_proprio or OBS_STATE not in batch:
+            return torch.zeros(batch_size, 0, device=device)
+        state = batch[OBS_STATE]
+        if state.ndim > 2:
+            state = state[:, -1, :]
+        return pad_vector(state, self.model.dim_proprio)
+
+    def _prepare_images(self, batch: dict[str, Tensor]) -> tuple[Tensor, Tensor]:
+        present_img_keys = [key for key in self.config.image_features if key in batch]
+        if len(present_img_keys) == 0:
+            raise ValueError(
+                "All image features are missing from the batch. "
+                f"Batch keys: {list(batch.keys())}, expected at least one of {list(self.config.image_features)}."
+            )
+
+        images = []
+        masks = []
+        for key in present_img_keys:
+            img = batch[key][:, -1] if batch[key].ndim == 5 else batch[key]
+            if self.config.resize_imgs_with_padding is not None:
+                img = resize_with_pad(img, *self.config.resize_imgs_with_padding)
+            images.append(img)
+            masks.append(torch.ones(img.size(0), dtype=torch.bool, device=img.device))
+
+        stacked_imgs = torch.stack(images, dim=1)
+        stacked_masks = torch.stack(masks, dim=1)
+
+        total_views = self.config.num_image_views or stacked_imgs.size(1)
+        total_views = max(total_views, stacked_imgs.size(1))
+        num_pad = total_views - stacked_imgs.size(1)
+        if num_pad > 0:
+            pad_shape = (stacked_imgs.size(0), num_pad, *stacked_imgs.shape[2:])
+            pad_imgs = stacked_imgs.new_zeros(pad_shape)
+            pad_masks = stacked_masks.new_zeros((stacked_masks.size(0), num_pad))
+            stacked_imgs = torch.cat([stacked_imgs, pad_imgs], dim=1)
+            stacked_masks = torch.cat([stacked_masks, pad_masks], dim=1)
+
+        return stacked_imgs, stacked_masks
+
+    def _get_domain_id(self, batch: dict[str, Tensor], batch_size: int, device: torch.device) -> Tensor:
+        candidate = None
+        if self.config.domain_feature_key and self.config.domain_feature_key in batch:
+            candidate = batch[self.config.domain_feature_key]
+        elif "domain_id" in batch:
+            candidate = batch["domain_id"]
+
+        if candidate is None:
+            return torch.zeros(batch_size, dtype=torch.long, device=device)
+
+        if not isinstance(candidate, torch.Tensor):
+            candidate = torch.as_tensor(candidate, device=device)
+        else:
+            candidate = candidate.to(device=device)
+
+        if candidate.ndim == 0:
+            candidate = candidate.expand(batch_size)
+        if candidate.ndim > 1:
+            candidate = candidate.view(candidate.shape[0], -1)[:, 0]
+        if candidate.shape[0] != batch_size:
+            candidate = candidate.expand(batch_size)
+        return candidate.to(dtype=torch.long)
+
+    def _prepare_action_targets(self, batch: dict[str, Tensor]) -> Tensor:
+        if ACTION not in batch:
+            raise ValueError("Batch is missing action targets required for training.")
+        actions = batch[ACTION]
+        if actions.ndim == 2:
+            actions = actions.unsqueeze(1)
+        actions = pad_tensor_along_dim(actions, self.config.chunk_size, dim=1)
+        if actions.shape[-1] != self.model.dim_action:
+            actions = pad_vector(actions, self.model.dim_action)
+        return actions
+
+    def _build_model_inputs(self, batch: dict[str, Tensor]) -> dict[str, Tensor]:
+        input_ids = batch[OBS_LANGUAGE_TOKENS]
+        batch_size = input_ids.shape[0]
+        images, image_mask = self._prepare_images(batch)
+        domain_id = self._get_domain_id(batch, batch_size, images.device)
+        proprio = self._prepare_state(batch, batch_size, images.device)
+        return {
+            "input_ids": input_ids,
+            "image_input": images,
+            "image_mask": image_mask,
+            "domain_id": domain_id,
+            "proprio": proprio,
+        }
+
+    def forward(self, batch: dict[str, Tensor]) -> tuple[Tensor, dict]:
+        inputs = self._build_model_inputs(batch)
+        targets = self._prepare_action_targets(batch)
+        losses = self.model(action=targets, **inputs)
+        total_loss = sum(losses.values())
+
+        log_dict = {k: v.detach().item() for k, v in losses.items()}
+        log_dict["loss"] = total_loss.detach().item()
+        return total_loss, log_dict
+
+    def _get_action_chunk(self, batch: dict[str, Tensor]) -> Tensor:
+        inputs = self._build_model_inputs(batch)
+        actions = self.model.generate_actions(**inputs, steps=self.config.num_denoising_steps)
+        return actions
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor], noise: Tensor | None = None) -> Tensor:  # noqa: ARG002
+        self.eval()
+        self._queues = populate_queues(self._queues, batch, exclude_keys=[ACTION])
+        return self._get_action_chunk(batch)
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor], noise: Tensor | None = None) -> Tensor:  # noqa: ARG002
+        self.eval()
+        self._queues = populate_queues(self._queues, batch, exclude_keys=[ACTION])
+
+        if len(self._queues[ACTION]) == 0:
+            actions = self._get_action_chunk(batch)
+            self._queues[ACTION].extend(actions.transpose(0, 1)[: self.config.n_action_steps])
+
+        return self._queues[ACTION].popleft()
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type[T],
+        pretrained_name_or_path: str | Path,
+        *,
+        config: PreTrainedConfig | None = None,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        strict: bool = False,
+        **kwargs,
+    ):
+        """
+        Loads XVLA model weights with:
+        - automatic prefix 'model.' added to all keys
+        - skip list for layers that should remain randomly initialized
+        """
+        import safetensors.torch
+
+        # step 1: load config
+        # TODO: jadechoghari, fix this
+        if config is None:
+            config = PreTrainedConfig.from_pretrained(
+                pretrained_name_or_path=pretrained_name_or_path,
+                force_download=force_download,
+                resume_download=resume_download,
+                proxies=proxies,
+                token=token,
+                cache_dir=cache_dir,
+                local_files_only=local_files_only,
+                revision=revision,
+                **kwargs,
+            )
+
+        model_id = str(pretrained_name_or_path)
+        instance = cls(config, **kwargs)
+        # step 2: locate model.safetensors
+        if os.path.isdir(model_id):
+            logging.info("Loading weights from local directory")
+            model_file = os.path.join(model_id, "model.safetensors")
+        else:
+            try:
+                from huggingface_hub import hf_hub_download
+                from huggingface_hub.utils import HfHubHTTPError
+
+                model_file = hf_hub_download(
+                    repo_id=model_id,
+                    filename="model.safetensors",
+                    revision=revision,
+                    cache_dir=cache_dir,
+                    force_download=force_download,
+                    proxies=proxies,
+                    resume_download=resume_download,
+                    token=token,
+                    local_files_only=local_files_only,
+                )
+            except HfHubHTTPError as e:
+                raise FileNotFoundError(f"model.safetensors not found on the Hub at {model_id}") from e
+
+        logging.info(f"Loading checkpoint from {model_file}")
+        # step 3: load state dict
+        state_dict = safetensors.torch.load_file(model_file)
+        encoder_key = "model.vlm.language_model.model.encoder.embed_tokens.weight"
+        shared_key = "model.vlm.language_model.model.shared.weight"
+        if encoder_key in state_dict:
+            state_dict[shared_key] = state_dict[encoder_key]
+            # or deepcopy
+        # step 4: load into instance
+        instance.load_state_dict(state_dict, strict=True)
+        logging.info("Loaded XVLA checkpoint")
+        # step 5: finalize
+        # Reapply dtype after loading state dict
+        instance.model._apply_dtype()
+        instance.to(config.device)
+        instance.eval()
+        return instance
+
+
+def resize_with_pad(img: torch.Tensor, height: int, width: int, pad_value: float = 0.0) -> torch.Tensor:
+    if img.ndim != 4:
+        raise ValueError(f"(b,c,h,w) expected, but got {img.shape}")
+
+    current_height, current_width = img.shape[2:]
+    if current_height == height and current_width == width:
+        return img
+
+    ratio = max(current_width / width, current_height / height)
+    resized_height = int(current_height / ratio)
+    resized_width = int(current_width / ratio)
+    resized_img = F.interpolate(
+        img, size=(resized_height, resized_width), mode="bilinear", align_corners=False
+    )
+
+    pad_height = max(0, height - resized_height)
+    pad_width = max(0, width - resized_width)
+    padded_img = F.pad(resized_img, (pad_width, 0, pad_height, 0), value=pad_value)
+    return padded_img
+
+
+def pad_vector(vector: Tensor, new_dim: int) -> Tensor:
+    if vector.shape[-1] == new_dim:
+        return vector
+    if new_dim == 0:
+        shape = list(vector.shape)
+        shape[-1] = 0
+        return vector.new_zeros(*shape)
+    shape = list(vector.shape)
+    current_dim = shape[-1]
+    shape[-1] = new_dim
+    new_vector = vector.new_zeros(*shape)
+    length = min(current_dim, new_dim)
+    new_vector[..., :length] = vector[..., :length]
+    return new_vector
+
+
+def pad_tensor_along_dim(tensor: Tensor, target_len: int, dim: int = 1) -> Tensor:
+    current_len = tensor.size(dim)
+    if current_len == target_len:
+        return tensor
+    if current_len > target_len:
+        slices = [slice(None)] * tensor.dim()
+        slices[dim] = slice(0, target_len)
+        return tensor[tuple(slices)]
+    pad_shape = list(tensor.shape)
+    pad_shape[dim] = target_len - current_len
+    pad_tensor = tensor.new_zeros(pad_shape)
+    return torch.cat([tensor, pad_tensor], dim=dim)
diff --git a/lerobot/src/lerobot/policies/xvla/processor_xvla.py b/lerobot/src/lerobot/policies/xvla/processor_xvla.py
new file mode 100644
index 0000000000000000000000000000000000000000..0fa9ffe3fb5b547031bdc60d00ba92c2c10382ab
--- /dev/null
+++ b/lerobot/src/lerobot/policies/xvla/processor_xvla.py
@@ -0,0 +1,556 @@
+# ------------------------------------------------------------------------------
+# Copyright 2025 The HuggingFace Inc. team and 2toINF (https://github.com/2toINF)
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# ------------------------------------------------------------------------------
+
+from dataclasses import dataclass
+from typing import Any
+
+import numpy as np
+import torch
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.datasets.factory import IMAGENET_STATS
+from lerobot.policies.xvla.configuration_xvla import XVLAConfig
+from lerobot.policies.xvla.utils import rotate6d_to_axis_angle
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    ObservationProcessorStep,
+    PolicyAction,
+    PolicyProcessorPipeline,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    RenameObservationsProcessorStep,
+    TokenizerProcessorStep,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import policy_action_to_transition, transition_to_policy_action
+from lerobot.types import EnvTransition, TransitionKey
+from lerobot.utils.constants import (
+    OBS_IMAGES,
+    OBS_PREFIX,
+    OBS_STATE,
+    POLICY_POSTPROCESSOR_DEFAULT_NAME,
+    POLICY_PREPROCESSOR_DEFAULT_NAME,
+)
+
+
+def make_xvla_pre_post_processors(
+    config: XVLAConfig,
+    dataset_stats: dict[str, dict[str, torch.Tensor]] | None = None,
+) -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Build the LeRobot processor pipelines for XVLA.
+    """
+
+    features = {**config.input_features, **config.output_features}
+    input_steps = [
+        RenameObservationsProcessorStep(rename_map={}),
+        AddBatchDimensionProcessorStep(),
+        TokenizerProcessorStep(
+            tokenizer_name=config.tokenizer_name,
+            max_length=config.tokenizer_max_length,
+            padding=config.pad_language_to,
+            padding_side=config.tokenizer_padding_side,
+        ),
+        XVLAImageToFloatProcessorStep(),
+        XVLAImageNetNormalizeProcessorStep(),
+        XVLAAddDomainIdProcessorStep(),
+        DeviceProcessorStep(device=config.device),
+        NormalizerProcessorStep(
+            features=features, norm_map=config.normalization_mapping, stats=dataset_stats
+        ),
+    ]
+    output_steps = [
+        UnnormalizerProcessorStep(
+            features=config.output_features,
+            norm_map=config.normalization_mapping,
+            stats=dataset_stats,
+        ),
+        DeviceProcessorStep(device="cpu"),
+    ]
+
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=input_steps,
+            name=POLICY_PREPROCESSOR_DEFAULT_NAME,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=output_steps,
+            name=POLICY_POSTPROCESSOR_DEFAULT_NAME,
+            to_transition=policy_action_to_transition,
+            to_output=transition_to_policy_action,
+        ),
+    )
+
+
+# Custom XVLA processor steps
+@dataclass
+class LiberoProcessorStep(ObservationProcessorStep):
+    """
+    Processes LIBERO observations into the LeRobot format.
+
+    This step handles the specific observation structure from LIBERO environments,
+    which includes nested robot_state dictionaries and image observations.
+
+    **State Processing:**
+    -   Processes the `robot_state` dictionary which contains nested end-effector,
+        gripper, and joint information.
+    -   Extracts and concatenates:
+        - End-effector position (3D)
+        - End-effector quaternion converted to axis-angle (3D)
+        - Gripper joint positions (2D)
+    -   Maps the concatenated state to `"observation.state"`.
+
+    **Image Processing:**
+    -   Rotates images by 180 degrees by flipping both height and width dimensions.
+    -   This accounts for the HuggingFaceVLA/libero camera orientation convention.
+    """
+
+    def _process_observation(self, observation):
+        """
+        Processes both image and robot_state observations from LIBERO.
+        """
+        processed_obs = observation.copy()
+        for key in list(processed_obs.keys()):
+            if key.startswith(f"{OBS_IMAGES}."):
+                img = processed_obs[key]
+
+                if key == f"{OBS_IMAGES}.image":
+                    # Flip both H and W
+                    img = torch.flip(img, dims=[2, 3])
+
+                processed_obs[key] = img
+        # Process robot_state into a flat state vector
+        robot_state_str = OBS_PREFIX + "robot_state"
+        if robot_state_str in processed_obs:
+            robot_state = processed_obs.pop(robot_state_str)
+
+            # Extract components
+            eef_pos = robot_state["eef"]["pos"]  # (B, 3,)
+            eef_mat = robot_state["eef"]["mat"]  # (B, 3, 3)
+            eef_rot6d = self._mat_to_rotate6d(eef_mat)  # (B, 6)
+
+            extra = torch.zeros((eef_pos.shape[0], 1), dtype=torch.float32, device=eef_pos.device)
+
+            proprio_state = torch.cat((eef_pos, eef_rot6d, extra), dim=-1)  # (B, 10)
+            state = torch.cat((proprio_state, torch.zeros_like(proprio_state)), dim=-1)  # (B, 20)
+            # ensure float32
+            state = state.float()
+            if state.dim() == 1:
+                state = state.unsqueeze(0)
+
+            processed_obs[OBS_STATE] = state
+        return processed_obs
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Transforms feature keys from the LIBERO format to the LeRobot standard.
+        """
+        new_features: dict[PipelineFeatureType, dict[str, PolicyFeature]] = {}
+
+        # copy over non-STATE features
+        for ft, feats in features.items():
+            if ft != PipelineFeatureType.STATE:
+                new_features[ft] = feats.copy()
+
+        # rebuild STATE features
+        state_feats = {}
+
+        # add our new flattened state
+        state_feats[OBS_STATE] = PolicyFeature(
+            key=OBS_STATE,
+            shape=(20,),
+            dtype="float32",
+        )
+
+        new_features[PipelineFeatureType.STATE] = state_feats
+
+        return new_features
+
+    def _mat_to_rotate6d(self, rot_mats: torch.Tensor) -> torch.Tensor:
+        """
+        Convert batched rotation matrices (B, 3, 3) into 6D rotation representation (B, 6).
+
+        Args:
+            rot_mats (Tensor): Rotation matrices of shape (B, 3, 3)
+
+        Returns:
+            Tensor: 6D rotation representation, shape (B, 6)
+
+        Raises:
+            TypeError: if input is not a torch tensor
+            ValueError: if shape is not (B, 3, 3)
+        """
+
+        if not isinstance(rot_mats, torch.Tensor):
+            raise TypeError(f"mat_to_rot6d expects a torch.Tensor, got {type(rot_mats)}")
+
+        if rot_mats.ndim != 3 or rot_mats.shape[1:] != (3, 3):
+            raise ValueError(f"mat_to_rot6d expects shape (B, 3, 3), got {tuple(rot_mats.shape)}")
+
+        rot_mats = rot_mats.to(torch.float32)
+
+        col1 = rot_mats[:, :3, 0]  # (B, 3)
+        col2 = rot_mats[:, :3, 1]  # (B, 3)
+
+        rot6d = torch.cat([col1, col2], dim=-1)  # (B, 6)
+
+        return rot6d
+
+    def observation(self, observation):
+        return self._process_observation(observation)
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="xvla_image_scale")
+class XVLAImageScaleProcessorStep(ProcessorStep):
+    """Scale image observations by 255 to convert from [0, 1] to [0, 255] range.
+
+    This processor step multiplies all image observations by 255, which is required
+    for XVLA models that expect images in uint8-like range.
+
+    Args:
+        image_keys: List of observation keys that contain images to scale.
+                   If None, will automatically detect keys starting with "observation.images."
+    """
+
+    image_keys: list[str] | None = None
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Scale image observations by 255."""
+        new_transition = transition.copy()
+        obs = new_transition.get(TransitionKey.OBSERVATION, {})
+        if obs is None:
+            return new_transition
+
+        # Make a copy of observations to avoid modifying the original
+        obs = obs.copy()
+
+        # Determine which keys to scale
+        keys_to_scale = self.image_keys
+        if keys_to_scale is None:
+            # Auto-detect image keys
+            keys_to_scale = [k for k in obs if k.startswith(OBS_IMAGES)]
+
+        # Scale each image
+        for key in keys_to_scale:
+            if key in obs and isinstance(obs[key], torch.Tensor):
+                obs[key] = obs[key] * 255
+
+        new_transition[TransitionKey.OBSERVATION] = obs
+        return new_transition
+
+    def transform_features(self, features):
+        """Image scaling doesn't change feature structure."""
+        return features
+
+    def get_config(self) -> dict[str, Any]:
+        """Return serializable configuration."""
+        return {
+            "image_keys": self.image_keys,
+        }
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="xvla_image_to_float")
+class XVLAImageToFloatProcessorStep(ProcessorStep):
+    """Convert image observations from [0, 255] to [0, 1] range.
+
+    This processor step divides image observations by 255 to convert from uint8-like
+    range [0, 255] to float range [0, 1]. This is typically used when loading images
+    that are stored as uint8 values.
+
+    Args:
+        image_keys: List of observation keys that contain images to convert.
+                   If None, will automatically detect keys starting with "observation.images."
+        validate_range: If True, validates that input values are in [0, 255] range (default: True)
+
+    Raises:
+        ValueError: If validate_range is True and image values are not in [0, 255] range.
+    """
+
+    image_keys: list[str] | None = None
+    validate_range: bool = True
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Convert image observations from [0, 255] to [0, 1]."""
+        new_transition = transition.copy()
+        obs = new_transition.get(TransitionKey.OBSERVATION, {})
+        if obs is None:
+            return new_transition
+
+        # Make a copy of observations to avoid modifying the original
+        obs = obs.copy()
+
+        # Determine which keys to convert
+        keys_to_convert = self.image_keys
+        if keys_to_convert is None:
+            # Auto-detect image keys
+            keys_to_convert = [k for k in obs if k.startswith(OBS_IMAGES)]
+
+        # Convert each image
+        for key in keys_to_convert:
+            if key in obs and isinstance(obs[key], torch.Tensor):
+                tensor = obs[key]
+
+                min_val = tensor.min().item()
+                max_val = tensor.max().item()
+
+                if max_val <= 1.0:
+                    obs[key] = tensor.float()  # ensure float dtype, but no division
+                    continue
+                # Validate that values are in [0, 255] range if requested
+                if self.validate_range and (min_val < 0.0 or max_val > 255.0):
+                    raise ValueError(
+                        f"Image '{key}' has values outside [0, 255] range: "
+                        f"min={min_val:.4f}, max={max_val:.4f}. "
+                        f"Cannot convert to [0, 1] range."
+                    )
+
+                # Convert to float and divide by 255
+                obs[key] = tensor.float() / 255.0
+
+        new_transition[TransitionKey.OBSERVATION] = obs
+        return new_transition
+
+    def transform_features(self, features):
+        """Image conversion doesn't change feature structure."""
+        return features
+
+    def get_config(self) -> dict[str, Any]:
+        """Return serializable configuration."""
+        return {
+            "image_keys": self.image_keys,
+            "validate_range": self.validate_range,
+        }
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="xvla_imagenet_normalize")
+class XVLAImageNetNormalizeProcessorStep(ProcessorStep):
+    """Normalize image observations using ImageNet statistics.
+
+    This processor step applies ImageNet normalization (mean and std) to image observations.
+    It validates that input values are in the [0, 1] range before normalizing.
+
+    The normalization formula is: (image - mean) / std
+
+    Args:
+        image_keys: List of observation keys that contain images to normalize.
+                   If None, will automatically detect keys starting with "observation.images."
+
+    Raises:
+        ValueError: If image values are not in the [0, 1] range.
+    """
+
+    image_keys: list[str] | None = None
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Normalize image observations using ImageNet statistics."""
+        new_transition = transition.copy()
+        obs = new_transition.get(TransitionKey.OBSERVATION, {})
+        if obs is None:
+            return new_transition
+
+        # Make a copy of observations to avoid modifying the original
+        obs = obs.copy()
+
+        # Determine which keys to normalize
+        keys_to_normalize = self.image_keys
+        if keys_to_normalize is None:
+            # Auto-detect image keys
+            keys_to_normalize = [k for k in obs if k.startswith(OBS_IMAGES)]
+
+        # Normalize each image
+        for key in keys_to_normalize:
+            if key in obs and isinstance(obs[key], torch.Tensor):
+                tensor = obs[key]
+
+                # Validate that values are in [0, 1] range
+                min_val = tensor.min().item()
+                max_val = tensor.max().item()
+                if min_val < 0.0 or max_val > 1.0:
+                    raise ValueError(
+                        f"Image '{key}' has values outside [0, 1] range: "
+                        f"min={min_val:.4f}, max={max_val:.4f}. "
+                        f"ImageNet normalization requires input values in [0, 1]."
+                    )
+
+                # Apply ImageNet normalization
+                mean = torch.tensor(IMAGENET_STATS["mean"], device=tensor.device, dtype=tensor.dtype)
+                std = torch.tensor(IMAGENET_STATS["std"], device=tensor.device, dtype=tensor.dtype)
+
+                # Expand mean/std to match tensor dims (e.g., BCHW or BNCHW)
+                while mean.dim() < tensor.dim():
+                    mean = mean.unsqueeze(0)
+                    std = std.unsqueeze(0)
+
+                # Normalize: (image - mean) / std
+                obs[key] = (tensor - mean) / std
+
+        new_transition[TransitionKey.OBSERVATION] = obs
+        return new_transition
+
+    def transform_features(self, features):
+        """ImageNet normalization doesn't change feature structure."""
+        return features
+
+    def get_config(self) -> dict[str, Any]:
+        """Return serializable configuration."""
+        return {
+            "image_keys": self.image_keys,
+        }
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="xvla_add_domain_id")
+class XVLAAddDomainIdProcessorStep(ProcessorStep):
+    """Add domain_id to complementary data.
+
+    This processor step adds a domain_id tensor to the complementary data,
+    which is used by XVLA to identify different robot embodiments or task domains.
+
+    Args:
+        domain_id: The domain ID to add (default: 3)
+    """
+
+    domain_id: int = 0
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Add domain_id to complementary data."""
+        new_transition = transition.copy()
+        comp = new_transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+        comp = {} if comp is None else comp.copy()
+
+        # Infer batch size from observation tensors
+        obs = new_transition.get(TransitionKey.OBSERVATION, {})
+        batch_size = 1
+        if obs:
+            for v in obs.values():
+                if isinstance(v, torch.Tensor):
+                    batch_size = v.shape[0]
+                    break
+
+        # Add domain_id tensor
+        comp["domain_id"] = torch.tensor([int(self.domain_id)] * batch_size, dtype=torch.long)
+
+        new_transition[TransitionKey.COMPLEMENTARY_DATA] = comp
+        return new_transition
+
+    def transform_features(self, features):
+        """Domain ID addition doesn't change feature structure."""
+        return features
+
+    def get_config(self) -> dict[str, Any]:
+        """Return serializable configuration."""
+        return {
+            "domain_id": self.domain_id,
+        }
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="xvla_rotation_6d_to_axis_angle")
+class XVLARotation6DToAxisAngleProcessorStep(ProcessorStep):
+    """Convert 6D rotation representation to axis-angle and reorganize action dimensions.
+
+    This processor step takes actions with 6D rotation representation and converts them to
+    axis-angle representation, reorganizing the action dimensions as:
+    - action[:, :3] -> target_eef (end-effector position)
+    - action[:, 3:9] -> 6D rotation (converted to axis-angle, 3D)
+    - action[:, 9:10] -> gripper action
+
+    Final output: [target_eef (3), axis_angle (3), gripper (1)] = 7D action
+
+    Args:
+        expected_action_dim: Expected input action dimension (default: 10, supports 6D rotation + extras)
+    """
+
+    expected_action_dim: int = 10
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Convert 6D rotation to axis-angle in action."""
+        new_transition = transition.copy()
+        action = new_transition.get(TransitionKey.ACTION)
+
+        if action is None or not isinstance(action, torch.Tensor):
+            return new_transition
+
+        # Convert to numpy for processing
+        device = action.device
+        dtype = action.dtype
+        action_np = action.cpu().numpy()
+
+        # Extract components
+        # action shape: (B, D) where D >= 10
+        target_eef = action_np[:, :3]  # (B, 3)
+        rotation_6d = action_np[:, 3:9]  # (B, 6)
+        target_act = action_np[:, 9:10]  # (B, 1)
+
+        # Convert 6D rotation to axis-angle
+        target_axis = rotate6d_to_axis_angle(rotation_6d)  # (B, 3)
+
+        # Concatenate: [eef (3), axis_angle (3), gripper (1)] = 7D
+        action_np = np.concatenate([target_eef, target_axis, target_act], axis=-1)
+
+        # Convert gripper action to -1 or 1
+        action_np[:, -1] = np.where(action_np[:, -1] > 0.5, 1.0, -1.0)
+
+        # Convert back to tensor
+        action = torch.from_numpy(action_np).to(device=device, dtype=dtype)
+
+        new_transition[TransitionKey.ACTION] = action
+        return new_transition
+
+    def transform_features(self, features):
+        """Rotation conversion changes action dimension from 10 to 7."""
+        # Note: This is a simplified version. In practice, you might want to
+        # update the action feature shape in the features dict.
+        return features
+
+    def get_config(self) -> dict[str, Any]:
+        """Return serializable configuration."""
+        return {
+            "expected_action_dim": self.expected_action_dim,
+        }
+
+
+def make_xvla_libero_pre_post_processors() -> tuple[
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """
+    Build the LeRobot processor pipelines for XVLA with LIBERO environment.
+    """
+    pre_processor_steps: list[ProcessorStep] = []
+    post_processor_steps: list[ProcessorStep] = []
+    pre_processor_steps.extend(
+        [LiberoProcessorStep(), XVLAImageNetNormalizeProcessorStep(), XVLAAddDomainIdProcessorStep()]
+    )
+    post_processor_steps.extend([XVLARotation6DToAxisAngleProcessorStep()])
+    return (
+        PolicyProcessorPipeline[dict[str, Any], dict[str, Any]](
+            steps=pre_processor_steps,
+        ),
+        PolicyProcessorPipeline[PolicyAction, PolicyAction](
+            steps=post_processor_steps,
+        ),
+    )
diff --git a/lerobot/src/lerobot/policies/xvla/soft_transformer.py b/lerobot/src/lerobot/policies/xvla/soft_transformer.py
new file mode 100644
index 0000000000000000000000000000000000000000..77ceb6e26a7b1587dcf6c4087f1a42dd292d2dc2
--- /dev/null
+++ b/lerobot/src/lerobot/policies/xvla/soft_transformer.py
@@ -0,0 +1,415 @@
+# ------------------------------------------------------------------------------
+# Copyright 2025 2toINF (https://github.com/2toINF)
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# ------------------------------------------------------------------------------
+
+from __future__ import annotations
+
+import math
+from collections.abc import Iterable
+from functools import partial
+from typing import Final
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as functional
+
+# ------------------------------- Small utils ----------------------------------
+
+
+def _to_2tuple(x) -> tuple:
+    """Minimal replacement for timm.layers.to_2tuple."""
+    if isinstance(x, Iterable) and not isinstance(x, (str, bytes)):
+        t = tuple(x)
+        return (t[0], t[1]) if len(t) >= 2 else (t[0], t[0])
+    return (x, x)
+
+
+def _has_sdp_attention() -> bool:
+    """Check if we can use PyTorch fused scaled_dot_product_attention."""
+    return hasattr(functional, "scaled_dot_product_attention")
+
+
+# ---------------------------------- MLP --------------------------------------
+
+
+class Mlp(nn.Module):
+    """
+    MLP used in ViT-style blocks.
+
+    Supports Linear or 1x1 Conv 'linear_layer' for token/channel mixing.
+    """
+
+    def __init__(
+        self,
+        in_features: int,
+        hidden_features: int | None = None,
+        out_features: int | None = None,
+        norm_layer: type[nn.Module] | None = None,
+        bias: bool | tuple[bool, bool] = True,
+        drop: float | tuple[float, float] = 0.0,
+        use_conv: bool = False,
+    ) -> None:
+        super().__init__()
+        out_features = out_features or in_features
+        hidden_features = hidden_features or in_features
+        bias = _to_2tuple(bias)
+        drop_probs = _to_2tuple(drop)
+        linear_layer = partial(nn.Conv2d, kernel_size=1) if use_conv else nn.Linear
+
+        self.fc1 = linear_layer(in_features, hidden_features, bias=bias[0])
+        self.act = nn.GELU(approximate="tanh")
+        self.drop1 = nn.Dropout(drop_probs[0])
+        self.norm = norm_layer(hidden_features) if norm_layer is not None else nn.Identity()
+        self.fc2 = linear_layer(hidden_features, out_features, bias=bias[1])
+        self.drop2 = nn.Dropout(drop_probs[1])
+
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        # Expect [B, T, C] for Linear variant; caller is responsible for shapes.
+        x = self.fc1(x)
+        x = self.act(x)
+        x = self.drop1(x)
+        x = self.norm(x)
+        x = self.fc2(x)
+        x = self.drop2(x)
+        return x
+
+
+# -------------------------------- Attention ----------------------------------
+
+
+class Attention(nn.Module):
+    """
+    Multi-Head Self-Attention with optional fused SDPA fallback.
+
+    If PyTorch provides `scaled_dot_product_attention`, it will be used
+    (usually faster and more stable); otherwise we use a manual implementation.
+    """
+
+    fused_attn: Final[bool]
+
+    def __init__(
+        self,
+        dim: int,
+        num_heads: int = 8,
+        qkv_bias: bool = False,
+        qk_norm: bool = False,
+        attn_drop: float = 0.0,
+        proj_drop: float = 0.0,
+        norm_layer: type[nn.Module] = nn.LayerNorm,
+    ) -> None:
+        super().__init__()
+        assert dim % num_heads == 0, "dim should be divisible by num_heads"
+        self.num_heads = num_heads
+        self.head_dim = dim // num_heads
+        self.scale = self.head_dim**-0.5
+        self.fused_attn = _has_sdp_attention()
+
+        self.qkv = nn.Linear(dim, dim * 3, bias=qkv_bias)
+        self.q_norm = norm_layer(self.head_dim) if qk_norm else nn.Identity()
+        self.k_norm = norm_layer(self.head_dim) if qk_norm else nn.Identity()
+        self.attn_drop = nn.Dropout(attn_drop)
+        self.proj = nn.Linear(dim, dim)
+        self.proj_drop = nn.Dropout(proj_drop)
+
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        """
+        Parameters
+        ----------
+        x : Tensor, shape [batch_size, seq_len, channels]
+            Input sequence.
+
+        Returns
+        -------
+        Tensor, shape [batch_size, seq_len, channels]
+            Output sequence after MHSA + projection.
+        """
+        batch_size, seq_len, channels = x.shape
+        qkv = (
+            self.qkv(x)
+            .reshape(batch_size, seq_len, 3, self.num_heads, self.head_dim)
+            .permute(2, 0, 3, 1, 4)  # 3 x [batch_size, num_heads, seq_len, head_dim]
+        )
+        q, k, v = qkv.unbind(0)  # each: [batch_size, num_heads, seq_len, head_dim]
+        q, k = self.q_norm(q), self.k_norm(k)
+
+        if self.fused_attn:
+            x = functional.scaled_dot_product_attention(
+                q,
+                k,
+                v,
+                dropout_p=self.attn_drop.p if self.training else 0.0,
+            )  # [batch_size, num_heads, seq_len, head_dim]
+        else:
+            q = q * self.scale
+            attn = q @ k.transpose(-2, -1)  # [batch_size, num_heads, seq_len, seq_len]
+            attn = attn.softmax(dim=-1)
+            attn = self.attn_drop(attn)
+            x = attn @ v  # [batch_size, num_heads, seq_len, head_dim]
+
+        x = x.transpose(1, 2).reshape(batch_size, seq_len, channels)  # [batch_size, seq_len, channels]
+        x = self.proj(x)
+        x = self.proj_drop(x)
+        return x
+
+
+# ------------------------------- Utilities -----------------------------------
+
+
+def basic_init(module: nn.Module) -> None:
+    """
+    Apply a basic initialization scheme to Linear layers.
+
+    - Weight: Xavier uniform initialization.
+    - Bias: Set to zero.
+    """
+    if isinstance(module, nn.Linear):
+        nn.init.xavier_uniform_(module.weight)
+        if module.bias is not None:
+            nn.init.constant_(module.bias, 0.0)
+
+
+def timestep_embedding(t: torch.Tensor, dim: int, max_period: int = 100) -> torch.Tensor:
+    """
+    Create sinusoidal timestep embeddings.
+
+    Parameters
+    ----------
+    t : torch.Tensor
+        Shape [B]. Each element is a timestep index, may be fractional.
+    dim : int
+        Dimensionality of the output embedding.
+    max_period : int, default=100
+        Controls the minimum frequency of the sinusoids.
+
+    Returns
+    -------
+    torch.Tensor
+        Shape [B, dim]. Sinusoidal embeddings.
+    """
+    half = dim // 2
+    freqs = torch.exp(
+        -math.log(max_period) * torch.arange(start=0, end=half, dtype=t.dtype, device=t.device) / half
+    )
+    args = t[:, None] * freqs[None]
+    embedding = torch.cat([torch.cos(args), torch.sin(args)], dim=-1)
+    if dim % 2 == 1:
+        embedding = torch.cat([embedding, torch.zeros_like(embedding[:, :1])], dim=-1)
+    return embedding
+
+
+# ------------------------------- Core Layers ----------------------------------
+
+
+class DomainAwareLinear(nn.Module):
+    """
+    Linear layer with domain-conditioned parameters (per-sample).
+
+    Each domain has its own weight and bias vectors, stored in embeddings.
+    """
+
+    def __init__(self, input_size: int, output_size: int, num_domains: int = 20) -> None:
+        super().__init__()
+        self.input_size = input_size
+        self.output_size = output_size
+        self.fc = nn.Embedding(num_domains, output_size * input_size)
+        self.bias = nn.Embedding(num_domains, output_size)
+        nn.init.xavier_uniform_(self.fc.weight)
+        nn.init.zeros_(self.bias.weight)
+
+    def forward(self, x: torch.Tensor, domain_id: torch.LongTensor) -> torch.Tensor:
+        """
+        Parameters
+        ----------
+        x : Tensor
+            [B, I] or [B, T, I]
+        domain_id : LongTensor
+            [B], domain indices.
+
+        Returns
+        -------
+        Tensor
+            [batch_size, output_size] or [batch_size, seq_len, output_size]
+        """
+        batch_size = domain_id.shape[0]
+        squeeze_seq = False
+        if x.dim() == 2:
+            x = x.unsqueeze(1)
+            squeeze_seq = True
+        weight = self.fc(domain_id).view(batch_size, self.input_size, self.output_size)
+        bias = self.bias(domain_id).view(batch_size, self.output_size)
+        y = torch.matmul(x, weight) + bias.view(batch_size, 1, self.output_size)
+        if squeeze_seq:
+            y = y.squeeze(1)
+        return y
+
+
+class TransformerBlock(nn.Module):
+    """
+    Standard Transformer block (pre-LN): LN → MHSA → residual, LN → MLP → residual.
+    """
+
+    def __init__(self, hidden_size: int, num_heads: int, mlp_ratio: float = 4.0) -> None:
+        super().__init__()
+        self.norm1 = nn.LayerNorm(hidden_size)
+        self.norm2 = nn.LayerNorm(hidden_size)
+        self.attn = Attention(hidden_size, num_heads=num_heads, qkv_bias=True, attn_drop=0.1)
+        self.mlp = Mlp(
+            in_features=hidden_size,
+            hidden_features=int(hidden_size * mlp_ratio),
+            drop=0.1,
+        )
+
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        """
+        Parameters
+        ----------
+        x : Tensor, [B, T, H]
+
+        Returns
+        -------
+        Tensor, [B, T, H]
+        """
+        x = x + self.attn(self.norm1(x))
+        x = x + self.mlp(self.norm2(x))
+        return x
+
+
+# --------------------------- Main Model ---------------------------------------
+
+
+class SoftPromptedTransformer(nn.Module):
+    """
+    Multi-modal, domain-aware Transformer with optional soft prompts.
+
+    See parameter and forward I/O descriptions inside the docstrings.
+    """
+
+    def __init__(
+        self,
+        hidden_size: int = 768,
+        multi_modal_input_size: int = 768,
+        depth: int = 24,
+        num_heads: int = 16,
+        mlp_ratio: float = 4.0,
+        num_domains: int = 20,
+        dim_action: int = 20,
+        dim_propio: int = 20,
+        dim_time: int = 32,
+        len_soft_prompts: int = 32,
+        max_len_seq: int = 512,
+        use_hetero_proj: bool = False,
+    ) -> None:
+        super().__init__()
+        self.hidden_size = hidden_size
+        self.dim_action = dim_action
+        self.dim_time = dim_time
+        self.len_soft_prompts = len_soft_prompts
+        self.use_hetero_proj = use_hetero_proj
+
+        self.blocks = nn.ModuleList(
+            [TransformerBlock(hidden_size, num_heads, mlp_ratio=mlp_ratio) for _ in range(depth)]
+        )
+
+        if use_hetero_proj:
+            self.vlm_proj = DomainAwareLinear(multi_modal_input_size, hidden_size, num_domains=num_domains)
+            self.aux_visual_proj = DomainAwareLinear(
+                multi_modal_input_size, hidden_size, num_domains=num_domains
+            )
+        else:
+            self.vlm_proj = nn.Linear(multi_modal_input_size, hidden_size)
+            self.aux_visual_proj = nn.Linear(multi_modal_input_size, hidden_size)
+
+        self.pos_emb = nn.Parameter(torch.zeros(1, max_len_seq, hidden_size), requires_grad=True)
+        nn.init.normal_(self.pos_emb, std=0.02)
+
+        self.norm = nn.LayerNorm(hidden_size)
+        self.action_encoder = DomainAwareLinear(
+            dim_action + dim_time + dim_propio, hidden_size, num_domains=num_domains
+        )
+        self.action_decoder = DomainAwareLinear(hidden_size, dim_action, num_domains=num_domains)
+
+        if len_soft_prompts > 0:
+            self.soft_prompt_hub = nn.Embedding(num_domains, len_soft_prompts * hidden_size)
+            nn.init.normal_(self.soft_prompt_hub.weight, std=0.02)
+
+        self.apply(basic_init)
+
+    def forward(
+        self,
+        domain_id: torch.LongTensor,
+        vlm_features: torch.Tensor,
+        aux_visual_inputs: torch.Tensor,
+        action_with_noise: torch.Tensor,
+        proprio: torch.Tensor,
+        t: torch.Tensor,
+    ) -> torch.Tensor:
+        """
+        Forward pass.
+
+        Inputs
+        ------
+        domain_id : [B]
+        vlm_features : [B, T_vlm, D]
+        aux_visual_inputs : [B, T_aux, D]
+        action_with_noise : [B, T_action, dim_action]
+        proprio : [B, dim_propio]
+        t : [B]
+
+        Returns
+        -------
+        Tensor
+            Predicted actions, [batch_size, num_actions, dim_action]
+        """
+        batch_size, num_actions = action_with_noise.shape[:2]
+
+        # Encode (action + proprio + time) → tokens
+        time_emb = timestep_embedding(t, self.dim_time)  # [batch_size, dim_time]
+        time_tokens = time_emb.unsqueeze(1).expand(batch_size, num_actions, self.dim_time)
+        proprio_tokens = proprio.unsqueeze(1).expand(batch_size, num_actions, proprio.shape[-1])
+        action_tokens = torch.cat([action_with_noise, proprio_tokens, time_tokens], dim=-1)
+        x = self.action_encoder(action_tokens, domain_id)  # [batch_size, num_actions, hidden_size]
+
+        # Project visual streams and concatenate
+        if self.use_hetero_proj:
+            x = torch.cat(
+                [
+                    x,
+                    self.vlm_proj(vlm_features, domain_id),
+                    self.aux_visual_proj(aux_visual_inputs, domain_id),
+                ],
+                dim=1,
+            )
+        else:
+            x = torch.cat([x, self.vlm_proj(vlm_features), self.aux_visual_proj(aux_visual_inputs)], dim=1)
+
+        # Add positional embeddings (truncate if needed)
+        seq_len = x.shape[1]
+        if seq_len > self.pos_emb.shape[1]:
+            raise ValueError(f"Sequence length {seq_len} exceeds max_len_seq={self.pos_emb.shape[1]}.")
+        x = x + self.pos_emb[:, :seq_len, :]
+
+        # Append soft prompts
+        if self.len_soft_prompts > 0:
+            soft_prompts = self.soft_prompt_hub(domain_id).view(
+                batch_size, self.len_soft_prompts, self.hidden_size
+            )
+            x = torch.cat([x, soft_prompts], dim=1)
+
+        # Transformer backbone
+        for block in self.blocks:
+            x = block(x)
+
+        # Decode only the action segment
+        return self.action_decoder(self.norm(x[:, :num_actions]), domain_id)
diff --git a/lerobot/src/lerobot/policies/xvla/utils.py b/lerobot/src/lerobot/policies/xvla/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..bf31ffd82cf2f7a8cb4080401f92171075cbb103
--- /dev/null
+++ b/lerobot/src/lerobot/policies/xvla/utils.py
@@ -0,0 +1,138 @@
+import math
+
+import numpy as np
+
+
+def mat2quat(rmat):
+    """
+    Converts given rotation matrix to quaternion.
+
+    Args:
+        rmat (np.array): 3x3 rotation matrix
+
+    Returns:
+        np.array: (x,y,z,w) float quaternion angles
+    """
+    mat = np.asarray(rmat).astype(np.float32)[:3, :3]
+
+    m00 = mat[0, 0]
+    m01 = mat[0, 1]
+    m02 = mat[0, 2]
+    m10 = mat[1, 0]
+    m11 = mat[1, 1]
+    m12 = mat[1, 2]
+    m20 = mat[2, 0]
+    m21 = mat[2, 1]
+    m22 = mat[2, 2]
+    # symmetric matrix k
+    k = np.array(
+        [
+            [m00 - m11 - m22, np.float32(0.0), np.float32(0.0), np.float32(0.0)],
+            [m01 + m10, m11 - m00 - m22, np.float32(0.0), np.float32(0.0)],
+            [m02 + m20, m12 + m21, m22 - m00 - m11, np.float32(0.0)],
+            [m21 - m12, m02 - m20, m10 - m01, m00 + m11 + m22],
+        ]
+    )
+    k /= 3.0
+    # quaternion is Eigen vector of k that corresponds to largest eigenvalue
+    w, v = np.linalg.eigh(k)
+    inds = np.array([3, 0, 1, 2])
+    q1 = v[inds, np.argmax(w)]
+    if q1[0] < 0.0:
+        np.negative(q1, q1)
+    inds = np.array([1, 2, 3, 0])
+    return q1[inds]
+
+
+def quat2axisangle(quat):
+    """
+    Converts quaternion to axis-angle format.
+    Returns a unit vector direction scaled by its angle in radians.
+
+    Args:
+        quat (np.array): (x,y,z,w) vec4 float angles
+
+    Returns:
+        np.array: (ax,ay,az) axis-angle exponential coordinates
+    """
+    # clip quaternion
+    if quat[3] > 1.0:
+        quat[3] = 1.0
+    elif quat[3] < -1.0:
+        quat[3] = -1.0
+
+    den = np.sqrt(1.0 - quat[3] * quat[3])
+    if math.isclose(den, 0.0):
+        # This is (close to) a zero degree rotation, immediately return
+        return np.zeros(3)
+
+    return (quat[:3] * 2.0 * math.acos(quat[3])) / den
+
+
+def rotate6d_to_axis_angle(r6d):
+    """
+    r6d: np.ndarray, shape (N, 6)
+    return: np.ndarray, shape (N, 3), axis-angle vectors
+    """
+    flag = 0
+    if len(r6d.shape) == 1:
+        r6d = r6d[None, ...]
+        flag = 1
+
+    a1 = r6d[:, 0:3]
+    a2 = r6d[:, 3:6]
+
+    # b1
+    b1 = a1 / (np.linalg.norm(a1, axis=-1, keepdims=True) + 1e-6)
+
+    # b2
+    dot_prod = np.sum(b1 * a2, axis=-1, keepdims=True)
+    b2_orth = a2 - dot_prod * b1
+    b2 = b2_orth / (np.linalg.norm(b2_orth, axis=-1, keepdims=True) + 1e-6)
+
+    # b3
+    b3 = np.cross(b1, b2, axis=-1)
+
+    rotation_matrix = np.stack([b1, b2, b3], axis=-1)  # shape: (N, 3, 3)
+
+    axis_angle_list = []
+    for i in range(rotation_matrix.shape[0]):
+        quat = mat2quat(rotation_matrix[i])
+        axis_angle = quat2axisangle(quat)
+        axis_angle_list.append(axis_angle)
+
+    axis_angle_array = np.stack(axis_angle_list, axis=0)  # shape: (N, 3)
+
+    if flag == 1:
+        axis_angle_array = axis_angle_array[0]
+
+    return axis_angle_array
+
+
+def mat_to_rotate6d(abs_action):
+    if len(abs_action.shape) == 2:
+        return np.concatenate([abs_action[:3, 0], abs_action[:3, 1]], axis=-1)
+    elif len(abs_action.shape) == 3:
+        return np.concatenate([abs_action[:, :3, 0], abs_action[:, :3, 1]], axis=-1)
+    else:
+        raise NotImplementedError
+
+
+def drop_path(x, drop_prob: float = 0.0, training: bool = False, scale_by_keep: bool = True):
+    """Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).
+
+    This is the same as the DropConnect impl I created for EfficientNet, etc networks, however,
+    the original name is misleading as 'Drop Connect' is a different form of dropout in a separate paper...
+    See discussion: https://github.com/tensorflow/tpu/issues/494#issuecomment-532968956 ... I've opted for
+    changing the layer and argument names to 'drop path' rather than mix DropConnect as a layer name and use
+    'survival rate' as the argument.
+
+    """
+    if drop_prob == 0.0 or not training:
+        return x
+    keep_prob = 1 - drop_prob
+    shape = (x.shape[0],) + (1,) * (x.ndim - 1)  # work with diff dim tensors, not just 2D ConvNets
+    random_tensor = x.new_empty(shape).bernoulli_(keep_prob)
+    if keep_prob > 0.0 and scale_by_keep:
+        random_tensor.div_(keep_prob)
+    return x * random_tensor
diff --git a/lerobot/src/lerobot/processor/__init__.py b/lerobot/src/lerobot/processor/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..12dcf0c6d3a3354b85c98effe65185ddee54f0c3
--- /dev/null
+++ b/lerobot/src/lerobot/processor/__init__.py
@@ -0,0 +1,134 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from lerobot.types import (
+    EnvAction,
+    EnvTransition,
+    PolicyAction,
+    RobotAction,
+    RobotObservation,
+    TransitionKey,
+)
+
+from .batch_processor import AddBatchDimensionProcessorStep
+from .converters import (
+    batch_to_transition,
+    create_transition,
+    transition_to_batch,
+)
+from .delta_action_processor import MapDeltaActionToRobotActionStep, MapTensorToDeltaActionDictStep
+from .device_processor import DeviceProcessorStep
+from .factory import (
+    make_default_processors,
+    make_default_robot_action_processor,
+    make_default_robot_observation_processor,
+    make_default_teleop_action_processor,
+)
+from .gym_action_processor import (
+    Numpy2TorchActionProcessorStep,
+    Torch2NumpyActionProcessorStep,
+)
+from .hil_processor import (
+    AddTeleopActionAsComplimentaryDataStep,
+    AddTeleopEventsAsInfoStep,
+    GripperPenaltyProcessorStep,
+    GymHILAdapterProcessorStep,
+    ImageCropResizeProcessorStep,
+    InterventionActionProcessorStep,
+    RewardClassifierProcessorStep,
+    TimeLimitProcessorStep,
+)
+from .normalize_processor import NormalizerProcessorStep, UnnormalizerProcessorStep, hotswap_stats
+from .observation_processor import VanillaObservationProcessorStep
+from .pipeline import (
+    ActionProcessorStep,
+    ComplementaryDataProcessorStep,
+    DataProcessorPipeline,
+    DoneProcessorStep,
+    IdentityProcessorStep,
+    InfoProcessorStep,
+    ObservationProcessorStep,
+    PolicyActionProcessorStep,
+    PolicyProcessorPipeline,
+    ProcessorKwargs,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    RewardProcessorStep,
+    RobotActionProcessorStep,
+    RobotProcessorPipeline,
+    TruncatedProcessorStep,
+)
+from .policy_robot_bridge import (
+    PolicyActionToRobotActionProcessorStep,
+    RobotActionToPolicyActionProcessorStep,
+)
+from .rename_processor import RenameObservationsProcessorStep
+from .tokenizer_processor import ActionTokenizerProcessorStep, TokenizerProcessorStep
+
+__all__ = [
+    "ActionProcessorStep",
+    "AddTeleopActionAsComplimentaryDataStep",
+    "AddTeleopEventsAsInfoStep",
+    "ComplementaryDataProcessorStep",
+    "batch_to_transition",
+    "create_transition",
+    "DeviceProcessorStep",
+    "DoneProcessorStep",
+    "EnvAction",
+    "EnvTransition",
+    "GymHILAdapterProcessorStep",
+    "GripperPenaltyProcessorStep",
+    "hotswap_stats",
+    "IdentityProcessorStep",
+    "ImageCropResizeProcessorStep",
+    "InfoProcessorStep",
+    "InterventionActionProcessorStep",
+    "make_default_processors",
+    "make_default_teleop_action_processor",
+    "make_default_robot_action_processor",
+    "make_default_robot_observation_processor",
+    "MapDeltaActionToRobotActionStep",
+    "MapTensorToDeltaActionDictStep",
+    "NormalizerProcessorStep",
+    "Numpy2TorchActionProcessorStep",
+    "ObservationProcessorStep",
+    "PolicyAction",
+    "PolicyActionProcessorStep",
+    "PolicyProcessorPipeline",
+    "ProcessorKwargs",
+    "ProcessorStep",
+    "ProcessorStepRegistry",
+    "RobotAction",
+    "RobotActionProcessorStep",
+    "RobotObservation",
+    "RenameObservationsProcessorStep",
+    "RewardClassifierProcessorStep",
+    "RewardProcessorStep",
+    "DataProcessorPipeline",
+    "TimeLimitProcessorStep",
+    "AddBatchDimensionProcessorStep",
+    "RobotProcessorPipeline",
+    "TokenizerProcessorStep",
+    "ActionTokenizerProcessorStep",
+    "Torch2NumpyActionProcessorStep",
+    "RobotActionToPolicyActionProcessorStep",
+    "PolicyActionToRobotActionProcessorStep",
+    "transition_to_batch",
+    "TransitionKey",
+    "TruncatedProcessorStep",
+    "UnnormalizerProcessorStep",
+    "VanillaObservationProcessorStep",
+]
diff --git a/lerobot/src/lerobot/processor/batch_processor.py b/lerobot/src/lerobot/processor/batch_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..c904acf8489a30e4bc4f0605c42874649b5d8009
--- /dev/null
+++ b/lerobot/src/lerobot/processor/batch_processor.py
@@ -0,0 +1,254 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+This script defines processor steps for adding a batch dimension to various components of an environment transition.
+
+These steps are designed to process actions, observations, and complementary data, making them suitable for batch processing by adding a leading dimension. This is a common requirement before feeding data into a neural network model.
+"""
+
+from dataclasses import dataclass, field
+
+from torch import Tensor
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.types import EnvTransition, PolicyAction
+from lerobot.utils.constants import OBS_ENV_STATE, OBS_IMAGE, OBS_IMAGES, OBS_STATE
+
+from .pipeline import (
+    ComplementaryDataProcessorStep,
+    ObservationProcessorStep,
+    PolicyActionProcessorStep,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    TransitionKey,
+)
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="to_batch_processor_action")
+class AddBatchDimensionActionStep(PolicyActionProcessorStep):
+    """
+    Processor step to add a batch dimension to a 1D tensor action.
+
+    This is useful for creating a batch of size 1 from a single action sample.
+    """
+
+    def action(self, action: PolicyAction) -> PolicyAction:
+        """
+        Adds a batch dimension to the action if it's a 1D tensor.
+
+        Args:
+            action: The action tensor.
+
+        Returns:
+            The action tensor with an added batch dimension.
+        """
+        if action.dim() != 1:
+            return action
+        return action.unsqueeze(0)
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Returns the input features unchanged.
+
+        Adding a batch dimension does not alter the feature definition.
+
+        Args:
+            features: A dictionary of policy features.
+
+        Returns:
+            The original dictionary of policy features.
+        """
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="to_batch_processor_observation")
+class AddBatchDimensionObservationStep(ObservationProcessorStep):
+    """
+    Processor step to add a batch dimension to observations.
+
+    It handles different types of observations:
+    - State vectors (1D tensors).
+    - Single images (3D tensors).
+    - Dictionaries of multiple images (3D tensors).
+    """
+
+    def observation(self, observation: dict[str, Tensor]) -> dict[str, Tensor]:
+        """
+        Adds a batch dimension to tensor-based observations in the observation dictionary.
+
+        Args:
+            observation: The observation dictionary.
+
+        Returns:
+            The observation dictionary with batch dimensions added to tensors.
+        """
+        # Process state observations - add batch dim if 1D
+        for state_key in [OBS_STATE, OBS_ENV_STATE]:
+            if state_key in observation:
+                state_value = observation[state_key]
+                if isinstance(state_value, Tensor) and state_value.dim() == 1:
+                    observation[state_key] = state_value.unsqueeze(0)
+
+        # Process single image observation - add batch dim if 3D
+        if OBS_IMAGE in observation:
+            image_value = observation[OBS_IMAGE]
+            if isinstance(image_value, Tensor) and image_value.dim() == 3:
+                observation[OBS_IMAGE] = image_value.unsqueeze(0)
+
+        # Process multiple image observations - add batch dim if 3D
+        for key, value in observation.items():
+            if key.startswith(f"{OBS_IMAGES}.") and isinstance(value, Tensor) and value.dim() == 3:
+                observation[key] = value.unsqueeze(0)
+        return observation
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Returns the input features unchanged.
+
+        Adding a batch dimension does not alter the feature definition.
+
+        Args:
+            features: A dictionary of policy features.
+
+        Returns:
+            The original dictionary of policy features.
+        """
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="to_batch_processor_complementary_data")
+class AddBatchDimensionComplementaryDataStep(ComplementaryDataProcessorStep):
+    """
+    Processor step to add a batch dimension to complementary data fields.
+
+    Handles specific keys like 'task', 'index', and 'task_index' to make them batched.
+    - 'task' (str) is wrapped in a list.
+    - 'index' and 'task_index' (0D tensors) get a batch dimension.
+    """
+
+    def complementary_data(self, complementary_data: dict) -> dict:
+        """
+        Adds a batch dimension to specific fields in the complementary data dictionary.
+
+        Args:
+            complementary_data: The complementary data dictionary.
+
+        Returns:
+            The complementary data dictionary with batch dimensions added.
+        """
+        # Process task field - wrap string in list to add batch dimension
+        if "task" in complementary_data:
+            task_value = complementary_data["task"]
+            if isinstance(task_value, str):
+                complementary_data["task"] = [task_value]
+
+        # Process index field - add batch dim if 0D
+        if "index" in complementary_data:
+            index_value = complementary_data["index"]
+            if isinstance(index_value, Tensor) and index_value.dim() == 0:
+                complementary_data["index"] = index_value.unsqueeze(0)
+
+        # Process task_index field - add batch dim if 0D
+        if "task_index" in complementary_data:
+            task_index_value = complementary_data["task_index"]
+            if isinstance(task_index_value, Tensor) and task_index_value.dim() == 0:
+                complementary_data["task_index"] = task_index_value.unsqueeze(0)
+        return complementary_data
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Returns the input features unchanged.
+
+        Adding a batch dimension does not alter the feature definition.
+
+        Args:
+            features: A dictionary of policy features.
+
+        Returns:
+            The original dictionary of policy features.
+        """
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="to_batch_processor")
+class AddBatchDimensionProcessorStep(ProcessorStep):
+    """
+    A composite processor step that adds a batch dimension to the entire environment transition.
+
+    This step combines individual processors for actions, observations, and complementary data
+    to create a batched transition (batch size 1) from a single-instance transition.
+
+    Attributes:
+        to_batch_action_processor: Processor for the action component.
+        to_batch_observation_processor: Processor for the observation component.
+        to_batch_complementary_data_processor: Processor for the complementary data component.
+    """
+
+    to_batch_action_processor: AddBatchDimensionActionStep = field(
+        default_factory=AddBatchDimensionActionStep
+    )
+    to_batch_observation_processor: AddBatchDimensionObservationStep = field(
+        default_factory=AddBatchDimensionObservationStep
+    )
+    to_batch_complementary_data_processor: AddBatchDimensionComplementaryDataStep = field(
+        default_factory=AddBatchDimensionComplementaryDataStep
+    )
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """
+        Applies the batching process to all relevant parts of an environment transition.
+
+        Args:
+            transition: The environment transition to process.
+
+        Returns:
+            The environment transition with a batch dimension added.
+        """
+        if transition[TransitionKey.ACTION] is not None:
+            transition = self.to_batch_action_processor(transition)
+        if transition[TransitionKey.OBSERVATION] is not None:
+            transition = self.to_batch_observation_processor(transition)
+        if transition[TransitionKey.COMPLEMENTARY_DATA] is not None:
+            transition = self.to_batch_complementary_data_processor(transition)
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Returns the input features unchanged.
+
+        Adding a batch dimension does not alter the feature definition.
+
+        Args:
+            features: A dictionary of policy features.
+
+        Returns:
+            The original dictionary of policy features.
+        """
+        # NOTE: We ignore the batch dimension when transforming features
+        return features
diff --git a/lerobot/src/lerobot/processor/converters.py b/lerobot/src/lerobot/processor/converters.py
new file mode 100644
index 0000000000000000000000000000000000000000..ffdf0098cd19214fb335702bc876a344182ce96f
--- /dev/null
+++ b/lerobot/src/lerobot/processor/converters.py
@@ -0,0 +1,415 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from __future__ import annotations
+
+from collections.abc import Sequence
+from functools import singledispatch
+from typing import Any
+
+import numpy as np
+import torch
+
+from lerobot.types import EnvTransition, PolicyAction, RobotAction, RobotObservation, TransitionKey
+from lerobot.utils.constants import ACTION, DONE, INFO, OBS_PREFIX, REWARD, TRUNCATED
+
+
+@singledispatch
+def to_tensor(
+    value: Any,
+    *,
+    dtype: torch.dtype | None = torch.float32,
+    device: torch.device | str | None = None,
+) -> torch.Tensor:
+    """
+    Convert various data types to PyTorch tensors with configurable options.
+
+    This is a unified tensor conversion function using single dispatch to handle
+    different input types appropriately.
+
+    Args:
+        value: Input value to convert (tensor, array, scalar, sequence, etc.).
+        dtype: Target tensor dtype. If None, preserves original dtype.
+        device: Target device for the tensor.
+
+    Returns:
+        A PyTorch tensor.
+
+    Raises:
+        TypeError: If the input type is not supported.
+    """
+    raise TypeError(f"Unsupported type for tensor conversion: {type(value)}")
+
+
+@to_tensor.register(torch.Tensor)
+def _(value: torch.Tensor, *, dtype=torch.float32, device=None, **kwargs) -> torch.Tensor:
+    """Handle conversion for existing PyTorch tensors."""
+    if dtype is not None:
+        value = value.to(dtype=dtype)
+    if device is not None:
+        value = value.to(device=device)
+    return value
+
+
+@to_tensor.register(np.ndarray)
+def _(
+    value: np.ndarray,
+    *,
+    dtype=torch.float32,
+    device=None,
+    **kwargs,
+) -> torch.Tensor:
+    """Handle conversion for numpy arrays."""
+    # Check for numpy scalars (0-dimensional arrays) and treat them as scalars.
+    if value.ndim == 0:
+        # Numpy scalars should be converted to 0-dimensional tensors.
+        scalar_value = value.item()
+        return torch.tensor(scalar_value, dtype=dtype, device=device)
+
+    # Create tensor from numpy array.
+    tensor = torch.from_numpy(value)
+
+    # Apply dtype and device conversion if specified.
+    if dtype is not None:
+        tensor = tensor.to(dtype=dtype)
+    if device is not None:
+        tensor = tensor.to(device=device)
+
+    return tensor
+
+
+@to_tensor.register(int)
+@to_tensor.register(float)
+@to_tensor.register(np.integer)
+@to_tensor.register(np.floating)
+def _(value, *, dtype=torch.float32, device=None, **kwargs) -> torch.Tensor:
+    """Handle conversion for scalar values including numpy scalars."""
+    return torch.tensor(value, dtype=dtype, device=device)
+
+
+@to_tensor.register(list)
+@to_tensor.register(tuple)
+def _(value: Sequence, *, dtype=torch.float32, device=None, **kwargs) -> torch.Tensor:
+    """Handle conversion for sequences (lists, tuples)."""
+    return torch.tensor(value, dtype=dtype, device=device)
+
+
+@to_tensor.register(dict)
+def _(value: dict, *, device=None, **kwargs) -> dict:
+    """Handle conversion for dictionaries by recursively converting their values to tensors."""
+    if not value:
+        return {}
+
+    result = {}
+    for key, sub_value in value.items():
+        if sub_value is None:
+            continue
+
+        if isinstance(sub_value, dict):
+            # Recursively process nested dictionaries.
+            result[key] = to_tensor(
+                sub_value,
+                device=device,
+                **kwargs,
+            )
+            continue
+
+        # Convert individual values to tensors.
+        result[key] = to_tensor(
+            sub_value,
+            device=device,
+            **kwargs,
+        )
+    return result
+
+
+def from_tensor_to_numpy(x: torch.Tensor | Any) -> np.ndarray | float | int | Any:
+    """
+    Convert a PyTorch tensor to a numpy array or scalar if applicable.
+
+    If the input is not a tensor, it is returned unchanged.
+
+    Args:
+        x: The input, which can be a tensor or any other type.
+
+    Returns:
+        A numpy array, a scalar, or the original input.
+    """
+    if isinstance(x, torch.Tensor):
+        return x.item() if x.numel() == 1 else x.detach().cpu().numpy()
+    return x
+
+
+def _extract_complementary_data(batch: dict[str, Any]) -> dict[str, Any]:
+    """
+    Extract complementary data from a batch dictionary.
+
+    This includes padding flags, task description, and indices.
+
+    Args:
+        batch: The batch dictionary.
+
+    Returns:
+        A dictionary with the extracted complementary data.
+    """
+    pad_keys = {k: v for k, v in batch.items() if "_is_pad" in k}
+    task_key = {"task": batch["task"]} if "task" in batch else {}
+    subtask_key = {"subtask": batch["subtask"]} if "subtask" in batch else {}
+    index_key = {"index": batch["index"]} if "index" in batch else {}
+    task_index_key = {"task_index": batch["task_index"]} if "task_index" in batch else {}
+    episode_index_key = {"episode_index": batch["episode_index"]} if "episode_index" in batch else {}
+
+    return {**pad_keys, **task_key, **subtask_key, **index_key, **task_index_key, **episode_index_key}
+
+
+def create_transition(
+    observation: RobotObservation | None = None,
+    action: PolicyAction | RobotAction | None = None,
+    reward: float = 0.0,
+    done: bool = False,
+    truncated: bool = False,
+    info: dict[str, Any] | None = None,
+    complementary_data: dict[str, Any] | None = None,
+) -> EnvTransition:
+    """
+    Create an `EnvTransition` dictionary with sensible defaults.
+
+    Args:
+        observation: Observation dictionary.
+        action: Action dictionary.
+        reward: Scalar reward value.
+        done: Episode termination flag.
+        truncated: Episode truncation flag.
+        info: Additional info dictionary.
+        complementary_data: Complementary data dictionary.
+
+    Returns:
+        A complete `EnvTransition` dictionary.
+    """
+    return {
+        TransitionKey.OBSERVATION: observation,
+        TransitionKey.ACTION: action,
+        TransitionKey.REWARD: reward,
+        TransitionKey.DONE: done,
+        TransitionKey.TRUNCATED: truncated,
+        TransitionKey.INFO: info if info is not None else {},
+        TransitionKey.COMPLEMENTARY_DATA: complementary_data if complementary_data is not None else {},
+    }
+
+
+def robot_action_observation_to_transition(
+    action_observation: tuple[RobotAction, RobotObservation],
+) -> EnvTransition:
+    """
+    Convert a raw robot action and observation dictionary into a standardized `EnvTransition`.
+
+    Args:
+        action: The raw action dictionary from a teleoperation device or controller.
+        observation: The raw observation dictionary from the environment.
+
+    Returns:
+        An `EnvTransition` containing the formatted observation.
+    """
+    if not isinstance(action_observation, tuple):
+        raise ValueError("action_observation should be a tuple type with an action and observation")
+
+    action, observation = action_observation
+
+    if action is not None and not isinstance(action, dict):
+        raise ValueError(f"Action should be a RobotAction type got {type(action)}")
+
+    if observation is not None and not isinstance(observation, dict):
+        raise ValueError(f"Observation should be a RobotObservation type got {type(observation)}")
+
+    return create_transition(action=action, observation=observation)
+
+
+def robot_action_to_transition(action: RobotAction) -> EnvTransition:
+    """
+    Convert a raw robot action dictionary into a standardized `EnvTransition`.
+
+    Args:
+        action: The raw action dictionary from a teleoperation device or controller.
+
+    Returns:
+        An `EnvTransition` containing the formatted action.
+    """
+    if not isinstance(action, dict):
+        raise ValueError(f"Action should be a RobotAction type got {type(action)}")
+    return create_transition(action=action)
+
+
+def observation_to_transition(observation: RobotObservation) -> EnvTransition:
+    """
+    Convert a raw robot observation dictionary into a standardized `EnvTransition`.
+
+    Args:
+        observation: The raw observation dictionary from the environment.
+
+    Returns:
+        An `EnvTransition` containing the formatted observation.
+    """
+    if not isinstance(observation, dict):
+        raise ValueError(f"Observation should be a RobotObservation type got {type(observation)}")
+    return create_transition(observation=observation)
+
+
+def transition_to_robot_action(transition: EnvTransition) -> RobotAction:
+    """
+    Extract a raw robot action dictionary for a robot from an `EnvTransition`.
+
+    This function searches for keys in the format "action.*.pos" or "action.*.vel"
+    and converts them into a flat dictionary suitable for sending to a robot controller.
+
+    Args:
+        transition: The `EnvTransition` containing the action.
+
+    Returns:
+        A dictionary representing the raw robot action.
+    """
+    if not isinstance(transition, dict):
+        raise ValueError(f"Transition should be a EnvTransition type (dict) got {type(transition)}")
+
+    action = transition.get(TransitionKey.ACTION)
+    if not isinstance(action, dict):
+        raise ValueError(f"Action should be a RobotAction type (dict) got {type(action)}")
+    return transition.get(TransitionKey.ACTION)
+
+
+def transition_to_policy_action(transition: EnvTransition) -> PolicyAction:
+    """
+    Convert an `EnvTransition` to a `PolicyAction`.
+    """
+    if not isinstance(transition, dict):
+        raise ValueError(f"Transition should be a EnvTransition type (dict) got {type(transition)}")
+
+    action = transition.get(TransitionKey.ACTION)
+    if not isinstance(action, PolicyAction):
+        raise ValueError(f"Action should be a PolicyAction type got {type(action)}")
+    return action
+
+
+def transition_to_observation(transition: EnvTransition) -> RobotObservation:
+    """
+    Convert an `EnvTransition` to a `RobotObservation`.
+    """
+    if not isinstance(transition, dict):
+        raise ValueError(f"Transition should be a EnvTransition type (dict) got {type(transition)}")
+
+    observation = transition.get(TransitionKey.OBSERVATION)
+    if not isinstance(observation, dict):
+        raise ValueError(f"Observation should be a RobotObservation (dict) type got {type(observation)}")
+    return observation
+
+
+def policy_action_to_transition(action: PolicyAction) -> EnvTransition:
+    """
+    Convert a `PolicyAction` to an `EnvTransition`.
+    """
+    if not isinstance(action, PolicyAction):
+        raise ValueError(f"Action should be a PolicyAction type got {type(action)}")
+    return create_transition(action=action)
+
+
+def batch_to_transition(batch: dict[str, Any]) -> EnvTransition:
+    """
+    Convert a batch dictionary from a dataset/dataloader into an `EnvTransition`.
+
+    This function maps recognized keys from a batch to the `EnvTransition` structure,
+    filling in missing keys with sensible defaults.
+
+    Args:
+        batch: A batch dictionary.
+
+    Returns:
+        An `EnvTransition` dictionary.
+
+    Raises:
+        ValueError: If the input is not a dictionary.
+    """
+
+    # Validate input type.
+    if not isinstance(batch, dict):
+        raise ValueError(f"EnvTransition must be a dictionary. Got {type(batch).__name__}")
+
+    action = batch.get(ACTION)
+    if action is not None and not isinstance(action, PolicyAction):
+        raise ValueError(f"Action should be a PolicyAction type got {type(action)}")
+
+    # Extract observation and complementary data keys.
+    observation_keys = {k: v for k, v in batch.items() if k.startswith(OBS_PREFIX)}
+    complementary_data = _extract_complementary_data(batch)
+
+    return create_transition(
+        observation=observation_keys if observation_keys else None,
+        action=batch.get(ACTION),
+        reward=batch.get(REWARD, 0.0),
+        done=batch.get(DONE, False),
+        truncated=batch.get(TRUNCATED, False),
+        info=batch.get("info", {}),
+        complementary_data=complementary_data if complementary_data else None,
+    )
+
+
+def transition_to_batch(transition: EnvTransition) -> dict[str, Any]:
+    """
+    Convert an `EnvTransition` back to the canonical batch format used in LeRobot.
+
+    This is the inverse of `batch_to_transition`.
+
+    Args:
+        transition: The `EnvTransition` to convert.
+
+    Returns:
+        A batch dictionary with canonical LeRobot field names.
+    """
+    if not isinstance(transition, dict):
+        raise ValueError(f"Transition should be a EnvTransition type (dict) got {type(transition)}")
+
+    batch = {
+        ACTION: transition.get(TransitionKey.ACTION),
+        REWARD: transition.get(TransitionKey.REWARD, 0.0),
+        DONE: transition.get(TransitionKey.DONE, False),
+        TRUNCATED: transition.get(TransitionKey.TRUNCATED, False),
+        INFO: transition.get(TransitionKey.INFO, {}),
+    }
+
+    # Add complementary data.
+    comp_data = transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+    if comp_data:
+        batch.update(comp_data)
+
+    # Flatten observation dictionary.
+    observation = transition.get(TransitionKey.OBSERVATION)
+    if isinstance(observation, dict):
+        batch.update(observation)
+
+    return batch
+
+
+def identity_transition(transition: EnvTransition) -> EnvTransition:
+    """
+    An identity function for transitions, returning the input unchanged.
+
+    Useful as a default or placeholder in processing pipelines.
+
+    Args:
+        tr: An `EnvTransition`.
+
+    Returns:
+        The same `EnvTransition`.
+    """
+    return transition
diff --git a/lerobot/src/lerobot/processor/delta_action_processor.py b/lerobot/src/lerobot/processor/delta_action_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..f7f5676ac7ff6bf9c985e885f1ef3453d5cbc488
--- /dev/null
+++ b/lerobot/src/lerobot/processor/delta_action_processor.py
@@ -0,0 +1,143 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from lerobot.types import PolicyAction, RobotAction
+
+from .pipeline import ActionProcessorStep, ProcessorStepRegistry, RobotActionProcessorStep
+
+
+@ProcessorStepRegistry.register("map_tensor_to_delta_action_dict")
+@dataclass
+class MapTensorToDeltaActionDictStep(ActionProcessorStep):
+    """
+    Maps a flat action tensor from a policy to a structured delta action dictionary.
+
+    This step is typically used after a policy outputs a continuous action vector.
+    It decomposes the vector into named components for delta movements of the
+    end-effector (x, y, z) and optionally the gripper.
+
+    Attributes:
+        use_gripper: If True, assumes the 4th element of the tensor is the
+                     gripper action.
+    """
+
+    use_gripper: bool = True
+
+    def action(self, action: PolicyAction) -> RobotAction:
+        if not isinstance(action, PolicyAction):
+            raise ValueError("Only PolicyAction is supported for this processor")
+
+        if action.dim() > 1:
+            action = action.squeeze(0)
+
+        # TODO (maractingi): add rotation
+        delta_action = {
+            "delta_x": action[0].item(),
+            "delta_y": action[1].item(),
+            "delta_z": action[2].item(),
+        }
+        if self.use_gripper:
+            delta_action["gripper"] = action[3].item()
+        return delta_action
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        for axis in ["x", "y", "z"]:
+            features[PipelineFeatureType.ACTION][f"delta_{axis}"] = PolicyFeature(
+                type=FeatureType.ACTION, shape=(1,)
+            )
+
+        if self.use_gripper:
+            features[PipelineFeatureType.ACTION]["gripper"] = PolicyFeature(
+                type=FeatureType.ACTION, shape=(1,)
+            )
+        return features
+
+
+@ProcessorStepRegistry.register("map_delta_action_to_robot_action")
+@dataclass
+class MapDeltaActionToRobotActionStep(RobotActionProcessorStep):
+    """
+    Maps delta actions from teleoperators to robot target actions for inverse kinematics.
+
+    This step converts a dictionary of delta movements (e.g., from a gamepad)
+    into a target action format that includes an "enabled" flag and target
+    end-effector positions. It also handles scaling and noise filtering.
+
+    Attributes:
+        position_scale: A factor to scale the delta position inputs.
+        noise_threshold: The magnitude below which delta inputs are considered noise
+                         and do not trigger an "enabled" state.
+    """
+
+    # Scale factors for delta movements
+    position_scale: float = 1.0
+    noise_threshold: float = 1e-3  # 1 mm threshold to filter out noise
+
+    def action(self, action: RobotAction) -> RobotAction:
+        # NOTE (maractingi): Action can be a dict from the teleop_devices or a tensor from the policy
+        # TODO (maractingi): changing this target_xyz naming convention from the teleop_devices
+        delta_x = action.pop("delta_x")
+        delta_y = action.pop("delta_y")
+        delta_z = action.pop("delta_z")
+        gripper = action.pop("gripper")
+
+        # Determine if the teleoperator is actively providing input
+        # Consider enabled if any significant movement delta is detected
+        position_magnitude = (delta_x**2 + delta_y**2 + delta_z**2) ** 0.5  # Use Euclidean norm for position
+        enabled = position_magnitude > self.noise_threshold  # Small threshold to avoid noise
+
+        # Scale the deltas appropriately
+        scaled_delta_x = delta_x * self.position_scale
+        scaled_delta_y = delta_y * self.position_scale
+        scaled_delta_z = delta_z * self.position_scale
+
+        # For gamepad/keyboard, we don't have rotation input, so set to 0
+        # These could be extended in the future for more sophisticated teleoperators
+        target_wx = 0.0
+        target_wy = 0.0
+        target_wz = 0.0
+
+        # Update action with robot target format
+        action = {
+            "enabled": enabled,
+            "target_x": scaled_delta_x,
+            "target_y": scaled_delta_y,
+            "target_z": scaled_delta_z,
+            "target_wx": target_wx,
+            "target_wy": target_wy,
+            "target_wz": target_wz,
+            "gripper_vel": float(gripper),
+        }
+
+        return action
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        for axis in ["x", "y", "z", "gripper"]:
+            features[PipelineFeatureType.ACTION].pop(f"delta_{axis}", None)
+
+        for feat in ["enabled", "target_x", "target_y", "target_z", "target_wx", "target_wy", "target_wz"]:
+            features[PipelineFeatureType.ACTION][f"{feat}"] = PolicyFeature(
+                type=FeatureType.ACTION, shape=(1,)
+            )
+
+        return features
diff --git a/lerobot/src/lerobot/processor/device_processor.py b/lerobot/src/lerobot/processor/device_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..36c80e58e34bd728417abd98177a6ea3633de84d
--- /dev/null
+++ b/lerobot/src/lerobot/processor/device_processor.py
@@ -0,0 +1,194 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+This script defines a processor step for moving environment transition data to a specific torch device and casting
+its floating-point precision.
+"""
+
+from dataclasses import dataclass
+from typing import Any
+
+import torch
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.types import EnvTransition, PolicyAction, TransitionKey
+from lerobot.utils.device_utils import get_safe_torch_device
+
+from .pipeline import ProcessorStep, ProcessorStepRegistry
+
+
+@ProcessorStepRegistry.register("device_processor")
+@dataclass
+class DeviceProcessorStep(ProcessorStep):
+    """
+    Processor step to move all tensors within an `EnvTransition` to a specified device and optionally cast their
+    floating-point data type.
+
+    This is crucial for preparing data for model training or inference on hardware like GPUs.
+
+    Attributes:
+        device: The target device for tensors (e.g., "cpu", "cuda", "cuda:0").
+        float_dtype: The target floating-point dtype as a string (e.g., "float32", "float16", "bfloat16").
+                     If None, the dtype is not changed.
+    """
+
+    device: str = "cpu"
+    float_dtype: str | None = None
+
+    DTYPE_MAPPING = {
+        "float16": torch.float16,
+        "float32": torch.float32,
+        "float64": torch.float64,
+        "bfloat16": torch.bfloat16,
+        "half": torch.float16,
+        "float": torch.float32,
+        "double": torch.float64,
+    }
+
+    def __post_init__(self):
+        """
+        Initializes the processor by converting string configurations to torch objects.
+
+        This method sets up the `torch.device`, determines if transfers can be non-blocking, and validates the
+        `float_dtype` string, converting it to a `torch.dtype` object.
+        """
+        self.tensor_device: torch.device = get_safe_torch_device(self.device)
+        # Update device string in case a specific GPU was selected (e.g., "cuda" -> "cuda:0")
+        self.device = self.tensor_device.type
+        self.non_blocking = "cuda" in str(self.device)
+
+        # Validate and convert float_dtype string to torch dtype
+        if self.float_dtype is not None:
+            if self.float_dtype not in self.DTYPE_MAPPING:
+                raise ValueError(
+                    f"Invalid float_dtype '{self.float_dtype}'. Available options: {list(self.DTYPE_MAPPING.keys())}"
+                )
+            self._target_float_dtype = self.DTYPE_MAPPING[self.float_dtype]
+        else:
+            self._target_float_dtype = None
+
+    def _process_tensor(self, tensor: torch.Tensor) -> torch.Tensor:
+        """
+        Moves a single tensor to the target device and casts its dtype.
+
+        Handles multi-GPU scenarios by not moving a tensor if it's already on a different CUDA device than
+        the target, which is useful when using frameworks like Accelerate.
+
+        Args:
+            tensor: The input torch.Tensor.
+
+        Returns:
+            The processed tensor on the correct device and with the correct dtype.
+        """
+        # Determine target device
+        if tensor.is_cuda and self.tensor_device.type == "cuda":
+            # Both tensor and target are on GPU - preserve tensor's GPU placement.
+            # This handles multi-GPU scenarios where Accelerate has already placed
+            # tensors on the correct GPU for each process.
+            target_device = tensor.device
+        else:
+            # Either tensor is on CPU, or we're configured for CPU.
+            # In both cases, use the configured device.
+            target_device = self.tensor_device
+
+        # MPS workaround: Convert float64 to float32 since MPS doesn't support float64
+        if target_device.type == "mps" and tensor.dtype == torch.float64:
+            tensor = tensor.to(dtype=torch.float32)
+
+        # Only move if necessary
+        if tensor.device != target_device:
+            tensor = tensor.to(target_device, non_blocking=self.non_blocking)
+
+        # Convert float dtype if specified and tensor is floating point
+        if self._target_float_dtype is not None and tensor.is_floating_point():
+            tensor = tensor.to(dtype=self._target_float_dtype)
+
+        return tensor
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """
+        Applies device and dtype conversion to all tensors in an environment transition.
+
+        It iterates through the transition, finds all `torch.Tensor` objects (including those nested in
+        dictionaries like `observation`), and processes them.
+
+        Args:
+            transition: The input `EnvTransition` object.
+
+        Returns:
+            A new `EnvTransition` object with all tensors moved to the target device and dtype.
+        """
+        new_transition = transition.copy()
+        action = new_transition.get(TransitionKey.ACTION)
+
+        if action is not None and not isinstance(action, PolicyAction):
+            raise ValueError(f"If action is not None should be a PolicyAction type got {type(action)}")
+
+        simple_tensor_keys = [
+            TransitionKey.ACTION,
+            TransitionKey.REWARD,
+            TransitionKey.DONE,
+            TransitionKey.TRUNCATED,
+        ]
+
+        dict_tensor_keys = [
+            TransitionKey.OBSERVATION,
+            TransitionKey.COMPLEMENTARY_DATA,
+        ]
+
+        # Process simple, top-level tensors
+        for key in simple_tensor_keys:
+            value = transition.get(key)
+            if isinstance(value, torch.Tensor):
+                new_transition[key] = self._process_tensor(value)
+
+        # Process tensors nested within dictionaries
+        for key in dict_tensor_keys:
+            data_dict = transition.get(key)
+            if data_dict is not None:
+                new_data_dict = {
+                    k: self._process_tensor(v) if isinstance(v, torch.Tensor) else v
+                    for k, v in data_dict.items()
+                }
+                new_transition[key] = new_data_dict
+
+        return new_transition
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns the serializable configuration of the processor.
+
+        Returns:
+            A dictionary containing the device and float_dtype settings.
+        """
+        return {"device": self.device, "float_dtype": self.float_dtype}
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Returns the input features unchanged.
+
+        Device and dtype transformations do not alter the fundamental definition of the features (e.g., shape).
+
+        Args:
+            features: A dictionary of policy features.
+
+        Returns:
+            The original dictionary of policy features.
+        """
+        return features
diff --git a/lerobot/src/lerobot/processor/env_processor.py b/lerobot/src/lerobot/processor/env_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..a77e066cfed7ad417a76685934ff44df44203eb8
--- /dev/null
+++ b/lerobot/src/lerobot/processor/env_processor.py
@@ -0,0 +1,228 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from dataclasses import dataclass
+
+import torch
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from lerobot.utils.constants import OBS_IMAGES, OBS_PREFIX, OBS_STATE, OBS_STR
+
+from .pipeline import ObservationProcessorStep, ProcessorStepRegistry
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="libero_processor")
+class LiberoProcessorStep(ObservationProcessorStep):
+    """
+    Processes LIBERO observations into the LeRobot format.
+
+    This step handles the specific observation structure from LIBERO environments,
+    which includes nested robot_state dictionaries and image observations.
+
+    **State Processing:**
+    -   Processes the `robot_state` dictionary which contains nested end-effector,
+        gripper, and joint information.
+    -   Extracts and concatenates:
+        - End-effector position (3D)
+        - End-effector quaternion converted to axis-angle (3D)
+        - Gripper joint positions (2D)
+    -   Maps the concatenated state to `"observation.state"`.
+
+    **Image Processing:**
+    -   Rotates images by 180 degrees by flipping both height and width dimensions.
+    -   This accounts for the HuggingFaceVLA/libero camera orientation convention.
+    """
+
+    def _process_observation(self, observation):
+        """
+        Processes both image and robot_state observations from LIBERO.
+        """
+        processed_obs = observation.copy()
+        for key in list(processed_obs.keys()):
+            if key.startswith(f"{OBS_IMAGES}."):
+                img = processed_obs[key]
+
+                # Flip both H and W
+                img = torch.flip(img, dims=[2, 3])
+
+                processed_obs[key] = img
+        # Process robot_state into a flat state vector
+        observation_robot_state_str = OBS_PREFIX + "robot_state"
+        if observation_robot_state_str in processed_obs:
+            robot_state = processed_obs.pop(observation_robot_state_str)
+
+            # Extract components
+            eef_pos = robot_state["eef"]["pos"]  # (B, 3,)
+            eef_quat = robot_state["eef"]["quat"]  # (B, 4,)
+            gripper_qpos = robot_state["gripper"]["qpos"]  # (B, 2,)
+
+            # Convert quaternion to axis-angle
+            eef_axisangle = self._quat2axisangle(eef_quat)  # (B, 3)
+            # Concatenate into a single state vector
+            state = torch.cat((eef_pos, eef_axisangle, gripper_qpos), dim=-1)
+
+            # ensure float32
+            state = state.float()
+            if state.dim() == 1:
+                state = state.unsqueeze(0)
+
+            processed_obs[OBS_STATE] = state
+        return processed_obs
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Transforms feature keys from the LIBERO format to the LeRobot standard.
+        """
+        new_features: dict[PipelineFeatureType, dict[str, PolicyFeature]] = {}
+
+        # copy over non-STATE features
+        for ft, feats in features.items():
+            if ft != FeatureType.STATE:
+                new_features[ft] = feats.copy()
+
+        # rebuild STATE features
+        state_feats = {}
+
+        # add our new flattened state
+        state_feats[OBS_STATE] = PolicyFeature(
+            type=FeatureType.STATE,
+            shape=(8,),  # [eef_pos(3), axis_angle(3), gripper(2)]
+        )
+
+        new_features[FeatureType.STATE] = state_feats
+
+        return new_features
+
+    def observation(self, observation):
+        return self._process_observation(observation)
+
+    def _quat2axisangle(self, quat: torch.Tensor) -> torch.Tensor:
+        """
+        Convert batched quaternions to axis-angle format.
+        Only accepts torch tensors of shape (B, 4).
+
+        Args:
+            quat (Tensor): (B, 4) tensor of quaternions in (x, y, z, w) format
+
+        Returns:
+            Tensor: (B, 3) axis-angle vectors
+
+        Raises:
+            TypeError: if input is not a torch tensor
+            ValueError: if shape is not (B, 4)
+        """
+
+        if not isinstance(quat, torch.Tensor):
+            raise TypeError(f"_quat2axisangle expected a torch.Tensor, got {type(quat)}")
+
+        if quat.ndim != 2 or quat.shape[1] != 4:
+            raise ValueError(f"_quat2axisangle expected shape (B, 4), got {tuple(quat.shape)}")
+
+        quat = quat.to(dtype=torch.float32)
+        device = quat.device
+        batch_size = quat.shape[0]
+
+        w = quat[:, 3].clamp(-1.0, 1.0)
+
+        den = torch.sqrt(torch.clamp(1.0 - w * w, min=0.0))
+
+        result = torch.zeros((batch_size, 3), device=device)
+
+        mask = den > 1e-10
+
+        if mask.any():
+            angle = 2.0 * torch.acos(w[mask])  # (M,)
+            axis = quat[mask, :3] / den[mask].unsqueeze(1)
+            result[mask] = axis * angle.unsqueeze(1)
+
+        return result
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="isaaclab_arena_processor")
+class IsaaclabArenaProcessorStep(ObservationProcessorStep):
+    """
+    Processes IsaacLab Arena observations into LeRobot format.
+
+    **State Processing:**
+    - Extracts state components from obs["policy"] based on `state_keys`.
+    - Concatenates into a flat vector mapped to "observation.state".
+
+    **Image Processing:**
+    - Extracts images from obs["camera_obs"] based on `camera_keys`.
+    - Converts from (B, H, W, C) uint8 to (B, C, H, W) float32 [0, 1].
+    - Maps to "observation.images.<camera_name>".
+    """
+
+    # Configurable from IsaacLabEnv config / cli args: --env.state_keys="robot_joint_pos,left_eef_pos"
+    state_keys: tuple[str, ...]
+
+    # Configurable from IsaacLabEnv config / cli args: --env.camera_keys="robot_pov_cam_rgb"
+    camera_keys: tuple[str, ...]
+
+    def _process_observation(self, observation):
+        """
+        Processes both image and policy state observations from IsaacLab Arena.
+        """
+        processed_obs = {}
+
+        if f"{OBS_STR}.camera_obs" in observation:
+            camera_obs = observation[f"{OBS_STR}.camera_obs"]
+
+            for cam_name, img in camera_obs.items():
+                if cam_name not in self.camera_keys:
+                    continue
+
+                img = img.permute(0, 3, 1, 2).contiguous()
+                if img.dtype == torch.uint8:
+                    img = img.float() / 255.0
+                elif img.dtype != torch.float32:
+                    img = img.float()
+
+                processed_obs[f"{OBS_IMAGES}.{cam_name}"] = img
+
+        # Process policy state -> observation.state
+        if f"{OBS_STR}.policy" in observation:
+            policy_obs = observation[f"{OBS_STR}.policy"]
+
+            # Collect state components in order
+            state_components = []
+            for key in self.state_keys:
+                if key in policy_obs:
+                    component = policy_obs[key]
+                    # Flatten extra dims: (B, N, M) -> (B, N*M)
+                    if component.dim() > 2:
+                        batch_size = component.shape[0]
+                        component = component.view(batch_size, -1)
+                    state_components.append(component)
+
+            if state_components:
+                state = torch.cat(state_components, dim=-1)
+                state = state.float()
+                processed_obs[OBS_STATE] = state
+
+        return processed_obs
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """Not used for policy evaluation."""
+        return features
+
+    def observation(self, observation):
+        return self._process_observation(observation)
diff --git a/lerobot/src/lerobot/processor/factory.py b/lerobot/src/lerobot/processor/factory.py
new file mode 100644
index 0000000000000000000000000000000000000000..5028122f1f35acb4890bce755e1f03e3646570dd
--- /dev/null
+++ b/lerobot/src/lerobot/processor/factory.py
@@ -0,0 +1,63 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from lerobot.types import RobotAction, RobotObservation
+
+from .converters import (
+    observation_to_transition,
+    robot_action_observation_to_transition,
+    transition_to_observation,
+    transition_to_robot_action,
+)
+from .pipeline import IdentityProcessorStep, RobotProcessorPipeline
+
+
+def make_default_teleop_action_processor() -> RobotProcessorPipeline[
+    tuple[RobotAction, RobotObservation], RobotAction
+]:
+    teleop_action_processor = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        steps=[IdentityProcessorStep()],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+    return teleop_action_processor
+
+
+def make_default_robot_action_processor() -> RobotProcessorPipeline[
+    tuple[RobotAction, RobotObservation], RobotAction
+]:
+    robot_action_processor = RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction](
+        steps=[IdentityProcessorStep()],
+        to_transition=robot_action_observation_to_transition,
+        to_output=transition_to_robot_action,
+    )
+    return robot_action_processor
+
+
+def make_default_robot_observation_processor() -> RobotProcessorPipeline[RobotObservation, RobotObservation]:
+    robot_observation_processor = RobotProcessorPipeline[RobotObservation, RobotObservation](
+        steps=[IdentityProcessorStep()],
+        to_transition=observation_to_transition,
+        to_output=transition_to_observation,
+    )
+    return robot_observation_processor
+
+
+def make_default_processors():
+    teleop_action_processor = make_default_teleop_action_processor()
+    robot_action_processor = make_default_robot_action_processor()
+    robot_observation_processor = make_default_robot_observation_processor()
+    return (teleop_action_processor, robot_action_processor, robot_observation_processor)
diff --git a/lerobot/src/lerobot/processor/gym_action_processor.py b/lerobot/src/lerobot/processor/gym_action_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..e756ded7f741ffeae3799884ae6376ec29cba65e
--- /dev/null
+++ b/lerobot/src/lerobot/processor/gym_action_processor.py
@@ -0,0 +1,105 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.types import EnvAction, EnvTransition, PolicyAction
+
+from .converters import to_tensor
+from .hil_processor import TELEOP_ACTION_KEY
+from .pipeline import ActionProcessorStep, ProcessorStep, ProcessorStepRegistry
+
+
+@ProcessorStepRegistry.register("torch2numpy_action_processor")
+@dataclass
+class Torch2NumpyActionProcessorStep(ActionProcessorStep):
+    """
+    Converts a PyTorch tensor action to a NumPy array.
+
+    This step is useful when the output of a policy (typically a torch.Tensor)
+    needs to be passed to an environment or component that expects a NumPy array.
+
+    Attributes:
+        squeeze_batch_dim: If True, removes the first dimension of the array
+                           if it is of size 1. This is useful for converting a
+                           batched action of size (1, D) to a single action of size (D,).
+    """
+
+    squeeze_batch_dim: bool = True
+
+    def action(self, action: PolicyAction) -> EnvAction:
+        if not isinstance(action, PolicyAction):
+            raise TypeError(
+                f"Expected PolicyAction or None, got {type(action).__name__}. "
+                "Use appropriate processor for non-tensor actions."
+            )
+
+        numpy_action = action.detach().cpu().numpy()
+
+        # Remove batch dimensions but preserve action dimensions.
+        # Only squeeze if there's a batch dimension (first dim == 1).
+        if (
+            self.squeeze_batch_dim
+            and numpy_action.shape
+            and len(numpy_action.shape) > 1
+            and numpy_action.shape[0] == 1
+        ):
+            numpy_action = numpy_action.squeeze(0)
+
+        return numpy_action
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+@ProcessorStepRegistry.register("numpy2torch_action_processor")
+@dataclass
+class Numpy2TorchActionProcessorStep(ProcessorStep):
+    """Converts a NumPy array action to a PyTorch tensor when action is present."""
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Converts numpy action to torch tensor if action exists, otherwise passes through."""
+        from lerobot.types import TransitionKey
+
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        action = new_transition.get(TransitionKey.ACTION)
+        if action is not None:
+            if not isinstance(action, EnvAction):
+                raise TypeError(
+                    f"Expected np.ndarray or None, got {type(action).__name__}. "
+                    "Use appropriate processor for non-tensor actions."
+                )
+            torch_action = to_tensor(action, dtype=None)  # Preserve original dtype
+            new_transition[TransitionKey.ACTION] = torch_action
+
+        complementary_data = new_transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+        if TELEOP_ACTION_KEY in complementary_data:
+            teleop_action = complementary_data[TELEOP_ACTION_KEY]
+            if isinstance(teleop_action, EnvAction):
+                complementary_data[TELEOP_ACTION_KEY] = to_tensor(teleop_action)
+            new_transition[TransitionKey.COMPLEMENTARY_DATA] = complementary_data
+
+        return new_transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
diff --git a/lerobot/src/lerobot/processor/hil_processor.py b/lerobot/src/lerobot/processor/hil_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..0b8521c2b5d4155e6c6fca7036dccb1fdf8727fb
--- /dev/null
+++ b/lerobot/src/lerobot/processor/hil_processor.py
@@ -0,0 +1,632 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import math
+import time
+from dataclasses import dataclass
+from typing import TYPE_CHECKING, Any, Protocol, TypeVar, runtime_checkable
+
+import numpy as np
+import torch
+import torchvision.transforms.functional as F  # noqa: N812
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.teleoperators.utils import TeleopEvents
+
+if TYPE_CHECKING:
+    from lerobot.teleoperators.teleoperator import Teleoperator
+
+from lerobot.types import EnvTransition, PolicyAction, TransitionKey
+
+from .pipeline import (
+    ComplementaryDataProcessorStep,
+    InfoProcessorStep,
+    ObservationProcessorStep,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    TruncatedProcessorStep,
+)
+
+GRIPPER_KEY = "gripper"
+DISCRETE_PENALTY_KEY = "discrete_penalty"
+TELEOP_ACTION_KEY = "teleop_action"
+
+
+@runtime_checkable
+class HasTeleopEvents(Protocol):
+    """
+    Minimal protocol for objects that provide teleoperation events.
+
+    This protocol defines the `get_teleop_events()` method, allowing processor
+    steps to interact with teleoperators that support event-based controls
+    (like episode termination or success flagging) without needing to know the
+    teleoperator's specific class.
+    """
+
+    def get_teleop_events(self) -> dict[str, Any]:
+        """
+        Get extra control events from the teleoperator.
+
+        Returns:
+            A dictionary containing control events such as:
+            - `is_intervention`: bool - Whether the human is currently intervening.
+            - `terminate_episode`: bool - Whether to terminate the current episode.
+            - `success`: bool - Whether the episode was successful.
+            - `rerecord_episode`: bool - Whether to rerecord the episode.
+        """
+        ...
+
+
+# Type variable constrained to Teleoperator subclasses that also implement events
+TeleopWithEvents = TypeVar("TeleopWithEvents", bound="Teleoperator")
+
+
+def _check_teleop_with_events(teleop: "Teleoperator") -> None:
+    """
+    Runtime check that a teleoperator implements the `HasTeleopEvents` protocol.
+
+    Args:
+        teleop: The teleoperator instance to check.
+
+    Raises:
+        TypeError: If the teleoperator does not have a `get_teleop_events` method.
+    """
+    if not isinstance(teleop, HasTeleopEvents):
+        raise TypeError(
+            f"Teleoperator {type(teleop).__name__} must implement get_teleop_events() method. "
+            f"Compatible teleoperators: GamepadTeleop, KeyboardEndEffectorTeleop"
+        )
+
+
+@ProcessorStepRegistry.register("add_teleop_action_as_complementary_data")
+@dataclass
+class AddTeleopActionAsComplimentaryDataStep(ComplementaryDataProcessorStep):
+    """
+    Adds the raw action from a teleoperator to the transition's complementary data.
+
+    This is useful for human-in-the-loop scenarios where the human's input needs to
+    be available to downstream processors, for example, to override a policy's action
+    during an intervention.
+
+    Attributes:
+        teleop_device: The teleoperator instance to get the action from.
+    """
+
+    teleop_device: "Teleoperator"
+
+    def complementary_data(self, complementary_data: dict) -> dict:
+        """
+        Retrieves the teleoperator's action and adds it to the complementary data.
+
+        Args:
+            complementary_data: The incoming complementary data dictionary.
+
+        Returns:
+            A new dictionary with the teleoperator action added under the
+            `teleop_action` key.
+        """
+        new_complementary_data = dict(complementary_data)
+        new_complementary_data[TELEOP_ACTION_KEY] = self.teleop_device.get_action()
+        return new_complementary_data
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+@ProcessorStepRegistry.register("add_teleop_action_as_info")
+@dataclass
+class AddTeleopEventsAsInfoStep(InfoProcessorStep):
+    """
+    Adds teleoperator control events (e.g., terminate, success) to the transition's info.
+
+    This step extracts control events from teleoperators that support event-based
+    interaction, making these signals available to other parts of the system.
+
+    Attributes:
+        teleop_device: An instance of a teleoperator that implements the
+                       `HasTeleopEvents` protocol.
+    """
+
+    teleop_device: TeleopWithEvents
+
+    def __post_init__(self):
+        """Validates that the provided teleoperator supports events after initialization."""
+        _check_teleop_with_events(self.teleop_device)
+
+    def info(self, info: dict) -> dict:
+        """
+        Retrieves teleoperator events and updates the info dictionary.
+
+        Args:
+            info: The incoming info dictionary.
+
+        Returns:
+            A new dictionary including the teleoperator events.
+        """
+        new_info = dict(info)
+
+        teleop_events = self.teleop_device.get_teleop_events()
+        new_info.update(teleop_events)
+        return new_info
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+@ProcessorStepRegistry.register("image_crop_resize_processor")
+@dataclass
+class ImageCropResizeProcessorStep(ObservationProcessorStep):
+    """
+    Crops and/or resizes image observations.
+
+    This step iterates through all image keys in an observation dictionary and applies
+    the specified transformations. It handles device placement, moving tensors to the
+    CPU if necessary for operations not supported on certain accelerators like MPS.
+
+    Attributes:
+        crop_params_dict: A dictionary mapping image keys to cropping parameters
+                          (top, left, height, width).
+        resize_size: A tuple (height, width) to resize all images to.
+    """
+
+    crop_params_dict: dict[str, tuple[int, int, int, int]] | None = None
+    resize_size: tuple[int, int] | None = None
+
+    def observation(self, observation: dict) -> dict:
+        """
+        Applies cropping and resizing to all images in the observation dictionary.
+
+        Args:
+            observation: The observation dictionary, potentially containing image tensors.
+
+        Returns:
+            A new observation dictionary with transformed images.
+        """
+        if self.resize_size is None and not self.crop_params_dict:
+            return observation
+
+        new_observation = dict(observation)
+
+        # Process all image keys in the observation
+        for key in observation:
+            if "image" not in key:
+                continue
+
+            image = observation[key]
+            device = image.device
+            # NOTE (maractingi): No mps kernel for crop and resize, so we need to move to cpu
+            if device.type == "mps":
+                image = image.cpu()
+            # Crop if crop params are provided for this key
+            if self.crop_params_dict is not None and key in self.crop_params_dict:
+                crop_params = self.crop_params_dict[key]
+                image = F.crop(image, *crop_params)
+            if self.resize_size is not None:
+                image = F.resize(image, self.resize_size)
+                image = image.clamp(0.0, 1.0)
+            new_observation[key] = image.to(device)
+
+        return new_observation
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns the configuration of the step for serialization.
+
+        Returns:
+            A dictionary with the crop parameters and resize dimensions.
+        """
+        return {
+            "crop_params_dict": self.crop_params_dict,
+            "resize_size": self.resize_size,
+        }
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Updates the image feature shapes in the policy features dictionary if resizing is applied.
+
+        Args:
+            features: The policy features dictionary.
+
+        Returns:
+            The updated policy features dictionary with new image shapes.
+        """
+        if self.resize_size is None:
+            return features
+        for key in features[PipelineFeatureType.OBSERVATION]:
+            if "image" in key:
+                nb_channel = features[PipelineFeatureType.OBSERVATION][key].shape[0]
+                features[PipelineFeatureType.OBSERVATION][key] = PolicyFeature(
+                    type=features[PipelineFeatureType.OBSERVATION][key].type,
+                    shape=(nb_channel, *self.resize_size),
+                )
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register("time_limit_processor")
+class TimeLimitProcessorStep(TruncatedProcessorStep):
+    """
+    Tracks episode steps and enforces a time limit by truncating the episode.
+
+    Attributes:
+        max_episode_steps: The maximum number of steps allowed per episode.
+        current_step: The current step count for the active episode.
+    """
+
+    max_episode_steps: int
+    current_step: int = 0
+
+    def truncated(self, truncated: bool) -> bool:
+        """
+        Increments the step counter and sets the truncated flag if the time limit is reached.
+
+        Args:
+            truncated: The incoming truncated flag.
+
+        Returns:
+            True if the episode step limit is reached, otherwise the incoming value.
+        """
+        self.current_step += 1
+        if self.current_step >= self.max_episode_steps:
+            truncated = True
+        # TODO (steven): missing an else truncated = False?
+        return truncated
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns the configuration of the step for serialization.
+
+        Returns:
+            A dictionary containing the `max_episode_steps`.
+        """
+        return {
+            "max_episode_steps": self.max_episode_steps,
+        }
+
+    def reset(self) -> None:
+        """Resets the step counter, typically called at the start of a new episode."""
+        self.current_step = 0
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+@ProcessorStepRegistry.register("gym_hil_adapter_processor")
+class GymHILAdapterProcessorStep(ProcessorStep):
+    """
+    Adapts the output of the `gym-hil` environment to the format expected by `lerobot` processors.
+
+    This step normalizes the `transition` object by:
+    1. Copying `teleop_action` from `info` to `complementary_data`.
+    2. Copying `is_intervention` from `info` (using the string key) to `info` (using the enum key).
+    """
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        info = transition.get(TransitionKey.INFO, {})
+        complementary_data = transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+
+        if TELEOP_ACTION_KEY in info:
+            complementary_data[TELEOP_ACTION_KEY] = info[TELEOP_ACTION_KEY]
+
+        if "is_intervention" in info:
+            info[TeleopEvents.IS_INTERVENTION] = info["is_intervention"]
+
+        transition[TransitionKey.INFO] = info
+        transition[TransitionKey.COMPLEMENTARY_DATA] = complementary_data
+
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register("gripper_penalty_processor")
+class GripperPenaltyProcessorStep(ProcessorStep):
+    """
+    Applies a penalty for inefficient gripper usage.
+
+    This step penalizes actions that attempt to close an already closed gripper or
+    open an already open one, based on position thresholds.
+
+    Attributes:
+        penalty: The negative reward value to apply.
+        max_gripper_pos: The maximum position value for the gripper, used for normalization.
+    """
+
+    penalty: float = -0.01
+    max_gripper_pos: float = 30.0
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """
+        Calculates the gripper penalty and adds it to the complementary data.
+
+        Args:
+            transition: The incoming environment transition.
+
+        Returns:
+            The modified transition with the penalty added to complementary data.
+        """
+        new_transition = transition.copy()
+        action = new_transition.get(TransitionKey.ACTION)
+        complementary_data = new_transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+
+        raw_joint_positions = complementary_data.get("raw_joint_positions")
+        if raw_joint_positions is None:
+            return new_transition
+
+        current_gripper_pos = raw_joint_positions.get(GRIPPER_KEY, None)
+        if current_gripper_pos is None:
+            return new_transition
+
+        # Gripper action is a PolicyAction at this stage
+        gripper_action = action[-1].item()
+        gripper_action_normalized = gripper_action / self.max_gripper_pos
+
+        # Normalize gripper state and action
+        gripper_state_normalized = current_gripper_pos / self.max_gripper_pos
+
+        # Calculate penalty boolean as in original
+        gripper_penalty_bool = (gripper_state_normalized < 0.5 and gripper_action_normalized > 0.5) or (
+            gripper_state_normalized > 0.75 and gripper_action_normalized < 0.5
+        )
+
+        gripper_penalty = self.penalty * int(gripper_penalty_bool)
+
+        # Update complementary data with penalty info
+        new_complementary_data = dict(complementary_data)
+        new_complementary_data[DISCRETE_PENALTY_KEY] = gripper_penalty
+        new_transition[TransitionKey.COMPLEMENTARY_DATA] = new_complementary_data
+
+        return new_transition
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns the configuration of the step for serialization.
+
+        Returns:
+            A dictionary containing the penalty value and max gripper position.
+        """
+        return {
+            "penalty": self.penalty,
+            "max_gripper_pos": self.max_gripper_pos,
+        }
+
+    def reset(self) -> None:
+        """Resets the processor's internal state."""
+        pass
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register("intervention_action_processor")
+class InterventionActionProcessorStep(ProcessorStep):
+    """
+    Handles human intervention, overriding policy actions and managing episode termination.
+
+    When an intervention is detected (via teleoperator events in the `info` dict),
+    this step replaces the policy's action with the human's teleoperated action.
+    It also processes signals to terminate the episode or flag success.
+
+    Attributes:
+        use_gripper: Whether to include the gripper in the teleoperated action.
+        terminate_on_success: If True, automatically sets the `done` flag when a
+                              `success` event is received.
+    """
+
+    use_gripper: bool = False
+    terminate_on_success: bool = True
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """
+        Processes the transition to handle interventions.
+
+        Args:
+            transition: The incoming environment transition.
+
+        Returns:
+            The modified transition, potentially with an overridden action, updated
+            reward, and termination status.
+        """
+        action = transition.get(TransitionKey.ACTION)
+        if not isinstance(action, PolicyAction):
+            raise ValueError(f"Action should be a PolicyAction type got {type(action)}")
+
+        # Get intervention signals from complementary data
+        info = transition.get(TransitionKey.INFO, {})
+        complementary_data = transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+        teleop_action = complementary_data.get(TELEOP_ACTION_KEY, {})
+        is_intervention = info.get(TeleopEvents.IS_INTERVENTION, False)
+        terminate_episode = info.get(TeleopEvents.TERMINATE_EPISODE, False)
+        success = info.get(TeleopEvents.SUCCESS, False)
+        rerecord_episode = info.get(TeleopEvents.RERECORD_EPISODE, False)
+
+        new_transition = transition.copy()
+
+        # Override action if intervention is active
+        if is_intervention and teleop_action is not None:
+            if isinstance(teleop_action, dict):
+                # Convert teleop_action dict to tensor format
+                action_list = [
+                    teleop_action.get("delta_x", 0.0),
+                    teleop_action.get("delta_y", 0.0),
+                    teleop_action.get("delta_z", 0.0),
+                ]
+                if self.use_gripper:
+                    action_list.append(teleop_action.get(GRIPPER_KEY, 1.0))
+            elif isinstance(teleop_action, np.ndarray):
+                action_list = teleop_action.tolist()
+            else:
+                action_list = teleop_action
+
+            teleop_action_tensor = torch.tensor(action_list, dtype=action.dtype, device=action.device)
+            new_transition[TransitionKey.ACTION] = teleop_action_tensor
+
+        # Handle episode termination
+        new_transition[TransitionKey.DONE] = bool(terminate_episode) or (
+            self.terminate_on_success and success
+        )
+        new_transition[TransitionKey.REWARD] = float(success)
+
+        # Update info with intervention metadata
+        info = new_transition.get(TransitionKey.INFO, {})
+        info[TeleopEvents.IS_INTERVENTION] = is_intervention
+        info[TeleopEvents.RERECORD_EPISODE] = rerecord_episode
+        info[TeleopEvents.SUCCESS] = success
+        new_transition[TransitionKey.INFO] = info
+
+        # Update complementary data with teleop action
+        complementary_data = new_transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+        complementary_data[TELEOP_ACTION_KEY] = new_transition.get(TransitionKey.ACTION)
+        new_transition[TransitionKey.COMPLEMENTARY_DATA] = complementary_data
+
+        return new_transition
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns the configuration of the step for serialization.
+
+        Returns:
+            A dictionary containing the step's configuration attributes.
+        """
+        return {
+            "use_gripper": self.use_gripper,
+            "terminate_on_success": self.terminate_on_success,
+        }
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register("reward_classifier_processor")
+class RewardClassifierProcessorStep(ProcessorStep):
+    """
+    Applies a pretrained reward classifier to image observations to predict success.
+
+    This step uses a model to determine if the current state is successful, updating
+    the reward and potentially terminating the episode.
+
+    Attributes:
+        pretrained_path: Path to the pretrained reward classifier model.
+        device: The device to run the classifier on.
+        success_threshold: The probability threshold to consider a prediction as successful.
+        success_reward: The reward value to assign on success.
+        terminate_on_success: If True, terminates the episode upon successful classification.
+        reward_classifier: The loaded classifier model instance.
+    """
+
+    pretrained_path: str | None = None
+    device: str = "cpu"
+    success_threshold: float = 0.5
+    success_reward: float = 1.0
+    terminate_on_success: bool = True
+
+    reward_classifier: Any = None
+
+    def __post_init__(self):
+        """Initializes the reward classifier model after the dataclass is created."""
+        if self.pretrained_path is not None:
+            from lerobot.policies.sac.reward_model.modeling_classifier import Classifier
+
+            self.reward_classifier = Classifier.from_pretrained(self.pretrained_path)
+            self.reward_classifier.to(self.device)
+            self.reward_classifier.eval()
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """
+        Processes a transition, applying the reward classifier to its image observations.
+
+        Args:
+            transition: The incoming environment transition.
+
+        Returns:
+            The modified transition with an updated reward and done flag based on the
+            classifier's prediction.
+        """
+        new_transition = transition.copy()
+        observation = new_transition.get(TransitionKey.OBSERVATION)
+        if observation is None or self.reward_classifier is None:
+            return new_transition
+
+        # Extract images from observation
+        images = {key: value for key, value in observation.items() if "image" in key}
+
+        if not images:
+            return new_transition
+
+        # Run reward classifier
+        start_time = time.perf_counter()
+        with torch.inference_mode():
+            success = self.reward_classifier.predict_reward(images, threshold=self.success_threshold)
+
+        classifier_frequency = 1 / (time.perf_counter() - start_time)
+
+        # Calculate reward and termination
+        reward = new_transition.get(TransitionKey.REWARD, 0.0)
+        terminated = new_transition.get(TransitionKey.DONE, False)
+
+        if math.isclose(success, 1, abs_tol=1e-2):
+            reward = self.success_reward
+            if self.terminate_on_success:
+                terminated = True
+
+        # Update transition
+        new_transition[TransitionKey.REWARD] = reward
+        new_transition[TransitionKey.DONE] = terminated
+
+        # Update info with classifier frequency
+        info = new_transition.get(TransitionKey.INFO, {})
+        info["reward_classifier_frequency"] = classifier_frequency
+        new_transition[TransitionKey.INFO] = info
+
+        return new_transition
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns the configuration of the step for serialization.
+
+        Returns:
+            A dictionary containing the step's configuration attributes.
+        """
+        return {
+            "device": self.device,
+            "success_threshold": self.success_threshold,
+            "success_reward": self.success_reward,
+            "terminate_on_success": self.terminate_on_success,
+        }
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
diff --git a/lerobot/src/lerobot/processor/migrate_policy_normalization.py b/lerobot/src/lerobot/processor/migrate_policy_normalization.py
new file mode 100644
index 0000000000000000000000000000000000000000..525b7431c94c5b4a090342e5d8b70ab27c3c7435
--- /dev/null
+++ b/lerobot/src/lerobot/processor/migrate_policy_normalization.py
@@ -0,0 +1,769 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+A generic script to migrate LeRobot policies with built-in normalization layers to the new
+pipeline-based processor system.
+
+This script performs the following steps:
+1.  Loads a pretrained policy model and its configuration from a local path or the
+    Hugging Face Hub.
+2.  Scans the model's state dictionary to extract normalization statistics (e.g., mean,
+    std, min, max) for all features.
+3.  Creates two new processor pipelines:
+    - A preprocessor that normalizes inputs (observations) and outputs (actions).
+    - A postprocessor that unnormalizes outputs (actions) for inference.
+4.  Removes the original normalization layers from the model's state dictionary,
+    creating a "clean" model.
+5.  Saves the new clean model, the preprocessor, the postprocessor, and a generated
+    model card to a new directory.
+6.  Optionally pushes all the new artifacts to the Hugging Face Hub.
+
+Usage:
+    python src/lerobot/processor/migrate_policy_normalization.py \
+        --pretrained-path lerobot/act_aloha_sim_transfer_cube_human \
+        --push-to-hub \
+        --branch main
+
+Note: This script now uses the modern `make_pre_post_processors` and `make_policy_config`
+factory functions from `lerobot.policies.factory` to create processors and configurations,
+ensuring consistency with the current codebase.
+
+The script extracts normalization statistics from the old model's state_dict, creates clean
+processor pipelines using the factory functions, and saves a migrated model that is compatible
+with the new PolicyProcessorPipeline architecture.
+"""
+
+import argparse
+import json
+import os
+from pathlib import Path
+from typing import Any
+
+import torch
+from huggingface_hub import HfApi, hf_hub_download
+from safetensors.torch import load_file as load_safetensors
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.policies.factory import get_policy_class, make_policy_config, make_pre_post_processors
+from lerobot.utils.constants import ACTION
+
+
+def extract_normalization_stats(state_dict: dict[str, torch.Tensor]) -> dict[str, dict[str, torch.Tensor]]:
+    """
+    Scans a model's state_dict to find and extract normalization statistics.
+
+    This function identifies keys corresponding to normalization layers (e.g., those
+    for mean, std, min, max) based on a set of predefined patterns and organizes
+    them into a nested dictionary.
+
+    Args:
+        state_dict: The state dictionary of a pretrained policy model.
+
+    Returns:
+        A nested dictionary where outer keys are feature names (e.g.,
+        'observation.state') and inner keys are statistic types ('mean', 'std'),
+        mapping to their corresponding tensor values.
+    """
+    stats = {}
+
+    # Define patterns to match and their prefixes to remove
+    normalization_patterns = [
+        "normalize_inputs.buffer_",
+        "unnormalize_outputs.buffer_",
+        "normalize_targets.buffer_",
+        "normalize.",  # Must come after normalize_* patterns
+        "unnormalize.",  # Must come after unnormalize_* patterns
+        "input_normalizer.",
+        "output_normalizer.",
+        "normalalize_inputs.",
+        "unnormalize_outputs.",
+        "normalize_targets.",
+        "unnormalize_targets.",
+    ]
+
+    # Process each key in state_dict
+    for key, tensor in state_dict.items():
+        # Try each pattern
+        for pattern in normalization_patterns:
+            if key.startswith(pattern):
+                # Extract the remaining part after the pattern
+                remaining = key[len(pattern) :]
+                parts = remaining.split(".")
+
+                # Need at least feature name and stat type
+                if len(parts) >= 2:
+                    # Last part is the stat type (mean, std, min, max, etc.)
+                    stat_type = parts[-1]
+                    # Everything else is the feature name
+                    feature_name = ".".join(parts[:-1]).replace("_", ".")
+
+                    # Add to stats
+                    if feature_name not in stats:
+                        stats[feature_name] = {}
+                    stats[feature_name][stat_type] = tensor.clone()
+
+                # Only process the first matching pattern
+                break
+
+    return stats
+
+
+def detect_features_and_norm_modes(
+    config: dict[str, Any], stats: dict[str, dict[str, torch.Tensor]]
+) -> tuple[dict[str, PolicyFeature], dict[FeatureType, NormalizationMode]]:
+    """
+    Infers policy features and normalization modes from the model config and stats.
+
+    This function first attempts to find feature definitions and normalization
+    mappings directly from the policy's configuration file. If this information is
+    not present, it infers it from the extracted normalization statistics, using
+    tensor shapes to determine feature shapes and the presence of specific stat
+    keys (e.g., 'mean'/'std' vs 'min'/'max') to determine the normalization mode.
+    It applies sensible defaults if inference is not possible.
+
+    Args:
+        config: The policy's configuration dictionary from `config.json`.
+        stats: The normalization statistics extracted from the model's state_dict.
+
+    Returns:
+        A tuple containing:
+        - A dictionary mapping feature names to `PolicyFeature` objects.
+        - A dictionary mapping `FeatureType` enums to `NormalizationMode` enums.
+    """
+    features = {}
+    norm_modes = {}
+
+    # First, check if there's a normalization_mapping in the config
+    if "normalization_mapping" in config:
+        print(f"Found normalization_mapping in config: {config['normalization_mapping']}")
+        # Extract normalization modes from config
+        for feature_type_str, mode_str in config["normalization_mapping"].items():
+            # Convert string to FeatureType enum
+            try:
+                if feature_type_str == "VISUAL":
+                    feature_type = FeatureType.VISUAL
+                elif feature_type_str == "STATE":
+                    feature_type = FeatureType.STATE
+                elif feature_type_str == "ACTION":
+                    feature_type = FeatureType.ACTION
+                else:
+                    print(f"Warning: Unknown feature type '{feature_type_str}', skipping")
+                    continue
+            except (AttributeError, ValueError):
+                print(f"Warning: Could not parse feature type '{feature_type_str}', skipping")
+                continue
+
+            # Convert string to NormalizationMode enum
+            try:
+                if mode_str == "MEAN_STD":
+                    mode = NormalizationMode.MEAN_STD
+                elif mode_str == "MIN_MAX":
+                    mode = NormalizationMode.MIN_MAX
+                elif mode_str == "IDENTITY":
+                    mode = NormalizationMode.IDENTITY
+                else:
+                    print(
+                        f"Warning: Unknown normalization mode '{mode_str}' for feature type '{feature_type_str}'"
+                    )
+                    continue
+            except (AttributeError, ValueError):
+                print(f"Warning: Could not parse normalization mode '{mode_str}', skipping")
+                continue
+
+            norm_modes[feature_type] = mode
+
+    # Try to extract from config
+    if "features" in config:
+        for key, feature_config in config["features"].items():
+            shape = feature_config.get("shape", feature_config.get("dim"))
+            shape = (shape,) if isinstance(shape, int) else tuple(shape)
+
+            # Determine feature type
+            if "image" in key or "visual" in key:
+                feature_type = FeatureType.VISUAL
+            elif "state" in key:
+                feature_type = FeatureType.STATE
+            elif ACTION in key:
+                feature_type = FeatureType.ACTION
+            else:
+                feature_type = FeatureType.STATE  # Default
+
+            features[key] = PolicyFeature(feature_type, shape)
+
+    # If no features in config, infer from stats
+    if not features:
+        for key, stat_dict in stats.items():
+            # Get shape from any stat tensor
+            tensor = next(iter(stat_dict.values()))
+            shape = tuple(tensor.shape)
+
+            # Determine feature type based on key
+            if "image" in key or "visual" in key or "pixels" in key:
+                feature_type = FeatureType.VISUAL
+            elif "state" in key or "joint" in key or "position" in key:
+                feature_type = FeatureType.STATE
+            elif ACTION in key:
+                feature_type = FeatureType.ACTION
+            else:
+                feature_type = FeatureType.STATE
+
+            features[key] = PolicyFeature(feature_type, shape)
+
+    # If normalization modes weren't in config, determine based on available stats
+    if not norm_modes:
+        for key, stat_dict in stats.items():
+            if key in features:
+                if "mean" in stat_dict and "std" in stat_dict:
+                    feature_type = features[key].type
+                    if feature_type not in norm_modes:
+                        norm_modes[feature_type] = NormalizationMode.MEAN_STD
+                elif "min" in stat_dict and "max" in stat_dict:
+                    feature_type = features[key].type
+                    if feature_type not in norm_modes:
+                        norm_modes[feature_type] = NormalizationMode.MIN_MAX
+
+    # Default normalization modes if not detected
+    if FeatureType.VISUAL not in norm_modes:
+        norm_modes[FeatureType.VISUAL] = NormalizationMode.MEAN_STD
+    if FeatureType.STATE not in norm_modes:
+        norm_modes[FeatureType.STATE] = NormalizationMode.MIN_MAX
+    if FeatureType.ACTION not in norm_modes:
+        norm_modes[FeatureType.ACTION] = NormalizationMode.MEAN_STD
+
+    return features, norm_modes
+
+
+def remove_normalization_layers(state_dict: dict[str, torch.Tensor]) -> dict[str, torch.Tensor]:
+    """
+    Creates a new state_dict with all normalization-related layers removed.
+
+    This function filters the original state dictionary, excluding any keys that
+    match a set of predefined patterns associated with normalization modules.
+
+    Args:
+        state_dict: The original model state dictionary.
+
+    Returns:
+        A new state dictionary containing only the core model weights, without
+        any normalization parameters.
+    """
+    new_state_dict = {}
+
+    # Patterns to remove
+    remove_patterns = [
+        "normalize_inputs.",
+        "unnormalize_outputs.",
+        "normalize_targets.",  # Added pattern for target normalization
+        "normalize.",
+        "unnormalize.",
+        "input_normalizer.",
+        "output_normalizer.",
+        "normalizer.",
+    ]
+
+    for key, tensor in state_dict.items():
+        should_remove = any(pattern in key for pattern in remove_patterns)
+        if not should_remove:
+            new_state_dict[key] = tensor
+
+    return new_state_dict
+
+
+def clean_state_dict(
+    state_dict: dict[str, torch.Tensor], remove_str: str = "._orig_mod"
+) -> dict[str, torch.Tensor]:
+    """
+    Remove a substring (e.g. '._orig_mod') from all keys in a state dict.
+
+    Args:
+        state_dict (dict): The original state dict.
+        remove_str (str): The substring to remove from the keys.
+
+    Returns:
+        dict: A new state dict with cleaned keys.
+    """
+    new_state_dict = {}
+    for k, v in state_dict.items():
+        new_k = k.replace(remove_str, "")
+        new_state_dict[new_k] = v
+    return new_state_dict
+
+
+def load_state_dict_with_missing_key_handling(
+    policy: torch.nn.Module,
+    state_dict: dict[str, torch.Tensor],
+    policy_type: str,
+    known_missing_keys_whitelist: dict[str, list[str]],
+) -> list[str]:
+    """
+    Load state dict into policy with graceful handling of missing keys.
+
+    This function loads the state dict with strict=False, filters out whitelisted
+    missing keys, and provides detailed reporting about any issues found.
+
+    Args:
+        policy: The policy model to load the state dict into.
+        state_dict: The cleaned state dictionary to load.
+        policy_type: The type of policy (used for whitelist lookup).
+        known_missing_keys_whitelist: Dictionary mapping policy types to lists of
+                                     known acceptable missing keys.
+
+    Returns:
+        List of problematic missing keys that weren't in the whitelist.
+    """
+    # Load the cleaned state dict with strict=False to capture missing/unexpected keys
+    load_result = policy.load_state_dict(state_dict, strict=False)
+
+    # Check for missing keys
+    missing_keys = load_result.missing_keys
+    unexpected_keys = load_result.unexpected_keys
+
+    # Filter out whitelisted missing keys
+    policy_type_lower = policy_type.lower()
+    whitelisted_keys = known_missing_keys_whitelist.get(policy_type_lower, [])
+    problematic_missing_keys = [key for key in missing_keys if key not in whitelisted_keys]
+
+    if missing_keys:
+        if problematic_missing_keys:
+            print(f"WARNING: Found {len(problematic_missing_keys)} unexpected missing keys:")
+            for key in problematic_missing_keys:
+                print(f"   - {key}")
+
+        if len(missing_keys) > len(problematic_missing_keys):
+            whitelisted_missing = [key for key in missing_keys if key in whitelisted_keys]
+            print(f"INFO: Found {len(whitelisted_missing)} expected missing keys (whitelisted):")
+            for key in whitelisted_missing:
+                print(f"   - {key}")
+
+    if unexpected_keys:
+        print(f"WARNING: Found {len(unexpected_keys)} unexpected keys:")
+        for key in unexpected_keys:
+            print(f"   - {key}")
+
+    if not missing_keys and not unexpected_keys:
+        print("Successfully loaded cleaned state dict into policy model (all keys matched)")
+    else:
+        print("State dict loaded with some missing/unexpected keys (see details above)")
+
+    return problematic_missing_keys
+
+
+def convert_features_to_policy_features(features_dict: dict[str, dict]) -> dict[str, PolicyFeature]:
+    """
+    Converts a feature dictionary from the old config format to the new `PolicyFeature` format.
+
+    Args:
+        features_dict: The feature dictionary in the old format, where values are
+                       simple dictionaries (e.g., `{"shape": [7]}`).
+
+    Returns:
+        A dictionary mapping feature names to `PolicyFeature` dataclass objects.
+    """
+    converted_features = {}
+
+    for key, feature_dict in features_dict.items():
+        # Determine feature type based on key
+        if "image" in key or "visual" in key:
+            feature_type = FeatureType.VISUAL
+        elif "state" in key:
+            feature_type = FeatureType.STATE
+        elif ACTION in key:
+            feature_type = FeatureType.ACTION
+        else:
+            feature_type = FeatureType.STATE
+
+        # Get shape from feature dict
+        shape = feature_dict.get("shape", feature_dict.get("dim"))
+        shape = (shape,) if isinstance(shape, int) else tuple(shape) if shape is not None else ()
+
+        converted_features[key] = PolicyFeature(feature_type, shape)
+
+    return converted_features
+
+
+def display_migration_summary_with_warnings(problematic_missing_keys: list[str]) -> None:
+    """
+    Display final migration summary with warnings about problematic missing keys.
+
+    Args:
+        problematic_missing_keys: List of missing keys that weren't in the whitelist.
+    """
+    if not problematic_missing_keys:
+        return
+
+    print("\n" + "=" * 60)
+    print("IMPORTANT: MIGRATION COMPLETED WITH WARNINGS")
+    print("=" * 60)
+    print(
+        f"The migration was successful, but {len(problematic_missing_keys)} unexpected missing keys were found:"
+    )
+    print()
+    for key in problematic_missing_keys:
+        print(f"   - {key}")
+    print()
+    print("These missing keys may indicate:")
+    print("  • The model architecture has changed")
+    print("  • Some components were not properly saved in the original model")
+    print("  • The migration script needs to be updated for this policy type")
+    print()
+    print("What to do next:")
+    print("  1. Test your migrated model carefully to ensure it works as expected")
+    print("  2. If you encounter issues, please open an issue at:")
+    print("     https://github.com/huggingface/lerobot/issues")
+    print("  3. Include this migration log and the missing keys listed above")
+    print()
+    print("If the model works correctly despite these warnings, the missing keys")
+    print("might be expected for your policy type and can be added to the whitelist.")
+    print("=" * 60)
+
+
+def load_model_from_hub(
+    repo_id: str, revision: str | None = None
+) -> tuple[dict[str, torch.Tensor], dict[str, Any], dict[str, Any] | None]:
+    """
+    Downloads and loads a model's state_dict and configs from the Hugging Face Hub.
+
+    Args:
+        repo_id: The repository ID on the Hub (e.g., 'lerobot/aloha').
+        revision: The specific git revision (branch, tag, or commit hash) to use.
+
+    Returns:
+        A tuple containing the model's state dictionary, the policy configuration,
+        and the training configuration (None if train_config.json is not found).
+    """
+    # Download files.
+    safetensors_path = hf_hub_download(repo_id=repo_id, filename="model.safetensors", revision=revision)
+
+    config_path = hf_hub_download(repo_id=repo_id, filename="config.json", revision=revision)
+
+    # Load state_dict
+    state_dict = load_safetensors(safetensors_path)
+
+    # Load config
+    with open(config_path) as f:
+        config = json.load(f)
+
+    # Try to load train_config (optional)
+    train_config = None
+    try:
+        train_config_path = hf_hub_download(repo_id=repo_id, filename="train_config.json", revision=revision)
+        with open(train_config_path) as f:
+            train_config = json.load(f)
+    except FileNotFoundError:
+        print("train_config.json not found - continuing without training configuration")
+
+    return state_dict, config, train_config
+
+
+def main():
+    parser = argparse.ArgumentParser(
+        description="Migrate policy models with normalization layers to new pipeline system"
+    )
+    parser.add_argument(
+        "--pretrained-path",
+        type=str,
+        required=True,
+        help="Path to pretrained model (hub repo or local directory)",
+    )
+    parser.add_argument(
+        "--output-dir",
+        type=str,
+        default=None,
+        help="Output directory for migrated model (default: same as pretrained-path)",
+    )
+    parser.add_argument("--push-to-hub", action="store_true", help="Push migrated model to hub")
+    parser.add_argument(
+        "--hub-repo-id",
+        type=str,
+        default=None,
+        help="Hub repository ID for pushing (default: same as pretrained-path)",
+    )
+    parser.add_argument("--revision", type=str, default=None, help="Revision of the model to load")
+    parser.add_argument("--private", action="store_true", help="Make the hub repository private")
+    parser.add_argument(
+        "--branch",
+        type=str,
+        default=None,
+        help="Git branch to use when pushing to hub. If specified, a PR will be created automatically (default: push directly to main)",
+    )
+
+    args = parser.parse_args()
+
+    # Load model and config
+    print(f"Loading model from {args.pretrained_path}...")
+    if os.path.isdir(args.pretrained_path):
+        # Local directory
+        state_dict = load_safetensors(os.path.join(args.pretrained_path, "model.safetensors"))
+        with open(os.path.join(args.pretrained_path, "config.json")) as f:
+            config = json.load(f)
+
+        # Try to load train_config (optional)
+        train_config = None
+        train_config_path = os.path.join(args.pretrained_path, "train_config.json")
+        if os.path.exists(train_config_path):
+            with open(train_config_path) as f:
+                train_config = json.load(f)
+        else:
+            print("train_config.json not found - continuing without training configuration")
+    else:
+        # Hub repository
+        state_dict, config, train_config = load_model_from_hub(args.pretrained_path, args.revision)
+
+    # Extract normalization statistics
+    print("Extracting normalization statistics...")
+    stats = extract_normalization_stats(state_dict)
+
+    print(f"Found normalization statistics for: {list(stats.keys())}")
+
+    # Detect input features and normalization modes
+    print("Detecting features and normalization modes...")
+    features, norm_map = detect_features_and_norm_modes(config, stats)
+
+    print(f"Detected features: {list(features.keys())}")
+    print(f"Normalization modes: {norm_map}")
+
+    # Remove normalization layers from state_dict
+    print("Removing normalization layers from model...")
+    new_state_dict = remove_normalization_layers(state_dict)
+    new_state_dict = clean_state_dict(new_state_dict, remove_str="._orig_mod")
+
+    removed_keys = set(state_dict.keys()) - set(new_state_dict.keys())
+    if removed_keys:
+        print(f"Removed {len(removed_keys)} normalization layer keys")
+
+    # Determine output path
+    if args.output_dir:
+        output_dir = Path(args.output_dir)
+    else:
+        if os.path.isdir(args.pretrained_path):
+            output_dir = Path(args.pretrained_path).parent / f"{Path(args.pretrained_path).name}_migrated"
+        else:
+            output_dir = Path(f"./{args.pretrained_path.replace('/', '_')}_migrated")
+
+    output_dir.mkdir(parents=True, exist_ok=True)
+
+    # Extract policy type from config
+    if "type" not in config:
+        raise ValueError("Policy type not found in config.json. The config must contain a 'type' field.")
+
+    policy_type = config["type"]
+    print(f"Detected policy type: {policy_type}")
+
+    # Clean up config - remove fields that shouldn't be passed to config constructor
+    cleaned_config = dict(config)
+
+    # Remove fields that are not part of the config class constructors
+    fields_to_remove = ["normalization_mapping", "type"]
+    for field in fields_to_remove:
+        if field in cleaned_config:
+            print(f"Removing '{field}' field from config")
+            del cleaned_config[field]
+
+    # Convert input_features and output_features to PolicyFeature objects if they exist
+    if "input_features" in cleaned_config:
+        cleaned_config["input_features"] = convert_features_to_policy_features(
+            cleaned_config["input_features"]
+        )
+    if "output_features" in cleaned_config:
+        cleaned_config["output_features"] = convert_features_to_policy_features(
+            cleaned_config["output_features"]
+        )
+
+    # Add normalization mapping to config
+    cleaned_config["normalization_mapping"] = norm_map
+
+    # Create policy configuration using the factory
+    print(f"Creating {policy_type} policy configuration...")
+    policy_config = make_policy_config(policy_type, **cleaned_config)
+
+    # Create policy instance using the factory
+    print(f"Instantiating {policy_type} policy...")
+    policy_class = get_policy_class(policy_type)
+    policy = policy_class(policy_config)
+
+    # Define whitelist of known missing keys that are acceptable (for example weight tie) for certain policy types
+    known_missing_keys_whitelist = {
+        "pi0": ["model.paligemma_with_expert.paligemma.model.language_model.embed_tokens.weight"],
+        # Add other policy types and their known missing keys here as needed
+    }
+
+    # Load state dict with graceful missing key handling
+    problematic_missing_keys = load_state_dict_with_missing_key_handling(
+        policy=policy,
+        state_dict=new_state_dict,
+        policy_type=policy_type,
+        known_missing_keys_whitelist=known_missing_keys_whitelist,
+    )
+    policy.to(torch.float32)
+    # Create preprocessor and postprocessor using the factory
+    print("Creating preprocessor and postprocessor using make_pre_post_processors...")
+    preprocessor, postprocessor = make_pre_post_processors(policy_cfg=policy_config, dataset_stats=stats)
+
+    # Determine hub repo ID if pushing to hub
+    hub_repo_id = None
+    if args.push_to_hub:
+        if args.hub_repo_id:
+            hub_repo_id = args.hub_repo_id
+        else:
+            if not os.path.isdir(args.pretrained_path):
+                # Use same repo with "_migrated" suffix
+                hub_repo_id = f"{args.pretrained_path}_migrated"
+            else:
+                raise ValueError("--hub-repo-id must be specified when pushing local model to hub")
+
+    # Save all components to local directory first
+    print(f"Saving preprocessor to {output_dir}...")
+    preprocessor.save_pretrained(output_dir)
+
+    print(f"Saving postprocessor to {output_dir}...")
+    postprocessor.save_pretrained(output_dir)
+
+    print(f"Saving model to {output_dir}...")
+    policy.save_pretrained(output_dir)
+
+    # Generate and save model card
+    print("Generating model card...")
+    # Get metadata from original config
+    dataset_repo_id = "unknown"
+    if train_config is not None:
+        dataset_repo_id = train_config.get("repo_id", "unknown")
+    license = config.get("license", "apache-2.0")
+
+    tags = config.get("tags", ["robotics", "lerobot", policy_type]) or ["robotics", "lerobot", policy_type]
+    tags = set(tags).union({"robotics", "lerobot", policy_type})
+    tags = list(tags)
+
+    # Generate model card
+    card = policy.generate_model_card(
+        dataset_repo_id=dataset_repo_id, model_type=policy_type, license=license, tags=tags
+    )
+
+    # Save model card locally
+    card.save(str(output_dir / "README.md"))
+    print(f"Model card saved to {output_dir / 'README.md'}")
+    # Push all files to hub in a single operation if requested
+    if args.push_to_hub and hub_repo_id:
+        api = HfApi()
+
+        # Determine if we should create a PR (automatically if branch is specified)
+        create_pr = args.branch is not None
+        target_location = f"branch '{args.branch}'" if args.branch else "main branch"
+
+        print(f"Pushing all migrated files to {hub_repo_id} on {target_location}...")
+
+        # Upload all files in a single commit with automatic PR creation if branch specified
+        commit_message = "Migrate policy to PolicyProcessorPipeline system"
+        commit_description = None
+
+        if create_pr:
+            # Separate commit description for PR body
+            commit_description = """**Automated Policy Migration to PolicyProcessorPipeline**
+
+This PR migrates your model to the new LeRobot policy format using the modern PolicyProcessorPipeline architecture.
+
+## What Changed
+
+### **New Architecture - PolicyProcessorPipeline**
+Your model now uses external PolicyProcessorPipeline components for data processing instead of built-in normalization layers. This provides:
+- **Modularity**: Separate preprocessing and postprocessing pipelines
+- **Flexibility**: Easy to swap, configure, and debug processing steps
+- **Compatibility**: Works with the latest LeRobot ecosystem
+
+### **Normalization Extraction**
+We've extracted normalization statistics from your model's state_dict and removed the built-in normalization layers:
+- **Extracted patterns**: `normalize_inputs.*`, `unnormalize_outputs.*`, `normalize.*`, `unnormalize.*`, `input_normalizer.*`, `output_normalizer.*`
+- **Statistics preserved**: Mean, std, min, max values for all features
+- **Clean model**: State dict now contains only core model weights
+
+### **Files Added**
+- **preprocessor_config.json**: Configuration for input preprocessing pipeline
+- **postprocessor_config.json**: Configuration for output postprocessing pipeline
+- **model.safetensors**: Clean model weights without normalization layers
+- **config.json**: Updated model configuration
+- **train_config.json**: Training configuration
+- **README.md**: Updated model card with migration information
+
+### **Benefits**
+- **Backward Compatible**: Your model behavior remains identical
+- **Future Ready**: Compatible with latest LeRobot features and updates
+- **Debuggable**: Easy to inspect and modify processing steps
+- **Portable**: Processors can be shared and reused across models
+
+### **Usage**
+```python
+# Load your migrated model
+from lerobot.policies import get_policy_class
+from lerobot.processor import PolicyProcessorPipeline
+
+# The preprocessor and postprocessor are now external
+preprocessor = PolicyProcessorPipeline.from_pretrained("your-model-repo", config_filename="preprocessor_config.json")
+postprocessor = PolicyProcessorPipeline.from_pretrained("your-model-repo", config_filename="postprocessor_config.json")
+policy = get_policy_class("your-policy-type").from_pretrained("your-model-repo")
+
+# Process data through the pipeline
+processed_batch = preprocessor(raw_batch)
+action = policy(processed_batch)
+final_action = postprocessor(action)
+```
+
+*Generated automatically by the LeRobot policy migration script*"""
+
+        upload_kwargs = {
+            "repo_id": hub_repo_id,
+            "folder_path": output_dir,
+            "repo_type": "model",
+            "commit_message": commit_message,
+            "revision": args.branch,
+            "create_pr": create_pr,
+            "allow_patterns": ["*.json", "*.safetensors", "*.md"],
+            "ignore_patterns": ["*.tmp", "*.log"],
+        }
+
+        # Add commit_description for PR body if creating PR
+        if create_pr and commit_description:
+            upload_kwargs["commit_description"] = commit_description
+
+        api.upload_folder(**upload_kwargs)
+
+        if create_pr:
+            print("All files pushed and pull request created successfully!")
+        else:
+            print("All files pushed to main branch successfully!")
+
+    print("\nMigration complete!")
+    print(f"Migrated model saved to: {output_dir}")
+    if args.push_to_hub and hub_repo_id:
+        if args.branch:
+            print(
+                f"Successfully pushed all files to branch '{args.branch}' and created PR on https://huggingface.co/{hub_repo_id}"
+            )
+        else:
+            print(f"Successfully pushed to https://huggingface.co/{hub_repo_id}")
+        if args.branch:
+            print(f"\nView the branch at: https://huggingface.co/{hub_repo_id}/tree/{args.branch}")
+            print(
+                f"View the PR at: https://huggingface.co/{hub_repo_id}/discussions (look for the most recent PR)"
+            )
+        else:
+            print(f"\nView the changes at: https://huggingface.co/{hub_repo_id}")
+
+    # Display final summary about any problematic missing keys
+    display_migration_summary_with_warnings(problematic_missing_keys)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/processor/normalize_processor.py b/lerobot/src/lerobot/processor/normalize_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..8a7a1176aed410b5bac5dab0c26145e4156267cd
--- /dev/null
+++ b/lerobot/src/lerobot/processor/normalize_processor.py
@@ -0,0 +1,560 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from __future__ import annotations
+
+from copy import deepcopy
+from dataclasses import dataclass, field
+from typing import Any
+
+import torch
+from torch import Tensor
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PipelineFeatureType, PolicyFeature
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.types import EnvTransition, PolicyAction, TransitionKey
+from lerobot.utils.constants import ACTION
+
+from .converters import from_tensor_to_numpy, to_tensor
+from .pipeline import PolicyProcessorPipeline, ProcessorStep, ProcessorStepRegistry, RobotObservation
+
+
+@dataclass
+class _NormalizationMixin:
+    """
+    A mixin class providing core functionality for normalization and unnormalization.
+
+    This class manages normalization statistics (`stats`), converts them to tensors for
+    efficient computation, handles device placement, and implements the logic for
+    applying normalization transformations (mean/std and min/max). It is designed to
+    be inherited by concrete `ProcessorStep` implementations and should not be used
+    directly.
+
+    **Stats Override Preservation:**
+    When stats are explicitly provided during construction (e.g., via overrides in
+    `DataProcessorPipeline.from_pretrained()`), they are preserved even when
+    `load_state_dict()` is called. This allows users to override normalization
+    statistics from saved models while keeping the rest of the model state intact.
+
+    Examples:
+        ```python
+        # Common use case: Override with dataset stats
+        from lerobot.datasets import LeRobotDataset
+
+        dataset = LeRobotDataset("my_dataset")
+        pipeline = DataProcessorPipeline.from_pretrained(
+            "model_path", overrides={"normalizer_processor": {"stats": dataset.meta.stats}}
+        )
+        # dataset.meta.stats will be used, not the stats from the saved model
+
+        # Custom stats override
+        custom_stats = {"action": {"mean": [0.0], "std": [1.0]}}
+        pipeline = DataProcessorPipeline.from_pretrained(
+            "model_path", overrides={"normalizer_processor": {"stats": custom_stats}}
+        )
+        ```
+
+    Attributes:
+        features: A dictionary mapping feature names to `PolicyFeature` objects, defining
+            the data structure to be processed.
+        norm_map: A dictionary mapping `FeatureType` to `NormalizationMode`, specifying
+            which normalization method to use for each type of feature.
+        stats: A dictionary containing the normalization statistics (e.g., mean, std,
+            min, max) for each feature.
+        device: The PyTorch device on which to store and perform tensor operations.
+        eps: A small epsilon value to prevent division by zero in normalization
+            calculations.
+        normalize_observation_keys: An optional set of keys to selectively apply
+            normalization to specific observation features.
+        _tensor_stats: An internal dictionary holding the normalization statistics as
+            PyTorch tensors.
+        _stats_explicitly_provided: Internal flag tracking whether stats were explicitly
+            provided during construction (used for override preservation).
+    """
+
+    features: dict[str, PolicyFeature]
+    norm_map: dict[FeatureType, NormalizationMode]
+    stats: dict[str, dict[str, Any]] | None = None
+    device: torch.device | str | None = None
+    dtype: torch.dtype | None = None
+    eps: float = 1e-8
+    normalize_observation_keys: set[str] | None = None
+
+    _tensor_stats: dict[str, dict[str, Tensor]] = field(default_factory=dict, init=False, repr=False)
+    _stats_explicitly_provided: bool = field(default=False, init=False, repr=False)
+
+    def __post_init__(self):
+        """
+        Initializes the mixin after dataclass construction.
+
+        This method handles the robust deserialization of `features` and `norm_map`
+        from JSON-compatible formats (where enums become strings and tuples become
+        lists) and converts the provided `stats` dictionary into a dictionary of
+        tensors (`_tensor_stats`) on the specified device.
+        """
+        # Track if stats were explicitly provided (not None and not empty)
+        self._stats_explicitly_provided = self.stats is not None and bool(self.stats)
+        # Robust JSON deserialization handling (guard empty maps).
+        if self.features:
+            first_val = next(iter(self.features.values()))
+            if isinstance(first_val, dict):
+                reconstructed = {}
+                for key, ft_dict in self.features.items():
+                    reconstructed[key] = PolicyFeature(
+                        type=FeatureType(ft_dict["type"]), shape=tuple(ft_dict["shape"])
+                    )
+                self.features = reconstructed
+
+        # if keys are strings (JSON), rebuild enum map
+        if self.norm_map and all(isinstance(k, str) for k in self.norm_map):
+            reconstructed = {}
+            for ft_type_str, norm_mode_str in self.norm_map.items():
+                reconstructed[FeatureType(ft_type_str)] = NormalizationMode(norm_mode_str)
+            self.norm_map = reconstructed
+
+        # Convert stats to tensors and move to the target device once during initialization.
+        self.stats = self.stats or {}
+        if self.dtype is None:
+            self.dtype = torch.float32
+        self._tensor_stats = to_tensor(self.stats, device=self.device, dtype=self.dtype)
+
+    def to(
+        self, device: torch.device | str | None = None, dtype: torch.dtype | None = None
+    ) -> _NormalizationMixin:
+        """
+        Moves the processor's normalization stats to the specified device.
+
+        Args:
+            device: The target PyTorch device.
+
+        Returns:
+            The instance of the class, allowing for method chaining.
+        """
+        if device is not None:
+            self.device = device
+        if dtype is not None:
+            self.dtype = dtype
+        self._tensor_stats = to_tensor(self.stats, device=self.device, dtype=self.dtype)
+        return self
+
+    def state_dict(self) -> dict[str, Tensor]:
+        """
+        Returns the normalization statistics as a flat state dictionary.
+
+        All tensors are moved to the CPU before being returned, which is standard practice
+        for saving state dictionaries.
+
+        Returns:
+            A flat dictionary mapping from `'feature_name.stat_name'` to the
+            corresponding statistics tensor on the CPU.
+        """
+        flat: dict[str, Tensor] = {}
+        for key, sub in self._tensor_stats.items():
+            for stat_name, tensor in sub.items():
+                flat[f"{key}.{stat_name}"] = tensor.cpu()  # Always save to CPU
+        return flat
+
+    def load_state_dict(self, state: dict[str, Tensor]) -> None:
+        """
+        Loads normalization statistics from a state dictionary.
+
+        The loaded tensors are moved to the processor's configured device.
+
+        **Stats Override Preservation:**
+        If stats were explicitly provided during construction (e.g., via overrides in
+        `DataProcessorPipeline.from_pretrained()`), they are preserved and the state
+        dictionary is ignored. This allows users to override normalization statistics
+        while still loading the rest of the model state.
+
+        This behavior is crucial for scenarios where users want to adapt a pretrained
+        model to a new dataset with different statistics without retraining the entire
+        model.
+
+        Args:
+            state: A flat state dictionary with keys in the format
+                   `'feature_name.stat_name'`.
+
+        Note:
+            When stats are preserved due to explicit provision, only the tensor
+            representation is updated to ensure consistency with the current device
+            and dtype settings.
+        """
+        # If stats were explicitly provided during construction, preserve them
+        if self._stats_explicitly_provided and self.stats is not None:
+            # Don't load from state_dict, keep the explicitly provided stats
+            # But ensure _tensor_stats is properly initialized
+            self._tensor_stats = to_tensor(self.stats, device=self.device, dtype=self.dtype)  # type: ignore[assignment]
+            return
+
+        # Normal behavior: load stats from state_dict
+        self._tensor_stats.clear()
+        for flat_key, tensor in state.items():
+            key, stat_name = flat_key.rsplit(".", 1)
+            # Load to the processor's configured device.
+            self._tensor_stats.setdefault(key, {})[stat_name] = tensor.to(
+                dtype=torch.float32, device=self.device
+            )
+
+        # Reconstruct the original stats dict from tensor stats for compatibility with to() method
+        # and other functions that rely on self.stats
+        self.stats = {}
+        for key, tensor_dict in self._tensor_stats.items():
+            self.stats[key] = {}
+            for stat_name, tensor in tensor_dict.items():
+                # Convert tensor back to python/numpy format
+                self.stats[key][stat_name] = from_tensor_to_numpy(tensor)
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns a serializable dictionary of the processor's configuration.
+
+        This method is used when saving the processor to disk, ensuring that its
+        configuration can be reconstructed later.
+
+        Returns:
+            A JSON-serializable dictionary containing the configuration.
+        """
+        config = {
+            "eps": self.eps,
+            "features": {
+                key: {"type": ft.type.value, "shape": ft.shape} for key, ft in self.features.items()
+            },
+            "norm_map": {ft_type.value: norm_mode.value for ft_type, norm_mode in self.norm_map.items()},
+        }
+        if self.normalize_observation_keys is not None:
+            config["normalize_observation_keys"] = sorted(self.normalize_observation_keys)
+        return config
+
+    def _normalize_observation(self, observation: RobotObservation, inverse: bool) -> dict[str, Tensor]:
+        """
+        Applies (un)normalization to all relevant features in an observation dictionary.
+
+        Args:
+            observation: The observation dictionary to process.
+            inverse: If `True`, applies unnormalization; otherwise, applies normalization.
+
+        Returns:
+            A new observation dictionary with the transformed tensor values.
+        """
+        new_observation = dict(observation)
+        for key, feature in self.features.items():
+            if self.normalize_observation_keys is not None and key not in self.normalize_observation_keys:
+                continue
+            if feature.type != FeatureType.ACTION and key in new_observation:
+                # Convert to tensor but preserve original dtype for adaptation logic
+                tensor = torch.as_tensor(new_observation[key])
+                new_observation[key] = self._apply_transform(tensor, key, feature.type, inverse=inverse)
+        return new_observation
+
+    def _normalize_action(self, action: Tensor, inverse: bool) -> Tensor:
+        # Convert to tensor but preserve original dtype for adaptation logic
+        """
+        Applies (un)normalization to an action tensor.
+
+        Args:
+            action: The action tensor to process.
+            inverse: If `True`, applies unnormalization; otherwise, applies normalization.
+
+        Returns:
+            The transformed action tensor.
+        """
+        processed_action = self._apply_transform(action, ACTION, FeatureType.ACTION, inverse=inverse)
+        return processed_action
+
+    def _apply_transform(
+        self, tensor: Tensor, key: str, feature_type: FeatureType, *, inverse: bool = False
+    ) -> Tensor:
+        """
+        Core logic to apply a normalization or unnormalization transformation to a tensor.
+
+        This method selects the appropriate normalization mode based on the feature type
+        and applies the corresponding mathematical operation.
+
+        Normalization Modes:
+          - MEAN_STD: Centers data around zero with unit variance.
+          - MIN_MAX: Scales data to [-1, 1] range using actual min/max values.
+          - QUANTILES: Scales data to [-1, 1] range using 1st and 99th percentiles (q01/q99).
+          - QUANTILE10: Scales data to [-1, 1] range using 10th and 90th percentiles (q10/q90).
+
+        Args:
+            tensor: The input tensor to transform.
+            key: The feature key corresponding to the tensor.
+            feature_type: The `FeatureType` of the tensor.
+            inverse: If `True`, applies the inverse transformation (unnormalization).
+
+        Returns:
+            The transformed tensor.
+
+        Raises:
+            ValueError: If an unsupported normalization mode is encountered.
+        """
+        norm_mode = self.norm_map.get(feature_type, NormalizationMode.IDENTITY)
+        if norm_mode == NormalizationMode.IDENTITY or key not in self._tensor_stats:
+            return tensor
+
+        if norm_mode not in (
+            NormalizationMode.MEAN_STD,
+            NormalizationMode.MIN_MAX,
+            NormalizationMode.QUANTILES,
+            NormalizationMode.QUANTILE10,
+        ):
+            raise ValueError(f"Unsupported normalization mode: {norm_mode}")
+
+        # For Accelerate compatibility: Ensure stats are on the same device and dtype as the input tensor
+        if self._tensor_stats and key in self._tensor_stats:
+            first_stat = next(iter(self._tensor_stats[key].values()))
+            if first_stat.device != tensor.device or first_stat.dtype != tensor.dtype:
+                self.to(device=tensor.device, dtype=tensor.dtype)
+
+        stats = self._tensor_stats[key]
+
+        if norm_mode == NormalizationMode.MEAN_STD:
+            mean = stats.get("mean", None)
+            std = stats.get("std", None)
+            if mean is None or std is None:
+                raise ValueError(
+                    "MEAN_STD normalization mode requires mean and std stats, please update the dataset with the correct stats"
+                )
+
+            mean, std = stats["mean"], stats["std"]
+            # Avoid division by zero by adding a small epsilon.
+            denom = std + self.eps
+            if inverse:
+                return tensor * std + mean
+            return (tensor - mean) / denom
+
+        if norm_mode == NormalizationMode.MIN_MAX:
+            min_val = stats.get("min", None)
+            max_val = stats.get("max", None)
+            if min_val is None or max_val is None:
+                raise ValueError(
+                    "MIN_MAX normalization mode requires min and max stats, please update the dataset with the correct stats"
+                )
+
+            min_val, max_val = stats["min"], stats["max"]
+            denom = max_val - min_val
+            # When min_val == max_val, substitute the denominator with a small epsilon
+            # to prevent division by zero. This consistently maps an input equal to
+            # min_val to -1, ensuring a stable transformation.
+            denom = torch.where(
+                denom == 0, torch.tensor(self.eps, device=tensor.device, dtype=tensor.dtype), denom
+            )
+            if inverse:
+                # Map from [-1, 1] back to [min, max]
+                return (tensor + 1) / 2 * denom + min_val
+            # Map from [min, max] to [-1, 1]
+            return 2 * (tensor - min_val) / denom - 1
+
+        if norm_mode == NormalizationMode.QUANTILES:
+            q01 = stats.get("q01", None)
+            q99 = stats.get("q99", None)
+            if q01 is None or q99 is None:
+                raise ValueError(
+                    "QUANTILES normalization mode requires q01 and q99 stats, please update the dataset with the correct stats using the `augment_dataset_quantile_stats.py` script"
+                )
+
+            denom = q99 - q01
+            # Avoid division by zero by adding epsilon when quantiles are identical
+            denom = torch.where(
+                denom == 0, torch.tensor(self.eps, device=tensor.device, dtype=tensor.dtype), denom
+            )
+            if inverse:
+                return (tensor + 1.0) * denom / 2.0 + q01
+            return 2.0 * (tensor - q01) / denom - 1.0
+
+        if norm_mode == NormalizationMode.QUANTILE10:
+            q10 = stats.get("q10", None)
+            q90 = stats.get("q90", None)
+            if q10 is None or q90 is None:
+                raise ValueError(
+                    "QUANTILE10 normalization mode requires q10 and q90 stats, please update the dataset with the correct stats using the `augment_dataset_quantile_stats.py` script"
+                )
+
+            denom = q90 - q10
+            # Avoid division by zero by adding epsilon when quantiles are identical
+            denom = torch.where(
+                denom == 0, torch.tensor(self.eps, device=tensor.device, dtype=tensor.dtype), denom
+            )
+            if inverse:
+                return (tensor + 1.0) * denom / 2.0 + q10
+            return 2.0 * (tensor - q10) / denom - 1.0
+
+        # If necessary stats are missing, return input unchanged.
+        return tensor
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="normalizer_processor")
+class NormalizerProcessorStep(_NormalizationMixin, ProcessorStep):
+    """
+    A processor step that applies normalization to observations and actions in a transition.
+
+    This class uses the logic from `_NormalizationMixin` to perform forward normalization
+    (e.g., scaling data to have zero mean and unit variance, or to the range [-1, 1]).
+    It is typically used in the pre-processing pipeline before feeding data to a policy.
+    """
+
+    @classmethod
+    def from_lerobot_dataset(
+        cls,
+        dataset: LeRobotDataset,
+        features: dict[str, PolicyFeature],
+        norm_map: dict[FeatureType, NormalizationMode],
+        *,
+        normalize_observation_keys: set[str] | None = None,
+        eps: float = 1e-8,
+        device: torch.device | str | None = None,
+    ) -> NormalizerProcessorStep:
+        """
+        Creates a `NormalizerProcessorStep` instance using statistics from a `LeRobotDataset`.
+
+        Args:
+            dataset: The dataset from which to extract normalization statistics.
+            features: The feature definition for the processor.
+            norm_map: The mapping from feature types to normalization modes.
+            normalize_observation_keys: An optional set of observation keys to normalize.
+            eps: A small epsilon value for numerical stability.
+            device: The target device for the processor.
+
+        Returns:
+            A new instance of `NormalizerProcessorStep`.
+        """
+        return cls(
+            features=features,
+            norm_map=norm_map,
+            stats=dataset.meta.stats,
+            normalize_observation_keys=normalize_observation_keys,
+            eps=eps,
+            device=device,
+        )
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        new_transition = transition.copy()
+
+        # Handle observation normalization.
+        observation = new_transition.get(TransitionKey.OBSERVATION)
+        if observation is not None:
+            new_transition[TransitionKey.OBSERVATION] = self._normalize_observation(
+                observation, inverse=False
+            )
+
+        # Handle action normalization.
+        action = new_transition.get(TransitionKey.ACTION)
+
+        if action is None:
+            return new_transition
+
+        if not isinstance(action, PolicyAction):
+            raise ValueError(f"Action should be a PolicyAction type got {type(action)}")
+
+        new_transition[TransitionKey.ACTION] = self._normalize_action(action, inverse=False)
+
+        return new_transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="unnormalizer_processor")
+class UnnormalizerProcessorStep(_NormalizationMixin, ProcessorStep):
+    """
+    A processor step that applies unnormalization to observations and actions.
+
+    This class inverts the normalization process, scaling data back to its original
+    range. It is typically used in the post-processing pipeline to convert a policy's
+    normalized action output into a format that can be executed by a robot or
+    environment.
+    """
+
+    @classmethod
+    def from_lerobot_dataset(
+        cls,
+        dataset: LeRobotDataset,
+        features: dict[str, PolicyFeature],
+        norm_map: dict[FeatureType, NormalizationMode],
+        *,
+        device: torch.device | str | None = None,
+    ) -> UnnormalizerProcessorStep:
+        """
+        Creates an `UnnormalizerProcessorStep` using statistics from a `LeRobotDataset`.
+
+        Args:
+            dataset: The dataset from which to extract normalization statistics.
+            features: The feature definition for the processor.
+            norm_map: The mapping from feature types to normalization modes.
+            device: The target device for the processor.
+
+        Returns:
+            A new instance of `UnnormalizerProcessorStep`.
+        """
+        return cls(features=features, norm_map=norm_map, stats=dataset.meta.stats, device=device)
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        new_transition = transition.copy()
+
+        # Handle observation unnormalization.
+        observation = new_transition.get(TransitionKey.OBSERVATION)
+        if observation is not None:
+            new_transition[TransitionKey.OBSERVATION] = self._normalize_observation(observation, inverse=True)
+
+        # Handle action unnormalization.
+        action = new_transition.get(TransitionKey.ACTION)
+
+        if action is None:
+            return new_transition
+        if not isinstance(action, PolicyAction):
+            raise ValueError(f"Action should be a PolicyAction type got {type(action)}")
+
+        new_transition[TransitionKey.ACTION] = self._normalize_action(action, inverse=True)
+
+        return new_transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+def hotswap_stats(
+    policy_processor: PolicyProcessorPipeline, stats: dict[str, dict[str, Any]]
+) -> PolicyProcessorPipeline:
+    """
+    Replaces normalization statistics in an existing `PolicyProcessorPipeline` instance.
+
+    This function creates a deep copy of the provided pipeline and updates the
+    statistics of any `NormalizerProcessorStep` or `UnnormalizerProcessorStep` it
+    contains. This is useful for adapting a trained policy to a new environment or
+    dataset with different data distributions without having to reconstruct the entire
+    pipeline.
+
+    Args:
+        policy_processor: The policy processor pipeline to modify.
+        stats: The new dictionary of normalization statistics to apply.
+
+    Returns:
+        A new `PolicyProcessorPipeline` instance with the updated statistics.
+    """
+    rp = deepcopy(policy_processor)
+    for step in rp.steps:
+        if isinstance(step, _NormalizationMixin):
+            step.stats = stats
+            # Re-initialize tensor_stats on the correct device.
+            step._tensor_stats = to_tensor(stats, device=step.device, dtype=step.dtype)  # type: ignore[assignment]
+    return rp
diff --git a/lerobot/src/lerobot/processor/observation_processor.py b/lerobot/src/lerobot/processor/observation_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..d22d8fb96ecf1f58edabb0547f72a8579f714abb
--- /dev/null
+++ b/lerobot/src/lerobot/processor/observation_processor.py
@@ -0,0 +1,206 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from dataclasses import dataclass
+
+import einops
+import numpy as np
+import torch
+from torch import Tensor
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.utils.constants import OBS_ENV_STATE, OBS_IMAGE, OBS_IMAGES, OBS_STATE, OBS_STR
+
+from .pipeline import ObservationProcessorStep, ProcessorStepRegistry
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="observation_processor")
+class VanillaObservationProcessorStep(ObservationProcessorStep):
+    """
+    Processes standard Gymnasium observations into the LeRobot format.
+
+    This step handles both image and state data from a typical observation dictionary,
+    preparing it for use in a LeRobot policy.
+
+    **Image Processing:**
+    -   Converts channel-last (H, W, C), `uint8` images to channel-first (C, H, W),
+        `float32` tensors.
+    -   Normalizes pixel values from the [0, 255] range to [0, 1].
+    -   Adds a batch dimension if one is not already present.
+    -   Recognizes a single image under the key `"pixels"` and maps it to
+        `"observation.image"`.
+    -   Recognizes a dictionary of images under the key `"pixels"` and maps them
+        to `"observation.images.{camera_name}"`.
+
+    **State Processing:**
+    -   Maps the `"environment_state"` key to `"observation.environment_state"`.
+    -   Maps the `"agent_pos"` key to `"observation.state"`.
+    -   Converts NumPy arrays to PyTorch tensors.
+    -   Adds a batch dimension if one is not already present.
+    """
+
+    def _process_single_image(self, img: np.ndarray) -> Tensor:
+        """
+        Processes a single NumPy image array into a channel-first, normalized tensor.
+
+        Args:
+            img: A NumPy array representing the image, expected to be in channel-last
+                 (H, W, C) format with a `uint8` dtype.
+
+        Returns:
+            A `float32` PyTorch tensor in channel-first (B, C, H, W) format, with
+            pixel values normalized to the [0, 1] range.
+
+        Raises:
+            ValueError: If the input image does not appear to be in channel-last
+                        format or is not of `uint8` dtype.
+        """
+        # Convert to tensor
+        img_tensor = torch.from_numpy(img)
+
+        # Add batch dimension if needed
+        if img_tensor.ndim == 3:
+            img_tensor = img_tensor.unsqueeze(0)
+
+        # Validate image format
+        _, h, w, c = img_tensor.shape
+        if not (c < h and c < w):
+            raise ValueError(f"Expected channel-last images, but got shape {img_tensor.shape}")
+
+        if img_tensor.dtype != torch.uint8:
+            raise ValueError(f"Expected torch.uint8 images, but got {img_tensor.dtype}")
+
+        # Convert to channel-first format
+        img_tensor = einops.rearrange(img_tensor, "b h w c -> b c h w").contiguous()
+
+        # Convert to float32 and normalize to [0, 1]
+        img_tensor = img_tensor.type(torch.float32) / 255.0
+
+        return img_tensor
+
+    def _process_observation(self, observation):
+        """
+        Processes both image and state observations.
+        """
+
+        processed_obs = observation.copy()
+
+        if "pixels" in processed_obs:
+            pixels = processed_obs.pop("pixels")
+
+            if isinstance(pixels, dict):
+                imgs = {f"{OBS_IMAGES}.{key}": img for key, img in pixels.items()}
+            else:
+                imgs = {OBS_IMAGE: pixels}
+
+            for imgkey, img in imgs.items():
+                processed_obs[imgkey] = self._process_single_image(img)
+
+        if "environment_state" in processed_obs:
+            env_state_np = processed_obs.pop("environment_state")
+            env_state = torch.from_numpy(env_state_np).float()
+            if env_state.dim() == 1:
+                env_state = env_state.unsqueeze(0)
+            processed_obs[OBS_ENV_STATE] = env_state
+
+        if "agent_pos" in processed_obs:
+            agent_pos_np = processed_obs.pop("agent_pos")
+            agent_pos = torch.from_numpy(agent_pos_np).float()
+            if agent_pos.dim() == 1:
+                agent_pos = agent_pos.unsqueeze(0)
+            processed_obs[OBS_STATE] = agent_pos
+
+        return processed_obs
+
+    def observation(self, observation):
+        return self._process_observation(observation)
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Transforms feature keys from the Gym standard to the LeRobot standard.
+
+        This method standardizes the feature dictionary by renaming keys according
+        to LeRobot's conventions, ensuring that policies can be constructed correctly.
+        It handles various raw key formats, including those with an "observation." prefix.
+
+        **Renaming Rules:**
+        - `pixels` or `observation.pixels` -> `observation.image`
+        - `pixels.{cam}` or `observation.pixels.{cam}` -> `observation.images.{cam}`
+        - `environment_state` or `observation.environment_state` -> `observation.environment_state`
+        - `agent_pos` or `observation.agent_pos` -> `observation.state`
+
+        Args:
+            features: The policy features dictionary with Gym-style keys.
+
+        Returns:
+            The policy features dictionary with standardized LeRobot keys.
+        """
+        # Build a new features mapping keyed by the same FeatureType buckets
+        # We assume callers already placed features in the correct FeatureType.
+        new_features: dict[PipelineFeatureType, dict[str, PolicyFeature]] = {ft: {} for ft in features}
+
+        exact_pairs = {
+            "pixels": OBS_IMAGE,
+            "environment_state": OBS_ENV_STATE,
+            "agent_pos": OBS_STATE,
+        }
+
+        prefix_pairs = {
+            "pixels.": f"{OBS_IMAGES}.",
+        }
+
+        # Iterate over all incoming feature buckets and normalize/move each entry
+        for src_ft, bucket in features.items():
+            for key, feat in list(bucket.items()):
+                handled = False
+
+                # Prefix-based rules (e.g. pixels.cam1 -> OBS_IMAGES.cam1)
+                for old_prefix, new_prefix in prefix_pairs.items():
+                    prefixed_old = f"{OBS_STR}.{old_prefix}"
+                    if key.startswith(prefixed_old):
+                        suffix = key[len(prefixed_old) :]
+                        new_key = f"{new_prefix}{suffix}"
+                        new_features[src_ft][new_key] = feat
+                        handled = True
+                        break
+
+                    if key.startswith(old_prefix):
+                        suffix = key[len(old_prefix) :]
+                        new_key = f"{new_prefix}{suffix}"
+                        new_features[src_ft][new_key] = feat
+                        handled = True
+                        break
+
+                if handled:
+                    continue
+
+                # Exact-name rules (pixels, environment_state, agent_pos)
+                for old, new in exact_pairs.items():
+                    if key == old or key == f"{OBS_STR}.{old}":
+                        new_key = new
+                        new_features[src_ft][new_key] = feat
+                        handled = True
+                        break
+
+                if handled:
+                    continue
+
+                # Default: keep key in the same source FeatureType bucket
+                new_features[src_ft][key] = feat
+
+        return new_features
diff --git a/lerobot/src/lerobot/processor/pipeline.py b/lerobot/src/lerobot/processor/pipeline.py
new file mode 100644
index 0000000000000000000000000000000000000000..abfb314210dffebb5a3b0eaa8499988ca3e37bda
--- /dev/null
+++ b/lerobot/src/lerobot/processor/pipeline.py
@@ -0,0 +1,1716 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+This module defines a generic, sequential data processing pipeline framework, primarily designed for
+transforming robotics data (observations, actions, rewards, etc.).
+
+The core components are:
+- ProcessorStep: An abstract base class for a single data transformation operation.
+- ProcessorStepRegistry: A mechanism to register and retrieve ProcessorStep classes by name.
+- DataProcessorPipeline: A class that chains multiple ProcessorStep instances together to form a complete
+  data processing workflow. It integrates with the Hugging Face Hub for easy sharing and versioning of
+  pipelines, including their configuration and state.
+- Specialized abstract ProcessorStep subclasses (e.g., ObservationProcessorStep, ActionProcessorStep)
+  to simplify the creation of steps that target specific parts of a data transition.
+"""
+
+from __future__ import annotations
+
+import importlib
+import json
+import os
+import re
+from abc import ABC, abstractmethod
+from collections.abc import Callable, Iterable, Sequence
+from copy import deepcopy
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import Any, TypedDict, TypeVar, cast
+
+import torch
+from huggingface_hub import hf_hub_download
+from safetensors.torch import load_file, save_file
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.types import EnvAction, EnvTransition, PolicyAction, RobotAction, RobotObservation, TransitionKey
+from lerobot.utils.hub import HubMixin
+
+from .converters import batch_to_transition, create_transition, transition_to_batch
+
+# Generic type variables for pipeline input and output.
+TInput = TypeVar("TInput")
+TOutput = TypeVar("TOutput")
+
+
+class ProcessorStepRegistry:
+    """A registry for ProcessorStep classes to allow instantiation from a string name.
+
+    This class provides a way to map string identifiers to `ProcessorStep` classes,
+    which is useful for deserializing pipelines from configuration files without
+
+    hardcoding class imports.
+    """
+
+    _registry: dict[str, type] = {}
+
+    @classmethod
+    def register(cls, name: str | None = None):
+        """A class decorator to register a ProcessorStep.
+
+        Args:
+            name: The name to register the class under. If None, the class's `__name__` is used.
+
+        Returns:
+            A decorator function that registers the class and returns it.
+
+        Raises:
+            ValueError: If a step with the same name is already registered.
+        """
+
+        def decorator(step_class: type) -> type:
+            """The actual decorator that performs the registration."""
+            registration_name = name if name is not None else step_class.__name__
+
+            if registration_name in cls._registry:
+                raise ValueError(
+                    f"Processor step '{registration_name}' is already registered. "
+                    f"Use a different name or unregister the existing one first."
+                )
+
+            cls._registry[registration_name] = step_class
+            # Store the registration name on the class for easy lookup during serialization.
+            step_class._registry_name = registration_name
+            return step_class
+
+        return decorator
+
+    @classmethod
+    def get(cls, name: str) -> type:
+        """Retrieves a processor step class from the registry by its name.
+
+        Args:
+            name: The name of the step to retrieve.
+
+        Returns:
+            The processor step class corresponding to the given name.
+
+        Raises:
+            KeyError: If the name is not found in the registry.
+        """
+        if name not in cls._registry:
+            available = list(cls._registry.keys())
+            raise KeyError(
+                f"Processor step '{name}' not found in registry. "
+                f"Available steps: {available}. "
+                f"Make sure the step is registered using @ProcessorStepRegistry.register()"
+            )
+        return cls._registry[name]
+
+    @classmethod
+    def unregister(cls, name: str) -> None:
+        """Removes a processor step from the registry.
+
+        Args:
+            name: The name of the step to unregister.
+        """
+        cls._registry.pop(name, None)
+
+    @classmethod
+    def list(cls) -> list[str]:
+        """Returns a list of all registered processor step names."""
+        return list(cls._registry.keys())
+
+    @classmethod
+    def clear(cls) -> None:
+        """Clears all processor steps from the registry."""
+        cls._registry.clear()
+
+
+class ProcessorStep(ABC):
+    """Abstract base class for a single step in a data processing pipeline.
+
+    Each step must implement the `__call__` method to perform its transformation
+    on a data transition and the `transform_features` method to describe how it
+    alters the shape or type of data features.
+
+    Subclasses can optionally be stateful by implementing `state_dict` and `load_state_dict`.
+    """
+
+    _current_transition: EnvTransition | None = None
+
+    @property
+    def transition(self) -> EnvTransition:
+        """Provides access to the most recent transition being processed.
+
+        This is useful for steps that need to access other parts of the transition
+        data beyond their primary target (e.g., an action processing step that
+        needs to look at the observation).
+
+        Raises:
+            ValueError: If accessed before the step has been called with a transition.
+        """
+        if self._current_transition is None:
+            raise ValueError("Transition is not set. Make sure to call the step with a transition first.")
+        return self._current_transition
+
+    @abstractmethod
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Processes an environment transition.
+
+        This method should contain the core logic of the processing step.
+
+        Args:
+            transition: The input data transition to be processed.
+
+        Returns:
+            The processed transition.
+        """
+        return transition
+
+    def get_config(self) -> dict[str, Any]:
+        """Returns the configuration of the step for serialization.
+
+        Returns:
+            A JSON-serializable dictionary of configuration parameters.
+        """
+        return {}
+
+    def state_dict(self) -> dict[str, torch.Tensor]:
+        """Returns the state of the step (e.g., learned parameters, running means).
+
+        Returns:
+            A dictionary mapping state names to tensors.
+        """
+        return {}
+
+    def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+        """Loads the step's state from a state dictionary.
+
+        Args:
+            state: A dictionary of state tensors.
+        """
+        return None
+
+    def reset(self) -> None:
+        """Resets the internal state of the processor step, if any."""
+        return None
+
+    @abstractmethod
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """Defines how this step modifies the description of pipeline features.
+
+        This method is used to track changes in data shapes, dtypes, or modalities
+        as data flows through the pipeline, without needing to process actual data.
+
+        Args:
+            features: A dictionary describing the input features for observations, actions, etc.
+
+        Returns:
+            A dictionary describing the output features after this step's transformation.
+        """
+        return features
+
+
+class ProcessorKwargs(TypedDict, total=False):
+    """A TypedDict for optional keyword arguments used in pipeline construction."""
+
+    to_transition: Callable[[dict[str, Any]], EnvTransition] | None
+    to_output: Callable[[EnvTransition], Any] | None
+    name: str | None
+    before_step_hooks: list[Callable[[int, EnvTransition], None]] | None
+    after_step_hooks: list[Callable[[int, EnvTransition], None]] | None
+
+
+class ProcessorMigrationError(Exception):
+    """Raised when a model needs migration to the processor format"""
+
+    def __init__(self, model_path: str | Path, migration_command: str, original_error: str):
+        self.model_path = model_path
+        self.migration_command = migration_command
+        self.original_error = original_error
+        super().__init__(
+            f"Model '{model_path}' requires migration to processor format. "
+            f"Run: {migration_command}\n\nOriginal error: {original_error}"
+        )
+
+
+@dataclass
+class DataProcessorPipeline[TInput, TOutput](HubMixin):
+    """A sequential pipeline for processing data, integrated with the Hugging Face Hub.
+
+    This class chains together multiple `ProcessorStep` instances to form a complete
+    data processing workflow. It's generic, allowing for custom input and output types,
+    which are handled by the `to_transition` and `to_output` converters.
+
+    Attributes:
+        steps: A sequence of `ProcessorStep` objects that make up the pipeline.
+        name: A descriptive name for the pipeline.
+        to_transition: A function to convert raw input data into the standardized `EnvTransition` format.
+        to_output: A function to convert the final `EnvTransition` into the desired output format.
+        before_step_hooks: A list of functions to be called before each step is executed.
+        after_step_hooks: A list of functions to be called after each step is executed.
+    """
+
+    steps: Sequence[ProcessorStep] = field(default_factory=list)
+    name: str = "DataProcessorPipeline"
+
+    to_transition: Callable[[TInput], EnvTransition] = field(
+        default_factory=lambda: cast(Callable[[TInput], EnvTransition], batch_to_transition), repr=False
+    )
+    to_output: Callable[[EnvTransition], TOutput] = field(
+        default_factory=lambda: cast(Callable[[EnvTransition], TOutput], transition_to_batch),
+        repr=False,
+    )
+
+    before_step_hooks: list[Callable[[int, EnvTransition], None]] = field(default_factory=list, repr=False)
+    after_step_hooks: list[Callable[[int, EnvTransition], None]] = field(default_factory=list, repr=False)
+
+    def __call__(self, data: TInput) -> TOutput:
+        """Processes input data through the full pipeline.
+
+        Args:
+            data: The input data to process.
+
+        Returns:
+            The processed data in the specified output format.
+        """
+        transition = self.to_transition(data)
+        transformed_transition = self._forward(transition)
+        return self.to_output(transformed_transition)
+
+    def _forward(self, transition: EnvTransition) -> EnvTransition:
+        """Executes all processing steps and hooks in sequence.
+
+        Args:
+            transition: The initial `EnvTransition` object.
+
+        Returns:
+            The final `EnvTransition` after all steps have been applied.
+        """
+        for idx, processor_step in enumerate(self.steps):
+            # Execute pre-hooks
+            for hook in self.before_step_hooks:
+                hook(idx, transition)
+
+            transition = processor_step(transition)
+
+            # Execute post-hooks
+            for hook in self.after_step_hooks:
+                hook(idx, transition)
+        return transition
+
+    def step_through(self, data: TInput) -> Iterable[EnvTransition]:
+        """Processes data step-by-step, yielding the transition at each stage.
+
+        This is a generator method useful for debugging and inspecting the intermediate
+        state of the data as it passes through the pipeline.
+
+        Args:
+            data: The input data.
+
+        Yields:
+            The `EnvTransition` object, starting with the initial state and then after
+            each processing step.
+        """
+        transition = self.to_transition(data)
+
+        # Yield the initial state before any processing.
+        yield transition
+
+        for processor_step in self.steps:
+            transition = processor_step(transition)
+            yield transition
+
+    def _save_pretrained(self, save_directory: Path, **kwargs):
+        """Internal method to comply with `HubMixin`'s saving mechanism.
+
+        This method does the actual saving work and is called by HubMixin.save_pretrained.
+        """
+        config_filename = kwargs.pop("config_filename", None)
+
+        # Sanitize the pipeline name to create a valid filename prefix.
+        sanitized_name = re.sub(r"[^a-zA-Z0-9_]", "_", self.name.lower())
+
+        if config_filename is None:
+            config_filename = f"{sanitized_name}.json"
+
+        config: dict[str, Any] = {
+            "name": self.name,
+            "steps": [],
+        }
+
+        # Iterate through each step to build its configuration entry.
+        for step_index, processor_step in enumerate(self.steps):
+            registry_name = getattr(processor_step.__class__, "_registry_name", None)
+
+            step_entry: dict[str, Any] = {}
+            # Prefer registry name for portability, otherwise fall back to full class path.
+            if registry_name:
+                step_entry["registry_name"] = registry_name
+            else:
+                step_entry["class"] = (
+                    f"{processor_step.__class__.__module__}.{processor_step.__class__.__name__}"
+                )
+
+            # Save step configuration if `get_config` is implemented.
+            if hasattr(processor_step, "get_config"):
+                step_entry["config"] = processor_step.get_config()
+
+            # Save step state if `state_dict` is implemented and returns a non-empty dict.
+            if hasattr(processor_step, "state_dict"):
+                state = processor_step.state_dict()
+                if state:
+                    # Clone tensors to avoid modifying the original state.
+                    cloned_state = {key: tensor.clone() for key, tensor in state.items()}
+
+                    # Create a unique filename for the state file.
+                    if registry_name:
+                        state_filename = f"{sanitized_name}_step_{step_index}_{registry_name}.safetensors"
+                    else:
+                        state_filename = f"{sanitized_name}_step_{step_index}.safetensors"
+
+                    save_file(cloned_state, os.path.join(str(save_directory), state_filename))
+                    step_entry["state_file"] = state_filename
+
+            config["steps"].append(step_entry)
+
+        # Write the main configuration JSON file.
+        with open(os.path.join(str(save_directory), config_filename), "w") as file_pointer:
+            json.dump(config, file_pointer, indent=2)
+
+    def save_pretrained(
+        self,
+        save_directory: str | Path | None = None,
+        *,
+        repo_id: str | None = None,
+        push_to_hub: bool = False,
+        card_kwargs: dict[str, Any] | None = None,
+        config_filename: str | None = None,
+        **push_to_hub_kwargs,
+    ):
+        """Saves the pipeline's configuration and state to a directory.
+
+        This method creates a JSON configuration file that defines the pipeline's structure
+        (name and steps). For each stateful step, it also saves a `.safetensors` file
+        containing its state dictionary.
+
+        Args:
+            save_directory: The directory where the pipeline will be saved. If None, saves to
+                HF_LEROBOT_HOME/processors/{sanitized_pipeline_name}.
+            repo_id: ID of your repository on the Hub. Used only if `push_to_hub=true`.
+            push_to_hub: Whether or not to push your object to the Hugging Face Hub after saving it.
+            card_kwargs: Additional arguments passed to the card template to customize the card.
+            config_filename: The name of the JSON configuration file. If None, a name is
+                generated from the pipeline's `name` attribute.
+            **push_to_hub_kwargs: Additional key word arguments passed along to the push_to_hub method.
+        """
+        if save_directory is None:
+            # Use default directory in HF_LEROBOT_HOME
+            from lerobot.utils.constants import HF_LEROBOT_HOME
+
+            sanitized_name = re.sub(r"[^a-zA-Z0-9_]", "_", self.name.lower())
+            save_directory = HF_LEROBOT_HOME / "processors" / sanitized_name
+
+        # For direct saves (not through hub), handle config_filename
+        if not push_to_hub and config_filename is not None:
+            # Call _save_pretrained directly with config_filename
+            save_directory = Path(save_directory)
+            save_directory.mkdir(parents=True, exist_ok=True)
+            self._save_pretrained(save_directory, config_filename=config_filename)
+            return None
+
+        # Pass config_filename through kwargs for _save_pretrained when using hub
+        if config_filename is not None:
+            push_to_hub_kwargs["config_filename"] = config_filename
+
+        # Call parent's save_pretrained which will call our _save_pretrained
+        return super().save_pretrained(
+            save_directory=save_directory,
+            repo_id=repo_id,
+            push_to_hub=push_to_hub,
+            card_kwargs=card_kwargs,
+            **push_to_hub_kwargs,
+        )
+
+    @classmethod
+    def from_pretrained(
+        cls,
+        pretrained_model_name_or_path: str | Path,
+        config_filename: str,
+        *,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict[str, str] | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        overrides: dict[str, Any] | None = None,
+        to_transition: Callable[[TInput], EnvTransition] | None = None,
+        to_output: Callable[[EnvTransition], TOutput] | None = None,
+        **kwargs,
+    ) -> DataProcessorPipeline[TInput, TOutput]:
+        """Loads a pipeline from a local directory, single file, or Hugging Face Hub repository.
+
+        This method implements a simplified loading pipeline with intelligent migration detection:
+
+        **Simplified Loading Strategy**:
+        1. **Config Loading** (_load_config):
+           - **Directory**: Load specified config_filename from directory
+           - **Single file**: Load file directly (config_filename ignored)
+           - **Hub repository**: Download specified config_filename from Hub
+
+        2. **Config Validation** (_validate_loaded_config):
+           - Format validation: Ensure config is valid processor format
+           - Migration detection: Guide users to migrate old LeRobot models
+           - Clear errors: Provide actionable error messages
+
+        3. **Step Construction** (_build_steps_with_overrides):
+           - Class resolution: Registry lookup or dynamic imports
+           - Override merging: User parameters override saved config
+           - State loading: Load .safetensors files for stateful steps
+
+        4. **Override Validation** (_validate_overrides_used):
+           - Ensure all user overrides were applied (catch typos)
+           - Provide helpful error messages with available keys
+
+        **Migration Detection**:
+        - **Smart detection**: Analyzes JSON files to detect old LeRobot models
+        - **Precise targeting**: Avoids false positives on other HuggingFace models
+        - **Clear guidance**: Provides exact migration command to run
+        - **Error mode**: Always raises ProcessorMigrationError for clear user action
+
+        **Loading Examples**:
+        ```python
+        # Directory loading
+        pipeline = DataProcessorPipeline.from_pretrained("/models/my_model", config_filename="processor.json")
+
+        # Single file loading
+        pipeline = DataProcessorPipeline.from_pretrained(
+            "/models/my_model/processor.json", config_filename="processor.json"
+        )
+
+        # Hub loading
+        pipeline = DataProcessorPipeline.from_pretrained("user/repo", config_filename="processor.json")
+
+        # Multiple configs (preprocessor/postprocessor)
+        preprocessor = DataProcessorPipeline.from_pretrained(
+            "model", config_filename="policy_preprocessor.json"
+        )
+        postprocessor = DataProcessorPipeline.from_pretrained(
+            "model", config_filename="policy_postprocessor.json"
+        )
+        ```
+
+        **Override System**:
+        - **Key matching**: Use registry names or class names as override keys
+        - **Config merging**: User overrides take precedence over saved config
+        - **Validation**: Ensure all override keys match actual steps (catch typos)
+        - **Example**: overrides={"NormalizeStep": {"device": "cuda"}}
+
+        Args:
+            pretrained_model_name_or_path: The identifier of the repository on the Hugging Face Hub,
+                a path to a local directory, or a path to a single config file.
+            config_filename: The name of the pipeline's JSON configuration file. Always required
+                to prevent ambiguity when multiple configs exist (e.g., preprocessor vs postprocessor).
+            force_download: Whether to force (re)downloading the files.
+            resume_download: Whether to resume a previously interrupted download.
+            proxies: A dictionary of proxy servers to use.
+            token: The token to use as HTTP bearer authorization for private Hub repositories.
+            cache_dir: The path to a specific cache folder to store downloaded files.
+            local_files_only: If True, avoid downloading files from the Hub.
+            revision: The specific model version to use (e.g., a branch name, tag name, or commit id).
+            overrides: A dictionary to override the configuration of specific steps. Keys should
+                match the step's class name or registry name.
+            to_transition: A custom function to convert input data to `EnvTransition`.
+            to_output: A custom function to convert the final `EnvTransition` to the output format.
+            **kwargs: Additional arguments (not used).
+
+        Returns:
+            An instance of `DataProcessorPipeline` loaded with the specified configuration and state.
+
+        Raises:
+            FileNotFoundError: If the config file cannot be found.
+            ValueError: If configuration is ambiguous or instantiation fails.
+            ImportError: If a step's class cannot be imported.
+            KeyError: If an override key doesn't match any step in the pipeline.
+            ProcessorMigrationError: If the model requires migration to processor format.
+        """
+        model_id = str(pretrained_model_name_or_path)
+        hub_download_kwargs = {
+            "force_download": force_download,
+            "resume_download": resume_download,
+            "proxies": proxies,
+            "token": token,
+            "cache_dir": cache_dir,
+            "local_files_only": local_files_only,
+            "revision": revision,
+        }
+
+        # 1. Load configuration using simplified 3-way logic
+        loaded_config, base_path = cls._load_config(model_id, config_filename, hub_download_kwargs)
+
+        # 2. Validate configuration and handle migration
+        cls._validate_loaded_config(model_id, loaded_config, config_filename)
+
+        # 3. Build steps with overrides
+        steps, validated_overrides = cls._build_steps_with_overrides(
+            loaded_config, overrides or {}, model_id, base_path, hub_download_kwargs
+        )
+
+        # 4. Validate that all overrides were used
+        cls._validate_overrides_used(validated_overrides, loaded_config)
+
+        # 5. Construct and return the final pipeline instance
+        return cls(
+            steps=steps,
+            name=loaded_config.get("name", "DataProcessorPipeline"),
+            to_transition=to_transition or cast(Callable[[TInput], EnvTransition], batch_to_transition),
+            to_output=to_output or cast(Callable[[EnvTransition], TOutput], transition_to_batch),
+        )
+
+    @classmethod
+    def _load_config(
+        cls,
+        model_id: str,
+        config_filename: str,
+        hub_download_kwargs: dict[str, Any],
+    ) -> tuple[dict[str, Any], Path]:
+        """Load configuration from local file or Hugging Face Hub.
+
+        This method implements a super-simplified 3-way loading strategy:
+
+        1. **Local directory**: Load config_filename from directory
+           - Example: model_id="/models/my_model", config_filename="processor.json"
+           - Loads: "/models/my_model/processor.json"
+
+        2. **Single file**: Load file directly (ignore config_filename)
+           - Example: model_id="/models/my_model/processor.json"
+           - Loads: "/models/my_model/processor.json" (config_filename ignored)
+
+        3. **Hub repository**: Download config_filename from Hub
+           - Example: model_id="user/repo", config_filename="processor.json"
+           - Downloads and loads: config_filename from Hub repo
+
+        **Benefits of Explicit config_filename**:
+        - No auto-detection complexity or edge cases
+        - No risk of loading wrong config (preprocessor vs postprocessor)
+        - Consistent behavior across local and Hub usage
+        - Clear, predictable errors
+
+        Args:
+            model_id: The model identifier (Hub repo ID, local directory, or file path)
+            config_filename: The explicit config filename to load (always required)
+            hub_download_kwargs: Parameters for hf_hub_download (tokens, cache, etc.)
+
+        Returns:
+            Tuple of (loaded_config, base_path)
+            - loaded_config: Parsed JSON config dict (always loaded, never None)
+            - base_path: Directory containing config file (for state file resolution)
+
+        Raises:
+            FileNotFoundError: If config file cannot be found locally or on Hub
+        """
+        model_path = Path(model_id)
+
+        if model_path.is_dir():
+            # Directory: load specified config from directory
+            config_path = model_path / config_filename
+            if not config_path.exists():
+                # Check for migration before giving clear error
+                if cls._should_suggest_migration(model_path):
+                    cls._suggest_processor_migration(model_id, f"Config file '{config_filename}' not found")
+                raise FileNotFoundError(
+                    f"Config file '{config_filename}' not found in directory '{model_id}'"
+                )
+
+            with open(config_path) as f:
+                return json.load(f), model_path
+
+        elif model_path.is_file():
+            # File: load file directly (config_filename is ignored for single files)
+            with open(model_path) as f:
+                return json.load(f), model_path.parent
+
+        else:
+            # Hub: download specified config
+            try:
+                config_path = hf_hub_download(
+                    repo_id=model_id,
+                    filename=config_filename,
+                    repo_type="model",
+                    **hub_download_kwargs,
+                )
+
+                with open(config_path) as f:
+                    return json.load(f), Path(config_path).parent
+
+            except Exception as e:
+                raise FileNotFoundError(
+                    f"Could not find '{config_filename}' on the HuggingFace Hub at '{model_id}'"
+                ) from e
+
+    @classmethod
+    def _validate_loaded_config(
+        cls, model_id: str, loaded_config: dict[str, Any], config_filename: str
+    ) -> None:
+        """Validate that a config was loaded and is a valid processor config.
+
+        This method validates processor config format with intelligent migration detection:
+
+        **Config Format Validation**:
+        - Use _is_processor_config() to validate structure
+          - Must have "steps" field with list of step configurations
+          - Each step needs "class" or "registry_name"
+        - If validation fails AND local directory: Check for migration need
+        - If migration needed: Raise ProcessorMigrationError with command
+        - If no migration: Raise ValueError with helpful error message
+
+        **Migration Detection Logic**:
+        - Only triggered for local directories (not Hub repos)
+        - Analyzes all JSON files in directory to detect old LeRobot models
+        - Provides exact migration command with model path
+
+        Args:
+            model_id: The model identifier (used for migration detection)
+            loaded_config: The loaded config dictionary (guaranteed non-None)
+            config_filename: The config filename that was loaded (for error messages)
+
+        Raises:
+            ValueError: If config format is invalid
+            ProcessorMigrationError: If model needs migration to processor format
+        """
+        # Validate that this is actually a processor config
+        if not cls._is_processor_config(loaded_config):
+            if Path(model_id).is_dir() and cls._should_suggest_migration(Path(model_id)):
+                cls._suggest_processor_migration(
+                    model_id,
+                    f"Config file '{config_filename}' is not a valid processor configuration",
+                )
+            raise ValueError(
+                f"Config file '{config_filename}' is not a valid processor configuration. "
+                f"Expected a config with 'steps' field, but got: {list(loaded_config.keys())}"
+            )
+
+    @classmethod
+    def _build_steps_with_overrides(
+        cls,
+        loaded_config: dict[str, Any],
+        overrides: dict[str, Any],
+        model_id: str,
+        base_path: Path | None,
+        hub_download_kwargs: dict[str, Any],
+    ) -> tuple[list[ProcessorStep], set[str]]:
+        """Build all processor steps with overrides and state loading.
+
+        This method orchestrates the complete step construction pipeline:
+
+        **For each step in loaded_config["steps"]**:
+
+        1. **Class Resolution** (via _resolve_step_class):
+           - **If "registry_name" exists**: Look up in ProcessorStepRegistry
+             Example: {"registry_name": "normalize_step"} -> Get registered class
+           - **Else use "class" field**: Dynamic import from full module path
+             Example: {"class": "lerobot.processor.normalize.NormalizeStep"}
+           - **Result**: (step_class, step_key) where step_key is used for overrides
+
+        2. **Step Instantiation** (via _instantiate_step):
+           - **Merge configs**: saved_config + user_overrides
+           - **Override priority**: User overrides take precedence over saved config
+           - **Example**: saved={"mean": 0.0}, override={"mean": 1.0} -> final={"mean": 1.0}
+           - **Result**: Instantiated ProcessorStep object
+
+        3. **State Loading** (via _load_step_state):
+           - **If step has "state_file"**: Load tensor state from .safetensors
+           - **Local first**: Check base_path/state_file.safetensors
+           - **Hub fallback**: Download state file if not found locally
+           - **Optional**: Only load if step has load_state_dict method
+
+        4. **Override Tracking**:
+           - **Track used overrides**: Remove step_key from remaining set
+           - **Purpose**: Validate all user overrides were applied (detect typos)
+
+        **Error Handling**:
+        - Class resolution errors -> ImportError with helpful message
+        - Instantiation errors -> ValueError with config details
+        - State loading errors -> Propagated from load_state_dict
+
+        Args:
+            loaded_config: The loaded processor configuration (must have "steps" field)
+            overrides: User-provided parameter overrides (keyed by class/registry name)
+            model_id: The model identifier (needed for Hub state file downloads)
+            base_path: Local directory path for finding state files
+            hub_download_kwargs: Parameters for hf_hub_download (tokens, cache, etc.)
+
+        Returns:
+            Tuple of (instantiated_steps_list, unused_override_keys)
+            - instantiated_steps_list: List of ready-to-use ProcessorStep instances
+            - unused_override_keys: Override keys that didn't match any step (for validation)
+
+        Raises:
+            ImportError: If a step class cannot be imported or found in registry
+            ValueError: If a step cannot be instantiated with its configuration
+        """
+        steps: list[ProcessorStep] = []
+        override_keys = set(overrides.keys())
+
+        for step_entry in loaded_config["steps"]:
+            # 1. Get step class and key
+            step_class, step_key = cls._resolve_step_class(step_entry)
+
+            # 2. Instantiate step with overrides
+            step_instance = cls._instantiate_step(step_entry, step_class, step_key, overrides)
+
+            # 3. Load step state if available
+            cls._load_step_state(step_instance, step_entry, model_id, base_path, hub_download_kwargs)
+
+            # 4. Track used overrides
+            if step_key in override_keys:
+                override_keys.discard(step_key)
+
+            steps.append(step_instance)
+
+        return steps, override_keys
+
+    @classmethod
+    def _resolve_step_class(cls, step_entry: dict[str, Any]) -> tuple[type[ProcessorStep], str]:
+        """Resolve step class from registry or import path.
+
+        This method implements a two-tier resolution strategy:
+
+        **Tier 1: Registry-based resolution** (preferred):
+        - **If "registry_name" in step_entry**: Look up in ProcessorStepRegistry
+          - **Advantage**: Faster, no imports needed, guaranteed compatibility
+          - **Example**: {"registry_name": "normalize_step"} -> Get pre-registered class
+          - **Error**: KeyError if registry_name not found -> Convert to ImportError
+
+        **Tier 2: Dynamic import fallback**:
+        - **Else use "class" field**: Full module.ClassName import path
+          - **Process**: Split "module.path.ClassName" into module + class parts
+          - **Import**: Use importlib.import_module() + getattr()
+          - **Example**: "lerobot.processor.normalize.NormalizeStep"
+            a. Import module: "lerobot.processor.normalize"
+            b. Get class: getattr(module, "NormalizeStep")
+          - **step_key**: Use class_name ("NormalizeStep") for overrides
+
+        **Override Key Strategy**:
+        - Registry steps: Use registry_name ("normalize_step")
+        - Import steps: Use class_name ("NormalizeStep")
+        - This allows users to override with: {"normalize_step": {...}} or {"NormalizeStep": {...}}
+
+        **Error Handling**:
+        - Registry KeyError -> ImportError with registry context
+        - Import/Attribute errors -> ImportError with helpful suggestions
+        - All errors include troubleshooting guidance
+
+        Args:
+            step_entry: The step configuration dictionary (must have "registry_name" or "class")
+
+        Returns:
+            Tuple of (step_class, step_key)
+            - step_class: The resolved ProcessorStep class (ready for instantiation)
+            - step_key: The key used for user overrides (registry_name or class_name)
+
+        Raises:
+            ImportError: If step class cannot be loaded from registry or import path
+        """
+        if "registry_name" in step_entry:
+            try:
+                step_class = ProcessorStepRegistry.get(step_entry["registry_name"])
+                return step_class, step_entry["registry_name"]
+            except KeyError as e:
+                raise ImportError(f"Failed to load processor step from registry. {str(e)}") from e
+        else:
+            # Fallback to dynamic import using the full class path
+            full_class_path = step_entry["class"]
+            module_path, class_name = full_class_path.rsplit(".", 1)
+
+            try:
+                module = importlib.import_module(module_path)
+                step_class = getattr(module, class_name)
+                return step_class, class_name
+            except (ImportError, AttributeError) as e:
+                raise ImportError(
+                    f"Failed to load processor step '{full_class_path}'. "
+                    f"Make sure the module '{module_path}' is installed and contains class '{class_name}'. "
+                    f"Consider registering the step using @ProcessorStepRegistry.register() for better portability. "
+                    f"Error: {str(e)}"
+                ) from e
+
+    @classmethod
+    def _instantiate_step(
+        cls,
+        step_entry: dict[str, Any],
+        step_class: type[ProcessorStep],
+        step_key: str,
+        overrides: dict[str, Any],
+    ) -> ProcessorStep:
+        """Instantiate a single processor step with config overrides.
+
+        This method handles the configuration merging and instantiation logic:
+
+        **Configuration Merging Strategy**:
+        1. **Extract saved config**: Get step_entry.get("config", {}) from saved pipeline
+           - Example: {"config": {"mean": 0.0, "std": 1.0}}
+        2. **Extract user overrides**: Get overrides.get(step_key, {}) for this step
+           - Example: overrides = {"NormalizeStep": {"mean": 2.0, "device": "cuda"}}
+        3. **Merge with priority**: {**saved_cfg, **step_overrides}
+           - **Override priority**: User values override saved values
+           - **Result**: {"mean": 2.0, "std": 1.0, "device": "cuda"}
+
+        **Instantiation Process**:
+        - **Call constructor**: step_class(**merged_cfg)
+        - **Example**: NormalizeStep(mean=2.0, std=1.0, device="cuda")
+
+        **Error Handling**:
+        - **Any exception during instantiation**: Convert to ValueError
+        - **Include context**: step name, attempted config, original error
+        - **Purpose**: Help users debug configuration issues
+        - **Common causes**:
+          a. Invalid parameter types (str instead of float)
+          b. Missing required parameters
+          c. Incompatible parameter combinations
+
+        Args:
+            step_entry: The step configuration from saved config (contains "config" dict)
+            step_class: The step class to instantiate (already resolved)
+            step_key: The key used for overrides ("registry_name" or class name)
+            overrides: User-provided parameter overrides (keyed by step_key)
+
+        Returns:
+            The instantiated processor step (ready for use)
+
+        Raises:
+            ValueError: If step cannot be instantiated, with detailed error context
+        """
+        try:
+            saved_cfg = step_entry.get("config", {})
+            step_overrides = overrides.get(step_key, {})
+            merged_cfg = {**saved_cfg, **step_overrides}
+            return step_class(**merged_cfg)
+        except Exception as e:
+            step_name = step_entry.get("registry_name", step_entry.get("class", "Unknown"))
+            raise ValueError(
+                f"Failed to instantiate processor step '{step_name}' with config: {step_entry.get('config', {})}. "
+                f"Error: {str(e)}"
+            ) from e
+
+    @classmethod
+    def _load_step_state(
+        cls,
+        step_instance: ProcessorStep,
+        step_entry: dict[str, Any],
+        model_id: str,
+        base_path: Path | None,
+        hub_download_kwargs: dict[str, Any],
+    ) -> None:
+        """Load state dictionary for a processor step if available.
+
+        This method implements conditional state loading with local/Hub fallback:
+
+        **Precondition Checks** (early return if not met):
+        1. **"state_file" in step_entry**: Step config specifies a state file
+           - **If missing**: Step has no saved state (e.g., stateless transforms)
+        2. **hasattr(step_instance, "load_state_dict")**: Step supports state loading
+           - **If missing**: Step doesn't implement state loading (rare)
+
+        **State File Resolution Strategy**:
+        1. **Local file priority**: Check base_path/state_filename exists
+           - **Advantage**: Faster, no network calls
+           - **Example**: "/models/my_model/normalize_step_0.safetensors"
+           - **Use case**: Loading from local saved model directory
+
+        2. **Hub download fallback**: Download state file from repository
+           - **When triggered**: Local file not found or base_path is None
+           - **Process**: Use hf_hub_download with same parameters as config
+           - **Example**: Download "normalize_step_0.safetensors" from "user/repo"
+           - **Result**: Downloaded to local cache, path returned
+
+        **State Loading Process**:
+        - **Load tensors**: Use safetensors.torch.load_file()
+        - **Apply to step**: Call step_instance.load_state_dict(tensor_dict)
+        - **In-place modification**: Updates step's internal tensor state
+
+        **Common state file examples**:
+        - "normalize_step_0.safetensors" - normalization statistics
+        - "custom_step_1.safetensors" - learned parameters
+        - "tokenizer_step_2.safetensors" - vocabulary embeddings
+
+        Args:
+            step_instance: The step instance to load state into (must have load_state_dict)
+            step_entry: The step configuration dictionary (may contain "state_file")
+            model_id: The model identifier (used for Hub downloads if needed)
+            base_path: Local directory path for finding state files (None for Hub-only)
+            hub_download_kwargs: Parameters for hf_hub_download (tokens, cache, etc.)
+
+        Note:
+            This method modifies step_instance in-place and returns None.
+            If state loading fails, exceptions from load_state_dict propagate.
+        """
+        if "state_file" not in step_entry or not hasattr(step_instance, "load_state_dict"):
+            return
+
+        state_filename = step_entry["state_file"]
+
+        # Try local file first
+        if base_path and (base_path / state_filename).exists():
+            state_path = str(base_path / state_filename)
+        else:
+            # Download from Hub
+            state_path = hf_hub_download(
+                repo_id=model_id,
+                filename=state_filename,
+                repo_type="model",
+                **hub_download_kwargs,
+            )
+
+        step_instance.load_state_dict(load_file(state_path))
+
+    @classmethod
+    def _validate_overrides_used(
+        cls, remaining_override_keys: set[str], loaded_config: dict[str, Any]
+    ) -> None:
+        """Validate that all provided overrides were used.
+
+        This method ensures user overrides are valid to catch typos and configuration errors:
+
+        **Validation Logic**:
+        1. **If remaining_override_keys is empty**: All overrides were used -> Success
+           - **Early return**: No validation needed
+           - **Normal case**: User provided correct override keys
+
+        2. **If remaining_override_keys has entries**: Some overrides unused -> Error
+           - **Root cause**: User provided keys that don't match any step
+           - **Common issues**:
+             a. Typos in step names ("NormalizStep" vs "NormalizeStep")
+             b. Using wrong key type (class name vs registry name)
+             c. Step doesn't exist in saved pipeline
+
+        **Helpful Error Generation**:
+        - **Extract available keys**: Build list of valid override keys from config
+          a. **Registry steps**: Use "registry_name" directly
+          b. **Import steps**: Extract class name from "class" field
+          - Example: "lerobot.processor.normalize.NormalizeStep" -> "NormalizeStep"
+        - **Error message includes**:
+          a. Invalid keys provided by user
+          b. List of valid keys they can use
+          c. Guidance about registry vs class names
+
+        **Override Key Resolution Rules**:
+        - Steps with "registry_name": Use registry_name for overrides
+        - Steps with "class": Use final class name for overrides
+        - Users must match these exact keys in their overrides dict
+
+        Args:
+            remaining_override_keys: Override keys that weren't matched to any step
+            loaded_config: The loaded processor configuration (contains "steps" list)
+
+        Raises:
+            KeyError: If any override keys were not used, with helpful error message
+        """
+        if not remaining_override_keys:
+            return
+
+        available_keys = [
+            step.get("registry_name") or step["class"].rsplit(".", 1)[1] for step in loaded_config["steps"]
+        ]
+
+        raise KeyError(
+            f"Override keys {list(remaining_override_keys)} do not match any step in the saved configuration. "
+            f"Available step keys: {available_keys}. "
+            f"Make sure override keys match exact step class names or registry names."
+        )
+
+    @classmethod
+    def _should_suggest_migration(cls, model_path: Path) -> bool:
+        """Check if directory has JSON files but no processor configs.
+
+        This method implements smart migration detection to avoid false positives:
+
+        **Decision Logic**:
+        1. **No JSON files found**: Return False
+           - **Reason**: Empty directory or only non-config files
+           - **Example**: Directory with only .safetensors, .md files
+           - **Action**: No migration needed
+
+        2. **JSON files exist**: Analyze each file
+           - **Goal**: Determine if ANY file is a valid processor config
+           - **Process**:
+             a. Try to parse each .json file
+             b. Skip files with JSON parse errors (malformed)
+             c. Check if parsed config passes _is_processor_config()
+           - **If ANY valid processor found**: Return False (no migration)
+           - **If NO valid processors found**: Return True (migration needed)
+
+        **Examples**:
+        - **No migration**: ["processor.json", "config.json"] where processor.json is valid
+        - **Migration needed**: ["config.json", "train.json"] where both are model configs
+        - **No migration**: [] (empty directory)
+        - **Migration needed**: ["old_model_config.json"] with old LeRobot format
+
+        **Why this works**:
+        - **Precise detection**: Only suggests migration for actual old LeRobot models
+        - **Avoids false positives**: Won't trigger on other HuggingFace model types
+        - **Graceful handling**: Ignores malformed JSON files
+
+        Args:
+            model_path: Path to local directory to analyze
+
+        Returns:
+            True if directory has JSON configs but none are processor configs (migration needed)
+            False if no JSON files or at least one valid processor config exists
+        """
+        json_files = list(model_path.glob("*.json"))
+        if len(json_files) == 0:
+            return False
+
+        # Check if any JSON file is a processor config
+        for json_file in json_files:
+            try:
+                with open(json_file) as f:
+                    config = json.load(f)
+
+                if cls._is_processor_config(config):
+                    return False  # Found at least one processor config, no migration needed
+
+            except (json.JSONDecodeError, OSError):
+                # Skip files that can't be parsed as JSON
+                continue
+
+        # Have JSON files but no processor configs - suggest migration
+        return True
+
+    @classmethod
+    def _is_processor_config(cls, config: dict) -> bool:
+        """Check if config follows DataProcessorPipeline format.
+
+        This method validates the processor configuration structure:
+
+        **Required Structure Validation**:
+        1. **"steps" field existence**: Must have top-level "steps" key
+           - **If missing**: Not a processor config (e.g., model config, train config)
+           - **Example invalid**: {"type": "act", "hidden_dim": 256}
+
+        2. **"steps" field type**: Must be a list, not other types
+           - **If not list**: Invalid format
+           - **Example invalid**: {"steps": "some_string"} or {"steps": {"key": "value"}}
+
+        3. **Empty steps validation**: Empty list is valid
+           - **If len(steps) == 0**: Return True immediately
+           - **Use case**: Empty processor pipeline (no-op)
+           - **Example valid**: {"name": "EmptyProcessor", "steps": []}
+
+        **Individual Step Validation** (for non-empty steps):
+        For each step in the steps list:
+        1. **Step type**: Must be a dictionary
+           - **If not dict**: Invalid step format
+           - **Example invalid**: ["string_step", 123, true]
+
+        2. **Step identifier**: Must have either "class" OR "registry_name"
+           - **"registry_name"**: Registered step (preferred)
+             Example: {"registry_name": "normalize_step", "config": {...}}
+           - **"class"**: Full import path
+             Example: {"class": "lerobot.processor.normalize.NormalizeStep"}
+           - **If neither**: Invalid step (can't resolve class)
+           - **If both**: Also valid (registry_name takes precedence)
+
+        **Valid Processor Config Examples**:
+        - {"steps": []} - Empty processor
+        - {"steps": [{"registry_name": "normalize"}]} - Registry step
+        - {"steps": [{"class": "my.module.Step"}]} - Import step
+        - {"name": "MyProcessor", "steps": [...]} - With name
+
+        **Invalid Config Examples**:
+        - {"type": "act"} - Missing "steps"
+        - {"steps": "normalize"} - Steps not a list
+        - {"steps": [{}]} - Step missing class/registry_name
+        - {"steps": ["string"]} - Step not a dict
+
+        Args:
+            config: The configuration dictionary to validate
+
+        Returns:
+            True if config follows valid DataProcessorPipeline format, False otherwise
+        """
+        # Must have a "steps" field with a list of step configurations
+        if not isinstance(config.get("steps"), list):
+            return False
+
+        steps = config["steps"]
+        if len(steps) == 0:
+            return True  # Empty processor is valid
+
+        # Each step must be a dict with either "class" or "registry_name"
+        for step in steps:
+            if not isinstance(step, dict):
+                return False
+            if not ("class" in step or "registry_name" in step):
+                return False
+
+        return True
+
+    @classmethod
+    def _suggest_processor_migration(cls, model_path: str | Path, original_error: str) -> None:
+        """Raise migration error when we detect JSON files but no processor configs.
+
+        This method is called when migration detection determines that a model
+        directory contains configuration files but none are valid processor configs.
+        This typically indicates an old LeRobot model that needs migration.
+
+        **When this is called**:
+        - User tries to load DataProcessorPipeline from local directory
+        - Directory contains JSON configuration files
+        - None of the JSON files follow processor config format
+        - _should_suggest_migration() returned True
+
+        **Migration Command Generation**:
+        - Constructs exact command user needs to run
+        - Uses the migration script: migrate_policy_normalization.py
+        - Includes the model path automatically
+        - Example: "python src/lerobot/processor/migrate_policy_normalization.py --pretrained-path /models/old_model"
+
+        **Error Structure**:
+        - **Always raises**: ProcessorMigrationError (never returns)
+        - **Includes**: model_path, migration_command, original_error
+        - **Purpose**: Force user attention to migration need
+        - **User experience**: Clear actionable error with exact command to run
+
+        **Migration Process**:
+        The suggested command will:
+        1. Extract normalization stats from old model
+        2. Create new processor configs (preprocessor + postprocessor)
+        3. Remove normalization layers from model
+        4. Save migrated model with processor pipeline
+
+        Args:
+            model_path: Path to the model directory needing migration
+            original_error: The error that triggered migration detection (for context)
+
+        Raises:
+            ProcessorMigrationError: Always raised (this method never returns normally)
+        """
+        migration_command = (
+            f"python src/lerobot/processor/migrate_policy_normalization.py --pretrained-path {model_path}"
+        )
+
+        raise ProcessorMigrationError(model_path, migration_command, original_error)
+
+    def __len__(self) -> int:
+        """Returns the number of steps in the pipeline."""
+        return len(self.steps)
+
+    def __getitem__(self, idx: int | slice) -> ProcessorStep | DataProcessorPipeline[TInput, TOutput]:
+        """Retrieves a step or a sub-pipeline by index or slice.
+
+        Args:
+            idx: An integer index or a slice object.
+
+        Returns:
+            A `ProcessorStep` if `idx` is an integer, or a new `DataProcessorPipeline`
+            containing the sliced steps.
+        """
+        if isinstance(idx, slice):
+            # Return a new pipeline instance with the sliced steps.
+            return DataProcessorPipeline(
+                steps=self.steps[idx],
+                name=self.name,
+                to_transition=self.to_transition,
+                to_output=self.to_output,
+                before_step_hooks=self.before_step_hooks.copy(),
+                after_step_hooks=self.after_step_hooks.copy(),
+            )
+        return self.steps[idx]
+
+    def register_before_step_hook(self, fn: Callable[[int, EnvTransition], None]):
+        """Registers a function to be called before each step.
+
+        Args:
+            fn: A callable that accepts the step index and the current transition.
+        """
+        self.before_step_hooks.append(fn)
+
+    def unregister_before_step_hook(self, fn: Callable[[int, EnvTransition], None]):
+        """Unregisters a 'before_step' hook.
+
+        Args:
+            fn: The exact function object that was previously registered.
+
+        Raises:
+            ValueError: If the hook is not found in the list.
+        """
+        try:
+            self.before_step_hooks.remove(fn)
+        except ValueError:
+            raise ValueError(
+                f"Hook {fn} not found in before_step_hooks. Make sure to pass the exact same function reference."
+            ) from None
+
+    def register_after_step_hook(self, fn: Callable[[int, EnvTransition], None]):
+        """Registers a function to be called after each step.
+
+        Args:
+            fn: A callable that accepts the step index and the current transition.
+        """
+        self.after_step_hooks.append(fn)
+
+    def unregister_after_step_hook(self, fn: Callable[[int, EnvTransition], None]):
+        """Unregisters an 'after_step' hook.
+
+        Args:
+            fn: The exact function object that was previously registered.
+
+        Raises:
+            ValueError: If the hook is not found in the list.
+        """
+        try:
+            self.after_step_hooks.remove(fn)
+        except ValueError:
+            raise ValueError(
+                f"Hook {fn} not found in after_step_hooks. Make sure to pass the exact same function reference."
+            ) from None
+
+    def reset(self):
+        """Resets the state of all stateful steps in the pipeline."""
+        for step in self.steps:
+            if hasattr(step, "reset"):
+                step.reset()
+
+    def __repr__(self) -> str:
+        """Provides a concise string representation of the pipeline."""
+        step_names = [step.__class__.__name__ for step in self.steps]
+
+        if not step_names:
+            steps_repr = "steps=0: []"
+        elif len(step_names) <= 3:
+            steps_repr = f"steps={len(step_names)}: [{', '.join(step_names)}]"
+        else:
+            # For long pipelines, show the first, second, and last steps.
+            displayed = f"{step_names[0]}, {step_names[1]}, ..., {step_names[-1]}"
+            steps_repr = f"steps={len(step_names)}: [{displayed}]"
+
+        parts = [f"name='{self.name}'", steps_repr]
+
+        return f"DataProcessorPipeline({', '.join(parts)})"
+
+    def __post_init__(self):
+        """Validates that all provided steps are instances of `ProcessorStep`."""
+        for i, step in enumerate(self.steps):
+            if not isinstance(step, ProcessorStep):
+                raise TypeError(f"Step {i} ({type(step).__name__}) must inherit from ProcessorStep")
+
+    def transform_features(
+        self, initial_features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """Applies feature transformations from all steps sequentially.
+
+        This method propagates a feature description dictionary through each step's
+        `transform_features` method, allowing the pipeline to statically determine
+        the output feature specification without processing any real data.
+
+        Args:
+            initial_features: A dictionary describing the initial features.
+
+        Returns:
+            The final feature description after all transformations.
+        """
+        features: dict[PipelineFeatureType, dict[str, PolicyFeature]] = deepcopy(initial_features)
+
+        for _, step in enumerate(self.steps):
+            out = step.transform_features(features)
+            features = out
+        return features
+
+    # Convenience methods for processing individual parts of a transition.
+    def process_observation(self, observation: RobotObservation) -> RobotObservation:
+        """Processes only the observation part of a transition through the pipeline.
+
+        Args:
+            observation: The observation dictionary.
+
+        Returns:
+            The processed observation dictionary.
+        """
+        transition: EnvTransition = create_transition(observation=observation)
+        transformed_transition = self._forward(transition)
+        return transformed_transition[TransitionKey.OBSERVATION]
+
+    def process_action(
+        self, action: PolicyAction | RobotAction | EnvAction
+    ) -> PolicyAction | RobotAction | EnvAction:
+        """Processes only the action part of a transition through the pipeline.
+
+        Args:
+            action: The action data.
+
+        Returns:
+            The processed action.
+        """
+        transition: EnvTransition = create_transition(action=action)
+        transformed_transition = self._forward(transition)
+        return transformed_transition[TransitionKey.ACTION]
+
+    def process_reward(self, reward: float | torch.Tensor) -> float | torch.Tensor:
+        """Processes only the reward part of a transition through the pipeline.
+
+        Args:
+            reward: The reward value.
+
+        Returns:
+            The processed reward.
+        """
+        transition: EnvTransition = create_transition(reward=reward)
+        transformed_transition = self._forward(transition)
+        return transformed_transition[TransitionKey.REWARD]
+
+    def process_done(self, done: bool | torch.Tensor) -> bool | torch.Tensor:
+        """Processes only the done flag of a transition through the pipeline.
+
+        Args:
+            done: The done flag.
+
+        Returns:
+            The processed done flag.
+        """
+        transition: EnvTransition = create_transition(done=done)
+        transformed_transition = self._forward(transition)
+        return transformed_transition[TransitionKey.DONE]
+
+    def process_truncated(self, truncated: bool | torch.Tensor) -> bool | torch.Tensor:
+        """Processes only the truncated flag of a transition through the pipeline.
+
+        Args:
+            truncated: The truncated flag.
+
+        Returns:
+            The processed truncated flag.
+        """
+        transition: EnvTransition = create_transition(truncated=truncated)
+        transformed_transition = self._forward(transition)
+        return transformed_transition[TransitionKey.TRUNCATED]
+
+    def process_info(self, info: dict[str, Any]) -> dict[str, Any]:
+        """Processes only the info dictionary of a transition through the pipeline.
+
+        Args:
+            info: The info dictionary.
+
+        Returns:
+            The processed info dictionary.
+        """
+        transition: EnvTransition = create_transition(info=info)
+        transformed_transition = self._forward(transition)
+        return transformed_transition[TransitionKey.INFO]
+
+    def process_complementary_data(self, complementary_data: dict[str, Any]) -> dict[str, Any]:
+        """Processes only the complementary data part of a transition through the pipeline.
+
+        Args:
+            complementary_data: The complementary data dictionary.
+
+        Returns:
+            The processed complementary data dictionary.
+        """
+        transition: EnvTransition = create_transition(complementary_data=complementary_data)
+        transformed_transition = self._forward(transition)
+        return transformed_transition[TransitionKey.COMPLEMENTARY_DATA]
+
+
+# Type aliases for semantic clarity.
+RobotProcessorPipeline = DataProcessorPipeline[TInput, TOutput]
+PolicyProcessorPipeline = DataProcessorPipeline[TInput, TOutput]
+
+
+class ObservationProcessorStep(ProcessorStep, ABC):
+    """An abstract `ProcessorStep` that specifically targets the observation in a transition."""
+
+    @abstractmethod
+    def observation(self, observation: RobotObservation) -> RobotObservation:
+        """Processes an observation dictionary. Subclasses must implement this method.
+
+        Args:
+            observation: The input observation dictionary from the transition.
+
+        Returns:
+            The processed observation dictionary.
+        """
+        ...
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Applies the `observation` method to the transition's observation."""
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        observation = new_transition.get(TransitionKey.OBSERVATION)
+        if observation is None or not isinstance(observation, dict):
+            raise ValueError("ObservationProcessorStep requires an observation in the transition.")
+
+        processed_observation = self.observation(observation.copy())
+        new_transition[TransitionKey.OBSERVATION] = processed_observation
+        return new_transition
+
+
+class ActionProcessorStep(ProcessorStep, ABC):
+    """An abstract `ProcessorStep` that specifically targets the action in a transition."""
+
+    @abstractmethod
+    def action(
+        self, action: PolicyAction | RobotAction | EnvAction
+    ) -> PolicyAction | RobotAction | EnvAction:
+        """Processes an action. Subclasses must implement this method.
+
+        Args:
+            action: The input action from the transition.
+
+        Returns:
+            The processed action.
+        """
+        ...
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Applies the `action` method to the transition's action."""
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        action = new_transition.get(TransitionKey.ACTION)
+        if action is None:
+            raise ValueError("ActionProcessorStep requires an action in the transition.")
+
+        processed_action = self.action(action)
+        new_transition[TransitionKey.ACTION] = processed_action
+        return new_transition
+
+
+class RobotActionProcessorStep(ProcessorStep, ABC):
+    """An abstract `ProcessorStep` for processing a `RobotAction` (a dictionary)."""
+
+    @abstractmethod
+    def action(self, action: RobotAction) -> RobotAction:
+        """Processes a `RobotAction`. Subclasses must implement this method.
+
+        Args:
+            action: The input `RobotAction` dictionary.
+
+        Returns:
+            The processed `RobotAction`.
+        """
+        ...
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Applies the `action` method to the transition's action, ensuring it's a `RobotAction`."""
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        action = new_transition.get(TransitionKey.ACTION)
+        if action is None or not isinstance(action, dict):
+            raise ValueError(f"Action should be a RobotAction type (dict), but got {type(action)}")
+
+        processed_action = self.action(action.copy())
+        new_transition[TransitionKey.ACTION] = processed_action
+        return new_transition
+
+
+class PolicyActionProcessorStep(ProcessorStep, ABC):
+    """An abstract `ProcessorStep` for processing a `PolicyAction` (a tensor or dict of tensors)."""
+
+    @abstractmethod
+    def action(self, action: PolicyAction) -> PolicyAction:
+        """Processes a `PolicyAction`. Subclasses must implement this method.
+
+        Args:
+            action: The input `PolicyAction`.
+
+        Returns:
+            The processed `PolicyAction`.
+        """
+        ...
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Applies the `action` method to the transition's action, ensuring it's a `PolicyAction`."""
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        action = new_transition.get(TransitionKey.ACTION)
+        if not isinstance(action, PolicyAction):
+            raise ValueError(f"Action should be a PolicyAction type (tensor), but got {type(action)}")
+
+        processed_action = self.action(action)
+        new_transition[TransitionKey.ACTION] = processed_action
+        return new_transition
+
+
+class RewardProcessorStep(ProcessorStep, ABC):
+    """An abstract `ProcessorStep` that specifically targets the reward in a transition."""
+
+    @abstractmethod
+    def reward(self, reward) -> float | torch.Tensor:
+        """Processes a reward. Subclasses must implement this method.
+
+        Args:
+            reward: The input reward from the transition.
+
+        Returns:
+            The processed reward.
+        """
+        ...
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Applies the `reward` method to the transition's reward."""
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        reward = new_transition.get(TransitionKey.REWARD)
+        if reward is None:
+            raise ValueError("RewardProcessorStep requires a reward in the transition.")
+
+        processed_reward = self.reward(reward)
+        new_transition[TransitionKey.REWARD] = processed_reward
+        return new_transition
+
+
+class DoneProcessorStep(ProcessorStep, ABC):
+    """An abstract `ProcessorStep` that specifically targets the 'done' flag in a transition."""
+
+    @abstractmethod
+    def done(self, done) -> bool | torch.Tensor:
+        """Processes a 'done' flag. Subclasses must implement this method.
+
+        Args:
+            done: The input 'done' flag from the transition.
+
+        Returns:
+            The processed 'done' flag.
+        """
+        ...
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Applies the `done` method to the transition's 'done' flag."""
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        done = new_transition.get(TransitionKey.DONE)
+        if done is None:
+            raise ValueError("DoneProcessorStep requires a done flag in the transition.")
+
+        processed_done = self.done(done)
+        new_transition[TransitionKey.DONE] = processed_done
+        return new_transition
+
+
+class TruncatedProcessorStep(ProcessorStep, ABC):
+    """An abstract `ProcessorStep` that specifically targets the 'truncated' flag in a transition."""
+
+    @abstractmethod
+    def truncated(self, truncated) -> bool | torch.Tensor:
+        """Processes a 'truncated' flag. Subclasses must implement this method.
+
+        Args:
+            truncated: The input 'truncated' flag from the transition.
+
+        Returns:
+            The processed 'truncated' flag.
+        """
+        ...
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Applies the `truncated` method to the transition's 'truncated' flag."""
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        truncated = new_transition.get(TransitionKey.TRUNCATED)
+        if truncated is None:
+            raise ValueError("TruncatedProcessorStep requires a truncated flag in the transition.")
+
+        processed_truncated = self.truncated(truncated)
+        new_transition[TransitionKey.TRUNCATED] = processed_truncated
+        return new_transition
+
+
+class InfoProcessorStep(ProcessorStep, ABC):
+    """An abstract `ProcessorStep` that specifically targets the 'info' dictionary in a transition."""
+
+    @abstractmethod
+    def info(self, info) -> dict[str, Any]:
+        """Processes an 'info' dictionary. Subclasses must implement this method.
+
+        Args:
+            info: The input 'info' dictionary from the transition.
+
+        Returns:
+            The processed 'info' dictionary.
+        """
+        ...
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Applies the `info` method to the transition's 'info' dictionary."""
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        info = new_transition.get(TransitionKey.INFO)
+        if info is None or not isinstance(info, dict):
+            raise ValueError("InfoProcessorStep requires an info dictionary in the transition.")
+
+        processed_info = self.info(info.copy())
+        new_transition[TransitionKey.INFO] = processed_info
+        return new_transition
+
+
+class ComplementaryDataProcessorStep(ProcessorStep, ABC):
+    """An abstract `ProcessorStep` that targets the 'complementary_data' in a transition."""
+
+    @abstractmethod
+    def complementary_data(self, complementary_data) -> dict[str, Any]:
+        """Processes a 'complementary_data' dictionary. Subclasses must implement this method.
+
+        Args:
+            complementary_data: The input 'complementary_data' from the transition.
+
+        Returns:
+            The processed 'complementary_data' dictionary.
+        """
+        ...
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Applies the `complementary_data` method to the transition's data."""
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        complementary_data = new_transition.get(TransitionKey.COMPLEMENTARY_DATA)
+        if complementary_data is None or not isinstance(complementary_data, dict):
+            raise ValueError("ComplementaryDataProcessorStep requires complementary data in the transition.")
+
+        processed_complementary_data = self.complementary_data(complementary_data.copy())
+        new_transition[TransitionKey.COMPLEMENTARY_DATA] = processed_complementary_data
+        return new_transition
+
+
+class IdentityProcessorStep(ProcessorStep):
+    """A no-op processor step that returns the input transition and features unchanged.
+
+    This can be useful as a placeholder or for debugging purposes.
+    """
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Returns the transition without modification."""
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """Returns the features without modification."""
+        return features
diff --git a/lerobot/src/lerobot/processor/policy_robot_bridge.py b/lerobot/src/lerobot/processor/policy_robot_bridge.py
new file mode 100644
index 0000000000000000000000000000000000000000..25887d414ee0321ec5fe2e46703eb557e16bb233
--- /dev/null
+++ b/lerobot/src/lerobot/processor/policy_robot_bridge.py
@@ -0,0 +1,69 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import asdict, dataclass
+from typing import Any
+
+import torch
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from lerobot.processor import ActionProcessorStep, PolicyAction, ProcessorStepRegistry, RobotAction
+from lerobot.utils.constants import ACTION
+
+
+@dataclass
+@ProcessorStepRegistry.register("robot_action_to_policy_action_processor")
+class RobotActionToPolicyActionProcessorStep(ActionProcessorStep):
+    """Processor step to map a dictionary to a tensor action."""
+
+    motor_names: list[str]
+
+    def action(self, action: RobotAction) -> PolicyAction:
+        if len(self.motor_names) != len(action):
+            raise ValueError(f"Action must have {len(self.motor_names)} elements, got {len(action)}")
+        return torch.tensor([action[f"{name}.pos"] for name in self.motor_names])
+
+    def get_config(self) -> dict[str, Any]:
+        return asdict(self)
+
+    def transform_features(self, features):
+        features[PipelineFeatureType.ACTION][ACTION] = PolicyFeature(
+            type=FeatureType.ACTION, shape=(len(self.motor_names),)
+        )
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register("policy_action_to_robot_action_processor")
+class PolicyActionToRobotActionProcessorStep(ActionProcessorStep):
+    """Processor step to map a policy action to a robot action."""
+
+    motor_names: list[str]
+
+    def action(self, action: PolicyAction) -> RobotAction:
+        if len(self.motor_names) != len(action):
+            raise ValueError(f"Action must have {len(self.motor_names)} elements, got {len(action)}")
+        return {f"{name}.pos": action[i] for i, name in enumerate(self.motor_names)}
+
+    def get_config(self) -> dict[str, Any]:
+        return asdict(self)
+
+    def transform_features(self, features):
+        for name in self.motor_names:
+            features[PipelineFeatureType.ACTION][f"{name}.pos"] = PolicyFeature(
+                type=FeatureType.ACTION, shape=(1,)
+            )
+        return features
diff --git a/lerobot/src/lerobot/processor/rename_processor.py b/lerobot/src/lerobot/processor/rename_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..6cae5921ff30ca11166e87c4d9c5cd6c3cec47c1
--- /dev/null
+++ b/lerobot/src/lerobot/processor/rename_processor.py
@@ -0,0 +1,93 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from copy import deepcopy
+from dataclasses import dataclass, field
+from typing import Any
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+
+from .pipeline import ObservationProcessorStep, ProcessorStepRegistry
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="rename_observations_processor")
+class RenameObservationsProcessorStep(ObservationProcessorStep):
+    """
+    A processor step that renames keys in an observation dictionary.
+
+    This step is useful for creating a standardized data interface by mapping keys
+    from an environment's format to the format expected by a LeRobot policy or
+    other downstream components.
+
+    Attributes:
+        rename_map: A dictionary mapping from old key names to new key names.
+                    Keys present in an observation that are not in this map will
+                    be kept with their original names.
+    """
+
+    rename_map: dict[str, str] = field(default_factory=dict)
+
+    def observation(self, observation):
+        processed_obs = {}
+        for key, value in observation.items():
+            if key in self.rename_map:
+                processed_obs[self.rename_map[key]] = value
+            else:
+                processed_obs[key] = value
+
+        return processed_obs
+
+    def get_config(self) -> dict[str, Any]:
+        return {"rename_map": self.rename_map}
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """Transforms:
+        - Each key in the observation that appears in `rename_map` is renamed to its value.
+        - Keys not in `rename_map` remain unchanged.
+        """
+        new_features: dict[PipelineFeatureType, dict[str, PolicyFeature]] = features.copy()
+        new_features[PipelineFeatureType.OBSERVATION] = {
+            self.rename_map.get(k, k): v for k, v in features[PipelineFeatureType.OBSERVATION].items()
+        }
+        return new_features
+
+
+def rename_stats(stats: dict[str, dict[str, Any]], rename_map: dict[str, str]) -> dict[str, dict[str, Any]]:
+    """
+    Renames the top-level keys in a statistics dictionary using a provided mapping.
+
+    This is a helper function typically used to keep normalization statistics
+    consistent with renamed observation or action features. It performs a defensive
+    deep copy to avoid modifying the original `stats` dictionary.
+
+    Args:
+        stats: A nested dictionary of statistics, where top-level keys are
+               feature names (e.g., `{"observation.state": {"mean": 0.5}}`).
+        rename_map: A dictionary mapping old feature names to new feature names.
+
+    Returns:
+        A new statistics dictionary with its top-level keys renamed. Returns an
+        empty dictionary if the input `stats` is empty.
+    """
+    if not stats:
+        return {}
+    renamed: dict[str, dict[str, Any]] = {}
+    for old_key, sub_stats in stats.items():
+        new_key = rename_map.get(old_key, old_key)
+        renamed[new_key] = deepcopy(sub_stats) if sub_stats is not None else {}
+    return renamed
diff --git a/lerobot/src/lerobot/processor/tokenizer_processor.py b/lerobot/src/lerobot/processor/tokenizer_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..2a972ecc86d0a9eca397b048136197bd7a3b4b1a
--- /dev/null
+++ b/lerobot/src/lerobot/processor/tokenizer_processor.py
@@ -0,0 +1,576 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+This script defines a processor for tokenizing natural language instructions from an environment transition.
+
+It uses a tokenizer from the Hugging Face `transformers` library to convert task descriptions (text) into
+token IDs and attention masks, which are then added to the observation dictionary.
+"""
+
+from __future__ import annotations
+
+import logging
+from dataclasses import dataclass, field
+from typing import TYPE_CHECKING, Any
+
+import torch
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from lerobot.types import EnvTransition, RobotObservation, TransitionKey
+from lerobot.utils.constants import (
+    ACTION_TOKEN_MASK,
+    ACTION_TOKENS,
+    OBS_LANGUAGE_ATTENTION_MASK,
+    OBS_LANGUAGE_SUBTASK_ATTENTION_MASK,
+    OBS_LANGUAGE_SUBTASK_TOKENS,
+    OBS_LANGUAGE_TOKENS,
+)
+from lerobot.utils.import_utils import _transformers_available
+
+from .pipeline import ActionProcessorStep, ObservationProcessorStep, ProcessorStepRegistry
+
+# Conditional import for type checking and lazy loading
+if TYPE_CHECKING or _transformers_available:
+    from transformers import AutoProcessor, AutoTokenizer
+else:
+    AutoProcessor = None
+    AutoTokenizer = None
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="tokenizer_processor")
+class TokenizerProcessorStep(ObservationProcessorStep):
+    """
+    Processor step to tokenize a natural language task description.
+
+    This step extracts a task string from the `complementary_data` of an `EnvTransition`,
+    tokenizes it using a Hugging Face `transformers` tokenizer, and adds the resulting
+    token IDs and attention mask to the `observation` dictionary.
+
+    Requires the `transformers` library to be installed.
+
+    Attributes:
+        tokenizer_name: The name of a pretrained tokenizer from the Hugging Face Hub (e.g., "bert-base-uncased").
+        tokenizer: A pre-initialized tokenizer object. If provided, `tokenizer_name` is ignored.
+        max_length: The maximum length to pad or truncate sequences to.
+        task_key: The key in `complementary_data` where the task string is stored.
+        padding_side: The side to pad on ('left' or 'right').
+        padding: The padding strategy ('max_length', 'longest', etc.).
+        truncation: Whether to truncate sequences longer than `max_length`.
+        input_tokenizer: The internal tokenizer instance, loaded during initialization.
+    """
+
+    tokenizer_name: str | None = None
+    tokenizer: Any | None = None  # Use `Any` for compatibility without a hard dependency
+    max_length: int = 512
+    task_key: str = "task"
+    padding_side: str = "right"
+    padding: str = "max_length"
+    truncation: bool = True
+
+    # Internal tokenizer instance (not part of the config)
+    input_tokenizer: Any = field(default=None, init=False, repr=False)
+
+    def __post_init__(self):
+        """
+        Initializes the tokenizer after the dataclass is created.
+
+        It checks for the availability of the `transformers` library and loads the tokenizer
+        either from a provided object or by name from the Hugging Face Hub.
+
+        Raises:
+            ImportError: If the `transformers` library is not installed.
+            ValueError: If neither `tokenizer` nor `tokenizer_name` is provided.
+        """
+        if not _transformers_available:
+            raise ImportError(
+                "The 'transformers' library is not installed. "
+                "Please install it with `pip install 'lerobot[transformers-dep]'` to use TokenizerProcessorStep."
+            )
+
+        if self.tokenizer is not None:
+            # Use provided tokenizer object directly
+            self.input_tokenizer = self.tokenizer
+        elif self.tokenizer_name is not None:
+            if AutoTokenizer is None:
+                raise ImportError("AutoTokenizer is not available")
+            self.input_tokenizer = AutoTokenizer.from_pretrained(self.tokenizer_name)
+        else:
+            raise ValueError(
+                "Either 'tokenizer' or 'tokenizer_name' must be provided. "
+                "Pass a tokenizer object directly or a tokenizer name to auto-load."
+            )
+
+    def get_task(self, transition: EnvTransition) -> list[str] | None:
+        """
+        Extracts the task description(s) from the transition's complementary data.
+
+        Args:
+            transition: The environment transition.
+
+        Returns:
+            A list of task strings, or None if the task key is not found or the value is None.
+        """
+        complementary_data = transition.get(TransitionKey.COMPLEMENTARY_DATA)
+        if complementary_data is None:
+            raise ValueError("Complementary data is None so no task can be extracted from it")
+
+        task = complementary_data[self.task_key]
+        if task is None:
+            raise ValueError("Task extracted from Complementary data is None")
+
+        # Standardize to a list of strings for the tokenizer
+        if isinstance(task, str):
+            return [task]
+        elif isinstance(task, list) and all(isinstance(t, str) for t in task):
+            return task
+
+        return None
+
+    def get_subtask(self, transition: EnvTransition) -> list[str] | None:
+        """
+        Extracts the subtask from the transition's complementary data.
+
+        Args:
+            transition: The environment transition.
+
+        Returns:
+            A list of subtask strings, or None if the subtask key is not found or the value is None.
+        """
+        complementary_data = transition.get(TransitionKey.COMPLEMENTARY_DATA)
+        if complementary_data is None:
+            return None
+
+        subtask = complementary_data.get("subtask")
+        if subtask is None:
+            return None
+
+        # Standardize to a list of strings for the tokenizer
+        if isinstance(subtask, str):
+            return [subtask]
+        elif isinstance(subtask, list) and all(isinstance(t, str) for t in subtask):
+            return subtask
+
+        return None
+
+    def observation(self, observation: RobotObservation) -> RobotObservation:
+        """
+        Tokenizes the task description and adds it to the observation dictionary.
+
+        This method retrieves the task, tokenizes it, moves the resulting tensors to the
+        same device as other data in the transition, and updates the observation.
+
+        Args:
+            observation: The original observation dictionary.
+
+        Returns:
+            The updated observation dictionary including token IDs and an attention mask.
+        """
+        task = self.get_task(self.transition)
+        if task is None:
+            raise ValueError("Task cannot be None")
+
+        # Tokenize the task (this will create CPU tensors)
+        tokenized_prompt = self._tokenize_text(task)
+
+        # Detect the device from existing tensors in the transition to ensure consistency
+        target_device = self._detect_device(self.transition)
+
+        # Move new tokenized tensors to the detected device
+        if target_device is not None:
+            tokenized_prompt = {
+                k: v.to(target_device) if isinstance(v, torch.Tensor) else v
+                for k, v in tokenized_prompt.items()
+            }
+
+        # Create a new observation dict to avoid modifying the original in place
+        new_observation = dict(observation)
+
+        # Add tokenized data to the observation
+        new_observation[OBS_LANGUAGE_TOKENS] = tokenized_prompt["input_ids"]
+        new_observation[OBS_LANGUAGE_ATTENTION_MASK] = tokenized_prompt["attention_mask"].to(dtype=torch.bool)
+
+        # Tokenize subtask if available
+        subtask = self.get_subtask(self.transition)
+        if subtask is not None:
+            tokenized_subtask = self._tokenize_text(subtask)
+
+            # Move new tokenized tensors to the detected device
+            if target_device is not None:
+                tokenized_subtask = {
+                    k: v.to(target_device) if isinstance(v, torch.Tensor) else v
+                    for k, v in tokenized_subtask.items()
+                }
+
+            # Add tokenized subtask to the observation
+            new_observation[OBS_LANGUAGE_SUBTASK_TOKENS] = tokenized_subtask["input_ids"]
+            new_observation[OBS_LANGUAGE_SUBTASK_ATTENTION_MASK] = tokenized_subtask["attention_mask"].to(
+                dtype=torch.bool
+            )
+
+        return new_observation
+
+    def _detect_device(self, transition: EnvTransition) -> torch.device | None:
+        """
+        Detects the torch.device from existing tensors in the transition.
+
+        It checks tensors in the observation dictionary first, then the action tensor.
+
+        Args:
+            transition: The environment transition.
+
+        Returns:
+            The detected `torch.device`, or None if no tensors are found.
+        """
+        # Check observation tensors first (most likely place to find tensors)
+        observation = transition.get(TransitionKey.OBSERVATION)
+        if observation:
+            for value in observation.values():
+                if isinstance(value, torch.Tensor):
+                    return value.device
+
+        # Fallback to checking the action tensor
+        action = transition.get(TransitionKey.ACTION)
+        if isinstance(action, torch.Tensor):
+            return action.device
+
+        return None  # No tensors found, default will be CPU
+
+    def _tokenize_text(self, text: str | list[str]) -> dict[str, torch.Tensor]:
+        """
+        A wrapper around the tokenizer call.
+
+        Args:
+            text: A string or list of strings to tokenize.
+
+        Returns:
+            A dictionary containing tokenized 'input_ids' and 'attention_mask' as PyTorch tensors.
+        """
+        return self.input_tokenizer(
+            text,
+            max_length=self.max_length,
+            truncation=self.truncation,
+            padding=self.padding,
+            padding_side=self.padding_side,
+            return_tensors="pt",
+        )
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns the serializable configuration of the processor.
+
+        Note: The tokenizer object itself is not serialized. If the processor was initialized
+        with a tokenizer name, that name will be included in the config.
+
+        Returns:
+            A dictionary with the processor's configuration parameters.
+        """
+        config = {
+            "max_length": self.max_length,
+            "task_key": self.task_key,
+            "padding_side": self.padding_side,
+            "padding": self.padding,
+            "truncation": self.truncation,
+        }
+
+        # Only save tokenizer_name if it was used to create the tokenizer
+        if self.tokenizer_name is not None and self.tokenizer is None:
+            config["tokenizer_name"] = self.tokenizer_name
+
+        return config
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Adds feature definitions for the language tokens and attention mask.
+
+        This updates the policy features dictionary to include the new data added to the
+        observation, ensuring downstream components are aware of their shape and type.
+
+        Args:
+            features: The dictionary of existing policy features.
+
+        Returns:
+            The updated dictionary of policy features.
+        """
+        # Add a feature for the token IDs if it doesn't already exist
+        if OBS_LANGUAGE_TOKENS not in features[PipelineFeatureType.OBSERVATION]:
+            features[PipelineFeatureType.OBSERVATION][OBS_LANGUAGE_TOKENS] = PolicyFeature(
+                type=FeatureType.LANGUAGE, shape=(self.max_length,)
+            )
+
+        # Add a feature for the attention mask if it doesn't already exist
+        if OBS_LANGUAGE_ATTENTION_MASK not in features[PipelineFeatureType.OBSERVATION]:
+            features[PipelineFeatureType.OBSERVATION][OBS_LANGUAGE_ATTENTION_MASK] = PolicyFeature(
+                type=FeatureType.LANGUAGE, shape=(self.max_length,)
+            )
+
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register(name="action_tokenizer_processor")
+class ActionTokenizerProcessorStep(ActionProcessorStep):
+    """
+    Processor step to tokenize action data using a fast action tokenizer.
+
+    This step takes action tensors from an `EnvTransition`, tokenizes them using
+    a Hugging Face `transformers` AutoProcessor (such as the Physical Intelligence "fast" tokenizer),
+    and returns the tokenized action.
+
+    Requires the `transformers` library to be installed.
+
+    Attributes:
+        tokenizer_name: The name of a pretrained processor from the Hugging Face Hub (e.g., "lerobot/fast-action-tokenizer").
+        tokenizer: A pre-initialized processor/tokenizer object. If provided, `tokenizer_name` is ignored.
+        trust_remote_code: Whether to trust remote code when loading the tokenizer (required for some tokenizers).
+        action_tokenizer: The internal tokenizer/processor instance, loaded during initialization.
+        paligemma_tokenizer_name: The name of a pretrained PaliGemma tokenizer from the Hugging Face Hub (e.g., "google/paligemma-3b-pt-224").
+    """
+
+    action_tokenizer_name: str | None = None
+    action_tokenizer_input_object: Any | None = None
+    trust_remote_code: bool = True
+    max_action_tokens: int = 256
+    fast_skip_tokens: int = 128
+    paligemma_tokenizer_name: str = "google/paligemma-3b-pt-224"
+    # Internal tokenizer instance (not part of the config)
+    action_tokenizer: Any = field(default=None, init=False, repr=False)
+    _paligemma_tokenizer: Any = field(default=None, init=False, repr=False)
+
+    def __post_init__(self):
+        """
+        Initializes the action tokenizer after the dataclass is created.
+
+        It checks for the availability of the `transformers` library and loads the tokenizer
+        either from a provided object or by name from the Hugging Face Hub.
+
+        Raises:
+            ImportError: If the `transformers` library is not installed.
+            ValueError: If neither `tokenizer` nor `tokenizer_name` is provided.
+        """
+        if not _transformers_available:
+            raise ImportError(
+                "The 'transformers' library is not installed. "
+                "Please install it with `pip install 'lerobot[transformers-dep]'` to use ActionTokenizerProcessorStep."
+            )
+
+        if self.action_tokenizer_input_object is not None:
+            self.action_tokenizer = self.action_tokenizer_input_object
+
+        elif self.action_tokenizer_name is not None:
+            if AutoProcessor is None:
+                raise ImportError("AutoProcessor is not available")
+            self.action_tokenizer = AutoProcessor.from_pretrained(
+                self.action_tokenizer_name, trust_remote_code=self.trust_remote_code
+            )
+        else:
+            raise ValueError(
+                "Either 'action_tokenizer' or 'action_tokenizer_name' must be provided. "
+                "Pass a tokenizer object directly or a tokenizer name to auto-load."
+            )
+
+        self._paligemma_tokenizer = AutoTokenizer.from_pretrained(
+            self.paligemma_tokenizer_name,
+            trust_remote_code=self.trust_remote_code,
+            add_eos_token=True,
+            add_bos_token=False,
+        )
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """
+        Applies action tokenization to the transition.
+
+        This overrides the base class to handle both tokens and mask.
+
+        Args:
+            transition: The input transition with action data.
+
+        Returns:
+            The processed transition with tokenized actions and mask in complementary data.
+        """
+        self._current_transition = transition.copy()
+        new_transition = self._current_transition
+
+        action = new_transition.get(TransitionKey.ACTION)
+        if action is None:
+            # During inference, no action is available, skip tokenization
+            return new_transition
+
+        # Tokenize and get both tokens and mask
+        tokens, mask = self._tokenize_action(action)
+
+        # Store mask in complementary data
+        complementary_data = new_transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+        if complementary_data is None:
+            complementary_data = {}
+        complementary_data[ACTION_TOKEN_MASK] = mask
+        complementary_data[ACTION_TOKENS] = tokens
+        new_transition[TransitionKey.COMPLEMENTARY_DATA] = complementary_data
+        return new_transition
+
+    def _act_tokens_to_paligemma_tokens(self, tokens: torch.Tensor) -> torch.Tensor:
+        """
+        Converts action tokens to PaliGemma tokens.
+        """
+        return self._paligemma_tokenizer.vocab_size - 1 - self.fast_skip_tokens - tokens
+
+    def _tokenize_action(self, action: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]:
+        """
+        Tokenizes the action tensor and creates a mask.
+
+        Args:
+            action: The input action tensor to tokenize. Shape: (B, H, action_dim) or (H, action_dim,)
+
+        Returns:
+            A tuple of (tokens, mask) where:
+            - tokens: Tensor of token IDs with shape (B, max_action_tokens)
+            - mask: Boolean mask with shape (B, max_action_tokens), True for real tokens, False for padding
+        """
+        if action is None:
+            raise ValueError("Action cannot be None")
+
+        # Get the device and dtype of the input action
+        device = action.device if isinstance(action, torch.Tensor) else None
+
+        # Handle single sample (add batch dimension)
+        single_sample = action.dim() == 1
+        if single_sample:
+            action = action.unsqueeze(0)
+
+        batch_size = action.shape[0]
+
+        # Tokenize the action batch
+        # The fast tokenizer expects action data and returns token IDs
+        tokens_list = []
+        masks_list = []
+
+        for i in range(batch_size):
+            # Tokenize single action (move to CPU first as tokenizer uses scipy which requires numpy)
+            action_cpu = action[i : i + 1].cpu()
+            tokens = self.action_tokenizer(action_cpu)
+
+            # Convert to numpy array if it's a list
+            if isinstance(tokens, list) or not isinstance(tokens, torch.Tensor):
+                tokens = torch.tensor(tokens, dtype=torch.long, device=action.device)
+            else:
+                # Move tokens back to the same device as input action
+                tokens = tokens.to(device=action.device)
+
+            # Flatten to 1D if needed
+            if tokens.dim() > 1:
+                tokens = tokens.flatten()
+
+            bos_id = self._paligemma_tokenizer.bos_token_id
+            # add bos
+            tokens = torch.cat(
+                [
+                    torch.tensor([bos_id], device=action.device),
+                    torch.tensor(
+                        self._paligemma_tokenizer.encode("Action: ", add_special_tokens=False),
+                        device=action.device,
+                    ),
+                    self._act_tokens_to_paligemma_tokens(tokens),
+                    torch.tensor(self._paligemma_tokenizer.encode("|"), device=action.device),
+                ]
+            )
+
+            # Truncate or pad to max_action_tokens
+            if len(tokens) > self.max_action_tokens:
+                logging.warning(
+                    f"Token length ({len(tokens)}) exceeds max length ({self.max_action_tokens}), truncating. "
+                    "Consider increasing the `max_action_tokens` in your model config if this happens frequently."
+                )
+                tokens = tokens[: self.max_action_tokens]
+                mask = torch.ones(self.max_action_tokens, dtype=torch.bool, device=action.device)
+            else:
+                mask = torch.cat(
+                    [
+                        torch.ones(len(tokens), dtype=torch.bool, device=action.device),
+                        torch.zeros(
+                            self.max_action_tokens - len(tokens), dtype=torch.bool, device=action.device
+                        ),
+                    ]
+                )
+                # Pad tokens with zeros
+                tokens = torch.nn.functional.pad(tokens, (0, self.max_action_tokens - len(tokens)), value=0)
+
+            tokens_list.append(tokens)
+            masks_list.append(mask)
+
+        # Stack into batched tensors
+        tokens_batch = torch.stack(tokens_list, dim=0)  # (B, max_action_tokens)
+        masks_batch = torch.stack(masks_list, dim=0)  # (B, max_action_tokens)
+
+        # Remove batch dimension if input was single sample
+        if single_sample:
+            tokens_batch = tokens_batch.squeeze(0)
+            masks_batch = masks_batch.squeeze(0)
+
+        # Move to the same device as the input
+        if device is not None:
+            tokens_batch = tokens_batch.to(device)
+            masks_batch = masks_batch.to(device)
+
+        return tokens_batch, masks_batch
+
+    def action(self, action: torch.Tensor) -> torch.Tensor:
+        """
+        This method is not used since we override __call__.
+        Required by ActionProcessorStep ABC.
+        """
+        tokens, _ = self._tokenize_action(action)
+        return tokens
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns the serializable configuration of the processor.
+
+        Note: The tokenizer object itself is not serialized. If the processor was initialized
+        with a tokenizer name, that name will be included in the config.
+
+        Returns:
+            A dictionary with the processor's configuration parameters.
+        """
+        config = {
+            "trust_remote_code": self.trust_remote_code,
+            "max_action_tokens": self.max_action_tokens,
+        }
+
+        # Only save tokenizer_name if it was used to create the tokenizer
+        if self.action_tokenizer_name is not None and self.action_tokenizer_input_object is None:
+            config["action_tokenizer_name"] = self.action_tokenizer_name
+
+        return config
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Updates feature definitions to reflect tokenized actions.
+
+        This updates the policy features dictionary to indicate that the action
+        has been tokenized into a sequence of token IDs with shape (max_action_tokens,).
+
+        Args:
+            features: The dictionary of existing policy features.
+
+        Returns:
+            The updated dictionary of policy features.
+        """
+        return features
diff --git a/lerobot/src/lerobot/rl/actor.py b/lerobot/src/lerobot/rl/actor.py
new file mode 100644
index 0000000000000000000000000000000000000000..18c0ca1ea0d441b4058f9a27376f69b09ba6f77d
--- /dev/null
+++ b/lerobot/src/lerobot/rl/actor.py
@@ -0,0 +1,738 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""
+Actor server runner for distributed HILSerl robot policy training.
+
+This script implements the actor component of the distributed HILSerl architecture.
+It executes the policy in the robot environment, collects experience,
+and sends transitions to the learner server for policy updates.
+
+Examples of usage:
+
+- Start an actor server for real robot training with human-in-the-loop intervention:
+```bash
+python -m lerobot.rl.actor --config_path src/lerobot/configs/train_config_hilserl_so100.json
+```
+
+**NOTE**: The actor server requires a running learner server to connect to. Ensure the learner
+server is started before launching the actor.
+
+**NOTE**: Human intervention is key to HILSerl training. Press the upper right trigger button on the
+gamepad to take control of the robot during training. Initially intervene frequently, then gradually
+reduce interventions as the policy improves.
+
+**WORKFLOW**:
+1. Determine robot workspace bounds using `lerobot-find-joint-limits`
+2. Record demonstrations with `gym_manipulator.py` in record mode
+3. Process the dataset and determine camera crops with `crop_dataset_roi.py`
+4. Start the learner server with the training configuration
+5. Start this actor server with the same configuration
+6. Use human interventions to guide policy learning
+
+For more details on the complete HILSerl training workflow, see:
+https://github.com/michel-aractingi/lerobot-hilserl-guide
+"""
+
+import logging
+import os
+import time
+from functools import lru_cache
+from queue import Empty
+
+import grpc
+import torch
+from torch import nn
+from torch.multiprocessing import Event, Queue
+
+from lerobot.cameras import opencv  # noqa: F401
+from lerobot.configs import parser
+from lerobot.configs.train import TrainRLServerPipelineConfig
+from lerobot.policies.factory import make_policy
+from lerobot.policies.sac.modeling_sac import SACPolicy
+from lerobot.rl.process import ProcessSignalHandler
+from lerobot.rl.queue import get_last_item_from_queue
+from lerobot.robots import so_follower  # noqa: F401
+from lerobot.teleoperators import gamepad, so_leader  # noqa: F401
+from lerobot.teleoperators.utils import TeleopEvents
+from lerobot.transport import services_pb2, services_pb2_grpc
+from lerobot.transport.utils import (
+    bytes_to_state_dict,
+    grpc_channel_options,
+    python_object_to_bytes,
+    receive_bytes_in_chunks,
+    send_bytes_in_chunks,
+    transitions_to_bytes,
+)
+from lerobot.types import TransitionKey
+from lerobot.utils.device_utils import get_safe_torch_device
+from lerobot.utils.random_utils import set_seed
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.transition import (
+    Transition,
+    move_state_dict_to_device,
+    move_transition_to_device,
+)
+from lerobot.utils.utils import (
+    TimerManager,
+    init_logging,
+)
+
+from .gym_manipulator import (
+    create_transition,
+    make_processors,
+    make_robot_env,
+    step_env_and_process_transition,
+)
+
+# Main entry point
+
+
+@parser.wrap()
+def actor_cli(cfg: TrainRLServerPipelineConfig):
+    cfg.validate()
+    display_pid = False
+    if not use_threads(cfg):
+        import torch.multiprocessing as mp
+
+        mp.set_start_method("spawn")
+        display_pid = True
+
+    # Create logs directory to ensure it exists
+    log_dir = os.path.join(cfg.output_dir, "logs")
+    os.makedirs(log_dir, exist_ok=True)
+    log_file = os.path.join(log_dir, f"actor_{cfg.job_name}.log")
+
+    # Initialize logging with explicit log file
+    init_logging(log_file=log_file, display_pid=display_pid)
+    logging.info(f"Actor logging initialized, writing to {log_file}")
+
+    is_threaded = use_threads(cfg)
+    shutdown_event = ProcessSignalHandler(is_threaded, display_pid=display_pid).shutdown_event
+
+    learner_client, grpc_channel = learner_service_client(
+        host=cfg.policy.actor_learner_config.learner_host,
+        port=cfg.policy.actor_learner_config.learner_port,
+    )
+
+    logging.info("[ACTOR] Establishing connection with Learner")
+    if not establish_learner_connection(learner_client, shutdown_event):
+        logging.error("[ACTOR] Failed to establish connection with Learner")
+        return
+
+    if not use_threads(cfg):
+        # If we use multithreading, we can reuse the channel
+        grpc_channel.close()
+        grpc_channel = None
+
+    logging.info("[ACTOR] Connection with Learner established")
+
+    parameters_queue = Queue()
+    transitions_queue = Queue()
+    interactions_queue = Queue()
+
+    concurrency_entity = None
+    if use_threads(cfg):
+        from threading import Thread
+
+        concurrency_entity = Thread
+    else:
+        from multiprocessing import Process
+
+        concurrency_entity = Process
+
+    receive_policy_process = concurrency_entity(
+        target=receive_policy,
+        args=(cfg, parameters_queue, shutdown_event, grpc_channel),
+        daemon=True,
+    )
+
+    transitions_process = concurrency_entity(
+        target=send_transitions,
+        args=(cfg, transitions_queue, shutdown_event, grpc_channel),
+        daemon=True,
+    )
+
+    interactions_process = concurrency_entity(
+        target=send_interactions,
+        args=(cfg, interactions_queue, shutdown_event, grpc_channel),
+        daemon=True,
+    )
+
+    transitions_process.start()
+    interactions_process.start()
+    receive_policy_process.start()
+
+    act_with_policy(
+        cfg=cfg,
+        shutdown_event=shutdown_event,
+        parameters_queue=parameters_queue,
+        transitions_queue=transitions_queue,
+        interactions_queue=interactions_queue,
+    )
+    logging.info("[ACTOR] Policy process joined")
+
+    logging.info("[ACTOR] Closing queues")
+    transitions_queue.close()
+    interactions_queue.close()
+    parameters_queue.close()
+
+    transitions_process.join()
+    logging.info("[ACTOR] Transitions process joined")
+    interactions_process.join()
+    logging.info("[ACTOR] Interactions process joined")
+    receive_policy_process.join()
+    logging.info("[ACTOR] Receive policy process joined")
+
+    logging.info("[ACTOR] join queues")
+    transitions_queue.cancel_join_thread()
+    interactions_queue.cancel_join_thread()
+    parameters_queue.cancel_join_thread()
+
+    logging.info("[ACTOR] queues closed")
+
+
+# Core algorithm functions
+
+
+def act_with_policy(
+    cfg: TrainRLServerPipelineConfig,
+    shutdown_event: any,  # Event,
+    parameters_queue: Queue,
+    transitions_queue: Queue,
+    interactions_queue: Queue,
+):
+    """
+    Executes policy interaction within the environment.
+
+    This function rolls out the policy in the environment, collecting interaction data and pushing it to a queue for streaming to the learner.
+    Once an episode is completed, updated network parameters received from the learner are retrieved from a queue and loaded into the network.
+
+    Args:
+        cfg: Configuration settings for the interaction process.
+        shutdown_event: Event to check if the process should shutdown.
+        parameters_queue: Queue to receive updated network parameters from the learner.
+        transitions_queue: Queue to send transitions to the learner.
+        interactions_queue: Queue to send interactions to the learner.
+    """
+    # Initialize logging for multiprocessing
+    if not use_threads(cfg):
+        log_dir = os.path.join(cfg.output_dir, "logs")
+        os.makedirs(log_dir, exist_ok=True)
+        log_file = os.path.join(log_dir, f"actor_policy_{os.getpid()}.log")
+        init_logging(log_file=log_file, display_pid=True)
+        logging.info("Actor policy process logging initialized")
+
+    logging.info("make_env online")
+
+    online_env, teleop_device = make_robot_env(cfg=cfg.env)
+    env_processor, action_processor = make_processors(online_env, teleop_device, cfg.env, cfg.policy.device)
+
+    set_seed(cfg.seed)
+    device = get_safe_torch_device(cfg.policy.device, log=True)
+
+    torch.backends.cudnn.benchmark = True
+    torch.backends.cuda.matmul.allow_tf32 = True
+
+    logging.info("make_policy")
+
+    ### Instantiate the policy in both the actor and learner processes
+    ### To avoid sending a SACPolicy object through the port, we create a policy instance
+    ### on both sides, the learner sends the updated parameters every n steps to update the actor's parameters
+    policy: SACPolicy = make_policy(
+        cfg=cfg.policy,
+        env_cfg=cfg.env,
+    )
+    policy = policy.eval()
+    assert isinstance(policy, nn.Module)
+
+    obs, info = online_env.reset()
+    env_processor.reset()
+    action_processor.reset()
+
+    # Process initial observation
+    transition = create_transition(observation=obs, info=info)
+    transition = env_processor(transition)
+
+    # NOTE: For the moment we will solely handle the case of a single environment
+    sum_reward_episode = 0
+    list_transition_to_send_to_learner = []
+    episode_intervention = False
+    # Add counters for intervention rate calculation
+    episode_intervention_steps = 0
+    episode_total_steps = 0
+
+    policy_timer = TimerManager("Policy inference", log=False)
+
+    for interaction_step in range(cfg.policy.online_steps):
+        start_time = time.perf_counter()
+        if shutdown_event.is_set():
+            logging.info("[ACTOR] Shutting down act_with_policy")
+            return
+
+        observation = {
+            k: v for k, v in transition[TransitionKey.OBSERVATION].items() if k in cfg.policy.input_features
+        }
+
+        # Time policy inference and check if it meets FPS requirement
+        with policy_timer:
+            # Extract observation from transition for policy
+            action = policy.select_action(batch=observation)
+        policy_fps = policy_timer.fps_last
+
+        log_policy_frequency_issue(policy_fps=policy_fps, cfg=cfg, interaction_step=interaction_step)
+
+        # Use the new step function
+        new_transition = step_env_and_process_transition(
+            env=online_env,
+            transition=transition,
+            action=action,
+            env_processor=env_processor,
+            action_processor=action_processor,
+        )
+
+        # Extract values from processed transition
+        next_observation = {
+            k: v
+            for k, v in new_transition[TransitionKey.OBSERVATION].items()
+            if k in cfg.policy.input_features
+        }
+
+        # Teleop action is the action that was executed in the environment
+        # It is either the action from the teleop device or the action from the policy
+        executed_action = new_transition[TransitionKey.COMPLEMENTARY_DATA]["teleop_action"]
+
+        reward = new_transition[TransitionKey.REWARD]
+        done = new_transition.get(TransitionKey.DONE, False)
+        truncated = new_transition.get(TransitionKey.TRUNCATED, False)
+
+        sum_reward_episode += float(reward)
+        episode_total_steps += 1
+
+        # Check for intervention from transition info
+        intervention_info = new_transition[TransitionKey.INFO]
+        if intervention_info.get(TeleopEvents.IS_INTERVENTION, False):
+            episode_intervention = True
+            episode_intervention_steps += 1
+
+        complementary_info = {
+            "discrete_penalty": torch.tensor(
+                [new_transition[TransitionKey.COMPLEMENTARY_DATA].get("discrete_penalty", 0.0)]
+            ),
+        }
+        # Create transition for learner (convert to old format)
+        list_transition_to_send_to_learner.append(
+            Transition(
+                state=observation,
+                action=executed_action,
+                reward=reward,
+                next_state=next_observation,
+                done=done,
+                truncated=truncated,
+                complementary_info=complementary_info,
+            )
+        )
+
+        # Update transition for next iteration
+        transition = new_transition
+
+        if done or truncated:
+            logging.info(f"[ACTOR] Global step {interaction_step}: Episode reward: {sum_reward_episode}")
+
+            update_policy_parameters(policy=policy, parameters_queue=parameters_queue, device=device)
+
+            if len(list_transition_to_send_to_learner) > 0:
+                push_transitions_to_transport_queue(
+                    transitions=list_transition_to_send_to_learner,
+                    transitions_queue=transitions_queue,
+                )
+                list_transition_to_send_to_learner = []
+
+            stats = get_frequency_stats(policy_timer)
+            policy_timer.reset()
+
+            # Calculate intervention rate
+            intervention_rate = 0.0
+            if episode_total_steps > 0:
+                intervention_rate = episode_intervention_steps / episode_total_steps
+
+            # Send episodic reward to the learner
+            interactions_queue.put(
+                python_object_to_bytes(
+                    {
+                        "Episodic reward": sum_reward_episode,
+                        "Interaction step": interaction_step,
+                        "Episode intervention": int(episode_intervention),
+                        "Intervention rate": intervention_rate,
+                        **stats,
+                    }
+                )
+            )
+
+            # Reset intervention counters and environment
+            sum_reward_episode = 0.0
+            episode_intervention = False
+            episode_intervention_steps = 0
+            episode_total_steps = 0
+
+            # Reset environment and processors
+            obs, info = online_env.reset()
+            env_processor.reset()
+            action_processor.reset()
+
+            # Process initial observation
+            transition = create_transition(observation=obs, info=info)
+            transition = env_processor(transition)
+
+        if cfg.env.fps is not None:
+            dt_time = time.perf_counter() - start_time
+            precise_sleep(max(1 / cfg.env.fps - dt_time, 0.0))
+
+
+#  Communication Functions - Group all gRPC/messaging functions
+
+
+def establish_learner_connection(
+    stub: services_pb2_grpc.LearnerServiceStub,
+    shutdown_event: Event,  # type: ignore
+    attempts: int = 30,
+):
+    """Establish a connection with the learner.
+
+    Args:
+        stub (services_pb2_grpc.LearnerServiceStub): The stub to use for the connection.
+        shutdown_event (Event): The event to check if the connection should be established.
+        attempts (int): The number of attempts to establish the connection.
+    Returns:
+        bool: True if the connection is established, False otherwise.
+    """
+    for _ in range(attempts):
+        if shutdown_event.is_set():
+            logging.info("[ACTOR] Shutting down establish_learner_connection")
+            return False
+
+        # Force a connection attempt and check state
+        try:
+            logging.info("[ACTOR] Send ready message to Learner")
+            if stub.Ready(services_pb2.Empty()) == services_pb2.Empty():
+                return True
+        except grpc.RpcError as e:
+            logging.error(f"[ACTOR] Waiting for Learner to be ready... {e}")
+            time.sleep(2)
+    return False
+
+
+@lru_cache(maxsize=1)
+def learner_service_client(
+    host: str = "127.0.0.1",
+    port: int = 50051,
+) -> tuple[services_pb2_grpc.LearnerServiceStub, grpc.Channel]:
+    """
+    Returns a client for the learner service.
+
+    GRPC uses HTTP/2, which is a binary protocol and multiplexes requests over a single connection.
+    So we need to create only one client and reuse it.
+    """
+
+    channel = grpc.insecure_channel(
+        f"{host}:{port}",
+        grpc_channel_options(),
+    )
+    stub = services_pb2_grpc.LearnerServiceStub(channel)
+    logging.info("[ACTOR] Learner service client created")
+    return stub, channel
+
+
+def receive_policy(
+    cfg: TrainRLServerPipelineConfig,
+    parameters_queue: Queue,
+    shutdown_event: Event,  # type: ignore
+    learner_client: services_pb2_grpc.LearnerServiceStub | None = None,
+    grpc_channel: grpc.Channel | None = None,
+):
+    """Receive parameters from the learner.
+
+    Args:
+        cfg (TrainRLServerPipelineConfig): The configuration for the actor.
+        parameters_queue (Queue): The queue to receive the parameters.
+        shutdown_event (Event): The event to check if the process should shutdown.
+    """
+    logging.info("[ACTOR] Start receiving parameters from the Learner")
+    if not use_threads(cfg):
+        # Create a process-specific log file
+        log_dir = os.path.join(cfg.output_dir, "logs")
+        os.makedirs(log_dir, exist_ok=True)
+        log_file = os.path.join(log_dir, f"actor_receive_policy_{os.getpid()}.log")
+
+        # Initialize logging with explicit log file
+        init_logging(log_file=log_file, display_pid=True)
+        logging.info("Actor receive policy process logging initialized")
+
+        # Setup process handlers to handle shutdown signal
+        # But use shutdown event from the main process
+        _ = ProcessSignalHandler(use_threads=False, display_pid=True)
+
+    if grpc_channel is None or learner_client is None:
+        learner_client, grpc_channel = learner_service_client(
+            host=cfg.policy.actor_learner_config.learner_host,
+            port=cfg.policy.actor_learner_config.learner_port,
+        )
+
+    try:
+        iterator = learner_client.StreamParameters(services_pb2.Empty())
+        receive_bytes_in_chunks(
+            iterator,
+            parameters_queue,
+            shutdown_event,
+            log_prefix="[ACTOR] parameters",
+        )
+
+    except grpc.RpcError as e:
+        logging.error(f"[ACTOR] gRPC error: {e}")
+
+    if not use_threads(cfg):
+        grpc_channel.close()
+    logging.info("[ACTOR] Received policy loop stopped")
+
+
+def send_transitions(
+    cfg: TrainRLServerPipelineConfig,
+    transitions_queue: Queue,
+    shutdown_event: any,  # Event,
+    learner_client: services_pb2_grpc.LearnerServiceStub | None = None,
+    grpc_channel: grpc.Channel | None = None,
+) -> services_pb2.Empty:
+    """
+    Sends transitions to the learner.
+
+    This function continuously retrieves messages from the queue and processes:
+
+    - Transition Data:
+        - A batch of transitions (observation, action, reward, next observation) is collected.
+        - Transitions are moved to the CPU and serialized using PyTorch.
+        - The serialized data is wrapped in a `services_pb2.Transition` message and sent to the learner.
+    """
+
+    if not use_threads(cfg):
+        # Create a process-specific log file
+        log_dir = os.path.join(cfg.output_dir, "logs")
+        os.makedirs(log_dir, exist_ok=True)
+        log_file = os.path.join(log_dir, f"actor_transitions_{os.getpid()}.log")
+
+        # Initialize logging with explicit log file
+        init_logging(log_file=log_file, display_pid=True)
+        logging.info("Actor transitions process logging initialized")
+
+    if grpc_channel is None or learner_client is None:
+        learner_client, grpc_channel = learner_service_client(
+            host=cfg.policy.actor_learner_config.learner_host,
+            port=cfg.policy.actor_learner_config.learner_port,
+        )
+
+    try:
+        learner_client.SendTransitions(
+            transitions_stream(
+                shutdown_event, transitions_queue, cfg.policy.actor_learner_config.queue_get_timeout
+            )
+        )
+    except grpc.RpcError as e:
+        logging.error(f"[ACTOR] gRPC error: {e}")
+
+    logging.info("[ACTOR] Finished streaming transitions")
+
+    if not use_threads(cfg):
+        grpc_channel.close()
+    logging.info("[ACTOR] Transitions process stopped")
+
+
+def send_interactions(
+    cfg: TrainRLServerPipelineConfig,
+    interactions_queue: Queue,
+    shutdown_event: Event,  # type: ignore
+    learner_client: services_pb2_grpc.LearnerServiceStub | None = None,
+    grpc_channel: grpc.Channel | None = None,
+) -> services_pb2.Empty:
+    """
+    Sends interactions to the learner.
+
+    This function continuously retrieves messages from the queue and processes:
+
+    - Interaction Messages:
+        - Contains useful statistics about episodic rewards and policy timings.
+        - The message is serialized using `pickle` and sent to the learner.
+    """
+
+    if not use_threads(cfg):
+        # Create a process-specific log file
+        log_dir = os.path.join(cfg.output_dir, "logs")
+        os.makedirs(log_dir, exist_ok=True)
+        log_file = os.path.join(log_dir, f"actor_interactions_{os.getpid()}.log")
+
+        # Initialize logging with explicit log file
+        init_logging(log_file=log_file, display_pid=True)
+        logging.info("Actor interactions process logging initialized")
+
+        # Setup process handlers to handle shutdown signal
+        # But use shutdown event from the main process
+        _ = ProcessSignalHandler(use_threads=False, display_pid=True)
+
+    if grpc_channel is None or learner_client is None:
+        learner_client, grpc_channel = learner_service_client(
+            host=cfg.policy.actor_learner_config.learner_host,
+            port=cfg.policy.actor_learner_config.learner_port,
+        )
+
+    try:
+        learner_client.SendInteractions(
+            interactions_stream(
+                shutdown_event, interactions_queue, cfg.policy.actor_learner_config.queue_get_timeout
+            )
+        )
+    except grpc.RpcError as e:
+        logging.error(f"[ACTOR] gRPC error: {e}")
+
+    logging.info("[ACTOR] Finished streaming interactions")
+
+    if not use_threads(cfg):
+        grpc_channel.close()
+    logging.info("[ACTOR] Interactions process stopped")
+
+
+def transitions_stream(shutdown_event: Event, transitions_queue: Queue, timeout: float) -> services_pb2.Empty:  # type: ignore
+    while not shutdown_event.is_set():
+        try:
+            message = transitions_queue.get(block=True, timeout=timeout)
+        except Empty:
+            logging.debug("[ACTOR] Transition queue is empty")
+            continue
+
+        yield from send_bytes_in_chunks(
+            message, services_pb2.Transition, log_prefix="[ACTOR] Send transitions"
+        )
+
+    return services_pb2.Empty()
+
+
+def interactions_stream(
+    shutdown_event: Event,
+    interactions_queue: Queue,
+    timeout: float,  # type: ignore
+) -> services_pb2.Empty:
+    while not shutdown_event.is_set():
+        try:
+            message = interactions_queue.get(block=True, timeout=timeout)
+        except Empty:
+            logging.debug("[ACTOR] Interaction queue is empty")
+            continue
+
+        yield from send_bytes_in_chunks(
+            message,
+            services_pb2.InteractionMessage,
+            log_prefix="[ACTOR] Send interactions",
+        )
+
+    return services_pb2.Empty()
+
+
+#  Policy functions
+
+
+def update_policy_parameters(policy: SACPolicy, parameters_queue: Queue, device):
+    bytes_state_dict = get_last_item_from_queue(parameters_queue, block=False)
+    if bytes_state_dict is not None:
+        logging.info("[ACTOR] Load new parameters from Learner.")
+        state_dicts = bytes_to_state_dict(bytes_state_dict)
+
+        # TODO: check encoder parameter synchronization possible issues:
+        # 1. When shared_encoder=True, we're loading stale encoder params from actor's state_dict
+        #    instead of the updated encoder params from critic (which is optimized separately)
+        # 2. When freeze_vision_encoder=True, we waste bandwidth sending/loading frozen params
+        # 3. Need to handle encoder params correctly for both actor and discrete_critic
+        # Potential fixes:
+        # - Send critic's encoder state when shared_encoder=True
+        # - Skip encoder params entirely when freeze_vision_encoder=True
+        # - Ensure discrete_critic gets correct encoder state (currently uses encoder_critic)
+
+        # Load actor state dict
+        actor_state_dict = move_state_dict_to_device(state_dicts["policy"], device=device)
+        policy.actor.load_state_dict(actor_state_dict)
+
+        # Load discrete critic if present
+        if hasattr(policy, "discrete_critic") and "discrete_critic" in state_dicts:
+            discrete_critic_state_dict = move_state_dict_to_device(
+                state_dicts["discrete_critic"], device=device
+            )
+            policy.discrete_critic.load_state_dict(discrete_critic_state_dict)
+            logging.info("[ACTOR] Loaded discrete critic parameters from Learner.")
+
+
+#  Utilities functions
+
+
+def push_transitions_to_transport_queue(transitions: list, transitions_queue):
+    """Send transitions to learner in smaller chunks to avoid network issues.
+
+    Args:
+        transitions: List of transitions to send
+        message_queue: Queue to send messages to learner
+        chunk_size: Size of each chunk to send
+    """
+    transition_to_send_to_learner = []
+    for transition in transitions:
+        tr = move_transition_to_device(transition=transition, device="cpu")
+        for key, value in tr["state"].items():
+            if torch.isnan(value).any():
+                logging.warning(f"Found NaN values in transition {key}")
+
+        transition_to_send_to_learner.append(tr)
+
+    transitions_queue.put(transitions_to_bytes(transition_to_send_to_learner))
+
+
+def get_frequency_stats(timer: TimerManager) -> dict[str, float]:
+    """Get the frequency statistics of the policy.
+
+    Args:
+        timer (TimerManager): The timer with collected metrics.
+
+    Returns:
+        dict[str, float]: The frequency statistics of the policy.
+    """
+    stats = {}
+    if timer.count > 1:
+        avg_fps = timer.fps_avg
+        p90_fps = timer.fps_percentile(90)
+        logging.debug(f"[ACTOR] Average policy frame rate: {avg_fps}")
+        logging.debug(f"[ACTOR] Policy frame rate 90th percentile: {p90_fps}")
+        stats = {
+            "Policy frequency [Hz]": avg_fps,
+            "Policy frequency 90th-p [Hz]": p90_fps,
+        }
+    return stats
+
+
+def log_policy_frequency_issue(policy_fps: float, cfg: TrainRLServerPipelineConfig, interaction_step: int):
+    if policy_fps < cfg.env.fps:
+        logging.warning(
+            f"[ACTOR] Policy FPS {policy_fps:.1f} below required {cfg.env.fps} at step {interaction_step}"
+        )
+
+
+def use_threads(cfg: TrainRLServerPipelineConfig) -> bool:
+    return cfg.policy.concurrency.actor == "threads"
+
+
+if __name__ == "__main__":
+    actor_cli()
diff --git a/lerobot/src/lerobot/rl/buffer.py b/lerobot/src/lerobot/rl/buffer.py
new file mode 100644
index 0000000000000000000000000000000000000000..81aa29c48038c4528d7445a55abd8675965ef613
--- /dev/null
+++ b/lerobot/src/lerobot/rl/buffer.py
@@ -0,0 +1,834 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import functools
+from collections.abc import Callable, Sequence
+from contextlib import suppress
+from typing import TypedDict
+
+import torch
+import torch.nn.functional as F  # noqa: N812
+from tqdm import tqdm
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.utils.constants import ACTION, DONE, OBS_IMAGE, REWARD
+from lerobot.utils.transition import Transition
+
+
+class BatchTransition(TypedDict):
+    state: dict[str, torch.Tensor]
+    action: torch.Tensor
+    reward: torch.Tensor
+    next_state: dict[str, torch.Tensor]
+    done: torch.Tensor
+    truncated: torch.Tensor
+    complementary_info: dict[str, torch.Tensor | float | int] | None = None
+
+
+def random_crop_vectorized(images: torch.Tensor, output_size: tuple) -> torch.Tensor:
+    """
+    Perform a per-image random crop over a batch of images in a vectorized way.
+    (Same as shown previously.)
+    """
+    B, C, H, W = images.shape  # noqa: N806
+    crop_h, crop_w = output_size
+
+    if crop_h > H or crop_w > W:
+        raise ValueError(
+            f"Requested crop size ({crop_h}, {crop_w}) is bigger than the image size ({H}, {W})."
+        )
+
+    tops = torch.randint(0, H - crop_h + 1, (B,), device=images.device)
+    lefts = torch.randint(0, W - crop_w + 1, (B,), device=images.device)
+
+    rows = torch.arange(crop_h, device=images.device).unsqueeze(0) + tops.unsqueeze(1)
+    cols = torch.arange(crop_w, device=images.device).unsqueeze(0) + lefts.unsqueeze(1)
+
+    rows = rows.unsqueeze(2).expand(-1, -1, crop_w)  # (B, crop_h, crop_w)
+    cols = cols.unsqueeze(1).expand(-1, crop_h, -1)  # (B, crop_h, crop_w)
+
+    images_hwcn = images.permute(0, 2, 3, 1)  # (B, H, W, C)
+
+    # Gather pixels
+    cropped_hwcn = images_hwcn[torch.arange(B, device=images.device).view(B, 1, 1), rows, cols, :]
+    # cropped_hwcn => (B, crop_h, crop_w, C)
+
+    cropped = cropped_hwcn.permute(0, 3, 1, 2)  # (B, C, crop_h, crop_w)
+    return cropped
+
+
+def random_shift(images: torch.Tensor, pad: int = 4):
+    """Vectorized random shift, imgs: (B,C,H,W), pad: #pixels"""
+    _, _, h, w = images.shape
+    images = F.pad(input=images, pad=(pad, pad, pad, pad), mode="replicate")
+    return random_crop_vectorized(images=images, output_size=(h, w))
+
+
+class ReplayBuffer:
+    def __init__(
+        self,
+        capacity: int,
+        device: str = "cuda:0",
+        state_keys: Sequence[str] | None = None,
+        image_augmentation_function: Callable | None = None,
+        use_drq: bool = True,
+        storage_device: str = "cpu",
+        optimize_memory: bool = False,
+    ):
+        """
+        Replay buffer for storing transitions.
+        It will allocate tensors on the specified device, when the first transition is added.
+        NOTE: If you encounter memory issues, you can try to use the `optimize_memory` flag to save memory or
+        and use the `storage_device` flag to store the buffer on a different device.
+        Args:
+            capacity (int): Maximum number of transitions to store in the buffer.
+            device (str): The device where the tensors will be moved when sampling ("cuda:0" or "cpu").
+            state_keys (List[str]): The list of keys that appear in `state` and `next_state`.
+            image_augmentation_function (Optional[Callable]): A function that takes a batch of images
+                and returns a batch of augmented images. If None, a default augmentation function is used.
+            use_drq (bool): Whether to use the default DRQ image augmentation style, when sampling in the buffer.
+            storage_device: The device (e.g. "cpu" or "cuda:0") where the data will be stored.
+                Using "cpu" can help save GPU memory.
+            optimize_memory (bool): If True, optimizes memory by not storing duplicate next_states when
+                they can be derived from states. This is useful for large datasets where next_state[i] = state[i+1].
+        """
+        if capacity <= 0:
+            raise ValueError("Capacity must be greater than 0.")
+
+        self.capacity = capacity
+        self.device = device
+        self.storage_device = storage_device
+        self.position = 0
+        self.size = 0
+        self.initialized = False
+        self.optimize_memory = optimize_memory
+
+        # Track episode boundaries for memory optimization
+        self.episode_ends = torch.zeros(capacity, dtype=torch.bool, device=storage_device)
+
+        # If no state_keys provided, default to an empty list
+        self.state_keys = state_keys if state_keys is not None else []
+
+        self.image_augmentation_function = image_augmentation_function
+
+        if image_augmentation_function is None:
+            base_function = functools.partial(random_shift, pad=4)
+            self.image_augmentation_function = torch.compile(base_function)
+        self.use_drq = use_drq
+
+    def _initialize_storage(
+        self,
+        state: dict[str, torch.Tensor],
+        action: torch.Tensor,
+        complementary_info: dict[str, torch.Tensor] | None = None,
+    ):
+        """Initialize the storage tensors based on the first transition."""
+        # Determine shapes from the first transition
+        state_shapes = {key: val.squeeze(0).shape for key, val in state.items()}
+        action_shape = action.squeeze(0).shape
+
+        # Pre-allocate tensors for storage
+        self.states = {
+            key: torch.empty((self.capacity, *shape), device=self.storage_device)
+            for key, shape in state_shapes.items()
+        }
+        self.actions = torch.empty((self.capacity, *action_shape), device=self.storage_device)
+        self.rewards = torch.empty((self.capacity,), device=self.storage_device)
+
+        if not self.optimize_memory:
+            # Standard approach: store states and next_states separately
+            self.next_states = {
+                key: torch.empty((self.capacity, *shape), device=self.storage_device)
+                for key, shape in state_shapes.items()
+            }
+        else:
+            # Memory-optimized approach: don't allocate next_states buffer
+            # Just create a reference to states for consistent API
+            self.next_states = self.states  # Just a reference for API consistency
+
+        self.dones = torch.empty((self.capacity,), dtype=torch.bool, device=self.storage_device)
+        self.truncateds = torch.empty((self.capacity,), dtype=torch.bool, device=self.storage_device)
+
+        # Initialize storage for complementary_info
+        self.has_complementary_info = complementary_info is not None
+        self.complementary_info_keys = []
+        self.complementary_info = {}
+
+        if self.has_complementary_info:
+            self.complementary_info_keys = list(complementary_info.keys())
+            # Pre-allocate tensors for each key in complementary_info
+            for key, value in complementary_info.items():
+                if isinstance(value, torch.Tensor):
+                    value_shape = value.squeeze(0).shape
+                    self.complementary_info[key] = torch.empty(
+                        (self.capacity, *value_shape), device=self.storage_device
+                    )
+                elif isinstance(value, (int | float)):
+                    # Handle scalar values similar to reward
+                    self.complementary_info[key] = torch.empty((self.capacity,), device=self.storage_device)
+                else:
+                    raise ValueError(f"Unsupported type {type(value)} for complementary_info[{key}]")
+
+        self.initialized = True
+
+    def __len__(self):
+        return self.size
+
+    def add(
+        self,
+        state: dict[str, torch.Tensor],
+        action: torch.Tensor,
+        reward: float,
+        next_state: dict[str, torch.Tensor],
+        done: bool,
+        truncated: bool,
+        complementary_info: dict[str, torch.Tensor] | None = None,
+    ):
+        """Saves a transition, ensuring tensors are stored on the designated storage device."""
+        # Initialize storage if this is the first transition
+        if not self.initialized:
+            self._initialize_storage(state=state, action=action, complementary_info=complementary_info)
+
+        # Store the transition in pre-allocated tensors
+        for key in self.states:
+            self.states[key][self.position].copy_(state[key].squeeze(dim=0))
+
+            if not self.optimize_memory:
+                # Only store next_states if not optimizing memory
+                self.next_states[key][self.position].copy_(next_state[key].squeeze(dim=0))
+
+        self.actions[self.position].copy_(action.squeeze(dim=0))
+        self.rewards[self.position] = reward
+        self.dones[self.position] = done
+        self.truncateds[self.position] = truncated
+
+        # Handle complementary_info if provided and storage is initialized
+        if complementary_info is not None and self.has_complementary_info:
+            # Store the complementary_info
+            for key in self.complementary_info_keys:
+                if key in complementary_info:
+                    value = complementary_info[key]
+                    if isinstance(value, torch.Tensor):
+                        self.complementary_info[key][self.position].copy_(value.squeeze(dim=0))
+                    elif isinstance(value, (int | float)):
+                        self.complementary_info[key][self.position] = value
+
+        self.position = (self.position + 1) % self.capacity
+        self.size = min(self.size + 1, self.capacity)
+
+    def sample(self, batch_size: int) -> BatchTransition:
+        """Sample a random batch of transitions and collate them into batched tensors."""
+        if not self.initialized:
+            raise RuntimeError("Cannot sample from an empty buffer. Add transitions first.")
+
+        batch_size = min(batch_size, self.size)
+        high = max(0, self.size - 1) if self.optimize_memory and self.size < self.capacity else self.size
+
+        # Random indices for sampling - create on the same device as storage
+        idx = torch.randint(low=0, high=high, size=(batch_size,), device=self.storage_device)
+
+        # Identify image keys that need augmentation
+        image_keys = [k for k in self.states if k.startswith(OBS_IMAGE)] if self.use_drq else []
+
+        # Create batched state and next_state
+        batch_state = {}
+        batch_next_state = {}
+
+        # First pass: load all state tensors to target device
+        for key in self.states:
+            batch_state[key] = self.states[key][idx].to(self.device)
+
+            if not self.optimize_memory:
+                # Standard approach - load next_states directly
+                batch_next_state[key] = self.next_states[key][idx].to(self.device)
+            else:
+                # Memory-optimized approach - get next_state from the next index
+                next_idx = (idx + 1) % self.capacity
+                batch_next_state[key] = self.states[key][next_idx].to(self.device)
+
+        # Apply image augmentation in a batched way if needed
+        if self.use_drq and image_keys:
+            # Concatenate all images from state and next_state
+            all_images = []
+            for key in image_keys:
+                all_images.append(batch_state[key])
+                all_images.append(batch_next_state[key])
+
+            # Optimization: Batch all images and apply augmentation once
+            all_images_tensor = torch.cat(all_images, dim=0)
+            augmented_images = self.image_augmentation_function(all_images_tensor)
+
+            # Split the augmented images back to their sources
+            for i, key in enumerate(image_keys):
+                # Calculate offsets for the current image key:
+                # For each key, we have 2*batch_size images (batch_size for states, batch_size for next_states)
+                # States start at index i*2*batch_size and take up batch_size slots
+                batch_state[key] = augmented_images[i * 2 * batch_size : (i * 2 + 1) * batch_size]
+                # Next states start after the states at index (i*2+1)*batch_size and also take up batch_size slots
+                batch_next_state[key] = augmented_images[(i * 2 + 1) * batch_size : (i + 1) * 2 * batch_size]
+
+        # Sample other tensors
+        batch_actions = self.actions[idx].to(self.device)
+        batch_rewards = self.rewards[idx].to(self.device)
+        batch_dones = self.dones[idx].to(self.device).float()
+        batch_truncateds = self.truncateds[idx].to(self.device).float()
+
+        # Sample complementary_info if available
+        batch_complementary_info = None
+        if self.has_complementary_info:
+            batch_complementary_info = {}
+            for key in self.complementary_info_keys:
+                batch_complementary_info[key] = self.complementary_info[key][idx].to(self.device)
+
+        return BatchTransition(
+            state=batch_state,
+            action=batch_actions,
+            reward=batch_rewards,
+            next_state=batch_next_state,
+            done=batch_dones,
+            truncated=batch_truncateds,
+            complementary_info=batch_complementary_info,
+        )
+
+    def get_iterator(
+        self,
+        batch_size: int,
+        async_prefetch: bool = True,
+        queue_size: int = 2,
+    ):
+        """
+        Creates an infinite iterator that yields batches of transitions.
+        Will automatically restart when internal iterator is exhausted.
+
+        Args:
+            batch_size (int): Size of batches to sample
+            async_prefetch (bool): Whether to use asynchronous prefetching with threads (default: True)
+            queue_size (int): Number of batches to prefetch (default: 2)
+
+        Yields:
+            BatchTransition: Batched transitions
+        """
+        while True:  # Create an infinite loop
+            if async_prefetch:
+                # Get the standard iterator
+                iterator = self._get_async_iterator(queue_size=queue_size, batch_size=batch_size)
+            else:
+                iterator = self._get_naive_iterator(batch_size=batch_size, queue_size=queue_size)
+
+            # Yield all items from the iterator
+            with suppress(StopIteration):
+                yield from iterator
+
+    def _get_async_iterator(self, batch_size: int, queue_size: int = 2):
+        """
+        Create an iterator that continuously yields prefetched batches in a
+        background thread. The design is intentionally simple and avoids busy
+        waiting / complex state management.
+
+        Args:
+            batch_size (int): Size of batches to sample.
+            queue_size (int): Maximum number of prefetched batches to keep in
+                memory.
+
+        Yields:
+            BatchTransition: A batch sampled from the replay buffer.
+        """
+        import queue
+        import threading
+
+        data_queue: queue.Queue = queue.Queue(maxsize=queue_size)
+        shutdown_event = threading.Event()
+
+        def producer() -> None:
+            """Continuously put sampled batches into the queue until shutdown."""
+            while not shutdown_event.is_set():
+                try:
+                    batch = self.sample(batch_size)
+                    # The timeout ensures the thread unblocks if the queue is full
+                    # and the shutdown event gets set meanwhile.
+                    data_queue.put(batch, block=True, timeout=0.5)
+                except queue.Full:
+                    # Queue is full – loop again (will re-check shutdown_event)
+                    continue
+                except Exception:
+                    # Surface any unexpected error and terminate the producer.
+                    shutdown_event.set()
+
+        producer_thread = threading.Thread(target=producer, daemon=True)
+        producer_thread.start()
+
+        try:
+            while not shutdown_event.is_set():
+                try:
+                    yield data_queue.get(block=True)
+                except Exception:
+                    # If the producer already set the shutdown flag we exit.
+                    if shutdown_event.is_set():
+                        break
+        finally:
+            shutdown_event.set()
+            # Drain the queue quickly to help the thread exit if it's blocked on `put`.
+            while not data_queue.empty():
+                _ = data_queue.get_nowait()
+            # Give the producer thread a bit of time to finish.
+            producer_thread.join(timeout=1.0)
+
+    def _get_naive_iterator(self, batch_size: int, queue_size: int = 2):
+        """
+        Creates a simple non-threaded iterator that yields batches.
+
+        Args:
+            batch_size (int): Size of batches to sample
+            queue_size (int): Number of initial batches to prefetch
+
+        Yields:
+            BatchTransition: Batch transitions
+        """
+        import collections
+
+        queue = collections.deque()
+
+        def enqueue(n):
+            for _ in range(n):
+                data = self.sample(batch_size)
+                queue.append(data)
+
+        enqueue(queue_size)
+        while queue:
+            yield queue.popleft()
+            enqueue(1)
+
+    @classmethod
+    def from_lerobot_dataset(
+        cls,
+        lerobot_dataset: LeRobotDataset,
+        device: str = "cuda:0",
+        state_keys: Sequence[str] | None = None,
+        capacity: int | None = None,
+        image_augmentation_function: Callable | None = None,
+        use_drq: bool = True,
+        storage_device: str = "cpu",
+        optimize_memory: bool = False,
+    ) -> "ReplayBuffer":
+        """
+        Convert a LeRobotDataset into a ReplayBuffer.
+
+        Args:
+            lerobot_dataset (LeRobotDataset): The dataset to convert.
+            device (str): The device for sampling tensors. Defaults to "cuda:0".
+            state_keys (Sequence[str] | None): The list of keys that appear in `state` and `next_state`.
+            capacity (int | None): Buffer capacity. If None, uses dataset length.
+            action_mask (Sequence[int] | None): Indices of action dimensions to keep.
+            image_augmentation_function (Callable | None): Function for image augmentation.
+                If None, uses default random shift with pad=4.
+            use_drq (bool): Whether to use DrQ image augmentation when sampling.
+            storage_device (str): Device for storing tensor data. Using "cpu" saves GPU memory.
+            optimize_memory (bool): If True, reduces memory usage by not duplicating state data.
+
+        Returns:
+            ReplayBuffer: The replay buffer with dataset transitions.
+        """
+        if capacity is None:
+            capacity = len(lerobot_dataset)
+
+        if capacity < len(lerobot_dataset):
+            raise ValueError(
+                "The capacity of the ReplayBuffer must be greater than or equal to the length of the LeRobotDataset."
+            )
+
+        # Create replay buffer with image augmentation and DrQ settings
+        replay_buffer = cls(
+            capacity=capacity,
+            device=device,
+            state_keys=state_keys,
+            image_augmentation_function=image_augmentation_function,
+            use_drq=use_drq,
+            storage_device=storage_device,
+            optimize_memory=optimize_memory,
+        )
+
+        # Convert dataset to transitions
+        list_transition = cls._lerobotdataset_to_transitions(dataset=lerobot_dataset, state_keys=state_keys)
+
+        # Initialize the buffer with the first transition to set up storage tensors
+        if list_transition:
+            first_transition = list_transition[0]
+            first_state = {k: v.to(device) for k, v in first_transition["state"].items()}
+            first_action = first_transition[ACTION].to(device)
+
+            # Get complementary info if available
+            first_complementary_info = None
+            if (
+                "complementary_info" in first_transition
+                and first_transition["complementary_info"] is not None
+            ):
+                first_complementary_info = {
+                    k: v.to(device) for k, v in first_transition["complementary_info"].items()
+                }
+
+            replay_buffer._initialize_storage(
+                state=first_state, action=first_action, complementary_info=first_complementary_info
+            )
+
+        # Fill the buffer with all transitions
+        for data in list_transition:
+            for k, v in data.items():
+                if isinstance(v, dict):
+                    for key, tensor in v.items():
+                        v[key] = tensor.to(storage_device)
+                elif isinstance(v, torch.Tensor):
+                    data[k] = v.to(storage_device)
+
+            action = data[ACTION]
+
+            replay_buffer.add(
+                state=data["state"],
+                action=action,
+                reward=data["reward"],
+                next_state=data["next_state"],
+                done=data["done"],
+                truncated=False,  # NOTE: Truncation are not supported yet in lerobot dataset
+                complementary_info=data.get("complementary_info", None),
+            )
+
+        return replay_buffer
+
+    def to_lerobot_dataset(
+        self,
+        repo_id: str,
+        fps=1,
+        root=None,
+        task_name="from_replay_buffer",
+    ) -> LeRobotDataset:
+        """
+        Converts all transitions in this ReplayBuffer into a single LeRobotDataset object.
+        """
+        if self.size == 0:
+            raise ValueError("The replay buffer is empty. Cannot convert to a dataset.")
+
+        # Create features dictionary for the dataset
+        features = {
+            "index": {"dtype": "int64", "shape": [1]},  # global index across episodes
+            "episode_index": {"dtype": "int64", "shape": [1]},  # which episode
+            "frame_index": {"dtype": "int64", "shape": [1]},  # index inside an episode
+            "timestamp": {"dtype": "float32", "shape": [1]},  # for now we store dummy
+            "task_index": {"dtype": "int64", "shape": [1]},
+        }
+
+        # Add "action"
+        sample_action = self.actions[0]
+        act_info = guess_feature_info(t=sample_action, name=ACTION)
+        features[ACTION] = act_info
+
+        # Add "reward" and "done"
+        features[REWARD] = {"dtype": "float32", "shape": (1,)}
+        features[DONE] = {"dtype": "bool", "shape": (1,)}
+
+        # Add state keys
+        for key in self.states:
+            sample_val = self.states[key][0]
+            f_info = guess_feature_info(t=sample_val, name=key)
+            features[key] = f_info
+
+        # Add complementary_info keys if available
+        if self.has_complementary_info:
+            for key in self.complementary_info_keys:
+                sample_val = self.complementary_info[key][0]
+                if isinstance(sample_val, torch.Tensor) and sample_val.ndim == 0:
+                    sample_val = sample_val.unsqueeze(0)
+                f_info = guess_feature_info(t=sample_val, name=f"complementary_info.{key}")
+                features[f"complementary_info.{key}"] = f_info
+
+        # Create an empty LeRobotDataset
+        lerobot_dataset = LeRobotDataset.create(
+            repo_id=repo_id,
+            fps=fps,
+            root=root,
+            robot_type=None,
+            features=features,
+            use_videos=True,
+        )
+
+        # Start writing images if needed
+        lerobot_dataset.start_image_writer(num_processes=0, num_threads=3)
+
+        # Convert transitions into episodes and frames
+
+        for idx in range(self.size):
+            actual_idx = (self.position - self.size + idx) % self.capacity
+
+            frame_dict = {}
+
+            # Fill the data for state keys
+            for key in self.states:
+                frame_dict[key] = self.states[key][actual_idx].cpu()
+
+            # Fill action, reward, done
+            frame_dict[ACTION] = self.actions[actual_idx].cpu()
+            frame_dict[REWARD] = torch.tensor([self.rewards[actual_idx]], dtype=torch.float32).cpu()
+            frame_dict[DONE] = torch.tensor([self.dones[actual_idx]], dtype=torch.bool).cpu()
+            frame_dict["task"] = task_name
+
+            # Add complementary_info if available
+            if self.has_complementary_info:
+                for key in self.complementary_info_keys:
+                    val = self.complementary_info[key][actual_idx]
+                    # Convert tensors to CPU
+                    if isinstance(val, torch.Tensor):
+                        if val.ndim == 0:
+                            val = val.unsqueeze(0)
+                        frame_dict[f"complementary_info.{key}"] = val.cpu()
+                    # Non-tensor values can be used directly
+                    else:
+                        frame_dict[f"complementary_info.{key}"] = val
+
+            # Add to the dataset's buffer
+            lerobot_dataset.add_frame(frame_dict)
+
+            # If we reached an episode boundary, call save_episode, reset counters
+            if self.dones[actual_idx] or self.truncateds[actual_idx]:
+                lerobot_dataset.save_episode()
+
+        # Save any remaining frames in the buffer
+        if lerobot_dataset.episode_buffer["size"] > 0:
+            lerobot_dataset.save_episode()
+
+        lerobot_dataset.stop_image_writer()
+        lerobot_dataset.finalize()
+
+        return lerobot_dataset
+
+    @staticmethod
+    def _lerobotdataset_to_transitions(
+        dataset: LeRobotDataset,
+        state_keys: Sequence[str] | None = None,
+    ) -> list[Transition]:
+        """
+        Convert a LeRobotDataset into a list of RL (s, a, r, s', done) transitions.
+
+        Args:
+            dataset (LeRobotDataset):
+                The dataset to convert. Each item in the dataset is expected to have
+                at least the following keys:
+                {
+                    "action": ...
+                    "next.reward": ...
+                    "next.done": ...
+                    "episode_index": ...
+                }
+                plus whatever your 'state_keys' specify.
+
+            state_keys (Sequence[str] | None):
+                The dataset keys to include in 'state' and 'next_state'. Their names
+                will be kept as-is in the output transitions. E.g.
+                ["observation.state", "observation.environment_state"].
+                If None, you must handle or define default keys.
+
+        Returns:
+            transitions (List[Transition]):
+                A list of Transition dictionaries with the same length as `dataset`.
+        """
+        if state_keys is None:
+            raise ValueError("State keys must be provided when converting LeRobotDataset to Transitions.")
+
+        transitions = []
+        num_frames = len(dataset)
+
+        # Check if the dataset has "next.done" key
+        sample = dataset[0]
+        has_done_key = DONE in sample
+
+        # Check for complementary_info keys
+        complementary_info_keys = [key for key in sample if key.startswith("complementary_info.")]
+        has_complementary_info = len(complementary_info_keys) > 0
+
+        # If not, we need to infer it from episode boundaries
+        if not has_done_key:
+            print("'next.done' key not found in dataset. Inferring from episode boundaries...")
+
+        for i in tqdm(range(num_frames)):
+            current_sample = dataset[i]
+
+            # ----- 1) Current state -----
+            current_state: dict[str, torch.Tensor] = {}
+            for key in state_keys:
+                val = current_sample[key]
+                current_state[key] = val.unsqueeze(0)  # Add batch dimension
+
+            # ----- 2) Action -----
+            action = current_sample[ACTION].unsqueeze(0)  # Add batch dimension
+
+            # ----- 3) Reward and done -----
+            reward = float(current_sample[REWARD].item())  # ensure float
+
+            # Determine done flag - use next.done if available, otherwise infer from episode boundaries
+            if has_done_key:
+                done = bool(current_sample[DONE].item())  # ensure bool
+            else:
+                # If this is the last frame or if next frame is in a different episode, mark as done
+                done = False
+                if i == num_frames - 1:
+                    done = True
+                elif i < num_frames - 1:
+                    next_sample = dataset[i + 1]
+                    if next_sample["episode_index"] != current_sample["episode_index"]:
+                        done = True
+
+            # TODO: (azouitine) Handle truncation (using the same value as done for now)
+            truncated = done
+
+            # ----- 4) Next state -----
+            # If not done and the next sample is in the same episode, we pull the next sample's state.
+            # Otherwise (done=True or next sample crosses to a new episode), next_state = current_state.
+            next_state = current_state  # default
+            if not done and (i < num_frames - 1):
+                next_sample = dataset[i + 1]
+                if next_sample["episode_index"] == current_sample["episode_index"]:
+                    # Build next_state from the same keys
+                    next_state_data: dict[str, torch.Tensor] = {}
+                    for key in state_keys:
+                        val = next_sample[key]
+                        next_state_data[key] = val.unsqueeze(0)  # Add batch dimension
+                    next_state = next_state_data
+
+            # ----- 5) Complementary info (if available) -----
+            complementary_info = None
+            if has_complementary_info:
+                complementary_info = {}
+                for key in complementary_info_keys:
+                    # Strip the "complementary_info." prefix to get the actual key
+                    clean_key = key[len("complementary_info.") :]
+                    val = current_sample[key]
+                    # Handle tensor and non-tensor values differently
+                    if isinstance(val, torch.Tensor):
+                        complementary_info[clean_key] = val.unsqueeze(0)  # Add batch dimension
+                    else:
+                        # TODO: (azouitine) Check if it's necessary to convert to tensor
+                        # For non-tensor values, use directly
+                        complementary_info[clean_key] = val
+
+            # ----- Construct the Transition -----
+            transition = Transition(
+                state=current_state,
+                action=action,
+                reward=reward,
+                next_state=next_state,
+                done=done,
+                truncated=truncated,
+                complementary_info=complementary_info,
+            )
+            transitions.append(transition)
+
+        return transitions
+
+
+# Utility function to guess shapes/dtypes from a tensor
+def guess_feature_info(t, name: str):
+    """
+    Return a dictionary with the 'dtype' and 'shape' for a given tensor or scalar value.
+    If it looks like a 3D (C,H,W) shape, we might consider it an 'image'.
+    Otherwise default to appropriate dtype for numeric.
+    """
+
+    shape = tuple(t.shape)
+    # Basic guess: if we have exactly 3 dims and shape[0] in {1, 3}, guess 'image'
+    if len(shape) == 3 and shape[0] in [1, 3]:
+        return {
+            "dtype": "image",
+            "shape": shape,
+        }
+    else:
+        # Otherwise treat as numeric
+        return {
+            "dtype": "float32",
+            "shape": shape,
+        }
+
+
+def concatenate_batch_transitions(
+    left_batch_transitions: BatchTransition, right_batch_transition: BatchTransition
+) -> BatchTransition:
+    """
+    Concatenates two BatchTransition objects into one.
+
+    This function merges the right BatchTransition into the left one by concatenating
+    all corresponding tensors along dimension 0. The operation modifies the left_batch_transitions
+    in place and also returns it.
+
+    Args:
+        left_batch_transitions (BatchTransition): The first batch to concatenate and the one
+            that will be modified in place.
+        right_batch_transition (BatchTransition): The second batch to append to the first one.
+
+    Returns:
+        BatchTransition: The concatenated batch (same object as left_batch_transitions).
+
+    Warning:
+        This function modifies the left_batch_transitions object in place.
+    """
+    # Concatenate state fields
+    left_batch_transitions["state"] = {
+        key: torch.cat(
+            [left_batch_transitions["state"][key], right_batch_transition["state"][key]],
+            dim=0,
+        )
+        for key in left_batch_transitions["state"]
+    }
+
+    # Concatenate basic fields
+    left_batch_transitions[ACTION] = torch.cat(
+        [left_batch_transitions[ACTION], right_batch_transition[ACTION]], dim=0
+    )
+    left_batch_transitions["reward"] = torch.cat(
+        [left_batch_transitions["reward"], right_batch_transition["reward"]], dim=0
+    )
+
+    # Concatenate next_state fields
+    left_batch_transitions["next_state"] = {
+        key: torch.cat(
+            [left_batch_transitions["next_state"][key], right_batch_transition["next_state"][key]],
+            dim=0,
+        )
+        for key in left_batch_transitions["next_state"]
+    }
+
+    # Concatenate done and truncated fields
+    left_batch_transitions["done"] = torch.cat(
+        [left_batch_transitions["done"], right_batch_transition["done"]], dim=0
+    )
+    left_batch_transitions["truncated"] = torch.cat(
+        [left_batch_transitions["truncated"], right_batch_transition["truncated"]],
+        dim=0,
+    )
+
+    # Handle complementary_info
+    left_info = left_batch_transitions.get("complementary_info")
+    right_info = right_batch_transition.get("complementary_info")
+
+    # Only process if right_info exists
+    if right_info is not None:
+        # Initialize left complementary_info if needed
+        if left_info is None:
+            left_batch_transitions["complementary_info"] = right_info
+        else:
+            # Concatenate each field
+            for key in right_info:
+                if key in left_info:
+                    left_info[key] = torch.cat([left_info[key], right_info[key]], dim=0)
+                else:
+                    left_info[key] = right_info[key]
+
+    return left_batch_transitions
diff --git a/lerobot/src/lerobot/rl/crop_dataset_roi.py b/lerobot/src/lerobot/rl/crop_dataset_roi.py
new file mode 100644
index 0000000000000000000000000000000000000000..4345fed3c104cd514fd6bf9f0baca99736d2001c
--- /dev/null
+++ b/lerobot/src/lerobot/rl/crop_dataset_roi.py
@@ -0,0 +1,326 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import argparse
+import json
+from copy import deepcopy
+from pathlib import Path
+
+import cv2
+import torch
+import torchvision.transforms.functional as F  # type: ignore  # noqa: N812
+from tqdm import tqdm  # type: ignore
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.utils.constants import DONE, REWARD
+
+
+def select_rect_roi(img):
+    """
+    Allows the user to draw a rectangular ROI on the image.
+
+    The user must click and drag to draw the rectangle.
+    - While dragging, the rectangle is dynamically drawn.
+    - On mouse button release, the rectangle is fixed.
+    - Press 'c' to confirm the selection.
+    - Press 'r' to reset the selection.
+    - Press ESC to cancel.
+
+    Returns:
+        A tuple (top, left, height, width) representing the rectangular ROI,
+        or None if no valid ROI is selected.
+    """
+    # Create a working copy of the image
+    clone = img.copy()
+    working_img = clone.copy()
+
+    roi = None  # Will store the final ROI as (top, left, height, width)
+    drawing = False
+    index_x, index_y = -1, -1  # Initial click coordinates
+
+    def mouse_callback(event, x, y, flags, param):
+        nonlocal index_x, index_y, drawing, roi, working_img
+
+        if event == cv2.EVENT_LBUTTONDOWN:
+            # Start drawing: record starting coordinates
+            drawing = True
+            index_x, index_y = x, y
+
+        elif event == cv2.EVENT_MOUSEMOVE:
+            if drawing:
+                # Compute the top-left and bottom-right corners regardless of drag direction
+                top = min(index_y, y)
+                left = min(index_x, x)
+                bottom = max(index_y, y)
+                right = max(index_x, x)
+                # Show a temporary image with the current rectangle drawn
+                temp = working_img.copy()
+                cv2.rectangle(temp, (left, top), (right, bottom), (0, 255, 0), 2)
+                cv2.imshow("Select ROI", temp)
+
+        elif event == cv2.EVENT_LBUTTONUP:
+            # Finish drawing
+            drawing = False
+            top = min(index_y, y)
+            left = min(index_x, x)
+            bottom = max(index_y, y)
+            right = max(index_x, x)
+            height = bottom - top
+            width = right - left
+            roi = (top, left, height, width)  # (top, left, height, width)
+            # Draw the final rectangle on the working image and display it
+            working_img = clone.copy()
+            cv2.rectangle(working_img, (left, top), (right, bottom), (0, 255, 0), 2)
+            cv2.imshow("Select ROI", working_img)
+
+    # Create the window and set the callback
+    cv2.namedWindow("Select ROI")
+    cv2.setMouseCallback("Select ROI", mouse_callback)
+    cv2.imshow("Select ROI", working_img)
+
+    print("Instructions for ROI selection:")
+    print("  - Click and drag to draw a rectangular ROI.")
+    print("  - Press 'c' to confirm the selection.")
+    print("  - Press 'r' to reset and draw again.")
+    print("  - Press ESC to cancel the selection.")
+
+    # Wait until the user confirms with 'c', resets with 'r', or cancels with ESC
+    while True:
+        key = cv2.waitKey(1) & 0xFF
+        # Confirm ROI if one has been drawn
+        if key == ord("c") and roi is not None:
+            break
+        # Reset: clear the ROI and restore the original image
+        elif key == ord("r"):
+            working_img = clone.copy()
+            roi = None
+            cv2.imshow("Select ROI", working_img)
+        # Cancel selection for this image
+        elif key == 27:  # ESC key
+            roi = None
+            break
+
+    cv2.destroyWindow("Select ROI")
+    return roi
+
+
+def select_square_roi_for_images(images: dict) -> dict:
+    """
+    For each image in the provided dictionary, open a window to allow the user
+    to select a rectangular ROI. Returns a dictionary mapping each key to a tuple
+    (top, left, height, width) representing the ROI.
+
+    Parameters:
+        images (dict): Dictionary where keys are identifiers and values are OpenCV images.
+
+    Returns:
+        dict: Mapping of image keys to the selected rectangular ROI.
+    """
+    selected_rois = {}
+
+    for key, img in images.items():
+        if img is None:
+            print(f"Image for key '{key}' is None, skipping.")
+            continue
+
+        print(f"\nSelect rectangular ROI for image with key: '{key}'")
+        roi = select_rect_roi(img)
+
+        if roi is None:
+            print(f"No valid ROI selected for '{key}'.")
+        else:
+            selected_rois[key] = roi
+            print(f"ROI for '{key}': {roi}")
+
+    return selected_rois
+
+
+def get_image_from_lerobot_dataset(dataset: LeRobotDataset):
+    """
+    Find the first row in the dataset and extract the image in order to be used for the crop.
+    """
+    row = dataset[0]
+    image_dict = {}
+    for k in row:
+        if "image" in k:
+            image_dict[k] = deepcopy(row[k])
+    return image_dict
+
+
+def convert_lerobot_dataset_to_cropped_lerobot_dataset(
+    original_dataset: LeRobotDataset,
+    crop_params_dict: dict[str, tuple[int, int, int, int]],
+    new_repo_id: str,
+    new_dataset_root: str,
+    resize_size: tuple[int, int] = (128, 128),
+    push_to_hub: bool = False,
+    task: str = "",
+) -> LeRobotDataset:
+    """
+    Converts an existing LeRobotDataset by iterating over its episodes and frames,
+    applying cropping and resizing to image observations, and saving a new dataset
+    with the transformed data.
+
+    Args:
+        original_dataset (LeRobotDataset): The source dataset.
+        crop_params_dict (Dict[str, Tuple[int, int, int, int]]):
+            A dictionary mapping observation keys to crop parameters (top, left, height, width).
+        new_repo_id (str): Repository id for the new dataset.
+        new_dataset_root (str): The root directory where the new dataset will be written.
+        resize_size (Tuple[int, int], optional): The target size (height, width) after cropping.
+            Defaults to (128, 128).
+
+    Returns:
+        LeRobotDataset: A new LeRobotDataset where the specified image observations have been cropped
+                        and resized.
+    """
+    # 1. Create a new (empty) LeRobotDataset for writing.
+    new_dataset = LeRobotDataset.create(
+        repo_id=new_repo_id,
+        fps=int(original_dataset.fps),
+        root=new_dataset_root,
+        robot_type=original_dataset.meta.robot_type,
+        features=original_dataset.meta.info["features"],
+        use_videos=len(original_dataset.meta.video_keys) > 0,
+    )
+
+    # Update the metadata for every image key that will be cropped:
+    # (Here we simply set the shape to be the final resize_size.)
+    for key in crop_params_dict:
+        if key in new_dataset.meta.info["features"]:
+            new_dataset.meta.info["features"][key]["shape"] = [3] + list(resize_size)
+
+    # TODO:  Directly modify the mp4 video + meta info features, instead of recreating a dataset
+    prev_episode_index = 0
+    for frame_idx in tqdm(range(len(original_dataset))):
+        frame = original_dataset[frame_idx]
+
+        # Create a copy of the frame to add to the new dataset
+        new_frame = {}
+        for key, value in frame.items():
+            if key in ("task_index", "timestamp", "episode_index", "frame_index", "index", "task"):
+                continue
+            if key in (DONE, REWARD):
+                # if not isinstance(value, str) and len(value.shape) == 0:
+                value = value.unsqueeze(0)
+
+            if key in crop_params_dict:
+                top, left, height, width = crop_params_dict[key]
+                # Apply crop then resize.
+                cropped = F.crop(value, top, left, height, width)
+                value = F.resize(cropped, resize_size)
+                value = value.clamp(0, 1)
+            if key.startswith("complementary_info") and isinstance(value, torch.Tensor) and value.dim() == 0:
+                value = value.unsqueeze(0)
+            new_frame[key] = value
+
+        new_frame["task"] = task
+        new_dataset.add_frame(new_frame)
+
+        if frame["episode_index"].item() != prev_episode_index:
+            # Save the episode
+            new_dataset.save_episode()
+            prev_episode_index = frame["episode_index"].item()
+
+    # Save the last episode
+    new_dataset.save_episode()
+
+    if push_to_hub:
+        new_dataset.push_to_hub()
+
+    return new_dataset
+
+
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser(description="Crop rectangular ROIs from a LeRobot dataset.")
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        default="lerobot",
+        help="The repository id of the LeRobot dataset to process.",
+    )
+    parser.add_argument(
+        "--root",
+        type=str,
+        default=None,
+        help="The root directory of the LeRobot dataset.",
+    )
+    parser.add_argument(
+        "--crop-params-path",
+        type=str,
+        default=None,
+        help="The path to the JSON file containing the ROIs.",
+    )
+    parser.add_argument(
+        "--push-to-hub",
+        action="store_true",
+        help="Whether to push the new dataset to the hub.",
+    )
+    parser.add_argument(
+        "--task",
+        type=str,
+        default="",
+        help="The natural language task to describe the dataset.",
+    )
+    parser.add_argument(
+        "--new-repo-id",
+        type=str,
+        default=None,
+        help="The repository id for the new cropped and resized dataset. If not provided, it defaults to `repo_id` + '_cropped_resized'.",
+    )
+    args = parser.parse_args()
+
+    dataset = LeRobotDataset(repo_id=args.repo_id, root=args.root)
+
+    images = get_image_from_lerobot_dataset(dataset)
+    images = {k: v.cpu().permute(1, 2, 0).numpy() for k, v in images.items()}
+    images = {k: (v * 255).astype("uint8") for k, v in images.items()}
+
+    if args.crop_params_path is None:
+        rois = select_square_roi_for_images(images)
+    else:
+        with open(args.crop_params_path) as f:
+            rois = json.load(f)
+
+    # Print the selected rectangular ROIs
+    print("\nSelected Rectangular Regions of Interest (top, left, height, width):")
+    for key, roi in rois.items():
+        print(f"{key}: {roi}")
+
+    new_repo_id = args.new_repo_id if args.new_repo_id else args.repo_id + "_cropped_resized"
+
+    if args.new_repo_id:
+        new_dataset_name = args.new_repo_id.split("/")[-1]
+        # Parent 1: HF user, Parent 2: HF LeRobot Home
+        new_dataset_root = dataset.root.parent.parent / new_dataset_name
+    else:
+        new_dataset_root = Path(str(dataset.root) + "_cropped_resized")
+
+    cropped_resized_dataset = convert_lerobot_dataset_to_cropped_lerobot_dataset(
+        original_dataset=dataset,
+        crop_params_dict=rois,
+        new_repo_id=new_repo_id,
+        new_dataset_root=new_dataset_root,
+        resize_size=(128, 128),
+        push_to_hub=args.push_to_hub,
+        task=args.task,
+    )
+
+    meta_dir = new_dataset_root / "meta"
+    meta_dir.mkdir(exist_ok=True)
+
+    with open(meta_dir / "crop_params.json", "w") as f:
+        json.dump(rois, f, indent=4)
diff --git a/lerobot/src/lerobot/rl/eval_policy.py b/lerobot/src/lerobot/rl/eval_policy.py
new file mode 100644
index 0000000000000000000000000000000000000000..fb2504f2a6f13d7a7f6526b54c52ddab648e1a2a
--- /dev/null
+++ b/lerobot/src/lerobot/rl/eval_policy.py
@@ -0,0 +1,75 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+
+from lerobot.cameras import opencv  # noqa: F401
+from lerobot.configs import parser
+from lerobot.configs.train import TrainRLServerPipelineConfig
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.factory import make_policy
+from lerobot.robots import (  # noqa: F401
+    RobotConfig,
+    make_robot_from_config,
+    so_follower,
+)
+from lerobot.teleoperators import (
+    gamepad,  # noqa: F401
+    so_leader,  # noqa: F401
+)
+
+from .gym_manipulator import make_robot_env
+
+logging.basicConfig(level=logging.INFO)
+
+
+def eval_policy(env, policy, n_episodes):
+    sum_reward_episode = []
+    for _ in range(n_episodes):
+        obs, _ = env.reset()
+        episode_reward = 0.0
+        while True:
+            action = policy.select_action(obs)
+            obs, reward, terminated, truncated, _ = env.step(action)
+            episode_reward += reward
+            if terminated or truncated:
+                break
+        sum_reward_episode.append(episode_reward)
+
+    logging.info(f"Success after 20 steps {sum_reward_episode}")
+    logging.info(f"success rate {sum(sum_reward_episode) / len(sum_reward_episode)}")
+
+
+@parser.wrap()
+def main(cfg: TrainRLServerPipelineConfig):
+    env_cfg = cfg.env
+    env = make_robot_env(env_cfg)
+    dataset_cfg = cfg.dataset
+    dataset = LeRobotDataset(repo_id=dataset_cfg.repo_id)
+    dataset_meta = dataset.meta
+
+    policy = make_policy(
+        cfg=cfg.policy,
+        # env_cfg=cfg.env,
+        ds_meta=dataset_meta,
+    )
+    policy = policy.from_pretrained(env_cfg.pretrained_policy_name_or_path)
+    policy.eval()
+
+    eval_policy(env, policy=policy, n_episodes=10)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/rl/gym_manipulator.py b/lerobot/src/lerobot/rl/gym_manipulator.py
new file mode 100644
index 0000000000000000000000000000000000000000..f5fcb74372a5bd75444921ee84972ce0a18dd3ca
--- /dev/null
+++ b/lerobot/src/lerobot/rl/gym_manipulator.py
@@ -0,0 +1,790 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from dataclasses import dataclass
+from typing import Any
+
+import gymnasium as gym
+import numpy as np
+import torch
+
+from lerobot.cameras import opencv  # noqa: F401
+from lerobot.configs import parser
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.envs.configs import HILSerlRobotEnvConfig
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    AddTeleopActionAsComplimentaryDataStep,
+    AddTeleopEventsAsInfoStep,
+    DataProcessorPipeline,
+    DeviceProcessorStep,
+    EnvTransition,
+    GripperPenaltyProcessorStep,
+    GymHILAdapterProcessorStep,
+    ImageCropResizeProcessorStep,
+    InterventionActionProcessorStep,
+    MapDeltaActionToRobotActionStep,
+    MapTensorToDeltaActionDictStep,
+    Numpy2TorchActionProcessorStep,
+    RewardClassifierProcessorStep,
+    RobotActionToPolicyActionProcessorStep,
+    RobotObservation,
+    TimeLimitProcessorStep,
+    Torch2NumpyActionProcessorStep,
+    TransitionKey,
+    VanillaObservationProcessorStep,
+    create_transition,
+)
+from lerobot.processor.converters import identity_transition
+from lerobot.robots import (  # noqa: F401
+    RobotConfig,
+    make_robot_from_config,
+    so_follower,
+)
+from lerobot.robots.robot import Robot
+from lerobot.robots.so_follower.robot_kinematic_processor import (
+    EEBoundsAndSafety,
+    EEReferenceAndDelta,
+    ForwardKinematicsJointsToEEObservation,
+    GripperVelocityToJoint,
+    InverseKinematicsRLStep,
+)
+from lerobot.teleoperators import (
+    gamepad,  # noqa: F401
+    keyboard,  # noqa: F401
+    make_teleoperator_from_config,
+    so_leader,  # noqa: F401
+)
+from lerobot.teleoperators.teleoperator import Teleoperator
+from lerobot.teleoperators.utils import TeleopEvents
+from lerobot.utils.constants import ACTION, DONE, OBS_IMAGES, OBS_STATE, REWARD
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import log_say
+
+from .joint_observations_processor import JointVelocityProcessorStep, MotorCurrentProcessorStep
+
+logging.basicConfig(level=logging.INFO)
+
+
+@dataclass
+class DatasetConfig:
+    """Configuration for dataset creation and management."""
+
+    repo_id: str
+    task: str
+    root: str | None = None
+    num_episodes_to_record: int = 5
+    replay_episode: int | None = None
+    push_to_hub: bool = False
+
+
+@dataclass
+class GymManipulatorConfig:
+    """Main configuration for gym manipulator environment."""
+
+    env: HILSerlRobotEnvConfig
+    dataset: DatasetConfig
+    mode: str | None = None  # Either "record", "replay", None
+    device: str = "cpu"
+
+
+def reset_follower_position(robot_arm: Robot, target_position: np.ndarray) -> None:
+    """Reset robot arm to target position using smooth trajectory."""
+    current_position_dict = robot_arm.bus.sync_read("Present_Position")
+    current_position = np.array(
+        [current_position_dict[name] for name in current_position_dict], dtype=np.float32
+    )
+    trajectory = torch.from_numpy(
+        np.linspace(current_position, target_position, 50)
+    )  # NOTE: 30 is just an arbitrary number
+    for pose in trajectory:
+        action_dict = dict(zip(current_position_dict, pose, strict=False))
+        robot_arm.bus.sync_write("Goal_Position", action_dict)
+        precise_sleep(0.015)
+
+
+class RobotEnv(gym.Env):
+    """Gym environment for robotic control with human intervention support."""
+
+    def __init__(
+        self,
+        robot,
+        use_gripper: bool = False,
+        display_cameras: bool = False,
+        reset_pose: list[float] | None = None,
+        reset_time_s: float = 5.0,
+    ) -> None:
+        """Initialize robot environment with configuration options.
+
+        Args:
+            robot: Robot interface for hardware communication.
+            use_gripper: Whether to include gripper in action space.
+            display_cameras: Whether to show camera feeds during execution.
+            reset_pose: Joint positions for environment reset.
+            reset_time_s: Time to wait during reset.
+        """
+        super().__init__()
+
+        self.robot = robot
+        self.display_cameras = display_cameras
+
+        # Connect to the robot if not already connected.
+        if not self.robot.is_connected:
+            self.robot.connect()
+
+        # Episode tracking.
+        self.current_step = 0
+        self.episode_data = None
+
+        self._joint_names = [f"{key}.pos" for key in self.robot.bus.motors]
+        self._image_keys = self.robot.cameras.keys()
+
+        self.reset_pose = reset_pose
+        self.reset_time_s = reset_time_s
+
+        self.use_gripper = use_gripper
+
+        self._joint_names = list(self.robot.bus.motors.keys())
+        self._raw_joint_positions = None
+
+        self._setup_spaces()
+
+    def _get_observation(self) -> RobotObservation:
+        """Get current robot observation including joint positions and camera images."""
+        obs_dict = self.robot.get_observation()
+        raw_joint_joint_position = {f"{name}.pos": obs_dict[f"{name}.pos"] for name in self._joint_names}
+        joint_positions = np.array([raw_joint_joint_position[f"{name}.pos"] for name in self._joint_names])
+
+        images = {key: obs_dict[key] for key in self._image_keys}
+
+        return {"agent_pos": joint_positions, "pixels": images, **raw_joint_joint_position}
+
+    def _setup_spaces(self) -> None:
+        """Configure observation and action spaces based on robot capabilities."""
+        current_observation = self._get_observation()
+
+        observation_spaces = {}
+
+        # Define observation spaces for images and other states.
+        if current_observation is not None and "pixels" in current_observation:
+            prefix = OBS_IMAGES
+            observation_spaces = {
+                f"{prefix}.{key}": gym.spaces.Box(
+                    low=0, high=255, shape=current_observation["pixels"][key].shape, dtype=np.uint8
+                )
+                for key in current_observation["pixels"]
+            }
+
+        if current_observation is not None:
+            agent_pos = current_observation["agent_pos"]
+            observation_spaces[OBS_STATE] = gym.spaces.Box(
+                low=0,
+                high=10,
+                shape=agent_pos.shape,
+                dtype=np.float32,
+            )
+
+        self.observation_space = gym.spaces.Dict(observation_spaces)
+
+        # Define the action space for joint positions along with setting an intervention flag.
+        action_dim = 3
+        bounds = {}
+        bounds["min"] = -np.ones(action_dim)
+        bounds["max"] = np.ones(action_dim)
+
+        if self.use_gripper:
+            action_dim += 1
+            bounds["min"] = np.concatenate([bounds["min"], [0]])
+            bounds["max"] = np.concatenate([bounds["max"], [2]])
+
+        self.action_space = gym.spaces.Box(
+            low=bounds["min"],
+            high=bounds["max"],
+            shape=(action_dim,),
+            dtype=np.float32,
+        )
+
+    def reset(
+        self, *, seed: int | None = None, options: dict[str, Any] | None = None
+    ) -> tuple[RobotObservation, dict[str, Any]]:
+        """Reset environment to initial state.
+
+        Args:
+            seed: Random seed for reproducibility.
+            options: Additional reset options.
+
+        Returns:
+            Tuple of (observation, info) dictionaries.
+        """
+        # Reset the robot
+        # self.robot.reset()
+        start_time = time.perf_counter()
+        if self.reset_pose is not None:
+            log_say("Reset the environment.", play_sounds=True)
+            reset_follower_position(self.robot, np.array(self.reset_pose))
+            log_say("Reset the environment done.", play_sounds=True)
+
+        precise_sleep(max(self.reset_time_s - (time.perf_counter() - start_time), 0.0))
+
+        super().reset(seed=seed, options=options)
+
+        # Reset episode tracking variables.
+        self.current_step = 0
+        self.episode_data = None
+        obs = self._get_observation()
+        self._raw_joint_positions = {f"{key}.pos": obs[f"{key}.pos"] for key in self._joint_names}
+        return obs, {TeleopEvents.IS_INTERVENTION: False}
+
+    def step(self, action) -> tuple[RobotObservation, float, bool, bool, dict[str, Any]]:
+        """Execute one environment step with given action."""
+        joint_targets_dict = {f"{key}.pos": action[i] for i, key in enumerate(self.robot.bus.motors.keys())}
+
+        self.robot.send_action(joint_targets_dict)
+
+        obs = self._get_observation()
+
+        self._raw_joint_positions = {f"{key}.pos": obs[f"{key}.pos"] for key in self._joint_names}
+
+        if self.display_cameras:
+            self.render()
+
+        self.current_step += 1
+
+        reward = 0.0
+        terminated = False
+        truncated = False
+
+        return (
+            obs,
+            reward,
+            terminated,
+            truncated,
+            {TeleopEvents.IS_INTERVENTION: False},
+        )
+
+    def render(self) -> None:
+        """Display robot camera feeds."""
+        import cv2
+
+        current_observation = self._get_observation()
+        if current_observation is not None:
+            image_keys = [key for key in current_observation if "image" in key]
+
+            for key in image_keys:
+                cv2.imshow(key, cv2.cvtColor(current_observation[key].numpy(), cv2.COLOR_RGB2BGR))
+                cv2.waitKey(1)
+
+    def close(self) -> None:
+        """Close environment and disconnect robot."""
+        if self.robot.is_connected:
+            self.robot.disconnect()
+
+    def get_raw_joint_positions(self) -> dict[str, float]:
+        """Get raw joint positions."""
+        return self._raw_joint_positions
+
+
+def make_robot_env(cfg: HILSerlRobotEnvConfig) -> tuple[gym.Env, Any]:
+    """Create robot environment from configuration.
+
+    Args:
+        cfg: Environment configuration.
+
+    Returns:
+        Tuple of (gym environment, teleoperator device).
+    """
+    # Check if this is a GymHIL simulation environment
+    if cfg.name == "gym_hil":
+        assert cfg.robot is None and cfg.teleop is None, "GymHIL environment does not support robot or teleop"
+        import gym_hil  # noqa: F401
+
+        # Extract gripper settings with defaults
+        use_gripper = cfg.processor.gripper.use_gripper if cfg.processor.gripper is not None else True
+        gripper_penalty = cfg.processor.gripper.gripper_penalty if cfg.processor.gripper is not None else 0.0
+
+        env = gym.make(
+            f"gym_hil/{cfg.task}",
+            image_obs=True,
+            render_mode="human",
+            use_gripper=use_gripper,
+            gripper_penalty=gripper_penalty,
+        )
+
+        return env, None
+
+    # Real robot environment
+    assert cfg.robot is not None, "Robot config must be provided for real robot environment"
+    assert cfg.teleop is not None, "Teleop config must be provided for real robot environment"
+
+    robot = make_robot_from_config(cfg.robot)
+    teleop_device = make_teleoperator_from_config(cfg.teleop)
+    teleop_device.connect()
+
+    # Create base environment with safe defaults
+    use_gripper = cfg.processor.gripper.use_gripper if cfg.processor.gripper is not None else True
+    display_cameras = (
+        cfg.processor.observation.display_cameras if cfg.processor.observation is not None else False
+    )
+    reset_pose = cfg.processor.reset.fixed_reset_joint_positions if cfg.processor.reset is not None else None
+
+    env = RobotEnv(
+        robot=robot,
+        use_gripper=use_gripper,
+        display_cameras=display_cameras,
+        reset_pose=reset_pose,
+    )
+
+    return env, teleop_device
+
+
+def make_processors(
+    env: gym.Env, teleop_device: Teleoperator | None, cfg: HILSerlRobotEnvConfig, device: str = "cpu"
+) -> tuple[
+    DataProcessorPipeline[EnvTransition, EnvTransition], DataProcessorPipeline[EnvTransition, EnvTransition]
+]:
+    """Create environment and action processors.
+
+    Args:
+        env: Robot environment instance.
+        teleop_device: Teleoperator device for intervention.
+        cfg: Processor configuration.
+        device: Target device for computations.
+
+    Returns:
+        Tuple of (environment processor, action processor).
+    """
+    terminate_on_success = (
+        cfg.processor.reset.terminate_on_success if cfg.processor.reset is not None else True
+    )
+
+    if cfg.name == "gym_hil":
+        action_pipeline_steps = [
+            InterventionActionProcessorStep(terminate_on_success=terminate_on_success),
+            Torch2NumpyActionProcessorStep(),
+        ]
+
+        env_pipeline_steps = [
+            GymHILAdapterProcessorStep(),
+            Numpy2TorchActionProcessorStep(),
+            VanillaObservationProcessorStep(),
+            AddBatchDimensionProcessorStep(),
+            DeviceProcessorStep(device=device),
+        ]
+
+        return DataProcessorPipeline(
+            steps=env_pipeline_steps, to_transition=identity_transition, to_output=identity_transition
+        ), DataProcessorPipeline(
+            steps=action_pipeline_steps, to_transition=identity_transition, to_output=identity_transition
+        )
+
+    # Full processor pipeline for real robot environment
+    # Get robot and motor information for kinematics
+    motor_names = list(env.robot.bus.motors.keys())
+
+    # Set up kinematics solver if inverse kinematics is configured
+    kinematics_solver = None
+    if cfg.processor.inverse_kinematics is not None:
+        kinematics_solver = RobotKinematics(
+            urdf_path=cfg.processor.inverse_kinematics.urdf_path,
+            target_frame_name=cfg.processor.inverse_kinematics.target_frame_name,
+            joint_names=motor_names,
+        )
+
+    env_pipeline_steps = [VanillaObservationProcessorStep()]
+
+    if cfg.processor.observation is not None:
+        if cfg.processor.observation.add_joint_velocity_to_observation:
+            env_pipeline_steps.append(JointVelocityProcessorStep(dt=1.0 / cfg.fps))
+        if cfg.processor.observation.add_current_to_observation:
+            env_pipeline_steps.append(MotorCurrentProcessorStep(robot=env.robot))
+
+    add_ee_pose = (
+        cfg.processor.observation is not None and cfg.processor.observation.add_ee_pose_to_observation
+    )
+    if kinematics_solver is not None and add_ee_pose:
+        env_pipeline_steps.append(
+            ForwardKinematicsJointsToEEObservation(
+                kinematics=kinematics_solver,
+                motor_names=motor_names,
+            )
+        )
+
+    if cfg.processor.image_preprocessing is not None:
+        env_pipeline_steps.append(
+            ImageCropResizeProcessorStep(
+                crop_params_dict=cfg.processor.image_preprocessing.crop_params_dict,
+                resize_size=cfg.processor.image_preprocessing.resize_size,
+            )
+        )
+
+    # Add time limit processor if reset config exists
+    if cfg.processor.reset is not None:
+        env_pipeline_steps.append(
+            TimeLimitProcessorStep(max_episode_steps=int(cfg.processor.reset.control_time_s * cfg.fps))
+        )
+
+    # Add gripper penalty processor if gripper config exists and enabled
+    # Only add if max_gripper_pos is explicitly configured (required for normalization)
+    if (
+        cfg.processor.gripper is not None
+        and cfg.processor.gripper.use_gripper
+        and cfg.processor.max_gripper_pos is not None
+    ):
+        env_pipeline_steps.append(
+            GripperPenaltyProcessorStep(
+                penalty=cfg.processor.gripper.gripper_penalty,
+                max_gripper_pos=cfg.processor.max_gripper_pos,
+            )
+        )
+
+    if (
+        cfg.processor.reward_classifier is not None
+        and cfg.processor.reward_classifier.pretrained_path is not None
+    ):
+        env_pipeline_steps.append(
+            RewardClassifierProcessorStep(
+                pretrained_path=cfg.processor.reward_classifier.pretrained_path,
+                device=device,
+                success_threshold=cfg.processor.reward_classifier.success_threshold,
+                success_reward=cfg.processor.reward_classifier.success_reward,
+                terminate_on_success=terminate_on_success,
+            )
+        )
+
+    env_pipeline_steps.append(AddBatchDimensionProcessorStep())
+    env_pipeline_steps.append(DeviceProcessorStep(device=device))
+
+    action_pipeline_steps = [
+        AddTeleopActionAsComplimentaryDataStep(teleop_device=teleop_device),
+        AddTeleopEventsAsInfoStep(teleop_device=teleop_device),
+        InterventionActionProcessorStep(
+            use_gripper=cfg.processor.gripper.use_gripper if cfg.processor.gripper is not None else False,
+            terminate_on_success=terminate_on_success,
+        ),
+    ]
+
+    # Replace InverseKinematicsProcessor with new kinematic processors
+    if cfg.processor.inverse_kinematics is not None and kinematics_solver is not None:
+        # Add EE bounds and safety processor
+        inverse_kinematics_steps = [
+            MapTensorToDeltaActionDictStep(
+                use_gripper=cfg.processor.gripper.use_gripper if cfg.processor.gripper is not None else False
+            ),
+            MapDeltaActionToRobotActionStep(),
+            EEReferenceAndDelta(
+                kinematics=kinematics_solver,
+                end_effector_step_sizes=cfg.processor.inverse_kinematics.end_effector_step_sizes,
+                motor_names=motor_names,
+                use_latched_reference=False,
+                use_ik_solution=True,
+            ),
+            EEBoundsAndSafety(
+                end_effector_bounds=cfg.processor.inverse_kinematics.end_effector_bounds,
+            ),
+            GripperVelocityToJoint(
+                clip_max=cfg.processor.max_gripper_pos,
+                speed_factor=1.0,
+                discrete_gripper=True,
+            ),
+            InverseKinematicsRLStep(
+                kinematics=kinematics_solver, motor_names=motor_names, initial_guess_current_joints=False
+            ),
+        ]
+        action_pipeline_steps.extend(inverse_kinematics_steps)
+        action_pipeline_steps.append(RobotActionToPolicyActionProcessorStep(motor_names=motor_names))
+
+    return DataProcessorPipeline(
+        steps=env_pipeline_steps, to_transition=identity_transition, to_output=identity_transition
+    ), DataProcessorPipeline(
+        steps=action_pipeline_steps, to_transition=identity_transition, to_output=identity_transition
+    )
+
+
+def step_env_and_process_transition(
+    env: gym.Env,
+    transition: EnvTransition,
+    action: torch.Tensor,
+    env_processor: DataProcessorPipeline[EnvTransition, EnvTransition],
+    action_processor: DataProcessorPipeline[EnvTransition, EnvTransition],
+) -> EnvTransition:
+    """
+    Execute one step with processor pipeline.
+
+    Args:
+        env: The robot environment
+        transition: Current transition state
+        action: Action to execute
+        env_processor: Environment processor
+        action_processor: Action processor
+
+    Returns:
+        Processed transition with updated state.
+    """
+
+    # Create action transition
+    transition[TransitionKey.ACTION] = action
+    transition[TransitionKey.OBSERVATION] = (
+        env.get_raw_joint_positions() if hasattr(env, "get_raw_joint_positions") else {}
+    )
+    processed_action_transition = action_processor(transition)
+    processed_action = processed_action_transition[TransitionKey.ACTION]
+
+    obs, reward, terminated, truncated, info = env.step(processed_action)
+
+    reward = reward + processed_action_transition[TransitionKey.REWARD]
+    terminated = terminated or processed_action_transition[TransitionKey.DONE]
+    truncated = truncated or processed_action_transition[TransitionKey.TRUNCATED]
+    complementary_data = processed_action_transition[TransitionKey.COMPLEMENTARY_DATA].copy()
+    new_info = processed_action_transition[TransitionKey.INFO].copy()
+    new_info.update(info)
+
+    new_transition = create_transition(
+        observation=obs,
+        action=processed_action,
+        reward=reward,
+        done=terminated,
+        truncated=truncated,
+        info=new_info,
+        complementary_data=complementary_data,
+    )
+    new_transition = env_processor(new_transition)
+
+    return new_transition
+
+
+def control_loop(
+    env: gym.Env,
+    env_processor: DataProcessorPipeline[EnvTransition, EnvTransition],
+    action_processor: DataProcessorPipeline[EnvTransition, EnvTransition],
+    teleop_device: Teleoperator,
+    cfg: GymManipulatorConfig,
+) -> None:
+    """Main control loop for robot environment interaction.
+    if cfg.mode == "record": then a dataset will be created and recorded
+
+    Args:
+     env: The robot environment
+     env_processor: Environment processor
+     action_processor: Action processor
+     teleop_device: Teleoperator device
+     cfg: gym_manipulator configuration
+    """
+    dt = 1.0 / cfg.env.fps
+
+    print(f"Starting control loop at {cfg.env.fps} FPS")
+    print("Controls:")
+    print("- Use gamepad/teleop device for intervention")
+    print("- When not intervening, robot will stay still")
+    print("- Press Ctrl+C to exit")
+
+    # Reset environment and processors
+    obs, info = env.reset()
+    complementary_data = (
+        {"raw_joint_positions": info.pop("raw_joint_positions")} if "raw_joint_positions" in info else {}
+    )
+    env_processor.reset()
+    action_processor.reset()
+
+    # Process initial observation
+    transition = create_transition(observation=obs, info=info, complementary_data=complementary_data)
+    transition = env_processor(data=transition)
+
+    # Determine if gripper is used
+    use_gripper = cfg.env.processor.gripper.use_gripper if cfg.env.processor.gripper is not None else True
+
+    dataset = None
+    if cfg.mode == "record":
+        if teleop_device:
+            action_features = teleop_device.action_features
+        else:
+            action_features = {
+                "dtype": "float32",
+                "shape": (4,),
+                "names": ["delta_x", "delta_y", "delta_z", "gripper"],
+            }
+        features = {
+            ACTION: action_features,
+            REWARD: {"dtype": "float32", "shape": (1,), "names": None},
+            DONE: {"dtype": "bool", "shape": (1,), "names": None},
+        }
+        if use_gripper:
+            features["complementary_info.discrete_penalty"] = {
+                "dtype": "float32",
+                "shape": (1,),
+                "names": ["discrete_penalty"],
+            }
+
+        for key, value in transition[TransitionKey.OBSERVATION].items():
+            if key == OBS_STATE:
+                features[key] = {
+                    "dtype": "float32",
+                    "shape": value.squeeze(0).shape,
+                    "names": None,
+                }
+            if "image" in key:
+                features[key] = {
+                    "dtype": "video",
+                    "shape": value.squeeze(0).shape,
+                    "names": ["channels", "height", "width"],
+                }
+
+        # Create dataset
+        dataset = LeRobotDataset.create(
+            cfg.dataset.repo_id,
+            cfg.env.fps,
+            root=cfg.dataset.root,
+            use_videos=True,
+            image_writer_threads=4,
+            image_writer_processes=0,
+            features=features,
+        )
+
+    episode_idx = 0
+    episode_step = 0
+    episode_start_time = time.perf_counter()
+
+    while episode_idx < cfg.dataset.num_episodes_to_record:
+        step_start_time = time.perf_counter()
+
+        # Create a neutral action (no movement)
+        neutral_action = torch.tensor([0.0, 0.0, 0.0], dtype=torch.float32)
+        if use_gripper:
+            neutral_action = torch.cat([neutral_action, torch.tensor([0.0])])  # Gripper stay
+
+        # Use the new step function
+        transition = step_env_and_process_transition(
+            env=env,
+            transition=transition,
+            action=neutral_action,
+            env_processor=env_processor,
+            action_processor=action_processor,
+        )
+        terminated = transition.get(TransitionKey.DONE, False)
+        truncated = transition.get(TransitionKey.TRUNCATED, False)
+
+        if cfg.mode == "record":
+            observations = {
+                k: v.squeeze(0).cpu()
+                for k, v in transition[TransitionKey.OBSERVATION].items()
+                if isinstance(v, torch.Tensor)
+            }
+            # Use teleop_action if available, otherwise use the action from the transition
+            action_to_record = transition[TransitionKey.COMPLEMENTARY_DATA].get(
+                "teleop_action", transition[TransitionKey.ACTION]
+            )
+            frame = {
+                **observations,
+                ACTION: action_to_record.cpu(),
+                REWARD: np.array([transition[TransitionKey.REWARD]], dtype=np.float32),
+                DONE: np.array([terminated or truncated], dtype=bool),
+            }
+            if use_gripper:
+                discrete_penalty = transition[TransitionKey.COMPLEMENTARY_DATA].get("discrete_penalty", 0.0)
+                frame["complementary_info.discrete_penalty"] = np.array([discrete_penalty], dtype=np.float32)
+
+            if dataset is not None:
+                frame["task"] = cfg.dataset.task
+                dataset.add_frame(frame)
+
+        episode_step += 1
+
+        # Handle episode termination
+        if terminated or truncated:
+            episode_time = time.perf_counter() - episode_start_time
+            logging.info(
+                f"Episode ended after {episode_step} steps in {episode_time:.1f}s with reward {transition[TransitionKey.REWARD]}"
+            )
+            episode_step = 0
+            episode_idx += 1
+
+            if dataset is not None:
+                if transition[TransitionKey.INFO].get(TeleopEvents.RERECORD_EPISODE, False):
+                    logging.info(f"Re-recording episode {episode_idx}")
+                    dataset.clear_episode_buffer()
+                    episode_idx -= 1
+                else:
+                    logging.info(f"Saving episode {episode_idx}")
+                    dataset.save_episode()
+
+            # Reset for new episode
+            obs, info = env.reset()
+            env_processor.reset()
+            action_processor.reset()
+
+            transition = create_transition(observation=obs, info=info)
+            transition = env_processor(transition)
+
+        # Maintain fps timing
+        precise_sleep(max(dt - (time.perf_counter() - step_start_time), 0.0))
+
+    if dataset is not None and cfg.dataset.push_to_hub:
+        logging.info("Finalizing dataset before pushing to hub")
+        dataset.finalize()
+        logging.info("Pushing dataset to hub")
+        dataset.push_to_hub()
+
+
+def replay_trajectory(
+    env: gym.Env, action_processor: DataProcessorPipeline, cfg: GymManipulatorConfig
+) -> None:
+    """Replay recorded trajectory on robot environment."""
+    assert cfg.dataset.replay_episode is not None, "Replay episode must be provided for replay"
+
+    dataset = LeRobotDataset(
+        cfg.dataset.repo_id,
+        root=cfg.dataset.root,
+        episodes=[cfg.dataset.replay_episode],
+        download_videos=False,
+    )
+    episode_frames = dataset.hf_dataset.filter(lambda x: x["episode_index"] == cfg.dataset.replay_episode)
+    actions = episode_frames.select_columns(ACTION)
+
+    _, info = env.reset()
+
+    for action_data in actions:
+        start_time = time.perf_counter()
+        transition = create_transition(
+            observation=env.get_raw_joint_positions() if hasattr(env, "get_raw_joint_positions") else {},
+            action=action_data[ACTION],
+        )
+        transition = action_processor(transition)
+        env.step(transition[TransitionKey.ACTION])
+        precise_sleep(max(1 / cfg.env.fps - (time.perf_counter() - start_time), 0.0))
+
+
+@parser.wrap()
+def main(cfg: GymManipulatorConfig) -> None:
+    """Main entry point for gym manipulator script."""
+    env, teleop_device = make_robot_env(cfg.env)
+    env_processor, action_processor = make_processors(env, teleop_device, cfg.env, cfg.device)
+
+    print("Environment observation space:", env.observation_space)
+    print("Environment action space:", env.action_space)
+    print("Environment processor:", env_processor)
+    print("Action processor:", action_processor)
+
+    if cfg.mode == "replay":
+        replay_trajectory(env, action_processor, cfg)
+        exit()
+
+    control_loop(env, env_processor, action_processor, teleop_device, cfg)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/rl/joint_observations_processor.py b/lerobot/src/lerobot/rl/joint_observations_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..2fbcc7c4682f4a1546cb33ae4ac913c05133ff6c
--- /dev/null
+++ b/lerobot/src/lerobot/rl/joint_observations_processor.py
@@ -0,0 +1,211 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+from typing import Any
+
+import torch
+
+from lerobot.configs.types import PipelineFeatureType, PolicyFeature
+from lerobot.processor.pipeline import (
+    ObservationProcessorStep,
+    ProcessorStepRegistry,
+)
+from lerobot.robots import Robot
+from lerobot.utils.constants import OBS_STATE
+
+
+@dataclass
+@ProcessorStepRegistry.register("joint_velocity_processor")
+class JointVelocityProcessorStep(ObservationProcessorStep):
+    """
+    Calculates and appends joint velocity information to the observation state.
+
+    This step computes the velocity of each joint by calculating the finite
+    difference between the current and the last observed joint positions. The
+    resulting velocity vector is then concatenated to the original state vector.
+
+    Attributes:
+        dt: The time step (delta time) in seconds between observations, used for
+            calculating velocity.
+        last_joint_positions: Stores the joint positions from the previous step
+                              to enable velocity calculation.
+    """
+
+    dt: float = 0.1
+
+    last_joint_positions: torch.Tensor | None = None
+
+    def observation(self, observation: dict) -> dict:
+        """
+        Computes joint velocities and adds them to the observation state.
+
+        Args:
+            observation: The input observation dictionary, expected to contain
+                         an `observation.state` key with joint positions.
+
+        Returns:
+            A new observation dictionary with the `observation.state` tensor
+            extended to include joint velocities.
+
+        Raises:
+            ValueError: If `observation.state` is not found in the observation.
+        """
+        # Get current joint positions (assuming they're in observation.state)
+        current_positions = observation.get(OBS_STATE)
+        if current_positions is None:
+            raise ValueError(f"{OBS_STATE} is not in observation")
+
+        # Initialize last joint positions if not already set
+        if self.last_joint_positions is None:
+            self.last_joint_positions = current_positions.clone()
+            joint_velocities = torch.zeros_like(current_positions)
+        else:
+            # Compute velocities
+            joint_velocities = (current_positions - self.last_joint_positions) / self.dt
+
+        self.last_joint_positions = current_positions.clone()
+
+        # Extend observation with velocities
+        extended_state = torch.cat([current_positions, joint_velocities], dim=-1)
+
+        # Create new observation dict
+        new_observation = dict(observation)
+        new_observation[OBS_STATE] = extended_state
+
+        return new_observation
+
+    def get_config(self) -> dict[str, Any]:
+        """
+        Returns the configuration of the step for serialization.
+
+        Returns:
+            A dictionary containing the time step `dt`.
+        """
+        return {
+            "dt": self.dt,
+        }
+
+    def reset(self) -> None:
+        """Resets the internal state, clearing the last known joint positions."""
+        self.last_joint_positions = None
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Updates the `observation.state` feature to reflect the added velocities.
+
+        This method doubles the size of the first dimension of the `observation.state`
+        shape to account for the concatenation of position and velocity vectors.
+
+        Args:
+            features: The policy features dictionary.
+
+        Returns:
+            The updated policy features dictionary.
+        """
+        if OBS_STATE in features[PipelineFeatureType.OBSERVATION]:
+            original_feature = features[PipelineFeatureType.OBSERVATION][OBS_STATE]
+            # Double the shape to account for positions + velocities
+            new_shape = (original_feature.shape[0] * 2,) + original_feature.shape[1:]
+
+            features[PipelineFeatureType.OBSERVATION][OBS_STATE] = PolicyFeature(
+                type=original_feature.type, shape=new_shape
+            )
+        return features
+
+
+@dataclass
+@ProcessorStepRegistry.register("current_processor")
+class MotorCurrentProcessorStep(ObservationProcessorStep):
+    """
+    Reads motor currents from a robot and appends them to the observation state.
+
+    This step queries the robot's hardware interface to get the present current
+    for each motor and concatenates this information to the existing state vector.
+
+    Attributes:
+        robot: An instance of a `lerobot` Robot class that provides access to
+               the hardware bus.
+    """
+
+    robot: Robot | None = None
+
+    def observation(self, observation: dict) -> dict:
+        """
+        Fetches motor currents and adds them to the observation state.
+
+        Args:
+            observation: The input observation dictionary.
+
+        Returns:
+            A new observation dictionary with the `observation.state` tensor
+            extended to include motor currents.
+
+        Raises:
+            ValueError: If the `robot` attribute has not been set.
+        """
+        # Get current values from robot state
+        if self.robot is None:
+            raise ValueError("Robot is not set")
+
+        present_current_dict = self.robot.bus.sync_read("Present_Current")  # type: ignore[attr-defined]
+        motor_currents = torch.tensor(
+            [present_current_dict[name] for name in self.robot.bus.motors],  # type: ignore[attr-defined]
+            dtype=torch.float32,
+        ).unsqueeze(0)
+
+        current_state = observation.get(OBS_STATE)
+        if current_state is None:
+            return observation
+
+        extended_state = torch.cat([current_state, motor_currents], dim=-1)
+
+        # Create new observation dict
+        new_observation = dict(observation)
+        new_observation[OBS_STATE] = extended_state
+
+        return new_observation
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        """
+        Updates the `observation.state` feature to reflect the added motor currents.
+
+        This method increases the size of the first dimension of the `observation.state`
+        shape by the number of motors in the robot.
+
+        Args:
+            features: The policy features dictionary.
+
+        Returns:
+            The updated policy features dictionary.
+        """
+        if OBS_STATE in features[PipelineFeatureType.OBSERVATION] and self.robot is not None:
+            original_feature = features[PipelineFeatureType.OBSERVATION][OBS_STATE]
+            # Add motor current dimensions to the original state shape
+            num_motors = 0
+            if hasattr(self.robot, "bus") and hasattr(self.robot.bus, "motors"):  # type: ignore[attr-defined]
+                num_motors = len(self.robot.bus.motors)  # type: ignore[attr-defined]
+
+            if num_motors > 0:
+                new_shape = (original_feature.shape[0] + num_motors,) + original_feature.shape[1:]
+                features[PipelineFeatureType.OBSERVATION][OBS_STATE] = PolicyFeature(
+                    type=original_feature.type, shape=new_shape
+                )
+        return features
diff --git a/lerobot/src/lerobot/rl/learner.py b/lerobot/src/lerobot/rl/learner.py
new file mode 100644
index 0000000000000000000000000000000000000000..2853fbcb3fc7cea26f3f42f5f00dd88828713c44
--- /dev/null
+++ b/lerobot/src/lerobot/rl/learner.py
@@ -0,0 +1,1200 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team.
+# All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""
+Learner server runner for distributed HILSerl robot policy training.
+
+This script implements the learner component of the distributed HILSerl architecture.
+It initializes the policy network, maintains replay buffers, and updates
+the policy based on transitions received from the actor server.
+
+Examples of usage:
+
+- Start a learner server for training:
+```bash
+python -m lerobot.rl.learner --config_path src/lerobot/configs/train_config_hilserl_so100.json
+```
+
+**NOTE**: Start the learner server before launching the actor server. The learner opens a gRPC server
+to communicate with actors.
+
+**NOTE**: Training progress can be monitored through Weights & Biases if wandb.enable is set to true
+in your configuration.
+
+**WORKFLOW**:
+1. Create training configuration with proper policy, dataset, and environment settings
+2. Start this learner server with the configuration
+3. Start an actor server with the same configuration
+4. Monitor training progress through wandb dashboard
+
+For more details on the complete HILSerl training workflow, see:
+https://github.com/michel-aractingi/lerobot-hilserl-guide
+"""
+
+import logging
+import os
+import shutil
+import time
+from concurrent.futures import ThreadPoolExecutor
+from pathlib import Path
+from pprint import pformat
+
+import grpc
+import torch
+from termcolor import colored
+from torch import nn
+from torch.multiprocessing import Queue
+from torch.optim.optimizer import Optimizer
+
+from lerobot.cameras import opencv  # noqa: F401
+from lerobot.configs import parser
+from lerobot.configs.train import TrainRLServerPipelineConfig
+from lerobot.datasets.factory import make_dataset
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.factory import make_policy
+from lerobot.policies.sac.modeling_sac import SACPolicy
+from lerobot.rl.buffer import ReplayBuffer, concatenate_batch_transitions
+from lerobot.rl.process import ProcessSignalHandler
+from lerobot.rl.wandb_utils import WandBLogger
+from lerobot.robots import so_follower  # noqa: F401
+from lerobot.teleoperators import gamepad, so_leader  # noqa: F401
+from lerobot.teleoperators.utils import TeleopEvents
+from lerobot.transport import services_pb2_grpc
+from lerobot.transport.utils import (
+    MAX_MESSAGE_SIZE,
+    bytes_to_python_object,
+    bytes_to_transitions,
+    state_to_bytes,
+)
+from lerobot.utils.constants import (
+    ACTION,
+    CHECKPOINTS_DIR,
+    LAST_CHECKPOINT_LINK,
+    PRETRAINED_MODEL_DIR,
+    TRAINING_STATE_DIR,
+)
+from lerobot.utils.device_utils import get_safe_torch_device
+from lerobot.utils.random_utils import set_seed
+from lerobot.utils.train_utils import (
+    get_step_checkpoint_dir,
+    load_training_state as utils_load_training_state,
+    save_checkpoint,
+    update_last_checkpoint,
+)
+from lerobot.utils.transition import move_state_dict_to_device, move_transition_to_device
+from lerobot.utils.utils import (
+    format_big_number,
+    init_logging,
+)
+
+from .learner_service import MAX_WORKERS, SHUTDOWN_TIMEOUT, LearnerService
+
+
+@parser.wrap()
+def train_cli(cfg: TrainRLServerPipelineConfig):
+    if not use_threads(cfg):
+        import torch.multiprocessing as mp
+
+        mp.set_start_method("spawn")
+
+    # Use the job_name from the config
+    train(
+        cfg,
+        job_name=cfg.job_name,
+    )
+
+    logging.info("[LEARNER] train_cli finished")
+
+
+def train(cfg: TrainRLServerPipelineConfig, job_name: str | None = None):
+    """
+    Main training function that initializes and runs the training process.
+
+    Args:
+        cfg (TrainRLServerPipelineConfig): The training configuration
+        job_name (str | None, optional): Job name for logging. Defaults to None.
+    """
+
+    cfg.validate()
+
+    if job_name is None:
+        job_name = cfg.job_name
+
+    if job_name is None:
+        raise ValueError("Job name must be specified either in config or as a parameter")
+
+    display_pid = False
+    if not use_threads(cfg):
+        display_pid = True
+
+    # Create logs directory to ensure it exists
+    log_dir = os.path.join(cfg.output_dir, "logs")
+    os.makedirs(log_dir, exist_ok=True)
+    log_file = os.path.join(log_dir, f"learner_{job_name}.log")
+
+    # Initialize logging with explicit log file
+    init_logging(log_file=log_file, display_pid=display_pid)
+    logging.info(f"Learner logging initialized, writing to {log_file}")
+    logging.info(pformat(cfg.to_dict()))
+
+    # Setup WandB logging if enabled
+    if cfg.wandb.enable and cfg.wandb.project:
+        from lerobot.rl.wandb_utils import WandBLogger
+
+        wandb_logger = WandBLogger(cfg)
+    else:
+        wandb_logger = None
+        logging.info(colored("Logs will be saved locally.", "yellow", attrs=["bold"]))
+
+    # Handle resume logic
+    cfg = handle_resume_logic(cfg)
+
+    set_seed(seed=cfg.seed)
+
+    torch.backends.cudnn.benchmark = True
+    torch.backends.cuda.matmul.allow_tf32 = True
+
+    is_threaded = use_threads(cfg)
+    shutdown_event = ProcessSignalHandler(is_threaded, display_pid=display_pid).shutdown_event
+
+    start_learner_threads(
+        cfg=cfg,
+        wandb_logger=wandb_logger,
+        shutdown_event=shutdown_event,
+    )
+
+
+def start_learner_threads(
+    cfg: TrainRLServerPipelineConfig,
+    wandb_logger: WandBLogger | None,
+    shutdown_event: any,  # Event,
+) -> None:
+    """
+    Start the learner threads for training.
+
+    Args:
+        cfg (TrainRLServerPipelineConfig): Training configuration
+        wandb_logger (WandBLogger | None): Logger for metrics
+        shutdown_event: Event to signal shutdown
+    """
+    # Create multiprocessing queues
+    transition_queue = Queue()
+    interaction_message_queue = Queue()
+    parameters_queue = Queue()
+
+    concurrency_entity = None
+
+    if use_threads(cfg):
+        from threading import Thread
+
+        concurrency_entity = Thread
+    else:
+        from torch.multiprocessing import Process
+
+        concurrency_entity = Process
+
+    communication_process = concurrency_entity(
+        target=start_learner,
+        args=(
+            parameters_queue,
+            transition_queue,
+            interaction_message_queue,
+            shutdown_event,
+            cfg,
+        ),
+        daemon=True,
+    )
+    communication_process.start()
+
+    add_actor_information_and_train(
+        cfg=cfg,
+        wandb_logger=wandb_logger,
+        shutdown_event=shutdown_event,
+        transition_queue=transition_queue,
+        interaction_message_queue=interaction_message_queue,
+        parameters_queue=parameters_queue,
+    )
+    logging.info("[LEARNER] Training process stopped")
+
+    logging.info("[LEARNER] Closing queues")
+    transition_queue.close()
+    interaction_message_queue.close()
+    parameters_queue.close()
+
+    communication_process.join()
+    logging.info("[LEARNER] Communication process joined")
+
+    logging.info("[LEARNER] join queues")
+    transition_queue.cancel_join_thread()
+    interaction_message_queue.cancel_join_thread()
+    parameters_queue.cancel_join_thread()
+
+    logging.info("[LEARNER] queues closed")
+
+
+# Core algorithm functions
+
+
+def add_actor_information_and_train(
+    cfg: TrainRLServerPipelineConfig,
+    wandb_logger: WandBLogger | None,
+    shutdown_event: any,  # Event,
+    transition_queue: Queue,
+    interaction_message_queue: Queue,
+    parameters_queue: Queue,
+):
+    """
+    Handles data transfer from the actor to the learner, manages training updates,
+    and logs training progress in an online reinforcement learning setup.
+
+    This function continuously:
+    - Transfers transitions from the actor to the replay buffer.
+    - Logs received interaction messages.
+    - Ensures training begins only when the replay buffer has a sufficient number of transitions.
+    - Samples batches from the replay buffer and performs multiple critic updates.
+    - Periodically updates the actor, critic, and temperature optimizers.
+    - Logs training statistics, including loss values and optimization frequency.
+
+    NOTE: This function doesn't have a single responsibility, it should be split into multiple functions
+    in the future. The reason why we did that is the  GIL in Python. It's super slow the performance
+    are divided by 200. So we need to have a single thread that does all the work.
+
+    Args:
+        cfg (TrainRLServerPipelineConfig): Configuration object containing hyperparameters.
+        wandb_logger (WandBLogger | None): Logger for tracking training progress.
+        shutdown_event (Event): Event to signal shutdown.
+        transition_queue (Queue): Queue for receiving transitions from the actor.
+        interaction_message_queue (Queue): Queue for receiving interaction messages from the actor.
+        parameters_queue (Queue): Queue for sending policy parameters to the actor.
+    """
+    # Extract all configuration variables at the beginning, it improve the speed performance
+    # of 7%
+    device = get_safe_torch_device(try_device=cfg.policy.device, log=True)
+    storage_device = get_safe_torch_device(try_device=cfg.policy.storage_device)
+    clip_grad_norm_value = cfg.policy.grad_clip_norm
+    online_step_before_learning = cfg.policy.online_step_before_learning
+    utd_ratio = cfg.policy.utd_ratio
+    fps = cfg.env.fps
+    log_freq = cfg.log_freq
+    save_freq = cfg.save_freq
+    policy_update_freq = cfg.policy.policy_update_freq
+    policy_parameters_push_frequency = cfg.policy.actor_learner_config.policy_parameters_push_frequency
+    saving_checkpoint = cfg.save_checkpoint
+    online_steps = cfg.policy.online_steps
+    async_prefetch = cfg.policy.async_prefetch
+
+    # Initialize logging for multiprocessing
+    if not use_threads(cfg):
+        log_dir = os.path.join(cfg.output_dir, "logs")
+        os.makedirs(log_dir, exist_ok=True)
+        log_file = os.path.join(log_dir, f"learner_train_process_{os.getpid()}.log")
+        init_logging(log_file=log_file, display_pid=True)
+        logging.info("Initialized logging for actor information and training process")
+
+    logging.info("Initializing policy")
+
+    policy: SACPolicy = make_policy(
+        cfg=cfg.policy,
+        env_cfg=cfg.env,
+    )
+
+    assert isinstance(policy, nn.Module)
+
+    policy.train()
+
+    push_actor_policy_to_queue(parameters_queue=parameters_queue, policy=policy)
+
+    last_time_policy_pushed = time.time()
+
+    optimizers, lr_scheduler = make_optimizers_and_scheduler(cfg=cfg, policy=policy)
+
+    # If we are resuming, we need to load the training state
+    resume_optimization_step, resume_interaction_step = load_training_state(cfg=cfg, optimizers=optimizers)
+
+    log_training_info(cfg=cfg, policy=policy)
+
+    replay_buffer = initialize_replay_buffer(cfg, device, storage_device)
+    batch_size = cfg.batch_size
+    offline_replay_buffer = None
+
+    if cfg.dataset is not None:
+        offline_replay_buffer = initialize_offline_replay_buffer(
+            cfg=cfg,
+            device=device,
+            storage_device=storage_device,
+        )
+        batch_size: int = batch_size // 2  # We will sample from both replay buffer
+
+    logging.info("Starting learner thread")
+    interaction_message = None
+    optimization_step = resume_optimization_step if resume_optimization_step is not None else 0
+    interaction_step_shift = resume_interaction_step if resume_interaction_step is not None else 0
+
+    dataset_repo_id = None
+    if cfg.dataset is not None:
+        dataset_repo_id = cfg.dataset.repo_id
+
+    # Initialize iterators
+    online_iterator = None
+    offline_iterator = None
+
+    # NOTE: THIS IS THE MAIN LOOP OF THE LEARNER
+    while True:
+        # Exit the training loop if shutdown is requested
+        if shutdown_event is not None and shutdown_event.is_set():
+            logging.info("[LEARNER] Shutdown signal received. Exiting...")
+            break
+
+        # Process all available transitions to the replay buffer, send by the actor server
+        process_transitions(
+            transition_queue=transition_queue,
+            replay_buffer=replay_buffer,
+            offline_replay_buffer=offline_replay_buffer,
+            device=device,
+            dataset_repo_id=dataset_repo_id,
+            shutdown_event=shutdown_event,
+        )
+
+        # Process all available interaction messages sent by the actor server
+        interaction_message = process_interaction_messages(
+            interaction_message_queue=interaction_message_queue,
+            interaction_step_shift=interaction_step_shift,
+            wandb_logger=wandb_logger,
+            shutdown_event=shutdown_event,
+        )
+
+        # Wait until the replay buffer has enough samples to start training
+        if len(replay_buffer) < online_step_before_learning:
+            continue
+
+        if online_iterator is None:
+            online_iterator = replay_buffer.get_iterator(
+                batch_size=batch_size, async_prefetch=async_prefetch, queue_size=2
+            )
+
+        if offline_replay_buffer is not None and offline_iterator is None:
+            offline_iterator = offline_replay_buffer.get_iterator(
+                batch_size=batch_size, async_prefetch=async_prefetch, queue_size=2
+            )
+
+        time_for_one_optimization_step = time.time()
+        for _ in range(utd_ratio - 1):
+            # Sample from the iterators
+            batch = next(online_iterator)
+
+            if dataset_repo_id is not None:
+                batch_offline = next(offline_iterator)
+                batch = concatenate_batch_transitions(
+                    left_batch_transitions=batch, right_batch_transition=batch_offline
+                )
+
+            actions = batch[ACTION]
+            rewards = batch["reward"]
+            observations = batch["state"]
+            next_observations = batch["next_state"]
+            done = batch["done"]
+            check_nan_in_transition(observations=observations, actions=actions, next_state=next_observations)
+
+            observation_features, next_observation_features = get_observation_features(
+                policy=policy, observations=observations, next_observations=next_observations
+            )
+
+            # Create a batch dictionary with all required elements for the forward method
+            forward_batch = {
+                ACTION: actions,
+                "reward": rewards,
+                "state": observations,
+                "next_state": next_observations,
+                "done": done,
+                "observation_feature": observation_features,
+                "next_observation_feature": next_observation_features,
+                "complementary_info": batch["complementary_info"],
+            }
+
+            # Use the forward method for critic loss
+            critic_output = policy.forward(forward_batch, model="critic")
+
+            # Main critic optimization
+            loss_critic = critic_output["loss_critic"]
+            optimizers["critic"].zero_grad()
+            loss_critic.backward()
+            critic_grad_norm = torch.nn.utils.clip_grad_norm_(
+                parameters=policy.critic_ensemble.parameters(), max_norm=clip_grad_norm_value
+            )
+            optimizers["critic"].step()
+
+            # Discrete critic optimization (if available)
+            if policy.config.num_discrete_actions is not None:
+                discrete_critic_output = policy.forward(forward_batch, model="discrete_critic")
+                loss_discrete_critic = discrete_critic_output["loss_discrete_critic"]
+                optimizers["discrete_critic"].zero_grad()
+                loss_discrete_critic.backward()
+                discrete_critic_grad_norm = torch.nn.utils.clip_grad_norm_(
+                    parameters=policy.discrete_critic.parameters(), max_norm=clip_grad_norm_value
+                )
+                optimizers["discrete_critic"].step()
+
+            # Update target networks (main and discrete)
+            policy.update_target_networks()
+
+        # Sample for the last update in the UTD ratio
+        batch = next(online_iterator)
+
+        if dataset_repo_id is not None:
+            batch_offline = next(offline_iterator)
+            batch = concatenate_batch_transitions(
+                left_batch_transitions=batch, right_batch_transition=batch_offline
+            )
+
+        actions = batch[ACTION]
+        rewards = batch["reward"]
+        observations = batch["state"]
+        next_observations = batch["next_state"]
+        done = batch["done"]
+
+        check_nan_in_transition(observations=observations, actions=actions, next_state=next_observations)
+
+        observation_features, next_observation_features = get_observation_features(
+            policy=policy, observations=observations, next_observations=next_observations
+        )
+
+        # Create a batch dictionary with all required elements for the forward method
+        forward_batch = {
+            ACTION: actions,
+            "reward": rewards,
+            "state": observations,
+            "next_state": next_observations,
+            "done": done,
+            "observation_feature": observation_features,
+            "next_observation_feature": next_observation_features,
+        }
+
+        critic_output = policy.forward(forward_batch, model="critic")
+
+        loss_critic = critic_output["loss_critic"]
+        optimizers["critic"].zero_grad()
+        loss_critic.backward()
+        critic_grad_norm = torch.nn.utils.clip_grad_norm_(
+            parameters=policy.critic_ensemble.parameters(), max_norm=clip_grad_norm_value
+        ).item()
+        optimizers["critic"].step()
+
+        # Initialize training info dictionary
+        training_infos = {
+            "loss_critic": loss_critic.item(),
+            "critic_grad_norm": critic_grad_norm,
+        }
+
+        # Discrete critic optimization (if available)
+        if policy.config.num_discrete_actions is not None:
+            discrete_critic_output = policy.forward(forward_batch, model="discrete_critic")
+            loss_discrete_critic = discrete_critic_output["loss_discrete_critic"]
+            optimizers["discrete_critic"].zero_grad()
+            loss_discrete_critic.backward()
+            discrete_critic_grad_norm = torch.nn.utils.clip_grad_norm_(
+                parameters=policy.discrete_critic.parameters(), max_norm=clip_grad_norm_value
+            ).item()
+            optimizers["discrete_critic"].step()
+
+            # Add discrete critic info to training info
+            training_infos["loss_discrete_critic"] = loss_discrete_critic.item()
+            training_infos["discrete_critic_grad_norm"] = discrete_critic_grad_norm
+
+        # Actor and temperature optimization (at specified frequency)
+        if optimization_step % policy_update_freq == 0:
+            for _ in range(policy_update_freq):
+                # Actor optimization
+                actor_output = policy.forward(forward_batch, model="actor")
+                loss_actor = actor_output["loss_actor"]
+                optimizers["actor"].zero_grad()
+                loss_actor.backward()
+                actor_grad_norm = torch.nn.utils.clip_grad_norm_(
+                    parameters=policy.actor.parameters(), max_norm=clip_grad_norm_value
+                ).item()
+                optimizers["actor"].step()
+
+                # Add actor info to training info
+                training_infos["loss_actor"] = loss_actor.item()
+                training_infos["actor_grad_norm"] = actor_grad_norm
+
+                # Temperature optimization
+                temperature_output = policy.forward(forward_batch, model="temperature")
+                loss_temperature = temperature_output["loss_temperature"]
+                optimizers["temperature"].zero_grad()
+                loss_temperature.backward()
+                temp_grad_norm = torch.nn.utils.clip_grad_norm_(
+                    parameters=[policy.log_alpha], max_norm=clip_grad_norm_value
+                ).item()
+                optimizers["temperature"].step()
+
+                # Add temperature info to training info
+                training_infos["loss_temperature"] = loss_temperature.item()
+                training_infos["temperature_grad_norm"] = temp_grad_norm
+                training_infos["temperature"] = policy.temperature
+
+        # Push policy to actors if needed
+        if time.time() - last_time_policy_pushed > policy_parameters_push_frequency:
+            push_actor_policy_to_queue(parameters_queue=parameters_queue, policy=policy)
+            last_time_policy_pushed = time.time()
+
+        # Update target networks (main and discrete)
+        policy.update_target_networks()
+
+        # Log training metrics at specified intervals
+        if optimization_step % log_freq == 0:
+            training_infos["replay_buffer_size"] = len(replay_buffer)
+            if offline_replay_buffer is not None:
+                training_infos["offline_replay_buffer_size"] = len(offline_replay_buffer)
+            training_infos["Optimization step"] = optimization_step
+
+            # Log training metrics
+            if wandb_logger:
+                wandb_logger.log_dict(d=training_infos, mode="train", custom_step_key="Optimization step")
+
+        # Calculate and log optimization frequency
+        time_for_one_optimization_step = time.time() - time_for_one_optimization_step
+        frequency_for_one_optimization_step = 1 / (time_for_one_optimization_step + 1e-9)
+
+        logging.info(f"[LEARNER] Optimization frequency loop [Hz]: {frequency_for_one_optimization_step}")
+
+        # Log optimization frequency
+        if wandb_logger:
+            wandb_logger.log_dict(
+                {
+                    "Optimization frequency loop [Hz]": frequency_for_one_optimization_step,
+                    "Optimization step": optimization_step,
+                },
+                mode="train",
+                custom_step_key="Optimization step",
+            )
+
+        optimization_step += 1
+        if optimization_step % log_freq == 0:
+            logging.info(f"[LEARNER] Number of optimization step: {optimization_step}")
+
+        # Save checkpoint at specified intervals
+        if saving_checkpoint and (optimization_step % save_freq == 0 or optimization_step == online_steps):
+            save_training_checkpoint(
+                cfg=cfg,
+                optimization_step=optimization_step,
+                online_steps=online_steps,
+                interaction_message=interaction_message,
+                policy=policy,
+                optimizers=optimizers,
+                replay_buffer=replay_buffer,
+                offline_replay_buffer=offline_replay_buffer,
+                dataset_repo_id=dataset_repo_id,
+                fps=fps,
+            )
+
+
+def start_learner(
+    parameters_queue: Queue,
+    transition_queue: Queue,
+    interaction_message_queue: Queue,
+    shutdown_event: any,  # Event,
+    cfg: TrainRLServerPipelineConfig,
+):
+    """
+    Start the learner server for training.
+    It will receive transitions and interaction messages from the actor server,
+    and send policy parameters to the actor server.
+
+    Args:
+        parameters_queue: Queue for sending policy parameters to the actor
+        transition_queue: Queue for receiving transitions from the actor
+        interaction_message_queue: Queue for receiving interaction messages from the actor
+        shutdown_event: Event to signal shutdown
+        cfg: Training configuration
+    """
+    if not use_threads(cfg):
+        # Create a process-specific log file
+        log_dir = os.path.join(cfg.output_dir, "logs")
+        os.makedirs(log_dir, exist_ok=True)
+        log_file = os.path.join(log_dir, f"learner_process_{os.getpid()}.log")
+
+        # Initialize logging with explicit log file
+        init_logging(log_file=log_file, display_pid=True)
+        logging.info("Learner server process logging initialized")
+
+        # Setup process handlers to handle shutdown signal
+        # But use shutdown event from the main process
+        # Return back for MP
+        # TODO: Check if its useful
+        _ = ProcessSignalHandler(False, display_pid=True)
+
+    service = LearnerService(
+        shutdown_event=shutdown_event,
+        parameters_queue=parameters_queue,
+        seconds_between_pushes=cfg.policy.actor_learner_config.policy_parameters_push_frequency,
+        transition_queue=transition_queue,
+        interaction_message_queue=interaction_message_queue,
+        queue_get_timeout=cfg.policy.actor_learner_config.queue_get_timeout,
+    )
+
+    server = grpc.server(
+        ThreadPoolExecutor(max_workers=MAX_WORKERS),
+        options=[
+            ("grpc.max_receive_message_length", MAX_MESSAGE_SIZE),
+            ("grpc.max_send_message_length", MAX_MESSAGE_SIZE),
+        ],
+    )
+
+    services_pb2_grpc.add_LearnerServiceServicer_to_server(
+        service,
+        server,
+    )
+
+    host = cfg.policy.actor_learner_config.learner_host
+    port = cfg.policy.actor_learner_config.learner_port
+
+    server.add_insecure_port(f"{host}:{port}")
+    server.start()
+    logging.info("[LEARNER] gRPC server started")
+
+    shutdown_event.wait()
+    logging.info("[LEARNER] Stopping gRPC server...")
+    server.stop(SHUTDOWN_TIMEOUT)
+    logging.info("[LEARNER] gRPC server stopped")
+
+
+def save_training_checkpoint(
+    cfg: TrainRLServerPipelineConfig,
+    optimization_step: int,
+    online_steps: int,
+    interaction_message: dict | None,
+    policy: nn.Module,
+    optimizers: dict[str, Optimizer],
+    replay_buffer: ReplayBuffer,
+    offline_replay_buffer: ReplayBuffer | None = None,
+    dataset_repo_id: str | None = None,
+    fps: int = 30,
+) -> None:
+    """
+    Save training checkpoint and associated data.
+
+    This function performs the following steps:
+    1. Creates a checkpoint directory with the current optimization step
+    2. Saves the policy model, configuration, and optimizer states
+    3. Saves the current interaction step for resuming training
+    4. Updates the "last" checkpoint symlink to point to this checkpoint
+    5. Saves the replay buffer as a dataset for later use
+    6. If an offline replay buffer exists, saves it as a separate dataset
+
+    Args:
+        cfg: Training configuration
+        optimization_step: Current optimization step
+        online_steps: Total number of online steps
+        interaction_message: Dictionary containing interaction information
+        policy: Policy model to save
+        optimizers: Dictionary of optimizers
+        replay_buffer: Replay buffer to save as dataset
+        offline_replay_buffer: Optional offline replay buffer to save
+        dataset_repo_id: Repository ID for dataset
+        fps: Frames per second for dataset
+    """
+    logging.info(f"Checkpoint policy after step {optimization_step}")
+    _num_digits = max(6, len(str(online_steps)))
+    interaction_step = interaction_message["Interaction step"] if interaction_message is not None else 0
+
+    # Create checkpoint directory
+    checkpoint_dir = get_step_checkpoint_dir(cfg.output_dir, online_steps, optimization_step)
+
+    # Save checkpoint
+    save_checkpoint(
+        checkpoint_dir=checkpoint_dir,
+        step=optimization_step,
+        cfg=cfg,
+        policy=policy,
+        optimizer=optimizers,
+        scheduler=None,
+    )
+
+    # Save interaction step manually
+    training_state_dir = os.path.join(checkpoint_dir, TRAINING_STATE_DIR)
+    os.makedirs(training_state_dir, exist_ok=True)
+    training_state = {"step": optimization_step, "interaction_step": interaction_step}
+    torch.save(training_state, os.path.join(training_state_dir, "training_state.pt"))
+
+    # Update the "last" symlink
+    update_last_checkpoint(checkpoint_dir)
+
+    # TODO : temporary save replay buffer here, remove later when on the robot
+    # We want to control this with the keyboard inputs
+    dataset_dir = os.path.join(cfg.output_dir, "dataset")
+    if os.path.exists(dataset_dir) and os.path.isdir(dataset_dir):
+        shutil.rmtree(dataset_dir)
+
+    # Save dataset
+    # NOTE: Handle the case where the dataset repo id is not specified in the config
+    # eg. RL training without demonstrations data
+    repo_id_buffer_save = cfg.env.task if dataset_repo_id is None else dataset_repo_id
+    replay_buffer.to_lerobot_dataset(repo_id=repo_id_buffer_save, fps=fps, root=dataset_dir)
+
+    if offline_replay_buffer is not None:
+        dataset_offline_dir = os.path.join(cfg.output_dir, "dataset_offline")
+        if os.path.exists(dataset_offline_dir) and os.path.isdir(dataset_offline_dir):
+            shutil.rmtree(dataset_offline_dir)
+
+        offline_replay_buffer.to_lerobot_dataset(
+            cfg.dataset.repo_id,
+            fps=fps,
+            root=dataset_offline_dir,
+        )
+
+    logging.info("Resume training")
+
+
+def make_optimizers_and_scheduler(cfg: TrainRLServerPipelineConfig, policy: nn.Module):
+    """
+    Creates and returns optimizers for the actor, critic, and temperature components of a reinforcement learning policy.
+
+    This function sets up Adam optimizers for:
+    - The **actor network**, ensuring that only relevant parameters are optimized.
+    - The **critic ensemble**, which evaluates the value function.
+    - The **temperature parameter**, which controls the entropy in soft actor-critic (SAC)-like methods.
+
+    It also initializes a learning rate scheduler, though currently, it is set to `None`.
+
+    NOTE:
+    - If the encoder is shared, its parameters are excluded from the actor's optimization process.
+    - The policy's log temperature (`log_alpha`) is wrapped in a list to ensure proper optimization as a standalone tensor.
+
+    Args:
+        cfg: Configuration object containing hyperparameters.
+        policy (nn.Module): The policy model containing the actor, critic, and temperature components.
+
+    Returns:
+        Tuple[Dict[str, torch.optim.Optimizer], Optional[torch.optim.lr_scheduler._LRScheduler]]:
+        A tuple containing:
+        - `optimizers`: A dictionary mapping component names ("actor", "critic", "temperature") to their respective Adam optimizers.
+        - `lr_scheduler`: Currently set to `None` but can be extended to support learning rate scheduling.
+
+    """
+    optimizer_actor = torch.optim.Adam(
+        params=[
+            p
+            for n, p in policy.actor.named_parameters()
+            if not policy.config.shared_encoder or not n.startswith("encoder")
+        ],
+        lr=cfg.policy.actor_lr,
+    )
+    optimizer_critic = torch.optim.Adam(params=policy.critic_ensemble.parameters(), lr=cfg.policy.critic_lr)
+
+    if cfg.policy.num_discrete_actions is not None:
+        optimizer_discrete_critic = torch.optim.Adam(
+            params=policy.discrete_critic.parameters(), lr=cfg.policy.critic_lr
+        )
+    optimizer_temperature = torch.optim.Adam(params=[policy.log_alpha], lr=cfg.policy.critic_lr)
+    lr_scheduler = None
+    optimizers = {
+        "actor": optimizer_actor,
+        "critic": optimizer_critic,
+        "temperature": optimizer_temperature,
+    }
+    if cfg.policy.num_discrete_actions is not None:
+        optimizers["discrete_critic"] = optimizer_discrete_critic
+    return optimizers, lr_scheduler
+
+
+# Training setup functions
+
+
+def handle_resume_logic(cfg: TrainRLServerPipelineConfig) -> TrainRLServerPipelineConfig:
+    """
+    Handle the resume logic for training.
+
+    If resume is True:
+    - Verifies that a checkpoint exists
+    - Loads the checkpoint configuration
+    - Logs resumption details
+    - Returns the checkpoint configuration
+
+    If resume is False:
+    - Checks if an output directory exists (to prevent accidental overwriting)
+    - Returns the original configuration
+
+    Args:
+        cfg (TrainRLServerPipelineConfig): The training configuration
+
+    Returns:
+        TrainRLServerPipelineConfig: The updated configuration
+
+    Raises:
+        RuntimeError: If resume is True but no checkpoint found, or if resume is False but directory exists
+    """
+    out_dir = cfg.output_dir
+
+    # Case 1: Not resuming, but need to check if directory exists to prevent overwrites
+    if not cfg.resume:
+        checkpoint_dir = os.path.join(out_dir, CHECKPOINTS_DIR, LAST_CHECKPOINT_LINK)
+        if os.path.exists(checkpoint_dir):
+            raise RuntimeError(
+                f"Output directory {checkpoint_dir} already exists. Use `resume=true` to resume training."
+            )
+        return cfg
+
+    # Case 2: Resuming training
+    checkpoint_dir = os.path.join(out_dir, CHECKPOINTS_DIR, LAST_CHECKPOINT_LINK)
+    if not os.path.exists(checkpoint_dir):
+        raise RuntimeError(f"No model checkpoint found in {checkpoint_dir} for resume=True")
+
+    # Log that we found a valid checkpoint and are resuming
+    logging.info(
+        colored(
+            "Valid checkpoint found: resume=True detected, resuming previous run",
+            color="yellow",
+            attrs=["bold"],
+        )
+    )
+
+    # Load config using Draccus
+    checkpoint_cfg_path = os.path.join(checkpoint_dir, PRETRAINED_MODEL_DIR, "train_config.json")
+    checkpoint_cfg = TrainRLServerPipelineConfig.from_pretrained(checkpoint_cfg_path)
+
+    # Ensure resume flag is set in returned config
+    checkpoint_cfg.resume = True
+    return checkpoint_cfg
+
+
+def load_training_state(
+    cfg: TrainRLServerPipelineConfig,
+    optimizers: Optimizer | dict[str, Optimizer],
+):
+    """
+    Loads the training state (optimizers, step count, etc.) from a checkpoint.
+
+    Args:
+        cfg (TrainRLServerPipelineConfig): Training configuration
+        optimizers (Optimizer | dict): Optimizers to load state into
+
+    Returns:
+        tuple: (optimization_step, interaction_step) or (None, None) if not resuming
+    """
+    if not cfg.resume:
+        return None, None
+
+    # Construct path to the last checkpoint directory
+    checkpoint_dir = os.path.join(cfg.output_dir, CHECKPOINTS_DIR, LAST_CHECKPOINT_LINK)
+
+    logging.info(f"Loading training state from {checkpoint_dir}")
+
+    try:
+        # Use the utility function from train_utils which loads the optimizer state
+        step, optimizers, _ = utils_load_training_state(Path(checkpoint_dir), optimizers, None)
+
+        # Load interaction step separately from training_state.pt
+        training_state_path = os.path.join(checkpoint_dir, TRAINING_STATE_DIR, "training_state.pt")
+        interaction_step = 0
+        if os.path.exists(training_state_path):
+            training_state = torch.load(training_state_path, weights_only=False)  # nosec B614: Safe usage of torch.load
+            interaction_step = training_state.get("interaction_step", 0)
+
+        logging.info(f"Resuming from step {step}, interaction step {interaction_step}")
+        return step, interaction_step
+
+    except Exception as e:
+        logging.error(f"Failed to load training state: {e}")
+        return None, None
+
+
+def log_training_info(cfg: TrainRLServerPipelineConfig, policy: nn.Module) -> None:
+    """
+    Log information about the training process.
+
+    Args:
+        cfg (TrainRLServerPipelineConfig): Training configuration
+        policy (nn.Module): Policy model
+    """
+    num_learnable_params = sum(p.numel() for p in policy.parameters() if p.requires_grad)
+    num_total_params = sum(p.numel() for p in policy.parameters())
+
+    logging.info(colored("Output dir:", "yellow", attrs=["bold"]) + f" {cfg.output_dir}")
+    logging.info(f"{cfg.env.task=}")
+    logging.info(f"{cfg.policy.online_steps=}")
+    logging.info(f"{num_learnable_params=} ({format_big_number(num_learnable_params)})")
+    logging.info(f"{num_total_params=} ({format_big_number(num_total_params)})")
+
+
+def initialize_replay_buffer(
+    cfg: TrainRLServerPipelineConfig, device: str, storage_device: str
+) -> ReplayBuffer:
+    """
+    Initialize a replay buffer, either empty or from a dataset if resuming.
+
+    Args:
+        cfg (TrainRLServerPipelineConfig): Training configuration
+        device (str): Device to store tensors on
+        storage_device (str): Device for storage optimization
+
+    Returns:
+        ReplayBuffer: Initialized replay buffer
+    """
+    if not cfg.resume:
+        return ReplayBuffer(
+            capacity=cfg.policy.online_buffer_capacity,
+            device=device,
+            state_keys=cfg.policy.input_features.keys(),
+            storage_device=storage_device,
+            optimize_memory=True,
+        )
+
+    logging.info("Resume training load the online dataset")
+    dataset_path = os.path.join(cfg.output_dir, "dataset")
+
+    # NOTE: In RL is possible to not have a dataset.
+    repo_id = None
+    if cfg.dataset is not None:
+        repo_id = cfg.dataset.repo_id
+    dataset = LeRobotDataset(
+        repo_id=repo_id,
+        root=dataset_path,
+    )
+    return ReplayBuffer.from_lerobot_dataset(
+        lerobot_dataset=dataset,
+        capacity=cfg.policy.online_buffer_capacity,
+        device=device,
+        state_keys=cfg.policy.input_features.keys(),
+        optimize_memory=True,
+    )
+
+
+def initialize_offline_replay_buffer(
+    cfg: TrainRLServerPipelineConfig,
+    device: str,
+    storage_device: str,
+) -> ReplayBuffer:
+    """
+    Initialize an offline replay buffer from a dataset.
+
+    Args:
+        cfg (TrainRLServerPipelineConfig): Training configuration
+        device (str): Device to store tensors on
+        storage_device (str): Device for storage optimization
+
+    Returns:
+        ReplayBuffer: Initialized offline replay buffer
+    """
+    if not cfg.resume:
+        logging.info("make_dataset offline buffer")
+        offline_dataset = make_dataset(cfg)
+    else:
+        logging.info("load offline dataset")
+        dataset_offline_path = os.path.join(cfg.output_dir, "dataset_offline")
+        offline_dataset = LeRobotDataset(
+            repo_id=cfg.dataset.repo_id,
+            root=dataset_offline_path,
+        )
+
+    logging.info("Convert to a offline replay buffer")
+    offline_replay_buffer = ReplayBuffer.from_lerobot_dataset(
+        offline_dataset,
+        device=device,
+        state_keys=cfg.policy.input_features.keys(),
+        storage_device=storage_device,
+        optimize_memory=True,
+        capacity=cfg.policy.offline_buffer_capacity,
+    )
+    return offline_replay_buffer
+
+
+# Utilities/Helpers functions
+
+
+def get_observation_features(
+    policy: SACPolicy, observations: torch.Tensor, next_observations: torch.Tensor
+) -> tuple[torch.Tensor | None, torch.Tensor | None]:
+    """
+    Get observation features from the policy encoder. It act as cache for the observation features.
+    when the encoder is frozen, the observation features are not updated.
+    We can save compute by caching the observation features.
+
+    Args:
+        policy: The policy model
+        observations: The current observations
+        next_observations: The next observations
+
+    Returns:
+        tuple: observation_features, next_observation_features
+    """
+
+    if policy.config.vision_encoder_name is None or not policy.config.freeze_vision_encoder:
+        return None, None
+
+    with torch.no_grad():
+        observation_features = policy.actor.encoder.get_cached_image_features(observations)
+        next_observation_features = policy.actor.encoder.get_cached_image_features(next_observations)
+
+    return observation_features, next_observation_features
+
+
+def use_threads(cfg: TrainRLServerPipelineConfig) -> bool:
+    return cfg.policy.concurrency.learner == "threads"
+
+
+def check_nan_in_transition(
+    observations: torch.Tensor,
+    actions: torch.Tensor,
+    next_state: torch.Tensor,
+    raise_error: bool = False,
+) -> bool:
+    """
+    Check for NaN values in transition data.
+
+    Args:
+        observations: Dictionary of observation tensors
+        actions: Action tensor
+        next_state: Dictionary of next state tensors
+        raise_error: If True, raises ValueError when NaN is detected
+
+    Returns:
+        bool: True if NaN values were detected, False otherwise
+    """
+    nan_detected = False
+
+    # Check observations
+    for key, tensor in observations.items():
+        if torch.isnan(tensor).any():
+            logging.error(f"observations[{key}] contains NaN values")
+            nan_detected = True
+            if raise_error:
+                raise ValueError(f"NaN detected in observations[{key}]")
+
+    # Check next state
+    for key, tensor in next_state.items():
+        if torch.isnan(tensor).any():
+            logging.error(f"next_state[{key}] contains NaN values")
+            nan_detected = True
+            if raise_error:
+                raise ValueError(f"NaN detected in next_state[{key}]")
+
+    # Check actions
+    if torch.isnan(actions).any():
+        logging.error("actions contains NaN values")
+        nan_detected = True
+        if raise_error:
+            raise ValueError("NaN detected in actions")
+
+    return nan_detected
+
+
+def push_actor_policy_to_queue(parameters_queue: Queue, policy: nn.Module):
+    logging.debug("[LEARNER] Pushing actor policy to the queue")
+
+    # Create a dictionary to hold all the state dicts
+    state_dicts = {"policy": move_state_dict_to_device(policy.actor.state_dict(), device="cpu")}
+
+    # Add discrete critic if it exists
+    if hasattr(policy, "discrete_critic") and policy.discrete_critic is not None:
+        state_dicts["discrete_critic"] = move_state_dict_to_device(
+            policy.discrete_critic.state_dict(), device="cpu"
+        )
+        logging.debug("[LEARNER] Including discrete critic in state dict push")
+
+    state_bytes = state_to_bytes(state_dicts)
+    parameters_queue.put(state_bytes)
+
+
+def process_interaction_message(
+    message, interaction_step_shift: int, wandb_logger: WandBLogger | None = None
+):
+    """Process a single interaction message with consistent handling."""
+    message = bytes_to_python_object(message)
+    # Shift interaction step for consistency with checkpointed state
+    message["Interaction step"] += interaction_step_shift
+
+    # Log if logger available
+    if wandb_logger:
+        wandb_logger.log_dict(d=message, mode="train", custom_step_key="Interaction step")
+
+    return message
+
+
+def process_transitions(
+    transition_queue: Queue,
+    replay_buffer: ReplayBuffer,
+    offline_replay_buffer: ReplayBuffer,
+    device: str,
+    dataset_repo_id: str | None,
+    shutdown_event: any,
+):
+    """Process all available transitions from the queue.
+
+    Args:
+        transition_queue: Queue for receiving transitions from the actor
+        replay_buffer: Replay buffer to add transitions to
+        offline_replay_buffer: Offline replay buffer to add transitions to
+        device: Device to move transitions to
+        dataset_repo_id: Repository ID for dataset
+        shutdown_event: Event to signal shutdown
+    """
+    while not transition_queue.empty() and not shutdown_event.is_set():
+        transition_list = transition_queue.get()
+        transition_list = bytes_to_transitions(buffer=transition_list)
+
+        for transition in transition_list:
+            transition = move_transition_to_device(transition=transition, device=device)
+
+            # Skip transitions with NaN values
+            if check_nan_in_transition(
+                observations=transition["state"],
+                actions=transition[ACTION],
+                next_state=transition["next_state"],
+            ):
+                logging.warning("[LEARNER] NaN detected in transition, skipping")
+                continue
+
+            replay_buffer.add(**transition)
+
+            # Add to offline buffer if it's an intervention
+            if dataset_repo_id is not None and transition.get("complementary_info", {}).get(
+                TeleopEvents.IS_INTERVENTION
+            ):
+                offline_replay_buffer.add(**transition)
+
+
+def process_interaction_messages(
+    interaction_message_queue: Queue,
+    interaction_step_shift: int,
+    wandb_logger: WandBLogger | None,
+    shutdown_event: any,
+) -> dict | None:
+    """Process all available interaction messages from the queue.
+
+    Args:
+        interaction_message_queue: Queue for receiving interaction messages
+        interaction_step_shift: Amount to shift interaction step by
+        wandb_logger: Logger for tracking progress
+        shutdown_event: Event to signal shutdown
+
+    Returns:
+        dict | None: The last interaction message processed, or None if none were processed
+    """
+    last_message = None
+    while not interaction_message_queue.empty() and not shutdown_event.is_set():
+        message = interaction_message_queue.get()
+        last_message = process_interaction_message(
+            message=message,
+            interaction_step_shift=interaction_step_shift,
+            wandb_logger=wandb_logger,
+        )
+
+    return last_message
+
+
+if __name__ == "__main__":
+    train_cli()
+    logging.info("[LEARNER] main finished")
diff --git a/lerobot/src/lerobot/rl/learner_service.py b/lerobot/src/lerobot/rl/learner_service.py
new file mode 100644
index 0000000000000000000000000000000000000000..7ef38119b28d98b5075c5e3a6cc28345d80f6467
--- /dev/null
+++ b/lerobot/src/lerobot/rl/learner_service.py
@@ -0,0 +1,117 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team.
+# All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from multiprocessing import Event, Queue
+
+from lerobot.rl.queue import get_last_item_from_queue
+from lerobot.transport import services_pb2, services_pb2_grpc
+from lerobot.transport.utils import receive_bytes_in_chunks, send_bytes_in_chunks
+
+MAX_WORKERS = 3  # Stream parameters, send transitions and interactions
+SHUTDOWN_TIMEOUT = 10
+
+
+class LearnerService(services_pb2_grpc.LearnerServiceServicer):
+    """
+    Implementation of the LearnerService gRPC service
+    This service is used to send parameters to the Actor and receive transitions and interactions from the Actor
+    check transport.proto for the gRPC service definition
+    """
+
+    def __init__(
+        self,
+        shutdown_event: Event,  # type: ignore
+        parameters_queue: Queue,
+        seconds_between_pushes: float,
+        transition_queue: Queue,
+        interaction_message_queue: Queue,
+        queue_get_timeout: float = 0.001,
+    ):
+        self.shutdown_event = shutdown_event
+        self.parameters_queue = parameters_queue
+        self.seconds_between_pushes = seconds_between_pushes
+        self.transition_queue = transition_queue
+        self.interaction_message_queue = interaction_message_queue
+        self.queue_get_timeout = queue_get_timeout
+
+    def StreamParameters(self, request, context):  # noqa: N802
+        # TODO: authorize the request
+        logging.info("[LEARNER] Received request to stream parameters from the Actor")
+
+        last_push_time = 0
+
+        while not self.shutdown_event.is_set():
+            time_since_last_push = time.time() - last_push_time
+            if time_since_last_push < self.seconds_between_pushes:
+                self.shutdown_event.wait(self.seconds_between_pushes - time_since_last_push)
+                # Continue, because we could receive a shutdown event,
+                # and it's checked in the while loop
+                continue
+
+            logging.info("[LEARNER] Push parameters to the Actor")
+            buffer = get_last_item_from_queue(
+                self.parameters_queue, block=True, timeout=self.queue_get_timeout
+            )
+
+            if buffer is None:
+                continue
+
+            yield from send_bytes_in_chunks(
+                buffer,
+                services_pb2.Parameters,
+                log_prefix="[LEARNER] Sending parameters",
+                silent=True,
+            )
+
+            last_push_time = time.time()
+            logging.info("[LEARNER] Parameters sent")
+
+        logging.info("[LEARNER] Stream parameters finished")
+        return services_pb2.Empty()
+
+    def SendTransitions(self, request_iterator, _context):  # noqa: N802
+        # TODO: authorize the request
+        logging.info("[LEARNER] Received request to receive transitions from the Actor")
+
+        receive_bytes_in_chunks(
+            request_iterator,
+            self.transition_queue,
+            self.shutdown_event,
+            log_prefix="[LEARNER] transitions",
+        )
+
+        logging.debug("[LEARNER] Finished receiving transitions")
+        return services_pb2.Empty()
+
+    def SendInteractions(self, request_iterator, _context):  # noqa: N802
+        # TODO: authorize the request
+        logging.info("[LEARNER] Received request to receive interactions from the Actor")
+
+        receive_bytes_in_chunks(
+            request_iterator,
+            self.interaction_message_queue,
+            self.shutdown_event,
+            log_prefix="[LEARNER] interactions",
+        )
+
+        logging.debug("[LEARNER] Finished receiving interactions")
+        return services_pb2.Empty()
+
+    def Ready(self, request, context):  # noqa: N802
+        return services_pb2.Empty()
diff --git a/lerobot/src/lerobot/rl/process.py b/lerobot/src/lerobot/rl/process.py
new file mode 100644
index 0000000000000000000000000000000000000000..72438b6f98c3cbd78b1fc0d3fb781619eb5a476c
--- /dev/null
+++ b/lerobot/src/lerobot/rl/process.py
@@ -0,0 +1,83 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team.
+# All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import os
+import signal
+import sys
+
+
+class ProcessSignalHandler:
+    """Utility class to attach graceful shutdown signal handlers.
+
+    The class exposes a shutdown_event attribute that is set when a shutdown
+    signal is received. A counter tracks how many shutdown signals have been
+    caught. On the second signal the process exits with status 1.
+    """
+
+    _SUPPORTED_SIGNALS = ("SIGINT", "SIGTERM", "SIGHUP", "SIGQUIT")
+
+    def __init__(self, use_threads: bool, display_pid: bool = False):
+        # TODO: Check if we can use Event from threading since Event from
+        # multiprocessing is the a clone of threading.Event.
+        # https://docs.python.org/3/library/multiprocessing.html#multiprocessing.Event
+        if use_threads:
+            from threading import Event
+        else:
+            from multiprocessing import Event
+
+        self.shutdown_event = Event()
+        self._counter: int = 0
+        self._display_pid = display_pid
+
+        self._register_handlers()
+
+    @property
+    def counter(self) -> int:  # pragma: no cover – simple accessor
+        """Number of shutdown signals that have been intercepted."""
+        return self._counter
+
+    def _register_handlers(self):
+        """Attach the internal _signal_handler to a subset of POSIX signals."""
+
+        def _signal_handler(signum, frame):
+            pid_str = ""
+            if self._display_pid:
+                pid_str = f"[PID: {os.getpid()}]"
+            logging.info(f"{pid_str} Shutdown signal {signum} received. Cleaning up…")
+            self.shutdown_event.set()
+            self._counter += 1
+
+            # On a second Ctrl-C (or any supported signal) force the exit to
+            # mimic the previous behaviour while giving the caller one chance to
+            # shutdown gracefully.
+            # TODO: Investigate if we need it later
+            if self._counter > 1:
+                logging.info("Force shutdown")
+                sys.exit(1)
+
+        for sig_name in self._SUPPORTED_SIGNALS:
+            sig = getattr(signal, sig_name, None)
+            if sig is None:
+                # The signal is not available on this platform (Windows for
+                # instance does not provide SIGHUP, SIGQUIT…). Skip it.
+                continue
+            try:
+                signal.signal(sig, _signal_handler)
+            except (ValueError, OSError):  # pragma: no cover – unlikely but safe
+                # Signal not supported or we are in a non-main thread.
+                continue
diff --git a/lerobot/src/lerobot/rl/queue.py b/lerobot/src/lerobot/rl/queue.py
new file mode 100644
index 0000000000000000000000000000000000000000..864d798acadd13afec289f374b3167416c581843
--- /dev/null
+++ b/lerobot/src/lerobot/rl/queue.py
@@ -0,0 +1,52 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import platform
+from contextlib import suppress
+from queue import Empty
+from typing import Any
+
+from torch.multiprocessing import Queue
+
+
+def get_last_item_from_queue(queue: Queue, block=True, timeout: float = 0.1) -> Any:
+    if block:
+        try:
+            item = queue.get(timeout=timeout)
+        except Empty:
+            return None
+    else:
+        item = None
+
+    # Drain queue and keep only the most recent parameters
+    if platform.system() == "Darwin":
+        # On Mac, avoid using `qsize` due to unreliable implementation.
+        # There is a comment on `qsize` code in the Python source:
+        # Raises NotImplementedError on Mac OSX because of broken sem_getvalue()
+        try:
+            while True:
+                item = queue.get_nowait()
+        except Empty:
+            pass
+
+        return item
+
+    # Details about using qsize in https://github.com/huggingface/lerobot/issues/1523
+    while queue.qsize() > 0:
+        with suppress(Empty):
+            item = queue.get_nowait()
+
+    return item
diff --git a/lerobot/src/lerobot/rl/wandb_utils.py b/lerobot/src/lerobot/rl/wandb_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..e3190b6ce9db0d5019226c87d3300ce423aad802
--- /dev/null
+++ b/lerobot/src/lerobot/rl/wandb_utils.py
@@ -0,0 +1,203 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+import os
+import re
+from glob import glob
+from pathlib import Path
+
+from huggingface_hub.constants import SAFETENSORS_SINGLE_FILE
+from termcolor import colored
+
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.utils.constants import PRETRAINED_MODEL_DIR
+
+
+def cfg_to_group(
+    cfg: TrainPipelineConfig, return_list: bool = False, truncate_tags: bool = False, max_tag_length: int = 64
+) -> list[str] | str:
+    """Return a group name for logging. Optionally returns group name as list."""
+
+    def _maybe_truncate(tag: str) -> str:
+        """Truncate tag to max_tag_length characters if required.
+
+        wandb rejects tags longer than 64 characters.
+        See: https://github.com/wandb/wandb/blob/main/wandb/sdk/wandb_settings.py
+        """
+        if len(tag) <= max_tag_length:
+            return tag
+        return tag[:max_tag_length]
+
+    lst = [
+        f"policy:{cfg.policy.type}",
+        f"seed:{cfg.seed}",
+    ]
+    if cfg.dataset is not None:
+        lst.append(f"dataset:{cfg.dataset.repo_id}")
+    if cfg.env is not None:
+        lst.append(f"env:{cfg.env.type}")
+    if truncate_tags:
+        lst = [_maybe_truncate(tag) for tag in lst]
+    return lst if return_list else "-".join(lst)
+
+
+def get_wandb_run_id_from_filesystem(log_dir: Path) -> str:
+    # Get the WandB run ID.
+    paths = glob(str(log_dir / "wandb/latest-run/run-*"))
+    if len(paths) != 1:
+        raise RuntimeError("Couldn't get the previous WandB run ID for run resumption.")
+    match = re.search(r"run-([^\.]+).wandb", paths[0].split("/")[-1])
+    if match is None:
+        raise RuntimeError("Couldn't get the previous WandB run ID for run resumption.")
+    wandb_run_id = match.groups(0)[0]
+    return wandb_run_id
+
+
+def get_safe_wandb_artifact_name(name: str):
+    """WandB artifacts don't accept ":" or "/" in their name."""
+    return name.replace(":", "_").replace("/", "_")
+
+
+class WandBLogger:
+    """A helper class to log object using wandb."""
+
+    def __init__(self, cfg: TrainPipelineConfig):
+        self.cfg = cfg.wandb
+        self.log_dir = cfg.output_dir
+        self.job_name = cfg.job_name
+        self.env_fps = cfg.env.fps if cfg.env else None
+        self._group = cfg_to_group(cfg)
+
+        # Set up WandB.
+        os.environ["WANDB_SILENT"] = "True"
+        import wandb
+
+        wandb_run_id = (
+            cfg.wandb.run_id
+            if cfg.wandb.run_id
+            else get_wandb_run_id_from_filesystem(self.log_dir)
+            if cfg.resume
+            else None
+        )
+        wandb.init(
+            id=wandb_run_id,
+            project=self.cfg.project,
+            entity=self.cfg.entity,
+            name=self.job_name,
+            notes=self.cfg.notes,
+            tags=cfg_to_group(cfg, return_list=True, truncate_tags=True) if self.cfg.add_tags else None,
+            dir=self.log_dir,
+            config=cfg.to_dict(),
+            # TODO(rcadene): try set to True
+            save_code=False,
+            # TODO(rcadene): split train and eval, and run async eval with job_type="eval"
+            job_type="train_eval",
+            resume="must" if cfg.resume else None,
+            mode=self.cfg.mode if self.cfg.mode in ["online", "offline", "disabled"] else "online",
+        )
+        run_id = wandb.run.id
+        # NOTE: We will override the cfg.wandb.run_id with the wandb run id.
+        # This is because we want to be able to resume the run from the wandb run id.
+        cfg.wandb.run_id = run_id
+        # Handle custom step key for rl asynchronous training.
+        self._wandb_custom_step_key: set[str] | None = None
+        logging.info(colored("Logs will be synced with wandb.", "blue", attrs=["bold"]))
+        logging.info(f"Track this run --> {colored(wandb.run.get_url(), 'yellow', attrs=['bold'])}")
+        self._wandb = wandb
+
+    def log_policy(self, checkpoint_dir: Path):
+        """Checkpoints the policy to wandb."""
+        if self.cfg.disable_artifact:
+            return
+
+        step_id = checkpoint_dir.name
+        artifact_name = f"{self._group}-{step_id}"
+        artifact_name = get_safe_wandb_artifact_name(artifact_name)
+        artifact = self._wandb.Artifact(artifact_name, type="model")
+        pretrained_model_dir = checkpoint_dir / PRETRAINED_MODEL_DIR
+
+        # Check if this is a PEFT model (has adapter files instead of model.safetensors)
+        adapter_model_file = pretrained_model_dir / "adapter_model.safetensors"
+        standard_model_file = pretrained_model_dir / SAFETENSORS_SINGLE_FILE
+
+        if adapter_model_file.exists():
+            # PEFT model: add adapter files and configs
+            artifact.add_file(adapter_model_file)
+            adapter_config_file = pretrained_model_dir / "adapter_config.json"
+            if adapter_config_file.exists():
+                artifact.add_file(adapter_config_file)
+            # Also add the policy config which is needed for loading
+            config_file = pretrained_model_dir / "config.json"
+            if config_file.exists():
+                artifact.add_file(config_file)
+        elif standard_model_file.exists():
+            # Standard model: add the single safetensors file
+            artifact.add_file(standard_model_file)
+        else:
+            logging.warning(
+                f"No {SAFETENSORS_SINGLE_FILE} or adapter_model.safetensors found in {pretrained_model_dir}. "
+                "Skipping model artifact upload to WandB."
+            )
+            return
+
+        self._wandb.log_artifact(artifact)
+
+    def log_dict(
+        self, d: dict, step: int | None = None, mode: str = "train", custom_step_key: str | None = None
+    ):
+        if mode not in {"train", "eval"}:
+            raise ValueError(mode)
+        if step is None and custom_step_key is None:
+            raise ValueError("Either step or custom_step_key must be provided.")
+
+        # NOTE: This is not simple. Wandb step must always monotonically increase and it
+        # increases with each wandb.log call, but in the case of asynchronous RL for example,
+        # multiple time steps is possible. For example, the interaction step with the environment,
+        # the training step, the evaluation step, etc. So we need to define a custom step key
+        # to log the correct step for each metric.
+        if custom_step_key is not None:
+            if self._wandb_custom_step_key is None:
+                self._wandb_custom_step_key = set()
+            new_custom_key = f"{mode}/{custom_step_key}"
+            if new_custom_key not in self._wandb_custom_step_key:
+                self._wandb_custom_step_key.add(new_custom_key)
+                self._wandb.define_metric(new_custom_key, hidden=True)
+
+        for k, v in d.items():
+            if not isinstance(v, (int | float | str)):
+                logging.warning(
+                    f'WandB logging of key "{k}" was ignored as its type "{type(v)}" is not handled by this wrapper.'
+                )
+                continue
+
+            # Do not log the custom step key itself.
+            if self._wandb_custom_step_key is not None and k in self._wandb_custom_step_key:
+                continue
+
+            if custom_step_key is not None:
+                value_custom_step = d[custom_step_key]
+                data = {f"{mode}/{k}": v, f"{mode}/{custom_step_key}": value_custom_step}
+                self._wandb.log(data)
+                continue
+
+            self._wandb.log(data={f"{mode}/{k}": v}, step=step)
+
+    def log_video(self, video_path: str, step: int, mode: str = "train"):
+        if mode not in {"train", "eval"}:
+            raise ValueError(mode)
+
+        wandb_video = self._wandb.Video(video_path, fps=self.env_fps, format="mp4")
+        self._wandb.log({f"{mode}/video": wandb_video}, step=step)
diff --git a/lerobot/src/lerobot/robots/__init__.py b/lerobot/src/lerobot/robots/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..1dba0f1b089ca9e96479ea90e7f5b776cd2beb41
--- /dev/null
+++ b/lerobot/src/lerobot/robots/__init__.py
@@ -0,0 +1,19 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config import RobotConfig
+from .robot import Robot
+from .utils import make_robot_from_config
diff --git a/lerobot/src/lerobot/robots/bi_openarm_follower/__init__.py b/lerobot/src/lerobot/robots/bi_openarm_follower/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..b1dcce4313e3d5b7b42ff091948326739d180b30
--- /dev/null
+++ b/lerobot/src/lerobot/robots/bi_openarm_follower/__init__.py
@@ -0,0 +1,20 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .bi_openarm_follower import BiOpenArmFollower
+from .config_bi_openarm_follower import BiOpenArmFollowerConfig
+
+__all__ = ["BiOpenArmFollower", "BiOpenArmFollowerConfig"]
diff --git a/lerobot/src/lerobot/robots/bi_openarm_follower/bi_openarm_follower.py b/lerobot/src/lerobot/robots/bi_openarm_follower/bi_openarm_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..7f5e92271ac1beae14828b14eff872f01c8d8f9a
--- /dev/null
+++ b/lerobot/src/lerobot/robots/bi_openarm_follower/bi_openarm_follower.py
@@ -0,0 +1,180 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+from functools import cached_property
+
+from lerobot.robots.openarm_follower import OpenArmFollower, OpenArmFollowerConfig
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from .config_bi_openarm_follower import BiOpenArmFollowerConfig
+
+logger = logging.getLogger(__name__)
+
+
+class BiOpenArmFollower(Robot):
+    """
+    Bimanual OpenArm Follower Arms
+    """
+
+    config_class = BiOpenArmFollowerConfig
+    name = "bi_openarm_follower"
+
+    def __init__(self, config: BiOpenArmFollowerConfig):
+        super().__init__(config)
+        self.config = config
+
+        left_arm_config = OpenArmFollowerConfig(
+            id=f"{config.id}_left" if config.id else None,
+            calibration_dir=config.calibration_dir,
+            port=config.left_arm_config.port,
+            disable_torque_on_disconnect=config.left_arm_config.disable_torque_on_disconnect,
+            max_relative_target=config.left_arm_config.max_relative_target,
+            cameras=config.left_arm_config.cameras,
+            side=config.left_arm_config.side,
+            can_interface=config.left_arm_config.can_interface,
+            use_can_fd=config.left_arm_config.use_can_fd,
+            can_bitrate=config.left_arm_config.can_bitrate,
+            can_data_bitrate=config.left_arm_config.can_data_bitrate,
+            motor_config=config.left_arm_config.motor_config,
+            position_kd=config.left_arm_config.position_kd,
+            position_kp=config.left_arm_config.position_kp,
+            joint_limits=config.left_arm_config.joint_limits,
+        )
+
+        right_arm_config = OpenArmFollowerConfig(
+            id=f"{config.id}_right" if config.id else None,
+            calibration_dir=config.calibration_dir,
+            port=config.right_arm_config.port,
+            disable_torque_on_disconnect=config.right_arm_config.disable_torque_on_disconnect,
+            max_relative_target=config.right_arm_config.max_relative_target,
+            cameras=config.right_arm_config.cameras,
+            side=config.right_arm_config.side,
+            can_interface=config.right_arm_config.can_interface,
+            use_can_fd=config.right_arm_config.use_can_fd,
+            can_bitrate=config.right_arm_config.can_bitrate,
+            can_data_bitrate=config.right_arm_config.can_data_bitrate,
+            motor_config=config.right_arm_config.motor_config,
+            position_kd=config.right_arm_config.position_kd,
+            position_kp=config.right_arm_config.position_kp,
+            joint_limits=config.right_arm_config.joint_limits,
+        )
+
+        self.left_arm = OpenArmFollower(left_arm_config)
+        self.right_arm = OpenArmFollower(right_arm_config)
+
+        # Only for compatibility with other parts of the codebase that expect a `robot.cameras` attribute
+        self.cameras = {**self.left_arm.cameras, **self.right_arm.cameras}
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        left_arm_motors_ft = self.left_arm._motors_ft
+        right_arm_motors_ft = self.right_arm._motors_ft
+
+        return {
+            **{f"left_{k}": v for k, v in left_arm_motors_ft.items()},
+            **{f"right_{k}": v for k, v in right_arm_motors_ft.items()},
+        }
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        left_arm_cameras_ft = self.left_arm._cameras_ft
+        right_arm_cameras_ft = self.right_arm._cameras_ft
+
+        return {
+            **{f"left_{k}": v for k, v in left_arm_cameras_ft.items()},
+            **{f"right_{k}": v for k, v in right_arm_cameras_ft.items()},
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self.left_arm.is_connected and self.right_arm.is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.left_arm.connect(calibrate)
+        self.right_arm.connect(calibrate)
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.left_arm.is_calibrated and self.right_arm.is_calibrated
+
+    def calibrate(self) -> None:
+        self.left_arm.calibrate()
+        self.right_arm.calibrate()
+
+    def configure(self) -> None:
+        self.left_arm.configure()
+        self.right_arm.configure()
+
+    def setup_motors(self) -> None:
+        raise NotImplementedError(
+            "Motor ID configuration is typically done via manufacturer tools for CAN motors."
+        )
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        obs_dict = {}
+
+        # Add "left_" prefix
+        left_obs = self.left_arm.get_observation()
+        obs_dict.update({f"left_{key}": value for key, value in left_obs.items()})
+
+        # Add "right_" prefix
+        right_obs = self.right_arm.get_observation()
+        obs_dict.update({f"right_{key}": value for key, value in right_obs.items()})
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(
+        self,
+        action: RobotAction,
+        custom_kp: dict[str, float] | None = None,
+        custom_kd: dict[str, float] | None = None,
+    ) -> RobotAction:
+        # Remove "left_" prefix
+        left_action = {
+            key.removeprefix("left_"): value for key, value in action.items() if key.startswith("left_")
+        }
+        # Remove "right_" prefix
+        right_action = {
+            key.removeprefix("right_"): value for key, value in action.items() if key.startswith("right_")
+        }
+
+        sent_action_left = self.left_arm.send_action(left_action, custom_kp, custom_kd)
+        sent_action_right = self.right_arm.send_action(right_action, custom_kp, custom_kd)
+
+        # Add prefixes back
+        prefixed_sent_action_left = {f"left_{key}": value for key, value in sent_action_left.items()}
+        prefixed_sent_action_right = {f"right_{key}": value for key, value in sent_action_right.items()}
+
+        return {**prefixed_sent_action_left, **prefixed_sent_action_right}
+
+    @check_if_not_connected
+    def disconnect(self):
+        self.left_arm.disconnect()
+        self.right_arm.disconnect()
diff --git a/lerobot/src/lerobot/robots/bi_openarm_follower/config_bi_openarm_follower.py b/lerobot/src/lerobot/robots/bi_openarm_follower/config_bi_openarm_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..9d11f7b4e1fb237949e2a958536584856754cc45
--- /dev/null
+++ b/lerobot/src/lerobot/robots/bi_openarm_follower/config_bi_openarm_follower.py
@@ -0,0 +1,30 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from lerobot.robots.openarm_follower import OpenArmFollowerConfigBase
+
+from ..config import RobotConfig
+
+
+@RobotConfig.register_subclass("bi_openarm_follower")
+@dataclass
+class BiOpenArmFollowerConfig(RobotConfig):
+    """Configuration class for Bi OpenArm Follower robots."""
+
+    left_arm_config: OpenArmFollowerConfigBase
+    right_arm_config: OpenArmFollowerConfigBase
diff --git a/lerobot/src/lerobot/robots/bi_so_follower/__init__.py b/lerobot/src/lerobot/robots/bi_so_follower/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..f631a14db516b89fc62aead2bceefd76d27eadfe
--- /dev/null
+++ b/lerobot/src/lerobot/robots/bi_so_follower/__init__.py
@@ -0,0 +1,18 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .bi_so_follower import BiSOFollower
+from .config_bi_so_follower import BiSOFollowerConfig
diff --git a/lerobot/src/lerobot/robots/bi_so_follower/bi_so_follower.py b/lerobot/src/lerobot/robots/bi_so_follower/bi_so_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..ba1826e29b51f67dd21b88e821a4f86457fd35ca
--- /dev/null
+++ b/lerobot/src/lerobot/robots/bi_so_follower/bi_so_follower.py
@@ -0,0 +1,158 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+from functools import cached_property
+
+from lerobot.robots.so_follower import SOFollower, SOFollowerRobotConfig
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from .config_bi_so_follower import BiSOFollowerConfig
+
+logger = logging.getLogger(__name__)
+
+
+class BiSOFollower(Robot):
+    """
+    [Bimanual SO Follower Arms](https://github.com/TheRobotStudio/SO-ARM100) designed by TheRobotStudio
+    """
+
+    config_class = BiSOFollowerConfig
+    name = "bi_so_follower"
+
+    def __init__(self, config: BiSOFollowerConfig):
+        super().__init__(config)
+        self.config = config
+
+        left_arm_config = SOFollowerRobotConfig(
+            id=f"{config.id}_left" if config.id else None,
+            calibration_dir=config.calibration_dir,
+            port=config.left_arm_config.port,
+            disable_torque_on_disconnect=config.left_arm_config.disable_torque_on_disconnect,
+            max_relative_target=config.left_arm_config.max_relative_target,
+            use_degrees=config.left_arm_config.use_degrees,
+            cameras=config.left_arm_config.cameras,
+        )
+
+        right_arm_config = SOFollowerRobotConfig(
+            id=f"{config.id}_right" if config.id else None,
+            calibration_dir=config.calibration_dir,
+            port=config.right_arm_config.port,
+            disable_torque_on_disconnect=config.right_arm_config.disable_torque_on_disconnect,
+            max_relative_target=config.right_arm_config.max_relative_target,
+            use_degrees=config.right_arm_config.use_degrees,
+            cameras=config.right_arm_config.cameras,
+        )
+
+        self.left_arm = SOFollower(left_arm_config)
+        self.right_arm = SOFollower(right_arm_config)
+
+        # Only for compatibility with other parts of the codebase that expect a `robot.cameras` attribute
+        self.cameras = {**self.left_arm.cameras, **self.right_arm.cameras}
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        left_arm_motors_ft = self.left_arm._motors_ft
+        right_arm_motors_ft = self.right_arm._motors_ft
+
+        return {
+            **{f"left_{k}": v for k, v in left_arm_motors_ft.items()},
+            **{f"right_{k}": v for k, v in right_arm_motors_ft.items()},
+        }
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        left_arm_cameras_ft = self.left_arm._cameras_ft
+        right_arm_cameras_ft = self.right_arm._cameras_ft
+
+        return {
+            **{f"left_{k}": v for k, v in left_arm_cameras_ft.items()},
+            **{f"right_{k}": v for k, v in right_arm_cameras_ft.items()},
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self.left_arm.is_connected and self.right_arm.is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.left_arm.connect(calibrate)
+        self.right_arm.connect(calibrate)
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.left_arm.is_calibrated and self.right_arm.is_calibrated
+
+    def calibrate(self) -> None:
+        self.left_arm.calibrate()
+        self.right_arm.calibrate()
+
+    def configure(self) -> None:
+        self.left_arm.configure()
+        self.right_arm.configure()
+
+    def setup_motors(self) -> None:
+        self.left_arm.setup_motors()
+        self.right_arm.setup_motors()
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        obs_dict = {}
+
+        # Add "left_" prefix
+        left_obs = self.left_arm.get_observation()
+        obs_dict.update({f"left_{key}": value for key, value in left_obs.items()})
+
+        # Add "right_" prefix
+        right_obs = self.right_arm.get_observation()
+        obs_dict.update({f"right_{key}": value for key, value in right_obs.items()})
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        # Remove "left_" prefix
+        left_action = {
+            key.removeprefix("left_"): value for key, value in action.items() if key.startswith("left_")
+        }
+        # Remove "right_" prefix
+        right_action = {
+            key.removeprefix("right_"): value for key, value in action.items() if key.startswith("right_")
+        }
+
+        sent_action_left = self.left_arm.send_action(left_action)
+        sent_action_right = self.right_arm.send_action(right_action)
+
+        # Add prefixes back
+        prefixed_sent_action_left = {f"left_{key}": value for key, value in sent_action_left.items()}
+        prefixed_sent_action_right = {f"right_{key}": value for key, value in sent_action_right.items()}
+
+        return {**prefixed_sent_action_left, **prefixed_sent_action_right}
+
+    @check_if_not_connected
+    def disconnect(self):
+        self.left_arm.disconnect()
+        self.right_arm.disconnect()
diff --git a/lerobot/src/lerobot/robots/bi_so_follower/config_bi_so_follower.py b/lerobot/src/lerobot/robots/bi_so_follower/config_bi_so_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..dca74fa2de4ffcf61e44181a3cea92875d209fb2
--- /dev/null
+++ b/lerobot/src/lerobot/robots/bi_so_follower/config_bi_so_follower.py
@@ -0,0 +1,30 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from lerobot.robots.so_follower import SOFollowerConfig
+
+from ..config import RobotConfig
+
+
+@RobotConfig.register_subclass("bi_so_follower")
+@dataclass
+class BiSOFollowerConfig(RobotConfig):
+    """Configuration class for Bi SO Follower robots."""
+
+    left_arm_config: SOFollowerConfig
+    right_arm_config: SOFollowerConfig
diff --git a/lerobot/src/lerobot/robots/config.py b/lerobot/src/lerobot/robots/config.py
new file mode 100644
index 0000000000000000000000000000000000000000..a85a831693c31fd13d3e19138887ee981a5cbc37
--- /dev/null
+++ b/lerobot/src/lerobot/robots/config.py
@@ -0,0 +1,40 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import abc
+from dataclasses import dataclass
+from pathlib import Path
+
+import draccus
+
+
+@dataclass(kw_only=True)
+class RobotConfig(draccus.ChoiceRegistry, abc.ABC):
+    # Allows to distinguish between different robots of the same type
+    id: str | None = None
+    # Directory to store calibration file
+    calibration_dir: Path | None = None
+
+    def __post_init__(self):
+        if hasattr(self, "cameras") and self.cameras:
+            for _, config in self.cameras.items():
+                for attr in ["width", "height", "fps"]:
+                    if getattr(config, attr) is None:
+                        raise ValueError(
+                            f"Specifying '{attr}' is required for the camera to be used in a robot"
+                        )
+
+    @property
+    def type(self) -> str:
+        return self.get_choice_name(self.__class__)
diff --git a/lerobot/src/lerobot/robots/earthrover_mini_plus/__init__.py b/lerobot/src/lerobot/robots/earthrover_mini_plus/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..60546cd41ad92fd4e77c059ff574cb1c9163082d
--- /dev/null
+++ b/lerobot/src/lerobot/robots/earthrover_mini_plus/__init__.py
@@ -0,0 +1,20 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_earthrover_mini_plus import EarthRoverMiniPlusConfig
+from .robot_earthrover_mini_plus import EarthRoverMiniPlus
+
+__all__ = ["EarthRoverMiniPlus", "EarthRoverMiniPlusConfig"]
diff --git a/lerobot/src/lerobot/robots/earthrover_mini_plus/config_earthrover_mini_plus.py b/lerobot/src/lerobot/robots/earthrover_mini_plus/config_earthrover_mini_plus.py
new file mode 100644
index 0000000000000000000000000000000000000000..5ed80476b752821b5aa715c8a05dc9d401fe8672
--- /dev/null
+++ b/lerobot/src/lerobot/robots/earthrover_mini_plus/config_earthrover_mini_plus.py
@@ -0,0 +1,35 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Configuration for EarthRover Mini Plus robot."""
+
+from dataclasses import dataclass
+
+from ..config import RobotConfig
+
+
+@RobotConfig.register_subclass("earthrover_mini_plus")
+@dataclass
+class EarthRoverMiniPlusConfig(RobotConfig):
+    """Configuration for EarthRover Mini Plus robot using Frodobots SDK.
+
+    This robot uses cloud-based control via the Frodobots SDK HTTP API.
+    Camera frames are accessed directly through SDK HTTP endpoints.
+
+    Attributes:
+        sdk_url: URL of the Frodobots SDK server (default: http://localhost:8000)
+    """
+
+    sdk_url: str = "http://localhost:8000"
diff --git a/lerobot/src/lerobot/robots/earthrover_mini_plus/earthrover_mini_plus.mdx b/lerobot/src/lerobot/robots/earthrover_mini_plus/earthrover_mini_plus.mdx
new file mode 120000
index 0000000000000000000000000000000000000000..37509e0a908ef8b69ca33fc8718f8052b63b0d79
--- /dev/null
+++ b/lerobot/src/lerobot/robots/earthrover_mini_plus/earthrover_mini_plus.mdx
@@ -0,0 +1 @@
+../../../../docs/source/earthrover_mini_plus.mdx
\ No newline at end of file
diff --git a/lerobot/src/lerobot/robots/earthrover_mini_plus/robot_earthrover_mini_plus.py b/lerobot/src/lerobot/robots/earthrover_mini_plus/robot_earthrover_mini_plus.py
new file mode 100644
index 0000000000000000000000000000000000000000..76707a80c81cecde040b2f31eb32d7e9f972f20c
--- /dev/null
+++ b/lerobot/src/lerobot/robots/earthrover_mini_plus/robot_earthrover_mini_plus.py
@@ -0,0 +1,561 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""EarthRover Mini Plus robot using Frodobots SDK."""
+
+import base64
+import logging
+from functools import cached_property
+
+import cv2
+import numpy as np
+import requests
+
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.errors import DeviceNotConnectedError
+
+from ..robot import Robot
+from .config_earthrover_mini_plus import EarthRoverMiniPlusConfig
+
+logger = logging.getLogger(__name__)
+
+# Action feature keys
+ACTION_LINEAR_VEL = "linear_velocity"
+ACTION_ANGULAR_VEL = "angular_velocity"
+
+# Observation feature keys — cameras
+OBS_FRONT = "front"
+OBS_REAR = "rear"
+
+# Observation feature keys — telemetry
+OBS_SPEED = "speed"
+OBS_BATTERY_LEVEL = "battery_level"
+OBS_ORIENTATION = "orientation"
+OBS_GPS_LATITUDE = "gps_latitude"
+OBS_GPS_LONGITUDE = "gps_longitude"
+OBS_GPS_SIGNAL = "gps_signal"
+OBS_SIGNAL_LEVEL = "signal_level"
+OBS_VIBRATION = "vibration"
+OBS_LAMP = "lamp"
+
+# Observation feature keys — IMU sensors
+OBS_ACCELEROMETER_X = "accelerometer_x"
+OBS_ACCELEROMETER_Y = "accelerometer_y"
+OBS_ACCELEROMETER_Z = "accelerometer_z"
+OBS_GYROSCOPE_X = "gyroscope_x"
+OBS_GYROSCOPE_Y = "gyroscope_y"
+OBS_GYROSCOPE_Z = "gyroscope_z"
+OBS_MAGNETOMETER_X = "magnetometer_filtered_x"
+OBS_MAGNETOMETER_Y = "magnetometer_filtered_y"
+OBS_MAGNETOMETER_Z = "magnetometer_filtered_z"
+
+# Observation feature keys — wheel RPMs
+OBS_WHEEL_RPM_0 = "wheel_rpm_0"
+OBS_WHEEL_RPM_1 = "wheel_rpm_1"
+OBS_WHEEL_RPM_2 = "wheel_rpm_2"
+OBS_WHEEL_RPM_3 = "wheel_rpm_3"
+
+
+class EarthRoverMiniPlus(Robot):
+    """
+    EarthRover Mini Plus robot controlled via Frodobots SDK HTTP API.
+
+    This robot uses cloud-based control through the Frodobots SDK instead of direct
+    hardware connection. Cameras stream via WebRTC through Agora cloud, and control
+    commands are sent via HTTP POST requests.
+
+    The robot supports:
+    - Dual cameras (front and rear) accessed via SDK HTTP endpoints
+    - Linear and angular velocity control
+    - Battery and orientation telemetry
+
+    Attributes:
+        config: Robot configuration
+        sdk_base_url: URL of the Frodobots SDK server (default: http://localhost:8000)
+    """
+
+    config_class = EarthRoverMiniPlusConfig
+    name = "earthrover_mini_plus"
+
+    def __init__(self, config: EarthRoverMiniPlusConfig):
+        """Initialize EarthRover Mini Plus robot.
+
+        Args:
+            config: Robot configuration including SDK URL
+        """
+        super().__init__(config)
+        self.config = config
+        self.sdk_base_url = "http://localhost:8000"
+
+        # Empty cameras dict for compatibility with recording script
+        # Cameras are accessed directly via SDK, not through Camera objects
+        self.cameras = {}
+        self._is_connected = False
+
+        # Cache for camera frames (fallback when requests fail)
+        self._last_front_frame = None
+        self._last_rear_frame = None
+
+        # Cache for robot telemetry data (fallback when requests fail)
+        self._last_robot_data = None
+
+        logger.info(f"Initialized {self.name} with SDK at {self.sdk_base_url}")
+
+    @property
+    def is_connected(self) -> bool:
+        """Check if robot is connected to SDK."""
+        return self._is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        """Connect to robot via Frodobots SDK.
+
+        Args:
+            calibrate: Not used for SDK-based robot (kept for API compatibility)
+
+        Raises:
+            DeviceAlreadyConnectedError: If robot is already connected
+            DeviceNotConnectedError: If cannot connect to SDK server
+        """
+
+        # Verify SDK is running and accessible
+        try:
+            response = requests.get(f"{self.sdk_base_url}/data", timeout=10.0)
+            if response.status_code != 200:
+                raise DeviceNotConnectedError(
+                    f"Cannot connect to SDK at {self.sdk_base_url}. "
+                    "Make sure it's running: hypercorn main:app --reload"
+                )
+        except requests.RequestException as e:
+            raise DeviceNotConnectedError(f"Cannot connect to SDK at {self.sdk_base_url}: {e}") from e
+
+        self._is_connected = True
+        logger.info(f"{self.name} connected to SDK")
+
+        if calibrate:
+            self.calibrate()
+
+    def calibrate(self) -> None:
+        """Calibration not needed for SDK-based robot."""
+        logger.info("Calibration not required for SDK-based robot")
+
+    @property
+    def is_calibrated(self) -> bool:
+        """SDK robot doesn't require calibration.
+
+        Returns:
+            bool: Always True for SDK-based robots
+        """
+        return True
+
+    def configure(self) -> None:
+        """Configure robot (no-op for SDK-based robot)."""
+        pass
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        """Define the observation space for dataset recording.
+
+        Returns:
+            dict: Observation features with types/shapes:
+                - front: (480, 640, 3) - Front camera RGB image
+                - rear: (480, 640, 3) - Rear camera RGB image
+                - speed: float - Current speed (raw SDK value)
+                - battery_level: float - Battery level (0-100)
+                - orientation: float - Robot orientation in degrees
+                - gps_latitude: float - GPS latitude coordinate
+                - gps_longitude: float - GPS longitude coordinate
+                - gps_signal: float - GPS signal strength (percentage)
+                - signal_level: float - Network signal level (0-5)
+                - vibration: float - Vibration sensor reading
+                - lamp: float - Lamp state (0=off, 1=on)
+                - accelerometer_x: float - Accelerometer X axis (raw SDK value)
+                - accelerometer_y: float - Accelerometer Y axis (raw SDK value)
+                - accelerometer_z: float - Accelerometer Z axis (raw SDK value)
+                - gyroscope_x: float - Gyroscope X axis (raw SDK value)
+                - gyroscope_y: float - Gyroscope Y axis (raw SDK value)
+                - gyroscope_z: float - Gyroscope Z axis (raw SDK value)
+                - magnetometer_filtered_x: float - Magnetometer X axis (raw SDK value)
+                - magnetometer_filtered_y: float - Magnetometer Y axis (raw SDK value)
+                - magnetometer_filtered_z: float - Magnetometer Z axis (raw SDK value)
+                - wheel_rpm_0: float - Wheel 0 RPM
+                - wheel_rpm_1: float - Wheel 1 RPM
+                - wheel_rpm_2: float - Wheel 2 RPM
+                - wheel_rpm_3: float - Wheel 3 RPM
+        """
+        return {
+            # Cameras (height, width, channels)
+            OBS_FRONT: (480, 640, 3),
+            OBS_REAR: (480, 640, 3),
+            # Telemetry
+            OBS_SPEED: float,
+            OBS_BATTERY_LEVEL: float,
+            OBS_ORIENTATION: float,
+            OBS_GPS_LATITUDE: float,
+            OBS_GPS_LONGITUDE: float,
+            OBS_GPS_SIGNAL: float,
+            OBS_SIGNAL_LEVEL: float,
+            OBS_VIBRATION: float,
+            OBS_LAMP: float,
+            # IMU — accelerometer
+            OBS_ACCELEROMETER_X: float,
+            OBS_ACCELEROMETER_Y: float,
+            OBS_ACCELEROMETER_Z: float,
+            # IMU — gyroscope
+            OBS_GYROSCOPE_X: float,
+            OBS_GYROSCOPE_Y: float,
+            OBS_GYROSCOPE_Z: float,
+            # IMU — magnetometer
+            OBS_MAGNETOMETER_X: float,
+            OBS_MAGNETOMETER_Y: float,
+            OBS_MAGNETOMETER_Z: float,
+            # Wheel RPMs
+            OBS_WHEEL_RPM_0: float,
+            OBS_WHEEL_RPM_1: float,
+            OBS_WHEEL_RPM_2: float,
+            OBS_WHEEL_RPM_3: float,
+        }
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        """Define the action space.
+
+        Returns:
+            dict: Action features with types:
+                - linear_velocity: float - Target linear velocity (-1 to 1)
+                - angular_velocity: float - Target angular velocity (-1 to 1)
+        """
+        return {
+            ACTION_LINEAR_VEL: float,
+            ACTION_ANGULAR_VEL: float,
+        }
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        """Get current robot observation from SDK.
+
+        Camera frames are retrieved from SDK endpoints /v2/front and /v2/rear.
+        Frames are decoded from base64 and converted from BGR to RGB format.
+        Robot telemetry is retrieved from /data endpoint.
+        Sensor arrays (accels, gyros, mags, rpms) each contain entries of
+        [values..., timestamp]; the latest reading from each array is used.
+
+        Returns:
+            RobotObservation: Observation containing:
+                - front: Front camera image (480, 640, 3) in RGB format
+                - rear: Rear camera image (480, 640, 3) in RGB format
+                - speed: float - Current speed (raw SDK value)
+                - battery_level: float - Battery level (0-100)
+                - orientation: float - Robot orientation in degrees
+                - gps_latitude: float - GPS latitude coordinate
+                - gps_longitude: float - GPS longitude coordinate
+                - gps_signal: float - GPS signal strength (percentage)
+                - signal_level: float - Network signal level (0-5)
+                - vibration: float - Vibration sensor reading
+                - lamp: float - Lamp state (0=off, 1=on)
+                - accelerometer_x/y/z: float - Accelerometer axes (raw SDK value)
+                - gyroscope_x/y/z: float - Gyroscope axes (raw SDK value)
+                - magnetometer_filtered_x/y/z: float - Magnetometer axes (raw SDK value)
+                - wheel_rpm_0/1/2/3: float - Wheel RPMs
+
+        Raises:
+            DeviceNotConnectedError: If robot is not connected
+
+        Note:
+            Camera frames are retrieved from SDK endpoints /v2/front and /v2/rear.
+            Frames are decoded from base64 and converted from BGR to RGB format.
+            Robot telemetry is retrieved from /data endpoint.
+            All SDK values are normalized to appropriate ranges for dataset recording.
+        """
+
+        observation = {}
+
+        # Get camera images from SDK
+        frames = self._get_camera_frames()
+        observation[OBS_FRONT] = frames["front"]
+        observation[OBS_REAR] = frames["rear"]
+
+        # Get robot state from SDK
+        robot_data = self._get_robot_data()
+
+        # Telemetry
+        observation[OBS_SPEED] = float(robot_data["speed"])
+        observation[OBS_BATTERY_LEVEL] = float(robot_data["battery"])
+        observation[OBS_ORIENTATION] = float(robot_data["orientation"])
+        observation[OBS_GPS_LATITUDE] = float(robot_data["latitude"])
+        observation[OBS_GPS_LONGITUDE] = float(robot_data["longitude"])
+        observation[OBS_GPS_SIGNAL] = float(robot_data["gps_signal"])
+        observation[OBS_SIGNAL_LEVEL] = float(robot_data["signal_level"])
+        observation[OBS_VIBRATION] = float(robot_data["vibration"])
+        observation[OBS_LAMP] = float(robot_data["lamp"])
+
+        # Accelerometer — latest reading from accels array [x, y, z, ts]
+        accel = self._latest_sensor_reading(robot_data, "accels", n_values=3)
+        observation[OBS_ACCELEROMETER_X] = accel[0]
+        observation[OBS_ACCELEROMETER_Y] = accel[1]
+        observation[OBS_ACCELEROMETER_Z] = accel[2]
+
+        # Gyroscope — latest reading from gyros array [x, y, z, ts]
+        gyro = self._latest_sensor_reading(robot_data, "gyros", n_values=3)
+        observation[OBS_GYROSCOPE_X] = gyro[0]
+        observation[OBS_GYROSCOPE_Y] = gyro[1]
+        observation[OBS_GYROSCOPE_Z] = gyro[2]
+
+        # Magnetometer — latest reading from mags array [x, y, z, ts]
+        mag = self._latest_sensor_reading(robot_data, "mags", n_values=3)
+        observation[OBS_MAGNETOMETER_X] = mag[0]
+        observation[OBS_MAGNETOMETER_Y] = mag[1]
+        observation[OBS_MAGNETOMETER_Z] = mag[2]
+
+        # Wheel RPMs — latest reading from rpms array [w0, w1, w2, w3, ts]
+        rpm = self._latest_sensor_reading(robot_data, "rpms", n_values=4)
+        observation[OBS_WHEEL_RPM_0] = rpm[0]
+        observation[OBS_WHEEL_RPM_1] = rpm[1]
+        observation[OBS_WHEEL_RPM_2] = rpm[2]
+        observation[OBS_WHEEL_RPM_3] = rpm[3]
+
+        return observation
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        """Send action to robot via SDK.
+
+        Args:
+            action: Action dict with keys:
+                - linear_velocity: Target linear velocity (-1 to 1)
+                - angular_velocity: Target angular velocity (-1 to 1)
+
+        Returns:
+            RobotAction: The action that was sent (matches action_features keys)
+
+        Raises:
+            DeviceNotConnectedError: If robot is not connected
+
+        Note:
+            Actions are sent to SDK via POST /control endpoint.
+            SDK expects commands in range [-1, 1].
+        """
+        linear = float(action.get(ACTION_LINEAR_VEL, 0.0))
+        angular = float(action.get(ACTION_ANGULAR_VEL, 0.0))
+
+        try:
+            self._send_command_to_sdk(linear, angular)
+        except Exception as e:
+            logger.error(f"Error sending action: {e}")
+
+        return {
+            ACTION_LINEAR_VEL: linear,
+            ACTION_ANGULAR_VEL: angular,
+        }
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        """Disconnect from robot.
+
+        Stops the robot and closes connection to SDK.
+
+        Raises:
+            DeviceNotConnectedError: If robot is not connected
+        """
+
+        # Stop the robot before disconnecting
+        try:
+            self._send_command_to_sdk(0.0, 0.0)
+        except Exception as e:
+            logger.warning(f"Failed to stop robot during disconnect: {e}")
+
+        self._is_connected = False
+        logger.info(f"{self.name} disconnected")
+
+    # Private helper methods for SDK communication
+
+    def _get_camera_frames(self) -> dict[str, np.ndarray]:
+        """Get camera frames from SDK using v2 endpoints with caching fallback.
+
+        Returns:
+            dict: Dictionary with 'front' and 'rear' keys containing:
+                - Current frame (if request succeeds)
+                - Cached frame (if request fails but cache exists)
+                - Zero array (if request fails and no cache exists yet)
+
+        Note:
+            Uses /v2/front and /v2/rear endpoints which are 15x faster than /screenshot.
+            Images are base64 encoded, resized to 640x480, and converted from BGR to RGB.
+            If request fails, returns the last successfully retrieved frame (cached).
+        """
+        frames = {}
+
+        # Get front camera
+        try:
+            response = requests.get(f"{self.sdk_base_url}/v2/front", timeout=2.0)
+            if response.status_code == 200:
+                data = response.json()
+                if "front_frame" in data and data["front_frame"]:
+                    front_img = self._decode_base64_image(data["front_frame"])
+                    if front_img is not None:
+                        # Resize and convert BGR to RGB
+                        front_img = cv2.resize(front_img, (640, 480))
+                        front_rgb = cv2.cvtColor(front_img, cv2.COLOR_BGR2RGB)
+                        frames["front"] = front_rgb
+                        # Cache the successful frame
+                        self._last_front_frame = front_rgb
+        except Exception as e:
+            logger.warning(f"Error fetching front camera: {e}")
+
+        # Fallback: use cache or zero array
+        if "front" not in frames:
+            if self._last_front_frame is not None:
+                frames["front"] = self._last_front_frame
+            else:
+                frames["front"] = np.zeros((480, 640, 3), dtype=np.uint8)
+
+        # Get rear camera
+        try:
+            response = requests.get(f"{self.sdk_base_url}/v2/rear", timeout=2.0)
+            if response.status_code == 200:
+                data = response.json()
+                if "rear_frame" in data and data["rear_frame"]:
+                    rear_img = self._decode_base64_image(data["rear_frame"])
+                    if rear_img is not None:
+                        # Resize and convert BGR to RGB
+                        rear_img = cv2.resize(rear_img, (640, 480))
+                        rear_rgb = cv2.cvtColor(rear_img, cv2.COLOR_BGR2RGB)
+                        frames["rear"] = rear_rgb
+                        # Cache the successful frame
+                        self._last_rear_frame = rear_rgb
+        except Exception as e:
+            logger.warning(f"Error fetching rear camera: {e}")
+
+        # Fallback: use cache or zero array
+        if "rear" not in frames:
+            if self._last_rear_frame is not None:
+                frames["rear"] = self._last_rear_frame
+            else:
+                frames["rear"] = np.zeros((480, 640, 3), dtype=np.uint8)
+
+        return frames
+
+    def _decode_base64_image(self, base64_string: str) -> np.ndarray | None:
+        """Decode base64 string to image.
+
+        Args:
+            base64_string: Base64 encoded image string
+
+        Returns:
+            np.ndarray: Decoded image in BGR format (OpenCV default), or None if decoding fails
+        """
+        try:
+            img_bytes = base64.b64decode(base64_string)
+            nparr = np.frombuffer(img_bytes, np.uint8)
+            img = cv2.imdecode(nparr, cv2.IMREAD_COLOR)
+            return img  # Return in BGR format (OpenCV default)
+        except Exception as e:
+            logger.error(f"Error decoding image: {e}")
+            return None
+
+    @staticmethod
+    def _latest_sensor_reading(robot_data: dict, key: str, n_values: int) -> list[float]:
+        """Extract the latest sensor reading from an SDK sensor array.
+
+        The SDK returns sensor arrays like ``accels``, ``gyros``, ``mags``,
+        ``rpms`` where each entry is ``[value_0, ..., value_n, timestamp]``.
+        This helper returns the *n_values* leading floats from the last entry,
+        falling back to zeros when the key is missing or the array is empty.
+        """
+        readings = robot_data.get(key)
+        if readings and len(readings) > 0:
+            latest = readings[-1]
+            return [float(v) for v in latest[:n_values]]
+        return [0.0] * n_values
+
+    def _get_robot_data(self) -> dict:
+        """Get robot telemetry data from SDK.
+
+        Returns:
+            dict: Robot telemetry data including battery, speed, orientation, GPS,
+                and sensor arrays (accels, gyros, mags, rpms):
+                - Current data (if request succeeds)
+                - Cached data (if request fails but cache exists)
+                - Default values (if request fails and no cache exists yet)
+
+        Note:
+            Uses /data endpoint which provides comprehensive robot state.
+            If request fails, returns the last successfully retrieved data (cached).
+        """
+        try:
+            response = requests.get(f"{self.sdk_base_url}/data", timeout=2.0)
+            if response.status_code == 200:
+                data = response.json()
+                # Cache the successful data
+                self._last_robot_data = data
+                return data
+        except Exception as e:
+            logger.warning(f"Error fetching robot data: {e}")
+
+        # Fallback: use cache or default values
+        if self._last_robot_data is not None:
+            return self._last_robot_data
+
+        # Return dict with default values (used only on first failure before any cache exists)
+        return {
+            "speed": 0,
+            "battery": 0,
+            "orientation": 0,
+            "latitude": 0.0,
+            "longitude": 0.0,
+            "gps_signal": 0,
+            "signal_level": 0,
+            "vibration": 0.0,
+            "lamp": 0,
+            "accels": [],
+            "gyros": [],
+            "mags": [],
+            "rpms": [],
+        }
+
+    def _send_command_to_sdk(self, linear: float, angular: float, lamp: int = 0) -> bool:
+        """Send control command to SDK.
+
+        Args:
+            linear: Linear velocity command (-1 to 1)
+            angular: Angular velocity command (-1 to 1)
+            lamp: Lamp control (0=off, 1=on)
+
+        Returns:
+            bool: True if command sent successfully, False otherwise
+
+        Note:
+            Uses POST /control endpoint. Commands are sent as JSON payload.
+        """
+        try:
+            payload = {
+                "command": {
+                    "linear": linear,
+                    "angular": angular,
+                    "lamp": lamp,
+                }
+            }
+
+            response = requests.post(
+                f"{self.sdk_base_url}/control",
+                json=payload,
+                timeout=1.0,
+            )
+
+            return response.status_code == 200
+        except Exception as e:
+            logger.error(f"Error sending command: {e}")
+            return False
diff --git a/lerobot/src/lerobot/robots/hope_jr/__init__.py b/lerobot/src/lerobot/robots/hope_jr/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..26603ebb080bc9fb3667acce6e1a8763602998eb
--- /dev/null
+++ b/lerobot/src/lerobot/robots/hope_jr/__init__.py
@@ -0,0 +1,19 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_hope_jr import HopeJrArmConfig, HopeJrHandConfig
+from .hope_jr_arm import HopeJrArm
+from .hope_jr_hand import HopeJrHand
diff --git a/lerobot/src/lerobot/robots/hope_jr/config_hope_jr.py b/lerobot/src/lerobot/robots/hope_jr/config_hope_jr.py
new file mode 100644
index 0000000000000000000000000000000000000000..f2af5f47c5faa00de73dc633366d863afce5b73d
--- /dev/null
+++ b/lerobot/src/lerobot/robots/hope_jr/config_hope_jr.py
@@ -0,0 +1,51 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.cameras import CameraConfig
+
+from ..config import RobotConfig
+
+
+@RobotConfig.register_subclass("hope_jr_hand")
+@dataclass
+class HopeJrHandConfig(RobotConfig):
+    port: str  # Port to connect to the hand
+    side: str  # "left" / "right"
+
+    disable_torque_on_disconnect: bool = True
+
+    cameras: dict[str, CameraConfig] = field(default_factory=dict)
+
+    def __post_init__(self):
+        super().__post_init__()
+        if self.side not in ["right", "left"]:
+            raise ValueError(self.side)
+
+
+@RobotConfig.register_subclass("hope_jr_arm")
+@dataclass
+class HopeJrArmConfig(RobotConfig):
+    port: str  # Port to connect to the hand
+    disable_torque_on_disconnect: bool = True
+
+    # `max_relative_target` limits the magnitude of the relative positional target vector for safety purposes.
+    # Set this to a positive scalar to have the same value for all motors, or a dictionary that maps motor
+    # names to the max_relative_target value for that motor.
+    max_relative_target: float | dict[str, float] | None = None
+
+    cameras: dict[str, CameraConfig] = field(default_factory=dict)
diff --git a/lerobot/src/lerobot/robots/hope_jr/hope_jr.mdx b/lerobot/src/lerobot/robots/hope_jr/hope_jr.mdx
new file mode 120000
index 0000000000000000000000000000000000000000..a076e4754acc0f88e53320da293f2fc9ec2d06cf
--- /dev/null
+++ b/lerobot/src/lerobot/robots/hope_jr/hope_jr.mdx
@@ -0,0 +1 @@
+../../../../docs/source/hope_jr.mdx
\ No newline at end of file
diff --git a/lerobot/src/lerobot/robots/hope_jr/hope_jr_arm.py b/lerobot/src/lerobot/robots/hope_jr/hope_jr_arm.py
new file mode 100644
index 0000000000000000000000000000000000000000..7f6492ef0e9f7920a003de611bb7dd0832f8b3c6
--- /dev/null
+++ b/lerobot/src/lerobot/robots/hope_jr/hope_jr_arm.py
@@ -0,0 +1,169 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from functools import cached_property
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.motors import Motor, MotorNormMode
+from lerobot.motors.calibration_gui import RangeFinderGUI
+from lerobot.motors.feetech import (
+    FeetechMotorsBus,
+)
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from ..utils import ensure_safe_goal_position
+from .config_hope_jr import HopeJrArmConfig
+
+logger = logging.getLogger(__name__)
+
+
+class HopeJrArm(Robot):
+    config_class = HopeJrArmConfig
+    name = "hope_jr_arm"
+
+    def __init__(self, config: HopeJrArmConfig):
+        super().__init__(config)
+        self.config = config
+        self.bus = FeetechMotorsBus(
+            port=self.config.port,
+            motors={
+                "shoulder_pitch": Motor(1, "sm8512bl", MotorNormMode.RANGE_M100_100),
+                "shoulder_yaw": Motor(2, "sts3250", MotorNormMode.RANGE_M100_100),
+                "shoulder_roll": Motor(3, "sts3250", MotorNormMode.RANGE_M100_100),
+                "elbow_flex": Motor(4, "sts3250", MotorNormMode.RANGE_M100_100),
+                "wrist_roll": Motor(5, "sts3250", MotorNormMode.RANGE_M100_100),
+                "wrist_yaw": Motor(6, "sts3250", MotorNormMode.RANGE_M100_100),
+                "wrist_pitch": Motor(7, "sts3250", MotorNormMode.RANGE_M100_100),
+            },
+            calibration=self.calibration,
+        )
+        self.cameras = make_cameras_from_configs(config.cameras)
+
+        # HACK
+        self.shoulder_pitch = "shoulder_pitch"
+        self.other_motors = [m for m in self.bus.motors if m != "shoulder_pitch"]
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.bus.motors}
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected and all(cam.is_connected for cam in self.cameras.values())
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        """
+        We assume that at connection time, arm is in a rest position,
+        and torque can be safely disabled to run calibration.
+        """
+
+        self.bus.connect(handshake=False)
+        if not self.is_calibrated and calibrate:
+            self.calibrate()
+
+        # Connect the cameras
+        for cam in self.cameras.values():
+            cam.connect()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        groups = {
+            "all": list(self.bus.motors.keys()),
+            "shoulder": ["shoulder_pitch", "shoulder_yaw", "shoulder_roll"],
+            "elbow": ["elbow_flex"],
+            "wrist": ["wrist_roll", "wrist_yaw", "wrist_pitch"],
+        }
+
+        self.calibration = RangeFinderGUI(self.bus, groups).run()
+        self._save_calibration()
+        print("Calibration saved to", self.calibration_fpath)
+
+    def configure(self) -> None:
+        with self.bus.torque_disabled():
+            self.bus.configure_motors(maximum_acceleration=30, acceleration=30)
+
+    def setup_motors(self) -> None:
+        # TODO: add docstring
+        for motor in reversed(self.bus.motors):
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        # Read arm position
+        start = time.perf_counter()
+        obs_dict = self.bus.sync_read("Present_Position", self.other_motors)
+        obs_dict[self.shoulder_pitch] = self.bus.read("Present_Position", self.shoulder_pitch)
+        obs_dict = {f"{motor}.pos": val for motor, val in obs_dict.items()}
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read state: {dt_ms:.1f}ms")
+
+        # Capture images from cameras
+        for cam_key, cam in self.cameras.items():
+            start = time.perf_counter()
+            obs_dict[cam_key] = cam.read_latest()
+            dt_ms = (time.perf_counter() - start) * 1e3
+            logger.debug(f"{self} read {cam_key}: {dt_ms:.1f}ms")
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        goal_pos = {key.removesuffix(".pos"): val for key, val in action.items() if key.endswith(".pos")}
+
+        # Cap goal position when too far away from present position.
+        # /!\ Slower fps expected due to reading from the follower.
+        if self.config.max_relative_target is not None:
+            present_pos = self.bus.sync_read("Present_Position")
+            goal_present_pos = {key: (g_pos, present_pos[key]) for key, g_pos in goal_pos.items()}
+            goal_pos = ensure_safe_goal_position(goal_present_pos, self.config.max_relative_target)
+
+        self.bus.sync_write("Goal_Position", goal_pos)
+        return {f"{motor}.pos": val for motor, val in goal_pos.items()}
+
+    @check_if_not_connected
+    def disconnect(self):
+        self.bus.disconnect(self.config.disable_torque_on_disconnect)
+        for cam in self.cameras.values():
+            cam.disconnect()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/robots/hope_jr/hope_jr_hand.py b/lerobot/src/lerobot/robots/hope_jr/hope_jr_hand.py
new file mode 100644
index 0000000000000000000000000000000000000000..78480483658ae267743bb6418c4c242d3dbde3ba
--- /dev/null
+++ b/lerobot/src/lerobot/robots/hope_jr/hope_jr_hand.py
@@ -0,0 +1,192 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from functools import cached_property
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.motors import Motor, MotorNormMode
+from lerobot.motors.calibration_gui import RangeFinderGUI
+from lerobot.motors.feetech import (
+    FeetechMotorsBus,
+)
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from .config_hope_jr import HopeJrHandConfig
+
+logger = logging.getLogger(__name__)
+
+RIGHT_HAND_INVERSIONS = [
+    "thumb_mcp",
+    "thumb_dip",
+    "index_ulnar_flexor",
+    "middle_ulnar_flexor",
+    "ring_ulnar_flexor",
+    "ring_pip_dip",
+    "pinky_ulnar_flexor",
+    "pinky_pip_dip",
+]
+
+LEFT_HAND_INVERSIONS = [
+    "thumb_cmc",
+    "thumb_mcp",
+    "thumb_dip",
+    "index_radial_flexor",
+    "index_pip_dip",
+    "middle_radial_flexor",
+    "middle_pip_dip",
+    "ring_radial_flexor",
+    "ring_pip_dip",
+    "pinky_radial_flexor",
+    # "pinky_pip_dip",
+]
+
+
+class HopeJrHand(Robot):
+    config_class = HopeJrHandConfig
+    name = "hope_jr_hand"
+
+    def __init__(self, config: HopeJrHandConfig):
+        super().__init__(config)
+        self.config = config
+        self.bus = FeetechMotorsBus(
+            port=self.config.port,
+            motors={
+                # Thumb
+                "thumb_cmc": Motor(1, "scs0009", MotorNormMode.RANGE_0_100),
+                "thumb_mcp": Motor(2, "scs0009", MotorNormMode.RANGE_0_100),
+                "thumb_pip": Motor(3, "scs0009", MotorNormMode.RANGE_0_100),
+                "thumb_dip": Motor(4, "scs0009", MotorNormMode.RANGE_0_100),
+                # Index
+                "index_radial_flexor": Motor(5, "scs0009", MotorNormMode.RANGE_0_100),
+                "index_ulnar_flexor": Motor(6, "scs0009", MotorNormMode.RANGE_0_100),
+                "index_pip_dip": Motor(7, "scs0009", MotorNormMode.RANGE_0_100),
+                # Middle
+                "middle_radial_flexor": Motor(8, "scs0009", MotorNormMode.RANGE_0_100),
+                "middle_ulnar_flexor": Motor(9, "scs0009", MotorNormMode.RANGE_0_100),
+                "middle_pip_dip": Motor(10, "scs0009", MotorNormMode.RANGE_0_100),
+                # Ring
+                "ring_radial_flexor": Motor(11, "scs0009", MotorNormMode.RANGE_0_100),
+                "ring_ulnar_flexor": Motor(12, "scs0009", MotorNormMode.RANGE_0_100),
+                "ring_pip_dip": Motor(13, "scs0009", MotorNormMode.RANGE_0_100),
+                # Pinky
+                "pinky_radial_flexor": Motor(14, "scs0009", MotorNormMode.RANGE_0_100),
+                "pinky_ulnar_flexor": Motor(15, "scs0009", MotorNormMode.RANGE_0_100),
+                "pinky_pip_dip": Motor(16, "scs0009", MotorNormMode.RANGE_0_100),
+            },
+            calibration=self.calibration,
+            protocol_version=1,
+        )
+        self.cameras = make_cameras_from_configs(config.cameras)
+        self.inverted_motors = RIGHT_HAND_INVERSIONS if config.side == "right" else LEFT_HAND_INVERSIONS
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.bus.motors}
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected and all(cam.is_connected for cam in self.cameras.values())
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.bus.connect()
+        if not self.is_calibrated and calibrate:
+            self.calibrate()
+
+        # Connect the cameras
+        for cam in self.cameras.values():
+            cam.connect()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        fingers = {}
+        for finger in ["thumb", "index", "middle", "ring", "pinky"]:
+            fingers[finger] = [motor for motor in self.bus.motors if motor.startswith(finger)]
+
+        self.calibration = RangeFinderGUI(self.bus, fingers).run()
+        for motor in self.inverted_motors:
+            self.calibration[motor].drive_mode = 1
+        self._save_calibration()
+        print("Calibration saved to", self.calibration_fpath)
+
+    def configure(self) -> None:
+        with self.bus.torque_disabled():
+            self.bus.configure_motors()
+
+    def setup_motors(self) -> None:
+        # TODO: add docstring
+        for motor in self.bus.motors:
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        obs_dict = {}
+
+        # Read hand position
+        start = time.perf_counter()
+        for motor in self.bus.motors:
+            obs_dict[f"{motor}.pos"] = self.bus.read("Present_Position", motor)
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read state: {dt_ms:.1f}ms")
+
+        # Capture images from cameras
+        for cam_key, cam in self.cameras.items():
+            start = time.perf_counter()
+            obs_dict[cam_key] = cam.read_latest()
+            dt_ms = (time.perf_counter() - start) * 1e3
+            logger.debug(f"{self} read {cam_key}: {dt_ms:.1f}ms")
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        goal_pos = {key.removesuffix(".pos"): val for key, val in action.items() if key.endswith(".pos")}
+        self.bus.sync_write("Goal_Position", goal_pos)
+        return action
+
+    @check_if_not_connected
+    def disconnect(self):
+        self.bus.disconnect(self.config.disable_torque_on_disconnect)
+        for cam in self.cameras.values():
+            cam.disconnect()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/robots/koch_follower/__init__.py b/lerobot/src/lerobot/robots/koch_follower/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..6271c4e55799fd0dacd6c72bc4056598a1bb12cf
--- /dev/null
+++ b/lerobot/src/lerobot/robots/koch_follower/__init__.py
@@ -0,0 +1,18 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_koch_follower import KochFollowerConfig
+from .koch_follower import KochFollower
diff --git a/lerobot/src/lerobot/robots/koch_follower/config_koch_follower.py b/lerobot/src/lerobot/robots/koch_follower/config_koch_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..02a95ef4eac1c6c7063133b77b7907b6a285f5b8
--- /dev/null
+++ b/lerobot/src/lerobot/robots/koch_follower/config_koch_follower.py
@@ -0,0 +1,39 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.cameras import CameraConfig
+
+from ..config import RobotConfig
+
+
+@RobotConfig.register_subclass("koch_follower")
+@dataclass
+class KochFollowerConfig(RobotConfig):
+    # Port to connect to the arm
+    port: str
+
+    disable_torque_on_disconnect: bool = True
+
+    # `max_relative_target` limits the magnitude of the relative positional target vector for safety purposes.
+    # Set this to a positive scalar to have the same value for all motors, or a dictionary that maps motor
+    # names to the max_relative_target value for that motor.
+    max_relative_target: float | dict[str, float] | None = None
+
+    # cameras
+    cameras: dict[str, CameraConfig] = field(default_factory=dict)
+
+    # Set to `True` for backward compatibility with previous policies/dataset
+    use_degrees: bool = False
diff --git a/lerobot/src/lerobot/robots/koch_follower/koch.mdx b/lerobot/src/lerobot/robots/koch_follower/koch.mdx
new file mode 120000
index 0000000000000000000000000000000000000000..ef43feb06680f79223d511915a46e352d33ac480
--- /dev/null
+++ b/lerobot/src/lerobot/robots/koch_follower/koch.mdx
@@ -0,0 +1 @@
+../../../../docs/source/koch.mdx
\ No newline at end of file
diff --git a/lerobot/src/lerobot/robots/koch_follower/koch_follower.py b/lerobot/src/lerobot/robots/koch_follower/koch_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..44e83f6a30c220e32f4fae419374723b79b5195e
--- /dev/null
+++ b/lerobot/src/lerobot/robots/koch_follower/koch_follower.py
@@ -0,0 +1,236 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from functools import cached_property
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.dynamixel import (
+    DynamixelMotorsBus,
+    OperatingMode,
+)
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from ..utils import ensure_safe_goal_position
+from .config_koch_follower import KochFollowerConfig
+
+logger = logging.getLogger(__name__)
+
+
+class KochFollower(Robot):
+    """
+    - [Koch v1.0](https://github.com/AlexanderKoch-Koch/low_cost_robot), with and without the wrist-to-elbow
+        expansion, developed by Alexander Koch from [Tau Robotics](https://tau-robotics.com)
+    - [Koch v1.1](https://github.com/jess-moss/koch-v1-1) developed by Jess Moss
+    """
+
+    config_class = KochFollowerConfig
+    name = "koch_follower"
+
+    def __init__(self, config: KochFollowerConfig):
+        super().__init__(config)
+        self.config = config
+        norm_mode_body = MotorNormMode.DEGREES if config.use_degrees else MotorNormMode.RANGE_M100_100
+        self.bus = DynamixelMotorsBus(
+            port=self.config.port,
+            motors={
+                "shoulder_pan": Motor(1, "xl430-w250", norm_mode_body),
+                "shoulder_lift": Motor(2, "xl430-w250", norm_mode_body),
+                "elbow_flex": Motor(3, "xl330-m288", norm_mode_body),
+                "wrist_flex": Motor(4, "xl330-m288", norm_mode_body),
+                "wrist_roll": Motor(5, "xl330-m288", norm_mode_body),
+                "gripper": Motor(6, "xl330-m288", MotorNormMode.RANGE_0_100),
+            },
+            calibration=self.calibration,
+        )
+        self.cameras = make_cameras_from_configs(config.cameras)
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.bus.motors}
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected and all(cam.is_connected for cam in self.cameras.values())
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        """
+        We assume that at connection time, arm is in a rest position,
+        and torque can be safely disabled to run calibration.
+        """
+
+        self.bus.connect()
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        for cam in self.cameras.values():
+            cam.connect()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        self.bus.disable_torque()
+        if self.calibration:
+            # Calibration file exists, ask user whether to use it or run new calibration
+            user_input = input(
+                f"Press ENTER to use provided calibration file associated with the id {self.id}, or type 'c' and press ENTER to run calibration: "
+            )
+            if user_input.strip().lower() != "c":
+                logger.info(f"Writing calibration file associated with the id {self.id} to the motors")
+                self.bus.write_calibration(self.calibration)
+                return
+        logger.info(f"\nRunning calibration of {self}")
+        for motor in self.bus.motors:
+            self.bus.write("Operating_Mode", motor, OperatingMode.EXTENDED_POSITION.value)
+
+        input(f"Move {self} to the middle of its range of motion and press ENTER....")
+        homing_offsets = self.bus.set_half_turn_homings()
+
+        full_turn_motors = ["shoulder_pan", "wrist_roll"]
+        unknown_range_motors = [motor for motor in self.bus.motors if motor not in full_turn_motors]
+        print(
+            f"Move all joints except {full_turn_motors} sequentially through their entire "
+            "ranges of motion.\nRecording positions. Press ENTER to stop..."
+        )
+        range_mins, range_maxes = self.bus.record_ranges_of_motion(unknown_range_motors)
+        for motor in full_turn_motors:
+            range_mins[motor] = 0
+            range_maxes[motor] = 4095
+
+        self.calibration = {}
+        for motor, m in self.bus.motors.items():
+            self.calibration[motor] = MotorCalibration(
+                id=m.id,
+                drive_mode=0,
+                homing_offset=homing_offsets[motor],
+                range_min=range_mins[motor],
+                range_max=range_maxes[motor],
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        logger.info(f"Calibration saved to {self.calibration_fpath}")
+
+    def configure(self) -> None:
+        with self.bus.torque_disabled():
+            self.bus.configure_motors()
+            # Use 'extended position mode' for all motors except gripper, because in joint mode the servos
+            # can't rotate more than 360 degrees (from 0 to 4095) And some mistake can happen while assembling
+            # the arm, you could end up with a servo with a position 0 or 4095 at a crucial point
+            for motor in self.bus.motors:
+                if motor != "gripper":
+                    self.bus.write("Operating_Mode", motor, OperatingMode.EXTENDED_POSITION.value)
+
+            # Use 'position control current based' for gripper to be limited by the limit of the current. For
+            # the follower gripper, it means it can grasp an object without forcing too much even tho, its
+            # goal position is a complete grasp (both gripper fingers are ordered to join and reach a touch).
+            # For the leader gripper, it means we can use it as a physical trigger, since we can force with
+            # our finger to make it move, and it will move back to its original target position when we
+            # release the force.
+            self.bus.write("Operating_Mode", "gripper", OperatingMode.CURRENT_POSITION.value)
+
+            # Set better PID values to close the gap between recorded states and actions
+            # TODO(rcadene): Implement an automatic procedure to set optimal PID values for each motor
+            self.bus.write("Position_P_Gain", "elbow_flex", 1500)
+            self.bus.write("Position_I_Gain", "elbow_flex", 0)
+            self.bus.write("Position_D_Gain", "elbow_flex", 600)
+
+    def setup_motors(self) -> None:
+        for motor in reversed(self.bus.motors):
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        # Read arm position
+        start = time.perf_counter()
+        obs_dict = self.bus.sync_read("Present_Position")
+        obs_dict = {f"{motor}.pos": val for motor, val in obs_dict.items()}
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read state: {dt_ms:.1f}ms")
+
+        # Capture images from cameras
+        for cam_key, cam in self.cameras.items():
+            start = time.perf_counter()
+            obs_dict[cam_key] = cam.read_latest()
+            dt_ms = (time.perf_counter() - start) * 1e3
+            logger.debug(f"{self} read {cam_key}: {dt_ms:.1f}ms")
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        """Command arm to move to a target joint configuration.
+
+        The relative action magnitude may be clipped depending on the configuration parameter
+        `max_relative_target`. In this case, the action sent differs from original action.
+        Thus, this function always returns the action actually sent.
+
+        Args:
+            action (RobotAction): The goal positions for the motors.
+
+        Returns:
+            RobotAction: The action sent to the motors, potentially clipped.
+        """
+
+        goal_pos = {key.removesuffix(".pos"): val for key, val in action.items() if key.endswith(".pos")}
+
+        # Cap goal position when too far away from present position.
+        # /!\ Slower fps expected due to reading from the follower.
+        if self.config.max_relative_target is not None:
+            present_pos = self.bus.sync_read("Present_Position")
+            goal_present_pos = {key: (g_pos, present_pos[key]) for key, g_pos in goal_pos.items()}
+            goal_pos = ensure_safe_goal_position(goal_present_pos, self.config.max_relative_target)
+
+        # Send goal position to the arm
+        self.bus.sync_write("Goal_Position", goal_pos)
+        return {f"{motor}.pos": val for motor, val in goal_pos.items()}
+
+    @check_if_not_connected
+    def disconnect(self):
+        self.bus.disconnect(self.config.disable_torque_on_disconnect)
+        for cam in self.cameras.values():
+            cam.disconnect()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/robots/lekiwi/__init__.py b/lerobot/src/lerobot/robots/lekiwi/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..ada2ff36846d1865cd3830a64b6a9804c3dbb499
--- /dev/null
+++ b/lerobot/src/lerobot/robots/lekiwi/__init__.py
@@ -0,0 +1,19 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_lekiwi import LeKiwiClientConfig, LeKiwiConfig
+from .lekiwi import LeKiwi
+from .lekiwi_client import LeKiwiClient
diff --git a/lerobot/src/lerobot/robots/lekiwi/config_lekiwi.py b/lerobot/src/lerobot/robots/lekiwi/config_lekiwi.py
new file mode 100644
index 0000000000000000000000000000000000000000..acaf5f0ecbadcf374779bd69220c99a6a4aa162f
--- /dev/null
+++ b/lerobot/src/lerobot/robots/lekiwi/config_lekiwi.py
@@ -0,0 +1,96 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.cameras.configs import CameraConfig, Cv2Rotation
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+
+from ..config import RobotConfig
+
+
+def lekiwi_cameras_config() -> dict[str, CameraConfig]:
+    return {
+        "front": OpenCVCameraConfig(
+            index_or_path="/dev/video0", fps=30, width=640, height=480, rotation=Cv2Rotation.ROTATE_180
+        ),
+        "wrist": OpenCVCameraConfig(
+            index_or_path="/dev/video2", fps=30, width=480, height=640, rotation=Cv2Rotation.ROTATE_90
+        ),
+    }
+
+
+@RobotConfig.register_subclass("lekiwi")
+@dataclass
+class LeKiwiConfig(RobotConfig):
+    port: str = "/dev/ttyACM0"  # port to connect to the bus
+
+    disable_torque_on_disconnect: bool = True
+
+    # `max_relative_target` limits the magnitude of the relative positional target vector for safety purposes.
+    # Set this to a positive scalar to have the same value for all motors, or a dictionary that maps motor
+    # names to the max_relative_target value for that motor.
+    max_relative_target: float | dict[str, float] | None = None
+
+    cameras: dict[str, CameraConfig] = field(default_factory=lekiwi_cameras_config)
+
+    # Set to `True` for backward compatibility with previous policies/dataset
+    use_degrees: bool = False
+
+
+@dataclass
+class LeKiwiHostConfig:
+    # Network Configuration
+    port_zmq_cmd: int = 5555
+    port_zmq_observations: int = 5556
+
+    # Duration of the application
+    connection_time_s: int = 30
+
+    # Watchdog: stop the robot if no command is received for over 0.5 seconds.
+    watchdog_timeout_ms: int = 500
+
+    # If robot jitters decrease the frequency and monitor cpu load with `top` in cmd
+    max_loop_freq_hz: int = 30
+
+
+@RobotConfig.register_subclass("lekiwi_client")
+@dataclass
+class LeKiwiClientConfig(RobotConfig):
+    # Network Configuration
+    remote_ip: str
+    port_zmq_cmd: int = 5555
+    port_zmq_observations: int = 5556
+
+    teleop_keys: dict[str, str] = field(
+        default_factory=lambda: {
+            # Movement
+            "forward": "w",
+            "backward": "s",
+            "left": "a",
+            "right": "d",
+            "rotate_left": "z",
+            "rotate_right": "x",
+            # Speed control
+            "speed_up": "r",
+            "speed_down": "f",
+            # quit teleop
+            "quit": "q",
+        }
+    )
+
+    cameras: dict[str, CameraConfig] = field(default_factory=lekiwi_cameras_config)
+
+    polling_timeout_ms: int = 15
+    connect_timeout_s: int = 5
diff --git a/lerobot/src/lerobot/robots/lekiwi/lekiwi.mdx b/lerobot/src/lerobot/robots/lekiwi/lekiwi.mdx
new file mode 120000
index 0000000000000000000000000000000000000000..f651589989a9ba21671a17f076a54d00f0256a5e
--- /dev/null
+++ b/lerobot/src/lerobot/robots/lekiwi/lekiwi.mdx
@@ -0,0 +1 @@
+../../../../docs/source/lekiwi.mdx
\ No newline at end of file
diff --git a/lerobot/src/lerobot/robots/lekiwi/lekiwi.py b/lerobot/src/lerobot/robots/lekiwi/lekiwi.py
new file mode 100644
index 0000000000000000000000000000000000000000..60fac89e56b31b631e9f7a7276fedf236dd18623
--- /dev/null
+++ b/lerobot/src/lerobot/robots/lekiwi/lekiwi.py
@@ -0,0 +1,417 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from functools import cached_property
+from itertools import chain
+from typing import Any
+
+import numpy as np
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.feetech import (
+    FeetechMotorsBus,
+    OperatingMode,
+)
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from ..utils import ensure_safe_goal_position
+from .config_lekiwi import LeKiwiConfig
+
+logger = logging.getLogger(__name__)
+
+
+class LeKiwi(Robot):
+    """
+    The robot includes a three omniwheel mobile base and a remote follower arm.
+    The leader arm is connected locally (on the laptop) and its joint positions are recorded and then
+    forwarded to the remote follower arm (after applying a safety clamp).
+    In parallel, keyboard teleoperation is used to generate raw velocity commands for the wheels.
+    """
+
+    config_class = LeKiwiConfig
+    name = "lekiwi"
+
+    def __init__(self, config: LeKiwiConfig):
+        super().__init__(config)
+        self.config = config
+        norm_mode_body = MotorNormMode.DEGREES if config.use_degrees else MotorNormMode.RANGE_M100_100
+        self.bus = FeetechMotorsBus(
+            port=self.config.port,
+            motors={
+                # arm
+                "arm_shoulder_pan": Motor(1, "sts3215", norm_mode_body),
+                "arm_shoulder_lift": Motor(2, "sts3215", norm_mode_body),
+                "arm_elbow_flex": Motor(3, "sts3215", norm_mode_body),
+                "arm_wrist_flex": Motor(4, "sts3215", norm_mode_body),
+                "arm_wrist_roll": Motor(5, "sts3215", norm_mode_body),
+                "arm_gripper": Motor(6, "sts3215", MotorNormMode.RANGE_0_100),
+                # base
+                "base_left_wheel": Motor(7, "sts3215", MotorNormMode.RANGE_M100_100),
+                "base_back_wheel": Motor(8, "sts3215", MotorNormMode.RANGE_M100_100),
+                "base_right_wheel": Motor(9, "sts3215", MotorNormMode.RANGE_M100_100),
+            },
+            calibration=self.calibration,
+        )
+        self.arm_motors = [motor for motor in self.bus.motors if motor.startswith("arm")]
+        self.base_motors = [motor for motor in self.bus.motors if motor.startswith("base")]
+        self.cameras = make_cameras_from_configs(config.cameras)
+
+    @property
+    def _state_ft(self) -> dict[str, type]:
+        return dict.fromkeys(
+            (
+                "arm_shoulder_pan.pos",
+                "arm_shoulder_lift.pos",
+                "arm_elbow_flex.pos",
+                "arm_wrist_flex.pos",
+                "arm_wrist_roll.pos",
+                "arm_gripper.pos",
+                "x.vel",
+                "y.vel",
+                "theta.vel",
+            ),
+            float,
+        )
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._state_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._state_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected and all(cam.is_connected for cam in self.cameras.values())
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.bus.connect()
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        for cam in self.cameras.values():
+            cam.connect()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        if self.calibration:
+            # Calibration file exists, ask user whether to use it or run new calibration
+            user_input = input(
+                f"Press ENTER to use provided calibration file associated with the id {self.id}, or type 'c' and press ENTER to run calibration: "
+            )
+            if user_input.strip().lower() != "c":
+                logger.info(f"Writing calibration file associated with the id {self.id} to the motors")
+                self.bus.write_calibration(self.calibration)
+                return
+        logger.info(f"\nRunning calibration of {self}")
+
+        motors = self.arm_motors + self.base_motors
+
+        self.bus.disable_torque(self.arm_motors)
+        for name in self.arm_motors:
+            self.bus.write("Operating_Mode", name, OperatingMode.POSITION.value)
+
+        input("Move robot to the middle of its range of motion and press ENTER....")
+        homing_offsets = self.bus.set_half_turn_homings(self.arm_motors)
+
+        homing_offsets.update(dict.fromkeys(self.base_motors, 0))
+
+        full_turn_motor = [
+            motor for motor in motors if any(keyword in motor for keyword in ["wheel", "wrist_roll"])
+        ]
+        unknown_range_motors = [motor for motor in motors if motor not in full_turn_motor]
+
+        print(
+            f"Move all arm joints except '{full_turn_motor}' sequentially through their "
+            "entire ranges of motion.\nRecording positions. Press ENTER to stop..."
+        )
+        range_mins, range_maxes = self.bus.record_ranges_of_motion(unknown_range_motors)
+        for name in full_turn_motor:
+            range_mins[name] = 0
+            range_maxes[name] = 4095
+
+        self.calibration = {}
+        for name, motor in self.bus.motors.items():
+            self.calibration[name] = MotorCalibration(
+                id=motor.id,
+                drive_mode=0,
+                homing_offset=homing_offsets[name],
+                range_min=range_mins[name],
+                range_max=range_maxes[name],
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        print("Calibration saved to", self.calibration_fpath)
+
+    def configure(self):
+        # Set-up arm actuators (position mode)
+        # We assume that at connection time, arm is in a rest position,
+        # and torque can be safely disabled to run calibration.
+        self.bus.disable_torque()
+        self.bus.configure_motors()
+        for name in self.arm_motors:
+            self.bus.write("Operating_Mode", name, OperatingMode.POSITION.value)
+            # Set P_Coefficient to lower value to avoid shakiness (Default is 32)
+            self.bus.write("P_Coefficient", name, 16)
+            # Set I_Coefficient and D_Coefficient to default value 0 and 32
+            self.bus.write("I_Coefficient", name, 0)
+            self.bus.write("D_Coefficient", name, 32)
+
+        for name in self.base_motors:
+            self.bus.write("Operating_Mode", name, OperatingMode.VELOCITY.value)
+
+        self.bus.enable_torque()
+
+    def setup_motors(self) -> None:
+        for motor in chain(reversed(self.arm_motors), reversed(self.base_motors)):
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @staticmethod
+    def _degps_to_raw(degps: float) -> int:
+        steps_per_deg = 4096.0 / 360.0
+        speed_in_steps = degps * steps_per_deg
+        speed_int = int(round(speed_in_steps))
+        # Cap the value to fit within signed 16-bit range (-32768 to 32767)
+        if speed_int > 0x7FFF:
+            speed_int = 0x7FFF  # 32767 -> maximum positive value
+        elif speed_int < -0x8000:
+            speed_int = -0x8000  # -32768 -> minimum negative value
+        return speed_int
+
+    @staticmethod
+    def _raw_to_degps(raw_speed: int) -> float:
+        steps_per_deg = 4096.0 / 360.0
+        magnitude = raw_speed
+        degps = magnitude / steps_per_deg
+        return degps
+
+    def _body_to_wheel_raw(
+        self,
+        x: float,
+        y: float,
+        theta: float,
+        wheel_radius: float = 0.05,
+        base_radius: float = 0.125,
+        max_raw: int = 3000,
+    ) -> dict:
+        """
+        Convert desired body-frame velocities into wheel raw commands.
+
+        Parameters:
+          x_cmd      : Linear velocity in x (m/s).
+          y_cmd      : Linear velocity in y (m/s).
+          theta_cmd  : Rotational velocity (deg/s).
+          wheel_radius: Radius of each wheel (meters).
+          base_radius : Distance from the center of rotation to each wheel (meters).
+          max_raw    : Maximum allowed raw command (ticks) per wheel.
+
+        Returns:
+          A dictionary with wheel raw commands:
+             {"base_left_wheel": value, "base_back_wheel": value, "base_right_wheel": value}.
+
+        Notes:
+          - Internally, the method converts theta_cmd to rad/s for the kinematics.
+          - The raw command is computed from the wheels angular speed in deg/s
+            using _degps_to_raw(). If any command exceeds max_raw, all commands
+            are scaled down proportionally.
+        """
+        # Convert rotational velocity from deg/s to rad/s.
+        theta_rad = theta * (np.pi / 180.0)
+        # Create the body velocity vector [x, y, theta_rad].
+        velocity_vector = np.array([x, y, theta_rad])
+
+        # Define the wheel mounting angles with a -90° offset.
+        angles = np.radians(np.array([240, 0, 120]) - 90)
+        # Build the kinematic matrix: each row maps body velocities to a wheel’s linear speed.
+        # The third column (base_radius) accounts for the effect of rotation.
+        m = np.array([[np.cos(a), np.sin(a), base_radius] for a in angles])
+
+        # Compute each wheel’s linear speed (m/s) and then its angular speed (rad/s).
+        wheel_linear_speeds = m.dot(velocity_vector)
+        wheel_angular_speeds = wheel_linear_speeds / wheel_radius
+
+        # Convert wheel angular speeds from rad/s to deg/s.
+        wheel_degps = wheel_angular_speeds * (180.0 / np.pi)
+
+        # Scaling
+        steps_per_deg = 4096.0 / 360.0
+        raw_floats = [abs(degps) * steps_per_deg for degps in wheel_degps]
+        max_raw_computed = max(raw_floats)
+        if max_raw_computed > max_raw:
+            scale = max_raw / max_raw_computed
+            wheel_degps = wheel_degps * scale
+
+        # Convert each wheel’s angular speed (deg/s) to a raw integer.
+        wheel_raw = [self._degps_to_raw(deg) for deg in wheel_degps]
+
+        return {
+            "base_left_wheel": wheel_raw[0],
+            "base_back_wheel": wheel_raw[1],
+            "base_right_wheel": wheel_raw[2],
+        }
+
+    def _wheel_raw_to_body(
+        self,
+        left_wheel_speed,
+        back_wheel_speed,
+        right_wheel_speed,
+        wheel_radius: float = 0.05,
+        base_radius: float = 0.125,
+    ) -> dict[str, Any]:
+        """
+        Convert wheel raw command feedback back into body-frame velocities.
+
+        Parameters:
+          wheel_raw   : Vector with raw wheel commands ("base_left_wheel", "base_back_wheel", "base_right_wheel").
+          wheel_radius: Radius of each wheel (meters).
+          base_radius : Distance from the robot center to each wheel (meters).
+
+        Returns:
+          A dict (x.vel, y.vel, theta.vel) all in m/s
+        """
+
+        # Convert each raw command back to an angular speed in deg/s.
+        wheel_degps = np.array(
+            [
+                self._raw_to_degps(left_wheel_speed),
+                self._raw_to_degps(back_wheel_speed),
+                self._raw_to_degps(right_wheel_speed),
+            ]
+        )
+
+        # Convert from deg/s to rad/s.
+        wheel_radps = wheel_degps * (np.pi / 180.0)
+        # Compute each wheel’s linear speed (m/s) from its angular speed.
+        wheel_linear_speeds = wheel_radps * wheel_radius
+
+        # Define the wheel mounting angles with a -90° offset.
+        angles = np.radians(np.array([240, 0, 120]) - 90)
+        m = np.array([[np.cos(a), np.sin(a), base_radius] for a in angles])
+
+        # Solve the inverse kinematics: body_velocity = M⁻¹ · wheel_linear_speeds.
+        m_inv = np.linalg.inv(m)
+        velocity_vector = m_inv.dot(wheel_linear_speeds)
+        x, y, theta_rad = velocity_vector
+        theta = theta_rad * (180.0 / np.pi)
+        return {
+            "x.vel": x,
+            "y.vel": y,
+            "theta.vel": theta,
+        }  # m/s and deg/s
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        # Read actuators position for arm and vel for base
+        start = time.perf_counter()
+        arm_pos = self.bus.sync_read("Present_Position", self.arm_motors)
+        base_wheel_vel = self.bus.sync_read("Present_Velocity", self.base_motors)
+
+        base_vel = self._wheel_raw_to_body(
+            base_wheel_vel["base_left_wheel"],
+            base_wheel_vel["base_back_wheel"],
+            base_wheel_vel["base_right_wheel"],
+        )
+
+        arm_state = {f"{k}.pos": v for k, v in arm_pos.items()}
+
+        obs_dict = {**arm_state, **base_vel}
+
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read state: {dt_ms:.1f}ms")
+
+        # Capture images from cameras
+        for cam_key, cam in self.cameras.items():
+            start = time.perf_counter()
+            obs_dict[cam_key] = cam.read_latest()
+            dt_ms = (time.perf_counter() - start) * 1e3
+            logger.debug(f"{self} read {cam_key}: {dt_ms:.1f}ms")
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        """Command lekiwi to move to a target joint configuration.
+
+        The relative action magnitude may be clipped depending on the configuration parameter
+        `max_relative_target`. In this case, the action sent differs from original action.
+        Thus, this function always returns the action actually sent.
+
+        Raises:
+            RobotDeviceNotConnectedError: if robot is not connected.
+
+        Returns:
+            RobotAction: the action sent to the motors, potentially clipped.
+        """
+
+        arm_goal_pos = {k: v for k, v in action.items() if k.endswith(".pos")}
+        base_goal_vel = {k: v for k, v in action.items() if k.endswith(".vel")}
+
+        base_wheel_goal_vel = self._body_to_wheel_raw(
+            base_goal_vel["x.vel"], base_goal_vel["y.vel"], base_goal_vel["theta.vel"]
+        )
+
+        # Cap goal position when too far away from present position.
+        # /!\ Slower fps expected due to reading from the follower.
+        if self.config.max_relative_target is not None:
+            present_pos = self.bus.sync_read("Present_Position", self.arm_motors)
+            goal_present_pos = {key: (g_pos, present_pos[key]) for key, g_pos in arm_goal_pos.items()}
+            arm_safe_goal_pos = ensure_safe_goal_position(goal_present_pos, self.config.max_relative_target)
+            arm_goal_pos = arm_safe_goal_pos
+
+        # Send goal position to the actuators
+        arm_goal_pos_raw = {k.replace(".pos", ""): v for k, v in arm_goal_pos.items()}
+        self.bus.sync_write("Goal_Position", arm_goal_pos_raw)
+        self.bus.sync_write("Goal_Velocity", base_wheel_goal_vel)
+
+        return {**arm_goal_pos, **base_goal_vel}
+
+    def stop_base(self):
+        self.bus.sync_write("Goal_Velocity", dict.fromkeys(self.base_motors, 0), num_retry=5)
+        logger.info("Base motors stopped")
+
+    @check_if_not_connected
+    def disconnect(self):
+        self.stop_base()
+        self.bus.disconnect(self.config.disable_torque_on_disconnect)
+        for cam in self.cameras.values():
+            cam.disconnect()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/robots/lekiwi/lekiwi_client.py b/lerobot/src/lerobot/robots/lekiwi/lekiwi_client.py
new file mode 100644
index 0000000000000000000000000000000000000000..fd43e84fed11729c4309b1897aac6dc4ed8f19b9
--- /dev/null
+++ b/lerobot/src/lerobot/robots/lekiwi/lekiwi_client.py
@@ -0,0 +1,335 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# TODO(aliberts, Steven, Pepijn): use gRPC calls instead of zmq?
+
+import base64
+import json
+import logging
+from functools import cached_property
+
+import cv2
+import numpy as np
+
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.constants import ACTION, OBS_STATE
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.errors import DeviceNotConnectedError
+
+from ..robot import Robot
+from .config_lekiwi import LeKiwiClientConfig
+
+
+class LeKiwiClient(Robot):
+    config_class = LeKiwiClientConfig
+    name = "lekiwi_client"
+
+    def __init__(self, config: LeKiwiClientConfig):
+        import zmq
+
+        self._zmq = zmq
+        super().__init__(config)
+        self.config = config
+        self.id = config.id
+        self.robot_type = config.type
+
+        self.remote_ip = config.remote_ip
+        self.port_zmq_cmd = config.port_zmq_cmd
+        self.port_zmq_observations = config.port_zmq_observations
+
+        self.teleop_keys = config.teleop_keys
+
+        self.polling_timeout_ms = config.polling_timeout_ms
+        self.connect_timeout_s = config.connect_timeout_s
+
+        self.zmq_context = None
+        self.zmq_cmd_socket = None
+        self.zmq_observation_socket = None
+
+        self.last_frames = {}
+
+        self.last_remote_state = {}
+
+        # Define three speed levels and a current index
+        self.speed_levels = [
+            {"xy": 0.1, "theta": 30},  # slow
+            {"xy": 0.2, "theta": 60},  # medium
+            {"xy": 0.3, "theta": 90},  # fast
+        ]
+        self.speed_index = 0  # Start at slow
+
+        self._is_connected = False
+        self.logs = {}
+
+    @cached_property
+    def _state_ft(self) -> dict[str, type]:
+        return dict.fromkeys(
+            (
+                "arm_shoulder_pan.pos",
+                "arm_shoulder_lift.pos",
+                "arm_elbow_flex.pos",
+                "arm_wrist_flex.pos",
+                "arm_wrist_roll.pos",
+                "arm_gripper.pos",
+                "x.vel",
+                "y.vel",
+                "theta.vel",
+            ),
+            float,
+        )
+
+    @cached_property
+    def _state_order(self) -> tuple[str, ...]:
+        return tuple(self._state_ft.keys())
+
+    @cached_property
+    def _cameras_ft(self) -> dict[str, tuple[int, int, int]]:
+        return {name: (cfg.height, cfg.width, 3) for name, cfg in self.config.cameras.items()}
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._state_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._state_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self._is_connected
+
+    @property
+    def is_calibrated(self) -> bool:
+        pass
+
+    @check_if_already_connected
+    def connect(self) -> None:
+        """Establishes ZMQ sockets with the remote mobile robot"""
+
+        zmq = self._zmq
+        self.zmq_context = zmq.Context()
+        self.zmq_cmd_socket = self.zmq_context.socket(zmq.PUSH)
+        zmq_cmd_locator = f"tcp://{self.remote_ip}:{self.port_zmq_cmd}"
+        self.zmq_cmd_socket.connect(zmq_cmd_locator)
+        self.zmq_cmd_socket.setsockopt(zmq.CONFLATE, 1)
+
+        self.zmq_observation_socket = self.zmq_context.socket(zmq.PULL)
+        zmq_observations_locator = f"tcp://{self.remote_ip}:{self.port_zmq_observations}"
+        self.zmq_observation_socket.connect(zmq_observations_locator)
+        self.zmq_observation_socket.setsockopt(zmq.CONFLATE, 1)
+
+        poller = zmq.Poller()
+        poller.register(self.zmq_observation_socket, zmq.POLLIN)
+        socks = dict(poller.poll(self.connect_timeout_s * 1000))
+        if self.zmq_observation_socket not in socks or socks[self.zmq_observation_socket] != zmq.POLLIN:
+            raise DeviceNotConnectedError("Timeout waiting for LeKiwi Host to connect expired.")
+
+        self._is_connected = True
+
+    def calibrate(self) -> None:
+        pass
+
+    def _poll_and_get_latest_message(self) -> str | None:
+        """Polls the ZMQ socket for a limited time and returns the latest message string."""
+        zmq = self._zmq
+        poller = zmq.Poller()
+        poller.register(self.zmq_observation_socket, zmq.POLLIN)
+
+        try:
+            socks = dict(poller.poll(self.polling_timeout_ms))
+        except zmq.ZMQError as e:
+            logging.error(f"ZMQ polling error: {e}")
+            return None
+
+        if self.zmq_observation_socket not in socks:
+            logging.info("No new data available within timeout.")
+            return None
+
+        last_msg = None
+        while True:
+            try:
+                msg = self.zmq_observation_socket.recv_string(zmq.NOBLOCK)
+                last_msg = msg
+            except zmq.Again:
+                break
+
+        if last_msg is None:
+            logging.warning("Poller indicated data, but failed to retrieve message.")
+
+        return last_msg
+
+    def _parse_observation_json(self, obs_string: str) -> RobotObservation | None:
+        """Parses the JSON observation string."""
+        try:
+            return json.loads(obs_string)
+        except json.JSONDecodeError as e:
+            logging.error(f"Error decoding JSON observation: {e}")
+            return None
+
+    def _decode_image_from_b64(self, image_b64: str) -> np.ndarray | None:
+        """Decodes a base64 encoded image string to an OpenCV image."""
+        if not image_b64:
+            return None
+        try:
+            jpg_data = base64.b64decode(image_b64)
+            np_arr = np.frombuffer(jpg_data, dtype=np.uint8)
+            frame = cv2.imdecode(np_arr, cv2.IMREAD_COLOR)
+            if frame is None:
+                logging.warning("cv2.imdecode returned None for an image.")
+            return frame
+        except (TypeError, ValueError) as e:
+            logging.error(f"Error decoding base64 image data: {e}")
+            return None
+
+    def _remote_state_from_obs(
+        self, observation: RobotObservation
+    ) -> tuple[dict[str, np.ndarray], RobotObservation]:
+        """Extracts frames, and state from the parsed observation."""
+
+        flat_state = {key: observation.get(key, 0.0) for key in self._state_order}
+
+        state_vec = np.array([flat_state[key] for key in self._state_order], dtype=np.float32)
+
+        obs_dict: RobotObservation = {**flat_state, OBS_STATE: state_vec}
+
+        # Decode images
+        current_frames: dict[str, np.ndarray] = {}
+        for cam_name, image_b64 in observation.items():
+            if cam_name not in self._cameras_ft:
+                continue
+            frame = self._decode_image_from_b64(image_b64)
+            if frame is not None:
+                current_frames[cam_name] = frame
+
+        return current_frames, obs_dict
+
+    def _get_data(self) -> tuple[dict[str, np.ndarray], RobotObservation]:
+        """
+        Polls the video socket for the latest observation data.
+
+        Attempts to retrieve and decode the latest message within a short timeout.
+        If successful, updates and returns the new frames, speed, and arm state.
+        If no new data arrives or decoding fails, returns the last known values.
+        """
+
+        # 1. Get the latest message string from the socket
+        latest_message_str = self._poll_and_get_latest_message()
+
+        # 2. If no message, return cached data
+        if latest_message_str is None:
+            return self.last_frames, self.last_remote_state
+
+        # 3. Parse the JSON message
+        observation = self._parse_observation_json(latest_message_str)
+
+        # 4. If JSON parsing failed, return cached data
+        if observation is None:
+            return self.last_frames, self.last_remote_state
+
+        # 5. Process the valid observation data
+        try:
+            new_frames, new_state = self._remote_state_from_obs(observation)
+        except Exception as e:
+            logging.error(f"Error processing observation data, serving last observation: {e}")
+            return self.last_frames, self.last_remote_state
+
+        self.last_frames = new_frames
+        self.last_remote_state = new_state
+
+        return new_frames, new_state
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        """
+        Capture observations from the remote robot: current follower arm positions,
+        present wheel speeds (converted to body-frame velocities: x, y, theta),
+        and a camera frame. Receives over ZMQ, translate to body-frame vel
+        """
+
+        frames, obs_dict = self._get_data()
+
+        # Loop over each configured camera
+        for cam_name, frame in frames.items():
+            if frame is None:
+                logging.warning("Frame is None")
+                frame = np.zeros((640, 480, 3), dtype=np.uint8)
+            obs_dict[cam_name] = frame
+
+        return obs_dict
+
+    def _from_keyboard_to_base_action(self, pressed_keys: np.ndarray):
+        # Speed control
+        if self.teleop_keys["speed_up"] in pressed_keys:
+            self.speed_index = min(self.speed_index + 1, 2)
+        if self.teleop_keys["speed_down"] in pressed_keys:
+            self.speed_index = max(self.speed_index - 1, 0)
+        speed_setting = self.speed_levels[self.speed_index]
+        xy_speed = speed_setting["xy"]  # e.g. 0.1, 0.25, or 0.4
+        theta_speed = speed_setting["theta"]  # e.g. 30, 60, or 90
+
+        x_cmd = 0.0  # m/s forward/backward
+        y_cmd = 0.0  # m/s lateral
+        theta_cmd = 0.0  # deg/s rotation
+
+        if self.teleop_keys["forward"] in pressed_keys:
+            x_cmd += xy_speed
+        if self.teleop_keys["backward"] in pressed_keys:
+            x_cmd -= xy_speed
+        if self.teleop_keys["left"] in pressed_keys:
+            y_cmd += xy_speed
+        if self.teleop_keys["right"] in pressed_keys:
+            y_cmd -= xy_speed
+        if self.teleop_keys["rotate_left"] in pressed_keys:
+            theta_cmd += theta_speed
+        if self.teleop_keys["rotate_right"] in pressed_keys:
+            theta_cmd -= theta_speed
+        return {
+            "x.vel": x_cmd,
+            "y.vel": y_cmd,
+            "theta.vel": theta_cmd,
+        }
+
+    def configure(self):
+        pass
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        """Command lekiwi to move to a target joint configuration. Translates to motor space + sends over ZMQ
+
+        Args:
+            action (RobotAction): array containing the goal positions for the motors.
+        Raises:
+            RobotDeviceNotConnectedError: if robot is not connected.
+
+        Returns:
+            np.ndarray: the action sent to the motors, potentially clipped.
+        """
+
+        self.zmq_cmd_socket.send_string(json.dumps(action))  # action is in motor space
+
+        # TODO(Steven): Remove the np conversion when it is possible to record a non-numpy array value
+        actions = np.array([action.get(k, 0.0) for k in self._state_order], dtype=np.float32)
+
+        action_sent = {key: actions[i] for i, key in enumerate(self._state_order)}
+        action_sent[ACTION] = actions
+        return action_sent
+
+    @check_if_not_connected
+    def disconnect(self):
+        """Cleans ZMQ comms"""
+
+        self.zmq_observation_socket.close()
+        self.zmq_cmd_socket.close()
+        self.zmq_context.term()
+        self._is_connected = False
diff --git a/lerobot/src/lerobot/robots/lekiwi/lekiwi_host.py b/lerobot/src/lerobot/robots/lekiwi/lekiwi_host.py
new file mode 100644
index 0000000000000000000000000000000000000000..ae7ac58c45019f1945b717d9125f7dbd676a33c7
--- /dev/null
+++ b/lerobot/src/lerobot/robots/lekiwi/lekiwi_host.py
@@ -0,0 +1,136 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import base64
+import json
+import logging
+import time
+from dataclasses import dataclass, field
+
+import cv2
+import draccus
+import zmq
+
+from .config_lekiwi import LeKiwiConfig, LeKiwiHostConfig
+from .lekiwi import LeKiwi
+
+
+@dataclass
+class LeKiwiServerConfig:
+    """Configuration for the LeKiwi host script."""
+
+    robot: LeKiwiConfig = field(default_factory=LeKiwiConfig)
+    host: LeKiwiHostConfig = field(default_factory=LeKiwiHostConfig)
+
+
+class LeKiwiHost:
+    def __init__(self, config: LeKiwiHostConfig):
+        self.zmq_context = zmq.Context()
+        self.zmq_cmd_socket = self.zmq_context.socket(zmq.PULL)
+        self.zmq_cmd_socket.setsockopt(zmq.CONFLATE, 1)
+        self.zmq_cmd_socket.bind(f"tcp://*:{config.port_zmq_cmd}")
+
+        self.zmq_observation_socket = self.zmq_context.socket(zmq.PUSH)
+        self.zmq_observation_socket.setsockopt(zmq.CONFLATE, 1)
+        self.zmq_observation_socket.bind(f"tcp://*:{config.port_zmq_observations}")
+
+        self.connection_time_s = config.connection_time_s
+        self.watchdog_timeout_ms = config.watchdog_timeout_ms
+        self.max_loop_freq_hz = config.max_loop_freq_hz
+
+    def disconnect(self):
+        self.zmq_observation_socket.close()
+        self.zmq_cmd_socket.close()
+        self.zmq_context.term()
+
+
+@draccus.wrap()
+def main(cfg: LeKiwiServerConfig):
+    logging.info("Configuring LeKiwi")
+    robot = LeKiwi(cfg.robot)
+
+    logging.info("Connecting LeKiwi")
+    robot.connect()
+
+    logging.info("Starting HostAgent")
+    host = LeKiwiHost(cfg.host)
+
+    last_cmd_time = time.time()
+    watchdog_active = False
+    logging.info("Waiting for commands...")
+    try:
+        # Business logic
+        start = time.perf_counter()
+        duration = 0
+        while duration < host.connection_time_s:
+            loop_start_time = time.time()
+            try:
+                msg = host.zmq_cmd_socket.recv_string(zmq.NOBLOCK)
+                data = dict(json.loads(msg))
+                _action_sent = robot.send_action(data)
+                last_cmd_time = time.time()
+                watchdog_active = False
+            except zmq.Again:
+                if not watchdog_active:
+                    logging.warning("No command available")
+            except Exception as e:
+                logging.error("Message fetching failed: %s", e)
+
+            now = time.time()
+            if (now - last_cmd_time > host.watchdog_timeout_ms / 1000) and not watchdog_active:
+                logging.warning(
+                    f"Command not received for more than {host.watchdog_timeout_ms} milliseconds. Stopping the base."
+                )
+                watchdog_active = True
+                robot.stop_base()
+
+            last_observation = robot.get_observation()
+
+            # Encode ndarrays to base64 strings
+            for cam_key, _ in robot.cameras.items():
+                ret, buffer = cv2.imencode(
+                    ".jpg", last_observation[cam_key], [int(cv2.IMWRITE_JPEG_QUALITY), 90]
+                )
+                if ret:
+                    last_observation[cam_key] = base64.b64encode(buffer).decode("utf-8")
+                else:
+                    last_observation[cam_key] = ""
+
+            # Send the observation to the remote agent
+            try:
+                host.zmq_observation_socket.send_string(json.dumps(last_observation), flags=zmq.NOBLOCK)
+            except zmq.Again:
+                logging.info("Dropping observation, no client connected")
+
+            # Ensure a short sleep to avoid overloading the CPU.
+            elapsed = time.time() - loop_start_time
+
+            time.sleep(max(1 / host.max_loop_freq_hz - elapsed, 0))
+            duration = time.perf_counter() - start
+        print("Cycle time reached.")
+
+    except KeyboardInterrupt:
+        print("Keyboard interrupt received. Exiting...")
+    finally:
+        print("Shutting down Lekiwi Host.")
+        robot.disconnect()
+        host.disconnect()
+
+    logging.info("Finished LeKiwi cleanly")
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/robots/omx_follower/__init__.py b/lerobot/src/lerobot/robots/omx_follower/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..db48dffe99a05c1572562fc99d1e4146af68e2d8
--- /dev/null
+++ b/lerobot/src/lerobot/robots/omx_follower/__init__.py
@@ -0,0 +1,21 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# OMX is a fully open-source robot from ROBOTIS.
+# More information at: https://ai.robotis.com/omx/introduction_omx.html
+
+from .config_omx_follower import OmxFollowerConfig
+from .omx_follower import OmxFollower
diff --git a/lerobot/src/lerobot/robots/omx_follower/config_omx_follower.py b/lerobot/src/lerobot/robots/omx_follower/config_omx_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..db4179fdf0197ea5485bcbd64e56ec8619335956
--- /dev/null
+++ b/lerobot/src/lerobot/robots/omx_follower/config_omx_follower.py
@@ -0,0 +1,39 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.cameras import CameraConfig
+
+from ..config import RobotConfig
+
+
+@RobotConfig.register_subclass("omx_follower")
+@dataclass
+class OmxFollowerConfig(RobotConfig):
+    # Port to connect to the arm
+    port: str
+
+    disable_torque_on_disconnect: bool = True
+
+    # `max_relative_target` limits the magnitude of the relative positional target vector for safety purposes.
+    # Set this to a positive scalar to have the same value for all motors, or a dictionary that maps motor
+    # names to the max_relative_target value for that motor.
+    max_relative_target: float | dict[str, float] | None = None
+
+    # cameras
+    cameras: dict[str, CameraConfig] = field(default_factory=dict)
+
+    # Set to `True` for backward compatibility with previous policies/dataset
+    use_degrees: bool = False
diff --git a/lerobot/src/lerobot/robots/omx_follower/omx_follower.py b/lerobot/src/lerobot/robots/omx_follower/omx_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..5d161daa270c4bd6161e7ac6229b5309c3a4c7d6
--- /dev/null
+++ b/lerobot/src/lerobot/robots/omx_follower/omx_follower.py
@@ -0,0 +1,219 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from functools import cached_property
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.dynamixel import (
+    DriveMode,
+    DynamixelMotorsBus,
+    OperatingMode,
+)
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from ..utils import ensure_safe_goal_position
+from .config_omx_follower import OmxFollowerConfig
+
+logger = logging.getLogger(__name__)
+
+
+class OmxFollower(Robot):
+    """
+    - [OMX](https://github.com/ROBOTIS-GIT/open_manipulator),
+        expansion, developed by Woojin Wie and Junha Cha from [ROBOTIS](https://ai.robotis.com/)
+    """
+
+    config_class = OmxFollowerConfig
+    name = "omx_follower"
+
+    def __init__(self, config: OmxFollowerConfig):
+        super().__init__(config)
+        self.config = config
+        norm_mode_body = MotorNormMode.DEGREES if config.use_degrees else MotorNormMode.RANGE_M100_100
+        self.bus = DynamixelMotorsBus(
+            port=self.config.port,
+            motors={
+                "shoulder_pan": Motor(11, "xl430-w250", norm_mode_body),
+                "shoulder_lift": Motor(12, "xl430-w250", norm_mode_body),
+                "elbow_flex": Motor(13, "xl430-w250", norm_mode_body),
+                "wrist_flex": Motor(14, "xl330-m288", norm_mode_body),
+                "wrist_roll": Motor(15, "xl330-m288", norm_mode_body),
+                "gripper": Motor(16, "xl330-m288", MotorNormMode.RANGE_0_100),
+            },
+            calibration=self.calibration,
+        )
+        self.cameras = make_cameras_from_configs(config.cameras)
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.bus.motors}
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected and all(cam.is_connected for cam in self.cameras.values())
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        """
+        For OMX robots that come pre-calibrated:
+        - If default calibration from package doesn't match motors, read from motors and save
+        - This allows using pre-calibrated robots without manual calibration
+        - If no calibration file exists, use factory default values (homing_offset=0, range_min=0, range_max=4095)
+        """
+
+        self.bus.connect()
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        for cam in self.cameras.values():
+            cam.connect()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        self.bus.disable_torque()
+        logger.info(f"\nUsing factory default calibration values for {self}")
+        logger.info(f"\nWriting default configuration of {self} to the motors")
+        for motor in self.bus.motors:
+            self.bus.write("Operating_Mode", motor, OperatingMode.EXTENDED_POSITION.value)
+
+        for motor in self.bus.motors:
+            self.bus.write("Drive_Mode", motor, DriveMode.NON_INVERTED.value)
+
+        self.calibration = {}
+        for motor, m in self.bus.motors.items():
+            self.calibration[motor] = MotorCalibration(
+                id=m.id,
+                drive_mode=0,
+                homing_offset=0,
+                range_min=0,
+                range_max=4095,
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        logger.info(f"Calibration saved to {self.calibration_fpath}")
+
+    def configure(self) -> None:
+        with self.bus.torque_disabled():
+            self.bus.configure_motors()
+            # Use 'extended position mode' for all motors except gripper, because in joint mode the servos
+            # can't rotate more than 360 degrees (from 0 to 4095) And some mistake can happen while assembling
+            # the arm, you could end up with a servo with a position 0 or 4095 at a crucial point
+            for motor in self.bus.motors:
+                if motor != "gripper":
+                    self.bus.write("Operating_Mode", motor, OperatingMode.EXTENDED_POSITION.value)
+
+            # Use 'position control current based' for gripper to be limited by the limit of the current. For
+            # the follower gripper, it means it can grasp an object without forcing too much even tho, its
+            # goal position is a complete grasp (both gripper fingers are ordered to join and reach a touch).
+            # For the leader gripper, it means we can use it as a physical trigger, since we can force with
+            # our finger to make it move, and it will move back to its original target position when we
+            # release the force.
+            self.bus.write("Operating_Mode", "gripper", OperatingMode.CURRENT_POSITION.value)
+
+            # Set better PID values to close the gap between recorded states and actions
+            # TODO(rcadene): Implement an automatic procedure to set optimal PID values for each motor
+            self.bus.write("Position_P_Gain", "elbow_flex", 1500)
+            self.bus.write("Position_I_Gain", "elbow_flex", 0)
+            self.bus.write("Position_D_Gain", "elbow_flex", 600)
+
+    def setup_motors(self) -> None:
+        for motor in reversed(self.bus.motors):
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        # Read arm position
+        start = time.perf_counter()
+        obs_dict = self.bus.sync_read("Present_Position")
+        obs_dict = {f"{motor}.pos": val for motor, val in obs_dict.items()}
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read state: {dt_ms:.1f}ms")
+
+        # Capture images from cameras
+        for cam_key, cam in self.cameras.items():
+            start = time.perf_counter()
+            obs_dict[cam_key] = cam.read_latest()
+            dt_ms = (time.perf_counter() - start) * 1e3
+            logger.debug(f"{self} read {cam_key}: {dt_ms:.1f}ms")
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        """Command arm to move to a target joint configuration.
+
+        The relative action magnitude may be clipped depending on the configuration parameter
+        `max_relative_target`. In this case, the action sent differs from original action.
+        Thus, this function always returns the action actually sent.
+
+        Args:
+            action (RobotAction): The goal positions for the motors.
+
+        Returns:
+            RobotAction: The action sent to the motors, potentially clipped.
+        """
+
+        goal_pos = {key.removesuffix(".pos"): val for key, val in action.items() if key.endswith(".pos")}
+
+        # Cap goal position when too far away from present position.
+        # /!\ Slower fps expected due to reading from the follower.
+        if self.config.max_relative_target is not None:
+            present_pos = self.bus.sync_read("Present_Position")
+            goal_present_pos = {key: (g_pos, present_pos[key]) for key, g_pos in goal_pos.items()}
+            goal_pos = ensure_safe_goal_position(goal_present_pos, self.config.max_relative_target)
+
+        # Send goal position to the arm
+        self.bus.sync_write("Goal_Position", goal_pos)
+        return {f"{motor}.pos": val for motor, val in goal_pos.items()}
+
+    @check_if_not_connected
+    def disconnect(self):
+        self.bus.disconnect(self.config.disable_torque_on_disconnect)
+        for cam in self.cameras.values():
+            cam.disconnect()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/robots/openarm_follower/__init__.py b/lerobot/src/lerobot/robots/openarm_follower/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..217432fd57522a5b712e53ab9598ef2022e5bebd
--- /dev/null
+++ b/lerobot/src/lerobot/robots/openarm_follower/__init__.py
@@ -0,0 +1,20 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_openarm_follower import OpenArmFollowerConfig, OpenArmFollowerConfigBase
+from .openarm_follower import OpenArmFollower
+
+__all__ = ["OpenArmFollower", "OpenArmFollowerConfig", "OpenArmFollowerConfigBase"]
diff --git a/lerobot/src/lerobot/robots/openarm_follower/config_openarm_follower.py b/lerobot/src/lerobot/robots/openarm_follower/config_openarm_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..88d81fd506e3b8341a27b940e5092597b97c1d74
--- /dev/null
+++ b/lerobot/src/lerobot/robots/openarm_follower/config_openarm_follower.py
@@ -0,0 +1,122 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.cameras import CameraConfig
+
+from ..config import RobotConfig
+
+LEFT_DEFAULT_JOINTS_LIMITS: dict[str, tuple[float, float]] = {
+    "joint_1": (-75.0, 75.0),
+    "joint_2": (-90.0, 9.0),
+    "joint_3": (-85.0, 85.0),
+    "joint_4": (0.0, 135.0),
+    "joint_5": (-85.0, 85.0),
+    "joint_6": (-40.0, 40.0),
+    "joint_7": (-80.0, 80.0),
+    "gripper": (-65.0, 0.0),
+}
+
+RIGHT_DEFAULT_JOINTS_LIMITS: dict[str, tuple[float, float]] = {
+    "joint_1": (-75.0, 75.0),
+    "joint_2": (-9.0, 90.0),
+    "joint_3": (-85.0, 85.0),
+    "joint_4": (0.0, 135.0),
+    "joint_5": (-85.0, 85.0),
+    "joint_6": (-40.0, 40.0),
+    "joint_7": (-80.0, 80.0),
+    "gripper": (-65.0, 0.0),
+}
+
+
+@dataclass
+class OpenArmFollowerConfigBase:
+    """Base configuration for the OpenArms follower robot with Damiao motors."""
+
+    # CAN interfaces - one per arm
+    # arm CAN interface (e.g., "can1")
+    # Linux: "can0", "can1", etc.
+    port: str
+
+    # side of the arm: "left" or "right". If "None" default values will be used
+    side: str | None = None
+
+    # CAN interface type: "socketcan" (Linux), "slcan" (serial), or "auto" (auto-detect)
+    can_interface: str = "socketcan"
+
+    # CAN FD settings (OpenArms uses CAN FD by default)
+    use_can_fd: bool = True
+    can_bitrate: int = 1000000  # Nominal bitrate (1 Mbps)
+    can_data_bitrate: int = 5000000  # Data bitrate for CAN FD (5 Mbps)
+
+    # Whether to disable torque when disconnecting
+    disable_torque_on_disconnect: bool = True
+
+    # Safety limit for relative target positions
+    # Set to a positive scalar for all motors, or a dict mapping motor names to limits
+    max_relative_target: float | dict[str, float] | None = None
+
+    # Camera configurations
+    cameras: dict[str, CameraConfig] = field(default_factory=dict)
+
+    # Motor configuration for OpenArms (7 DOF per arm)
+    # Maps motor names to (send_can_id, recv_can_id, motor_type)
+    # Based on: https://docs.openarm.dev/software/setup/configure-test
+    # OpenArms uses 4 types of motors:
+    # - DM8009 (DM-J8009P-2EC) for shoulders (high torque)
+    # - DM4340P and DM4340 for shoulder rotation and elbow
+    # - DM4310 (DM-J4310-2EC V1.1) for wrist and gripper
+    motor_config: dict[str, tuple[int, int, str]] = field(
+        default_factory=lambda: {
+            "joint_1": (0x01, 0x11, "dm8009"),  # J1 - Shoulder pan (DM8009)
+            "joint_2": (0x02, 0x12, "dm8009"),  # J2 - Shoulder lift (DM8009)
+            "joint_3": (0x03, 0x13, "dm4340"),  # J3 - Shoulder rotation (DM4340)
+            "joint_4": (0x04, 0x14, "dm4340"),  # J4 - Elbow flex (DM4340)
+            "joint_5": (0x05, 0x15, "dm4310"),  # J5 - Wrist roll (DM4310)
+            "joint_6": (0x06, 0x16, "dm4310"),  # J6 - Wrist pitch (DM4310)
+            "joint_7": (0x07, 0x17, "dm4310"),  # J7 - Wrist rotation (DM4310)
+            "gripper": (0x08, 0x18, "dm4310"),  # J8 - Gripper (DM4310)
+        }
+    )
+
+    # MIT control parameters for position control (used in send_action)
+    # List of 8 values: [joint_1, joint_2, joint_3, joint_4, joint_5, joint_6, joint_7, gripper]
+    position_kp: list[float] = field(
+        default_factory=lambda: [240.0, 240.0, 240.0, 240.0, 24.0, 31.0, 25.0, 25.0]
+    )
+    position_kd: list[float] = field(default_factory=lambda: [5.0, 5.0, 3.0, 5.0, 0.3, 0.3, 0.3, 0.3])
+
+    # Values for joint limits. Can be overridden via CLI (for custom values) or by setting config.side to either 'left' or 'right'.
+    # If config.side is left set to None and no CLI values are passed, the default joint limit values are small for safety.
+    joint_limits: dict[str, tuple[float, float]] = field(
+        default_factory=lambda: {
+            "joint_1": (-5.0, 5.0),
+            "joint_2": (-5.0, 5.0),
+            "joint_3": (-5.0, 5.0),
+            "joint_4": (0.0, 5.0),
+            "joint_5": (-5.0, 5.0),
+            "joint_6": (-5.0, 5.0),
+            "joint_7": (-5.0, 5.0),
+            "gripper": (-5.0, 0.0),
+        }
+    )
+
+
+@RobotConfig.register_subclass("openarm_follower")
+@dataclass
+class OpenArmFollowerConfig(RobotConfig, OpenArmFollowerConfigBase):
+    pass
diff --git a/lerobot/src/lerobot/robots/openarm_follower/openarm_follower.py b/lerobot/src/lerobot/robots/openarm_follower/openarm_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..99e8b920b076a7e0f59f12ef2c98bad715c04daa
--- /dev/null
+++ b/lerobot/src/lerobot/robots/openarm_follower/openarm_follower.py
@@ -0,0 +1,343 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from functools import cached_property
+from typing import Any
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.damiao import DamiaoMotorsBus
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from ..utils import ensure_safe_goal_position
+from .config_openarm_follower import (
+    LEFT_DEFAULT_JOINTS_LIMITS,
+    RIGHT_DEFAULT_JOINTS_LIMITS,
+    OpenArmFollowerConfig,
+)
+
+logger = logging.getLogger(__name__)
+
+
+class OpenArmFollower(Robot):
+    """
+    OpenArms Follower Robot which uses CAN bus communication to control 7 DOF arm with a gripper.
+    The arm uses Damiao motors in MIT control mode.
+    """
+
+    config_class = OpenArmFollowerConfig
+    name = "openarm_follower"
+
+    def __init__(self, config: OpenArmFollowerConfig):
+        super().__init__(config)
+        self.config = config
+
+        # Arm motors
+        motors: dict[str, Motor] = {}
+        for motor_name, (send_id, recv_id, motor_type_str) in config.motor_config.items():
+            motor = Motor(
+                send_id, motor_type_str, MotorNormMode.DEGREES
+            )  # Always use degrees for Damiao motors
+            motor.recv_id = recv_id
+            motor.motor_type_str = motor_type_str
+            motors[motor_name] = motor
+
+        self.bus = DamiaoMotorsBus(
+            port=self.config.port,
+            motors=motors,
+            calibration=self.calibration,
+            can_interface=self.config.can_interface,
+            use_can_fd=self.config.use_can_fd,
+            bitrate=self.config.can_bitrate,
+            data_bitrate=self.config.can_data_bitrate if self.config.use_can_fd else None,
+        )
+
+        if config.side is not None:
+            if config.side == "left":
+                config.joint_limits = LEFT_DEFAULT_JOINTS_LIMITS
+            elif config.side == "right":
+                config.joint_limits = RIGHT_DEFAULT_JOINTS_LIMITS
+            else:
+                raise ValueError(
+                    "config.side must be either 'left', 'right' (for default values) or 'None' (for CLI values)"
+                )
+        else:
+            logger.info(
+                "Set config.side to either 'left' or 'right' to use pre-configured values for joint limits."
+            )
+        logger.info(f"Values used for joint limits: {config.joint_limits}.")
+
+        # Initialize cameras
+        self.cameras = make_cameras_from_configs(config.cameras)
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        """Motor features for observation and action spaces."""
+        features: dict[str, type] = {}
+        for motor in self.bus.motors:
+            features[f"{motor}.pos"] = float
+            features[f"{motor}.vel"] = float  # Add this
+            features[f"{motor}.torque"] = float  # Add this
+        return features
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        """Camera features for observation space."""
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        """Combined observation features from motors and cameras."""
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        """Action features."""
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        """Check if robot is connected."""
+        return self.bus.is_connected and all(cam.is_connected for cam in self.cameras.values())
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        """
+        Connect to the robot and optionally calibrate.
+
+        We assume that at connection time, the arms are in a safe rest position,
+        and torque can be safely disabled to run calibration if needed.
+        """
+
+        # Connect to CAN bus
+        logger.info(f"Connecting arm on {self.config.port}...")
+        self.bus.connect()
+
+        # Run calibration if needed
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        for cam in self.cameras.values():
+            cam.connect()
+
+        self.configure()
+
+        if self.is_calibrated:
+            self.bus.set_zero_position()
+
+        self.bus.enable_torque()
+
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        """Check if robot is calibrated."""
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        """
+        Run calibration procedure for OpenArms robot.
+
+        The calibration procedure:
+        1. Disable torque
+        2. Ask user to position arms in hanging position with grippers closed
+        3. Set this as zero position
+        4. Record range of motion for each joint
+        5. Save calibration
+        """
+        if self.calibration:
+            # Calibration file exists, ask user whether to use it or run new calibration
+            user_input = input(
+                f"Press ENTER to use provided calibration file associated with the id {self.id}, or type 'c' and press ENTER to run calibration: "
+            )
+            if user_input.strip().lower() != "c":
+                logger.info(f"Writing calibration file associated with the id {self.id} to the motors")
+                self.bus.write_calibration(self.calibration)
+                return
+
+        logger.info(f"\nRunning calibration for {self}")
+        self.bus.disable_torque()
+
+        # Step 1: Set zero position
+        input(
+            "\nCalibration: Set Zero Position)\n"
+            "Position the arm in the following configuration:\n"
+            "  - Arm hanging straight down\n"
+            "  - Gripper closed\n"
+            "Press ENTER when ready..."
+        )
+
+        # Set current position as zero for all motors
+        self.bus.set_zero_position()
+        logger.info("Arm zero position set.")
+
+        logger.info("Setting range: -90° to +90° for safety by default for all joints")
+        for motor_name, motor in self.bus.motors.items():
+            self.calibration[motor_name] = MotorCalibration(
+                id=motor.id,
+                drive_mode=0,
+                homing_offset=0,
+                range_min=-90,
+                range_max=90,
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        print(f"Calibration saved to {self.calibration_fpath}")
+
+    def configure(self) -> None:
+        """Configure motors with appropriate settings."""
+        # TODO(Steven, Pepijn): Slightly different from what it is happening in the leader
+        with self.bus.torque_disabled():
+            self.bus.configure_motors()
+
+    def setup_motors(self) -> None:
+        raise NotImplementedError(
+            "Motor ID configuration is typically done via manufacturer tools for CAN motors."
+        )
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        """
+        Get current observation from robot including position, velocity, and torque.
+
+        Reads all motor states (pos/vel/torque) in one CAN refresh cycle
+        instead of 3 separate reads.
+        """
+        start = time.perf_counter()
+
+        obs_dict: dict[str, Any] = {}
+
+        states = self.bus.sync_read_all_states()
+
+        for motor in self.bus.motors:
+            state = states.get(motor, {})
+            obs_dict[f"{motor}.pos"] = state.get("position", 0.0)
+            obs_dict[f"{motor}.vel"] = state.get("velocity", 0.0)
+            obs_dict[f"{motor}.torque"] = state.get("torque", 0.0)
+
+        # Capture images from cameras
+        for cam_key, cam in self.cameras.items():
+            start = time.perf_counter()
+            obs_dict[cam_key] = cam.read_latest()
+            dt_ms = (time.perf_counter() - start) * 1e3
+            logger.debug(f"{self} read {cam_key}: {dt_ms:.1f}ms")
+
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} get_observation took: {dt_ms:.1f}ms")
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(
+        self,
+        action: RobotAction,
+        custom_kp: dict[str, float] | None = None,
+        custom_kd: dict[str, float] | None = None,
+    ) -> RobotAction:
+        """
+        Send action command to robot.
+
+        The action magnitude may be clipped based on safety limits.
+
+        Args:
+            action: Dictionary with motor positions (e.g., "joint_1.pos", "joint_2.pos")
+            custom_kp: Optional custom kp gains per motor (e.g., {"joint_1": 120.0, "joint_2": 150.0})
+            custom_kd: Optional custom kd gains per motor (e.g., {"joint_1": 1.5, "joint_2": 2.0})
+
+        Returns:
+            The action actually sent (potentially clipped)
+        """
+
+        goal_pos = {key.removesuffix(".pos"): val for key, val in action.items() if key.endswith(".pos")}
+
+        # Apply joint limit clipping to arm
+        for motor_name, position in goal_pos.items():
+            if motor_name in self.config.joint_limits:
+                min_limit, max_limit = self.config.joint_limits[motor_name]
+                clipped_position = max(min_limit, min(max_limit, position))
+                if clipped_position != position:
+                    logger.debug(f"Clipped {motor_name} from {position:.2f}° to {clipped_position:.2f}°")
+                goal_pos[motor_name] = clipped_position
+
+        # Cap goal position when too far away from present position.
+        # /!\ Slower fps expected due to reading from the follower.
+        if self.config.max_relative_target is not None:
+            present_pos = self.bus.sync_read("Present_Position")
+            goal_present_pos = {key: (g_pos, present_pos[key]) for key, g_pos in goal_pos.items()}
+            goal_pos = ensure_safe_goal_position(goal_present_pos, self.config.max_relative_target)
+
+        # TODO(Steven, Pepijn): Refactor writing
+        # Motor name to index mapping for gains
+        motor_index = {
+            "joint_1": 0,
+            "joint_2": 1,
+            "joint_3": 2,
+            "joint_4": 3,
+            "joint_5": 4,
+            "joint_6": 5,
+            "joint_7": 6,
+            "gripper": 7,
+        }
+
+        # Use batch MIT control for arm (sends all commands, then collects responses)
+        commands = {}
+        for motor_name, position_degrees in goal_pos.items():
+            idx = motor_index.get(motor_name, 0)
+            # Use custom gains if provided, otherwise use config defaults
+            if custom_kp is not None and motor_name in custom_kp:
+                kp = custom_kp[motor_name]
+            else:
+                kp = (
+                    self.config.position_kp[idx]
+                    if isinstance(self.config.position_kp, list)
+                    else self.config.position_kp
+                )
+            if custom_kd is not None and motor_name in custom_kd:
+                kd = custom_kd[motor_name]
+            else:
+                kd = (
+                    self.config.position_kd[idx]
+                    if isinstance(self.config.position_kd, list)
+                    else self.config.position_kd
+                )
+            commands[motor_name] = (kp, kd, position_degrees, 0.0, 0.0)
+
+        self.bus._mit_control_batch(commands)
+
+        return {f"{motor}.pos": val for motor, val in goal_pos.items()}
+
+    @check_if_not_connected
+    def disconnect(self):
+        """Disconnect from robot."""
+
+        # Disconnect CAN bus
+        self.bus.disconnect(self.config.disable_torque_on_disconnect)
+
+        # Disconnect cameras
+        for cam in self.cameras.values():
+            cam.disconnect()
+
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/robots/reachy2/__init__.py b/lerobot/src/lerobot/robots/reachy2/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..1a38fd03b29edd9898a453225fd6f7a6ee574cb3
--- /dev/null
+++ b/lerobot/src/lerobot/robots/reachy2/__init__.py
@@ -0,0 +1,25 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configuration_reachy2 import Reachy2RobotConfig
+from .robot_reachy2 import (
+    REACHY2_ANTENNAS_JOINTS,
+    REACHY2_L_ARM_JOINTS,
+    REACHY2_NECK_JOINTS,
+    REACHY2_R_ARM_JOINTS,
+    REACHY2_VEL,
+    Reachy2Robot,
+)
diff --git a/lerobot/src/lerobot/robots/reachy2/configuration_reachy2.py b/lerobot/src/lerobot/robots/reachy2/configuration_reachy2.py
new file mode 100644
index 0000000000000000000000000000000000000000..63293e675b00977971a9cdc1907b292a6b3ef59b
--- /dev/null
+++ b/lerobot/src/lerobot/robots/reachy2/configuration_reachy2.py
@@ -0,0 +1,117 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.cameras import CameraConfig
+from lerobot.cameras.configs import ColorMode
+from lerobot.cameras.reachy2_camera import Reachy2CameraConfig
+
+from ..config import RobotConfig
+
+
+@RobotConfig.register_subclass("reachy2")
+@dataclass
+class Reachy2RobotConfig(RobotConfig):
+    # `max_relative_target` limits the magnitude of the relative positional target vector for safety purposes.
+    # Set this to a positive scalar to have the same value for all motors.
+    max_relative_target: float | None = None
+
+    # IP address of the Reachy 2 robot
+    ip_address: str | None = "localhost"
+    # Port of the Reachy 2 robot
+    port: int = 50065
+
+    # If True, turn_off_smoothly() will be sent to the robot before disconnecting.
+    disable_torque_on_disconnect: bool = False
+
+    # Tag for external commands control
+    # Set to True if you use an external commands system to control the robot,
+    # such as the official teleoperation application: https://github.com/pollen-robotics/Reachy2Teleoperation
+    # If True, robot.send_action() will not send commands to the robot.
+    use_external_commands: bool = False
+
+    # Robot parts
+    # Set to False to not add the corresponding joints part to the robot list of joints.
+    # By default, all parts are set to True.
+    with_mobile_base: bool = True
+    with_l_arm: bool = True
+    with_r_arm: bool = True
+    with_neck: bool = True
+    with_antennas: bool = True
+
+    # Robot cameras
+    # Set to True if you want to use the corresponding cameras in the observations.
+    # By default, no camera is used.
+    with_left_teleop_camera: bool = False
+    with_right_teleop_camera: bool = False
+    with_torso_camera: bool = False
+
+    # Camera parameters
+    camera_width: int = 640
+    camera_height: int = 480
+
+    # For cameras other than the 3 default Reachy 2 cameras.
+    cameras: dict[str, CameraConfig] = field(default_factory=dict)
+
+    def __post_init__(self) -> None:
+        # Add cameras with same ip_address as the robot
+        if self.with_left_teleop_camera:
+            self.cameras["teleop_left"] = Reachy2CameraConfig(
+                name="teleop",
+                image_type="left",
+                ip_address=self.ip_address,
+                port=self.port,
+                width=self.camera_width,
+                height=self.camera_height,
+                fps=30,  # Not configurable for Reachy 2 cameras
+                color_mode=ColorMode.RGB,
+            )
+        if self.with_right_teleop_camera:
+            self.cameras["teleop_right"] = Reachy2CameraConfig(
+                name="teleop",
+                image_type="right",
+                ip_address=self.ip_address,
+                port=self.port,
+                width=self.camera_width,
+                height=self.camera_height,
+                fps=30,  # Not configurable for Reachy 2 cameras
+                color_mode=ColorMode.RGB,
+            )
+        if self.with_torso_camera:
+            self.cameras["torso_rgb"] = Reachy2CameraConfig(
+                name="depth",
+                image_type="rgb",
+                ip_address=self.ip_address,
+                port=self.port,
+                width=self.camera_width,
+                height=self.camera_height,
+                fps=30,  # Not configurable for Reachy 2 cameras
+                color_mode=ColorMode.RGB,
+            )
+
+        super().__post_init__()
+
+        if not (
+            self.with_mobile_base
+            or self.with_l_arm
+            or self.with_r_arm
+            or self.with_neck
+            or self.with_antennas
+        ):
+            raise ValueError(
+                "No Reachy2Robot part used.\n"
+                "At least one part of the robot must be set to True "
+                "(with_mobile_base, with_l_arm, with_r_arm, with_neck, with_antennas)"
+            )
diff --git a/lerobot/src/lerobot/robots/reachy2/robot_reachy2.py b/lerobot/src/lerobot/robots/reachy2/robot_reachy2.py
new file mode 100644
index 0000000000000000000000000000000000000000..5227a096ac40650742a68c5dfff3d51bb2fe9bf8
--- /dev/null
+++ b/lerobot/src/lerobot/robots/reachy2/robot_reachy2.py
@@ -0,0 +1,235 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from __future__ import annotations
+
+import time
+from typing import TYPE_CHECKING, Any
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.import_utils import _reachy2_sdk_available
+
+from ..robot import Robot
+from ..utils import ensure_safe_goal_position
+from .configuration_reachy2 import Reachy2RobotConfig
+
+if TYPE_CHECKING or _reachy2_sdk_available:
+    from reachy2_sdk import ReachySDK
+else:
+    ReachySDK = None
+
+# {lerobot_keys: reachy2_sdk_keys}
+REACHY2_NECK_JOINTS = {
+    "neck_yaw.pos": "head.neck.yaw",
+    "neck_pitch.pos": "head.neck.pitch",
+    "neck_roll.pos": "head.neck.roll",
+}
+
+REACHY2_ANTENNAS_JOINTS = {
+    "l_antenna.pos": "head.l_antenna",
+    "r_antenna.pos": "head.r_antenna",
+}
+
+REACHY2_R_ARM_JOINTS = {
+    "r_shoulder_pitch.pos": "r_arm.shoulder.pitch",
+    "r_shoulder_roll.pos": "r_arm.shoulder.roll",
+    "r_elbow_yaw.pos": "r_arm.elbow.yaw",
+    "r_elbow_pitch.pos": "r_arm.elbow.pitch",
+    "r_wrist_roll.pos": "r_arm.wrist.roll",
+    "r_wrist_pitch.pos": "r_arm.wrist.pitch",
+    "r_wrist_yaw.pos": "r_arm.wrist.yaw",
+    "r_gripper.pos": "r_arm.gripper",
+}
+
+REACHY2_L_ARM_JOINTS = {
+    "l_shoulder_pitch.pos": "l_arm.shoulder.pitch",
+    "l_shoulder_roll.pos": "l_arm.shoulder.roll",
+    "l_elbow_yaw.pos": "l_arm.elbow.yaw",
+    "l_elbow_pitch.pos": "l_arm.elbow.pitch",
+    "l_wrist_roll.pos": "l_arm.wrist.roll",
+    "l_wrist_pitch.pos": "l_arm.wrist.pitch",
+    "l_wrist_yaw.pos": "l_arm.wrist.yaw",
+    "l_gripper.pos": "l_arm.gripper",
+}
+
+REACHY2_VEL = {
+    "mobile_base.vx": "vx",
+    "mobile_base.vy": "vy",
+    "mobile_base.vtheta": "vtheta",
+}
+
+
+class Reachy2Robot(Robot):
+    """
+    [Reachy 2](https://www.pollen-robotics.com/reachy/), by Pollen Robotics.
+    """
+
+    config_class = Reachy2RobotConfig
+    name = "reachy2"
+
+    def __init__(self, config: Reachy2RobotConfig):
+        super().__init__(config)
+
+        self.config = config
+        self.robot_type = self.config.type
+        self.use_external_commands = self.config.use_external_commands
+
+        self.reachy: None | ReachySDK = None
+        self.cameras = make_cameras_from_configs(config.cameras)
+
+        self.logs: dict[str, float] = {}
+
+        self.joints_dict: dict[str, str] = self._generate_joints_dict()
+
+    @property
+    def observation_features(self) -> dict[str, Any]:
+        return {**self.motors_features, **self.camera_features}
+
+    @property
+    def action_features(self) -> dict[str, type]:
+        return self.motors_features
+
+    @property
+    def camera_features(self) -> dict[str, tuple[int | None, int | None, int]]:
+        return {cam: (self.cameras[cam].height, self.cameras[cam].width, 3) for cam in self.cameras}
+
+    @property
+    def motors_features(self) -> dict[str, type]:
+        if self.config.with_mobile_base:
+            return {
+                **dict.fromkeys(
+                    self.joints_dict.keys(),
+                    float,
+                ),
+                **dict.fromkeys(
+                    REACHY2_VEL.keys(),
+                    float,
+                ),
+            }
+        else:
+            return dict.fromkeys(self.joints_dict.keys(), float)
+
+    @property
+    def is_connected(self) -> bool:
+        return self.reachy.is_connected() if self.reachy is not None else False
+
+    def connect(self, calibrate: bool = False) -> None:
+        self.reachy = ReachySDK(self.config.ip_address)
+        if not self.is_connected:
+            raise ConnectionError()
+
+        for cam in self.cameras.values():
+            cam.connect()
+
+        self.configure()
+
+    def configure(self) -> None:
+        if self.reachy is not None:
+            self.reachy.turn_on()
+            self.reachy.reset_default_limits()
+
+    @property
+    def is_calibrated(self) -> bool:
+        return True
+
+    def calibrate(self) -> None:
+        pass
+
+    def _generate_joints_dict(self) -> dict[str, str]:
+        joints = {}
+        if self.config.with_neck:
+            joints.update(REACHY2_NECK_JOINTS)
+        if self.config.with_l_arm:
+            joints.update(REACHY2_L_ARM_JOINTS)
+        if self.config.with_r_arm:
+            joints.update(REACHY2_R_ARM_JOINTS)
+        if self.config.with_antennas:
+            joints.update(REACHY2_ANTENNAS_JOINTS)
+        return joints
+
+    def _get_state(self) -> dict[str, float]:
+        if self.reachy is not None:
+            pos_dict = {k: self.reachy.joints[v].present_position for k, v in self.joints_dict.items()}
+            if not self.config.with_mobile_base:
+                return pos_dict
+            vel_dict = {k: self.reachy.mobile_base.odometry[v] for k, v in REACHY2_VEL.items()}
+            return {**pos_dict, **vel_dict}
+        else:
+            return {}
+
+    def get_observation(self) -> RobotObservation:
+        obs_dict: RobotObservation = {}
+
+        # Read Reachy 2 state
+        before_read_t = time.perf_counter()
+        obs_dict.update(self._get_state())
+        self.logs["read_pos_dt_s"] = time.perf_counter() - before_read_t
+
+        # Capture images from cameras
+        for cam_key, cam in self.cameras.items():
+            obs_dict[cam_key] = cam.read_latest()
+
+        return obs_dict
+
+    def send_action(self, action: RobotAction) -> RobotAction:
+        if self.reachy is not None:
+            if not self.is_connected:
+                raise ConnectionError()
+
+            before_write_t = time.perf_counter()
+
+            vel = {}
+            goal_pos = {}
+            for key, val in action.items():
+                if key not in self.joints_dict:
+                    if key not in REACHY2_VEL:
+                        raise KeyError(f"Key '{key}' is not a valid motor key in Reachy 2.")
+                    else:
+                        vel[REACHY2_VEL[key]] = float(val)
+                else:
+                    if not self.use_external_commands and self.config.max_relative_target is not None:
+                        goal_pos[key] = float(val)
+                        goal_present_pos = {
+                            key: (
+                                goal_pos[key],
+                                self.reachy.joints[self.joints_dict[key]].present_position,
+                            )
+                        }
+                        safe_goal_pos = ensure_safe_goal_position(
+                            goal_present_pos, float(self.config.max_relative_target)
+                        )
+                        val = safe_goal_pos[key]
+                    self.reachy.joints[self.joints_dict[key]].goal_position = float(val)
+
+            if self.config.with_mobile_base:
+                self.reachy.mobile_base.set_goal_speed(vel["vx"], vel["vy"], vel["vtheta"])
+
+            # We don't send the goal positions if we control Reachy 2 externally
+            if not self.use_external_commands:
+                self.reachy.send_goal_positions()
+                if self.config.with_mobile_base:
+                    self.reachy.mobile_base.send_speed_command()
+
+            self.logs["write_pos_dt_s"] = time.perf_counter() - before_write_t
+        return action
+
+    def disconnect(self) -> None:
+        if self.reachy is not None:
+            for cam in self.cameras.values():
+                cam.disconnect()
+            if self.config.disable_torque_on_disconnect:
+                self.reachy.turn_off_smoothly()
+            self.reachy.disconnect()
diff --git a/lerobot/src/lerobot/robots/robot.py b/lerobot/src/lerobot/robots/robot.py
new file mode 100644
index 0000000000000000000000000000000000000000..1b556f9637c9c540e7b71a2812f5d6330bdda639
--- /dev/null
+++ b/lerobot/src/lerobot/robots/robot.py
@@ -0,0 +1,211 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import abc
+import builtins
+from pathlib import Path
+
+import draccus
+
+from lerobot.motors import MotorCalibration
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.constants import HF_LEROBOT_CALIBRATION, ROBOTS
+
+from .config import RobotConfig
+
+
+# TODO(aliberts): action/obs typing such as Generic[ObsType, ActType] similar to gym.Env ?
+# https://github.com/Farama-Foundation/Gymnasium/blob/3287c869f9a48d99454306b0d4b4ec537f0f35e3/gymnasium/core.py#L23
+class Robot(abc.ABC):
+    """
+    The base abstract class for all LeRobot-compatible robots.
+
+    This class provides a standardized interface for interacting with physical robots.
+    Subclasses must implement all abstract methods and properties to be usable.
+
+    Attributes:
+        config_class (RobotConfig): The expected configuration class for this robot.
+        name (str): The unique robot name used to identify this robot type.
+    """
+
+    # Set these in ALL subclasses
+    config_class: builtins.type[RobotConfig]
+    name: str
+
+    def __init__(self, config: RobotConfig):
+        self.robot_type = self.name
+        self.id = config.id
+        self.calibration_dir = (
+            config.calibration_dir if config.calibration_dir else HF_LEROBOT_CALIBRATION / ROBOTS / self.name
+        )
+        self.calibration_dir.mkdir(parents=True, exist_ok=True)
+        self.calibration_fpath = self.calibration_dir / f"{self.id}.json"
+        self.calibration: dict[str, MotorCalibration] = {}
+        if self.calibration_fpath.is_file():
+            self._load_calibration()
+
+    def __str__(self) -> str:
+        return f"{self.id} {self.__class__.__name__}"
+
+    def __enter__(self):
+        """
+        Context manager entry.
+        Automatically connects to the camera.
+        """
+        self.connect()
+        return self
+
+    def __exit__(self, exc_type, exc_value, traceback) -> None:
+        """
+        Context manager exit.
+        Automatically disconnects, ensuring resources are released even on error.
+        """
+        self.disconnect()
+
+    def __del__(self) -> None:
+        """
+        Destructor safety net.
+        Attempts to disconnect if the object is garbage collected without cleanup.
+        """
+        try:
+            if self.is_connected:
+                self.disconnect()
+        except Exception:  # nosec B110
+            pass
+
+    # TODO(aliberts): create a proper Feature class for this that links with datasets
+    @property
+    @abc.abstractmethod
+    def observation_features(self) -> dict:
+        """
+        A dictionary describing the structure and types of the observations produced by the robot.
+        Its structure (keys) should match the structure of what is returned by :pymeth:`get_observation`.
+        Values for the dict should either be:
+            - The type of the value if it's a simple value, e.g. `float` for single proprioceptive value (a joint's position/velocity)
+            - A tuple representing the shape if it's an array-type value, e.g. `(height, width, channel)` for images
+
+        Note: this property should be able to be called regardless of whether the robot is connected or not.
+        """
+        pass
+
+    @property
+    @abc.abstractmethod
+    def action_features(self) -> dict:
+        """
+        A dictionary describing the structure and types of the actions expected by the robot. Its structure
+        (keys) should match the structure of what is passed to :pymeth:`send_action`. Values for the dict
+        should be the type of the value if it's a simple value, e.g. `float` for single proprioceptive value
+        (a joint's goal position/velocity)
+
+        Note: this property should be able to be called regardless of whether the robot is connected or not.
+        """
+        pass
+
+    @property
+    @abc.abstractmethod
+    def is_connected(self) -> bool:
+        """
+        Whether the robot is currently connected or not. If `False`, calling :pymeth:`get_observation` or
+        :pymeth:`send_action` should raise an error.
+        """
+        pass
+
+    @abc.abstractmethod
+    def connect(self, calibrate: bool = True) -> None:
+        """
+        Establish communication with the robot.
+
+        Args:
+            calibrate (bool): If True, automatically calibrate the robot after connecting if it's not
+                calibrated or needs calibration (this is hardware-dependant).
+        """
+        pass
+
+    @property
+    @abc.abstractmethod
+    def is_calibrated(self) -> bool:
+        """Whether the robot is currently calibrated or not. Should be always `True` if not applicable"""
+        pass
+
+    @abc.abstractmethod
+    def calibrate(self) -> None:
+        """
+        Calibrate the robot if applicable. If not, this should be a no-op.
+
+        This method should collect any necessary data (e.g., motor offsets) and update the
+        :pyattr:`calibration` dictionary accordingly.
+        """
+        pass
+
+    def _load_calibration(self, fpath: Path | None = None) -> None:
+        """
+        Helper to load calibration data from the specified file.
+
+        Args:
+            fpath (Path | None): Optional path to the calibration file. Defaults to `self.calibration_fpath`.
+        """
+        fpath = self.calibration_fpath if fpath is None else fpath
+        with open(fpath) as f, draccus.config_type("json"):
+            self.calibration = draccus.load(dict[str, MotorCalibration], f)
+
+    def _save_calibration(self, fpath: Path | None = None) -> None:
+        """
+        Helper to save calibration data to the specified file.
+
+        Args:
+            fpath (Path | None): Optional path to save the calibration file. Defaults to `self.calibration_fpath`.
+        """
+        fpath = self.calibration_fpath if fpath is None else fpath
+        with open(fpath, "w") as f, draccus.config_type("json"):
+            draccus.dump(self.calibration, f, indent=4)
+
+    @abc.abstractmethod
+    def configure(self) -> None:
+        """
+        Apply any one-time or runtime configuration to the robot.
+        This may include setting motor parameters, control modes, or initial state.
+        """
+        pass
+
+    @abc.abstractmethod
+    def get_observation(self) -> RobotObservation:
+        """
+        Retrieve the current observation from the robot.
+
+        Returns:
+            RobotObservation: A flat dictionary representing the robot's current sensory state. Its structure
+                should match :pymeth:`observation_features`.
+        """
+
+        pass
+
+    @abc.abstractmethod
+    def send_action(self, action: RobotAction) -> RobotAction:
+        """
+        Send an action command to the robot.
+
+        Args:
+            action (RobotAction): Dictionary representing the desired action. Its structure should match
+                :pymeth:`action_features`.
+
+        Returns:
+            RobotAction: The action actually sent to the motors potentially clipped or modified, e.g. by
+                safety limits on velocity.
+        """
+        pass
+
+    @abc.abstractmethod
+    def disconnect(self) -> None:
+        """Disconnect from the robot and perform any necessary cleanup."""
+        pass
diff --git a/lerobot/src/lerobot/robots/so_follower/__init__.py b/lerobot/src/lerobot/robots/so_follower/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..eea2fcbdf95a2d427b7ba1e552f7a79fabcef6e0
--- /dev/null
+++ b/lerobot/src/lerobot/robots/so_follower/__init__.py
@@ -0,0 +1,23 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_so_follower import (
+    SO100FollowerConfig,
+    SO101FollowerConfig,
+    SOFollowerConfig,
+    SOFollowerRobotConfig,
+)
+from .so_follower import SO100Follower, SO101Follower, SOFollower
diff --git a/lerobot/src/lerobot/robots/so_follower/config_so_follower.py b/lerobot/src/lerobot/robots/so_follower/config_so_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..52f7953de3af1dd197604ccaa393363129e76c29
--- /dev/null
+++ b/lerobot/src/lerobot/robots/so_follower/config_so_follower.py
@@ -0,0 +1,53 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.cameras import CameraConfig
+
+from ..config import RobotConfig
+
+
+@dataclass
+class SOFollowerConfig:
+    """Base configuration class for SO Follower robots."""
+
+    # Port to connect to the arm
+    port: str
+
+    disable_torque_on_disconnect: bool = True
+
+    # `max_relative_target` limits the magnitude of the relative positional target vector for safety purposes.
+    # Set this to a positive scalar to have the same value for all motors, or a dictionary that maps motor
+    # names to the max_relative_target value for that motor.
+    max_relative_target: float | dict[str, float] | None = None
+
+    # cameras
+    cameras: dict[str, CameraConfig] = field(default_factory=dict)
+
+    # Set to `True` for backward compatibility with previous policies/dataset
+    use_degrees: bool = True
+
+
+@RobotConfig.register_subclass("so101_follower")
+@RobotConfig.register_subclass("so100_follower")
+@dataclass
+class SOFollowerRobotConfig(RobotConfig, SOFollowerConfig):
+    pass
+
+
+SO100FollowerConfig = SOFollowerRobotConfig
+SO101FollowerConfig = SOFollowerRobotConfig
diff --git a/lerobot/src/lerobot/robots/so_follower/robot_kinematic_processor.py b/lerobot/src/lerobot/robots/so_follower/robot_kinematic_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..2aa60e12a968191ec666327a5cbcd6c5dfb8c553
--- /dev/null
+++ b/lerobot/src/lerobot/robots/so_follower/robot_kinematic_processor.py
@@ -0,0 +1,611 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+from typing import Any
+
+import numpy as np
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.processor import (
+    EnvTransition,
+    ObservationProcessorStep,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    RobotAction,
+    RobotActionProcessorStep,
+    RobotObservation,
+    TransitionKey,
+)
+from lerobot.utils.rotation import Rotation
+
+
+@ProcessorStepRegistry.register("ee_reference_and_delta")
+@dataclass
+class EEReferenceAndDelta(RobotActionProcessorStep):
+    """
+    Computes a target end-effector pose from a relative delta command.
+
+    This step takes a desired change in position and orientation (`target_*`) and applies it to a
+    reference end-effector pose to calculate an absolute target pose. The reference pose is derived
+    from the current robot joint positions using forward kinematics.
+
+    The processor can operate in two modes:
+    1.  `use_latched_reference=True`: The reference pose is "latched" or saved at the moment the action
+        is first enabled. Subsequent commands are relative to this fixed reference.
+    2.  `use_latched_reference=False`: The reference pose is updated to the robot's current pose at
+        every step.
+
+    Attributes:
+        kinematics: The robot's kinematic model for forward kinematics.
+        end_effector_step_sizes: A dictionary scaling the input delta commands.
+        motor_names: A list of motor names required for forward kinematics.
+        use_latched_reference: If True, latch the reference pose on enable; otherwise, always use the
+            current pose as the reference.
+        reference_ee_pose: Internal state storing the latched reference pose.
+        _prev_enabled: Internal state to detect the rising edge of the enable signal.
+        _command_when_disabled: Internal state to hold the last command while disabled.
+    """
+
+    kinematics: RobotKinematics
+    end_effector_step_sizes: dict
+    motor_names: list[str]
+    use_latched_reference: bool = (
+        True  # If True, latch reference on enable; if False, always use current pose
+    )
+    use_ik_solution: bool = False
+
+    reference_ee_pose: np.ndarray | None = field(default=None, init=False, repr=False)
+    _prev_enabled: bool = field(default=False, init=False, repr=False)
+    _command_when_disabled: np.ndarray | None = field(default=None, init=False, repr=False)
+
+    def action(self, action: RobotAction) -> RobotAction:
+        observation = self.transition.get(TransitionKey.OBSERVATION).copy()
+
+        if observation is None:
+            raise ValueError("Joints observation is require for computing robot kinematics")
+
+        if self.use_ik_solution and "IK_solution" in self.transition.get(TransitionKey.COMPLEMENTARY_DATA):
+            q_raw = self.transition.get(TransitionKey.COMPLEMENTARY_DATA)["IK_solution"]
+        else:
+            q_raw = np.array(
+                [
+                    float(v)
+                    for k, v in observation.items()
+                    if isinstance(k, str)
+                    and k.endswith(".pos")
+                    and k.removesuffix(".pos") in self.motor_names
+                ],
+                dtype=float,
+            )
+
+        if q_raw is None:
+            raise ValueError("Joints observation is require for computing robot kinematics")
+
+        # Current pose from FK on measured joints
+        t_curr = self.kinematics.forward_kinematics(q_raw)
+
+        enabled = bool(action.pop("enabled"))
+        tx = float(action.pop("target_x"))
+        ty = float(action.pop("target_y"))
+        tz = float(action.pop("target_z"))
+        wx = float(action.pop("target_wx"))
+        wy = float(action.pop("target_wy"))
+        wz = float(action.pop("target_wz"))
+        gripper_vel = float(action.pop("gripper_vel"))
+
+        desired = None
+
+        if enabled:
+            ref = t_curr
+            if self.use_latched_reference:
+                # Latched reference mode: latch reference at the rising edge
+                if not self._prev_enabled or self.reference_ee_pose is None:
+                    self.reference_ee_pose = t_curr.copy()
+                ref = self.reference_ee_pose if self.reference_ee_pose is not None else t_curr
+
+            delta_p = np.array(
+                [
+                    tx * self.end_effector_step_sizes["x"],
+                    ty * self.end_effector_step_sizes["y"],
+                    tz * self.end_effector_step_sizes["z"],
+                ],
+                dtype=float,
+            )
+            r_abs = Rotation.from_rotvec([wx, wy, wz]).as_matrix()
+            desired = np.eye(4, dtype=float)
+            desired[:3, :3] = ref[:3, :3] @ r_abs
+            desired[:3, 3] = ref[:3, 3] + delta_p
+
+            self._command_when_disabled = desired.copy()
+        else:
+            # While disabled, keep sending the same command to avoid drift.
+            if self._command_when_disabled is None:
+                # If we've never had an enabled command yet, freeze current FK pose once.
+                self._command_when_disabled = t_curr.copy()
+            desired = self._command_when_disabled.copy()
+
+        # Write action fields
+        pos = desired[:3, 3]
+        tw = Rotation.from_matrix(desired[:3, :3]).as_rotvec()
+        action["ee.x"] = float(pos[0])
+        action["ee.y"] = float(pos[1])
+        action["ee.z"] = float(pos[2])
+        action["ee.wx"] = float(tw[0])
+        action["ee.wy"] = float(tw[1])
+        action["ee.wz"] = float(tw[2])
+        action["ee.gripper_vel"] = gripper_vel
+
+        self._prev_enabled = enabled
+        return action
+
+    def reset(self):
+        """Resets the internal state of the processor."""
+        self._prev_enabled = False
+        self.reference_ee_pose = None
+        self._command_when_disabled = None
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        for feat in [
+            "enabled",
+            "target_x",
+            "target_y",
+            "target_z",
+            "target_wx",
+            "target_wy",
+            "target_wz",
+            "gripper_vel",
+        ]:
+            features[PipelineFeatureType.ACTION].pop(f"{feat}", None)
+
+        for feat in ["x", "y", "z", "wx", "wy", "wz", "gripper_vel"]:
+            features[PipelineFeatureType.ACTION][f"ee.{feat}"] = PolicyFeature(
+                type=FeatureType.ACTION, shape=(1,)
+            )
+
+        return features
+
+
+@ProcessorStepRegistry.register("ee_bounds_and_safety")
+@dataclass
+class EEBoundsAndSafety(RobotActionProcessorStep):
+    """
+    Clips the end-effector pose to predefined bounds and checks for unsafe jumps.
+
+    This step ensures that the target end-effector pose remains within a safe operational workspace.
+    It also moderates the command to prevent large, sudden movements between consecutive steps.
+
+    Attributes:
+        end_effector_bounds: A dictionary with "min" and "max" keys for position clipping.
+        max_ee_step_m: The maximum allowed change in position (in meters) between steps.
+        _last_pos: Internal state storing the last commanded position.
+    """
+
+    end_effector_bounds: dict
+    max_ee_step_m: float = 0.05
+    _last_pos: np.ndarray | None = field(default=None, init=False, repr=False)
+
+    def action(self, action: RobotAction) -> RobotAction:
+        x = action["ee.x"]
+        y = action["ee.y"]
+        z = action["ee.z"]
+        wx = action["ee.wx"]
+        wy = action["ee.wy"]
+        wz = action["ee.wz"]
+        # TODO(Steven): ee.gripper_vel does not need to be bounded
+
+        if None in (x, y, z, wx, wy, wz):
+            raise ValueError(
+                "Missing required end-effector pose components: x, y, z, wx, wy, wz must all be present in action"
+            )
+
+        pos = np.array([x, y, z], dtype=float)
+        twist = np.array([wx, wy, wz], dtype=float)
+
+        # Clip position
+        pos = np.clip(pos, self.end_effector_bounds["min"], self.end_effector_bounds["max"])
+
+        # Check for jumps in position
+        if self._last_pos is not None:
+            dpos = pos - self._last_pos
+            n = float(np.linalg.norm(dpos))
+            if n > self.max_ee_step_m and n > 0:
+                pos = self._last_pos + dpos * (self.max_ee_step_m / n)
+                raise ValueError(f"EE jump {n:.3f}m > {self.max_ee_step_m}m")
+
+        self._last_pos = pos
+
+        action["ee.x"] = float(pos[0])
+        action["ee.y"] = float(pos[1])
+        action["ee.z"] = float(pos[2])
+        action["ee.wx"] = float(twist[0])
+        action["ee.wy"] = float(twist[1])
+        action["ee.wz"] = float(twist[2])
+        return action
+
+    def reset(self):
+        """Resets the last known position and orientation."""
+        self._last_pos = None
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+@ProcessorStepRegistry.register("inverse_kinematics_ee_to_joints")
+@dataclass
+class InverseKinematicsEEToJoints(RobotActionProcessorStep):
+    """
+    Computes desired joint positions from a target end-effector pose using inverse kinematics (IK).
+
+    This step translates a Cartesian command (position and orientation of the end-effector) into
+    the corresponding joint-space commands for each motor.
+
+    Attributes:
+        kinematics: The robot's kinematic model for inverse kinematics.
+        motor_names: A list of motor names for which to compute joint positions.
+        q_curr: Internal state storing the last joint positions, used as an initial guess for the IK solver.
+        initial_guess_current_joints: If True, use the robot's current joint state as the IK guess.
+            If False, use the solution from the previous step.
+    """
+
+    kinematics: RobotKinematics
+    motor_names: list[str]
+    q_curr: np.ndarray | None = field(default=None, init=False, repr=False)
+    initial_guess_current_joints: bool = True
+
+    def action(self, action: RobotAction) -> RobotAction:
+        x = action.pop("ee.x")
+        y = action.pop("ee.y")
+        z = action.pop("ee.z")
+        wx = action.pop("ee.wx")
+        wy = action.pop("ee.wy")
+        wz = action.pop("ee.wz")
+        gripper_pos = action.pop("ee.gripper_pos")
+
+        if None in (x, y, z, wx, wy, wz, gripper_pos):
+            raise ValueError(
+                "Missing required end-effector pose components: ee.x, ee.y, ee.z, ee.wx, ee.wy, ee.wz, ee.gripper_pos must all be present in action"
+            )
+
+        observation = self.transition.get(TransitionKey.OBSERVATION).copy()
+        if observation is None:
+            raise ValueError("Joints observation is require for computing robot kinematics")
+
+        q_raw = np.array(
+            [float(v) for k, v in observation.items() if isinstance(k, str) and k.endswith(".pos")],
+            dtype=float,
+        )
+        if q_raw is None:
+            raise ValueError("Joints observation is require for computing robot kinematics")
+
+        if self.initial_guess_current_joints:  # Use current joints as initial guess
+            self.q_curr = q_raw
+        else:  # Use previous ik solution as initial guess
+            if self.q_curr is None:
+                self.q_curr = q_raw
+
+        # Build desired 4x4 transform from pos + rotvec (twist)
+        t_des = np.eye(4, dtype=float)
+        t_des[:3, :3] = Rotation.from_rotvec([wx, wy, wz]).as_matrix()
+        t_des[:3, 3] = [x, y, z]
+
+        # Compute inverse kinematics
+        q_target = self.kinematics.inverse_kinematics(self.q_curr, t_des)
+        self.q_curr = q_target
+
+        # TODO: This is sentitive to order of motor_names = q_target mapping
+        for i, name in enumerate(self.motor_names):
+            if name != "gripper":
+                action[f"{name}.pos"] = float(q_target[i])
+            else:
+                action["gripper.pos"] = float(gripper_pos)
+
+        return action
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        for feat in ["x", "y", "z", "wx", "wy", "wz", "gripper_pos"]:
+            features[PipelineFeatureType.ACTION].pop(f"ee.{feat}", None)
+
+        for name in self.motor_names:
+            features[PipelineFeatureType.ACTION][f"{name}.pos"] = PolicyFeature(
+                type=FeatureType.ACTION, shape=(1,)
+            )
+
+        return features
+
+    def reset(self):
+        """Resets the initial guess for the IK solver."""
+        self.q_curr = None
+
+
+@ProcessorStepRegistry.register("gripper_velocity_to_joint")
+@dataclass
+class GripperVelocityToJoint(RobotActionProcessorStep):
+    """
+    Converts a gripper velocity command into a target gripper joint position.
+
+    This step integrates a normalized velocity command over time to produce a position command,
+    taking the current gripper position as a starting point. It also supports a discrete mode
+    where integer actions map to open, close, or no-op.
+
+    Attributes:
+        motor_names: A list of motor names, which must include 'gripper'.
+        speed_factor: A scaling factor to convert the normalized velocity command to a position change.
+        clip_min: The minimum allowed gripper joint position.
+        clip_max: The maximum allowed gripper joint position.
+        discrete_gripper: If True, treat the input action as discrete (0: open, 1: close, 2: stay).
+    """
+
+    speed_factor: float = 20.0
+    clip_min: float = 0.0
+    clip_max: float = 100.0
+    discrete_gripper: bool = False
+
+    def action(self, action: RobotAction) -> RobotAction:
+        observation = self.transition.get(TransitionKey.OBSERVATION).copy()
+
+        gripper_vel = action.pop("ee.gripper_vel")
+
+        if observation is None:
+            raise ValueError("Joints observation is require for computing robot kinematics")
+
+        q_raw = np.array(
+            [float(v) for k, v in observation.items() if isinstance(k, str) and k.endswith(".pos")],
+            dtype=float,
+        )
+        if q_raw is None:
+            raise ValueError("Joints observation is require for computing robot kinematics")
+
+        if self.discrete_gripper:
+            # Discrete gripper actions are in [0, 1, 2]
+            # 0: open, 1: close, 2: stay
+            # We need to shift them to [-1, 0, 1] and then scale them to clip_max
+            gripper_vel = (gripper_vel - 1) * self.clip_max
+
+        # Compute desired gripper position
+        delta = gripper_vel * float(self.speed_factor)
+        # TODO: This assumes gripper is the last specified joint in the robot
+        gripper_pos = float(np.clip(q_raw[-1] + delta, self.clip_min, self.clip_max))
+        action["ee.gripper_pos"] = gripper_pos
+
+        return action
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        features[PipelineFeatureType.ACTION].pop("ee.gripper_vel", None)
+        features[PipelineFeatureType.ACTION]["ee.gripper_pos"] = PolicyFeature(
+            type=FeatureType.ACTION, shape=(1,)
+        )
+
+        return features
+
+
+def compute_forward_kinematics_joints_to_ee(
+    joints: dict[str, Any], kinematics: RobotKinematics, motor_names: list[str]
+) -> dict[str, Any]:
+    motor_joint_values = [joints[f"{n}.pos"] for n in motor_names]
+
+    q = np.array(motor_joint_values, dtype=float)
+    t = kinematics.forward_kinematics(q)
+    pos = t[:3, 3]
+    tw = Rotation.from_matrix(t[:3, :3]).as_rotvec()
+    gripper_pos = joints["gripper.pos"]
+    for n in motor_names:
+        joints.pop(f"{n}.pos")
+    joints["ee.x"] = float(pos[0])
+    joints["ee.y"] = float(pos[1])
+    joints["ee.z"] = float(pos[2])
+    joints["ee.wx"] = float(tw[0])
+    joints["ee.wy"] = float(tw[1])
+    joints["ee.wz"] = float(tw[2])
+    joints["ee.gripper_pos"] = float(gripper_pos)
+    return joints
+
+
+@ProcessorStepRegistry.register("forward_kinematics_joints_to_ee_observation")
+@dataclass
+class ForwardKinematicsJointsToEEObservation(ObservationProcessorStep):
+    """
+    Computes the end-effector pose from joint positions using forward kinematics (FK).
+
+    This step is typically used to add the robot's Cartesian pose to the observation space,
+    which can be useful for visualization or as an input to a policy.
+
+    Attributes:
+        kinematics: The robot's kinematic model.
+    """
+
+    kinematics: RobotKinematics
+    motor_names: list[str]
+
+    def observation(self, observation: RobotObservation) -> RobotObservation:
+        return compute_forward_kinematics_joints_to_ee(observation, self.kinematics, self.motor_names)
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We only use the ee pose in the dataset, so we don't need the joint positions
+        for n in self.motor_names:
+            features[PipelineFeatureType.OBSERVATION].pop(f"{n}.pos", None)
+        # We specify the dataset features of this step that we want to be stored in the dataset
+        for k in ["x", "y", "z", "wx", "wy", "wz", "gripper_pos"]:
+            features[PipelineFeatureType.OBSERVATION][f"ee.{k}"] = PolicyFeature(
+                type=FeatureType.STATE, shape=(1,)
+            )
+        return features
+
+
+@ProcessorStepRegistry.register("forward_kinematics_joints_to_ee_action")
+@dataclass
+class ForwardKinematicsJointsToEEAction(RobotActionProcessorStep):
+    """
+    Computes the end-effector pose from joint positions using forward kinematics (FK).
+
+    This step is typically used to add the robot's Cartesian pose to the observation space,
+    which can be useful for visualization or as an input to a policy.
+
+    Attributes:
+        kinematics: The robot's kinematic model.
+    """
+
+    kinematics: RobotKinematics
+    motor_names: list[str]
+
+    def action(self, action: RobotAction) -> RobotAction:
+        return compute_forward_kinematics_joints_to_ee(action, self.kinematics, self.motor_names)
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We only use the ee pose in the dataset, so we don't need the joint positions
+        for n in self.motor_names:
+            features[PipelineFeatureType.ACTION].pop(f"{n}.pos", None)
+        # We specify the dataset features of this step that we want to be stored in the dataset
+        for k in ["x", "y", "z", "wx", "wy", "wz", "gripper_pos"]:
+            features[PipelineFeatureType.ACTION][f"ee.{k}"] = PolicyFeature(
+                type=FeatureType.STATE, shape=(1,)
+            )
+        return features
+
+
+@ProcessorStepRegistry.register(name="forward_kinematics_joints_to_ee")
+@dataclass
+class ForwardKinematicsJointsToEE(ProcessorStep):
+    kinematics: RobotKinematics
+    motor_names: list[str]
+
+    def __post_init__(self):
+        self.joints_to_ee_action_processor = ForwardKinematicsJointsToEEAction(
+            kinematics=self.kinematics, motor_names=self.motor_names
+        )
+        self.joints_to_ee_observation_processor = ForwardKinematicsJointsToEEObservation(
+            kinematics=self.kinematics, motor_names=self.motor_names
+        )
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        if transition.get(TransitionKey.ACTION) is not None:
+            transition = self.joints_to_ee_action_processor(transition)
+        if transition.get(TransitionKey.OBSERVATION) is not None:
+            transition = self.joints_to_ee_observation_processor(transition)
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        if features[PipelineFeatureType.ACTION] is not None:
+            features = self.joints_to_ee_action_processor.transform_features(features)
+        if features[PipelineFeatureType.OBSERVATION] is not None:
+            features = self.joints_to_ee_observation_processor.transform_features(features)
+        return features
+
+
+@ProcessorStepRegistry.register("inverse_kinematics_rl_step")
+@dataclass
+class InverseKinematicsRLStep(ProcessorStep):
+    """
+    Computes desired joint positions from a target end-effector pose using inverse kinematics (IK).
+
+    This is modified from the InverseKinematicsEEToJoints step to be used in the RL pipeline.
+    """
+
+    kinematics: RobotKinematics
+    motor_names: list[str]
+    q_curr: np.ndarray | None = field(default=None, init=False, repr=False)
+    initial_guess_current_joints: bool = True
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        new_transition = dict(transition)
+        action = new_transition.get(TransitionKey.ACTION)
+        if action is None:
+            raise ValueError("Action is required for InverseKinematicsEEToJoints")
+        action = dict(action)
+
+        x = action.pop("ee.x")
+        y = action.pop("ee.y")
+        z = action.pop("ee.z")
+        wx = action.pop("ee.wx")
+        wy = action.pop("ee.wy")
+        wz = action.pop("ee.wz")
+        gripper_pos = action.pop("ee.gripper_pos")
+
+        if None in (x, y, z, wx, wy, wz, gripper_pos):
+            raise ValueError(
+                "Missing required end-effector pose components: ee.x, ee.y, ee.z, ee.wx, ee.wy, ee.wz, ee.gripper_pos must all be present in action"
+            )
+
+        observation = new_transition.get(TransitionKey.OBSERVATION).copy()
+        if observation is None:
+            raise ValueError("Joints observation is require for computing robot kinematics")
+
+        q_raw = np.array(
+            [float(v) for k, v in observation.items() if isinstance(k, str) and k.endswith(".pos")],
+            dtype=float,
+        )
+        if q_raw is None:
+            raise ValueError("Joints observation is require for computing robot kinematics")
+
+        if self.initial_guess_current_joints:  # Use current joints as initial guess
+            self.q_curr = q_raw
+        else:  # Use previous ik solution as initial guess
+            if self.q_curr is None:
+                self.q_curr = q_raw
+
+        # Build desired 4x4 transform from pos + rotvec (twist)
+        t_des = np.eye(4, dtype=float)
+        t_des[:3, :3] = Rotation.from_rotvec([wx, wy, wz]).as_matrix()
+        t_des[:3, 3] = [x, y, z]
+
+        # Compute inverse kinematics
+        q_target = self.kinematics.inverse_kinematics(self.q_curr, t_des)
+        self.q_curr = q_target
+
+        # TODO: This is sentitive to order of motor_names = q_target mapping
+        for i, name in enumerate(self.motor_names):
+            if name != "gripper":
+                action[f"{name}.pos"] = float(q_target[i])
+            else:
+                action["gripper.pos"] = float(gripper_pos)
+
+        new_transition[TransitionKey.ACTION] = action
+        complementary_data = new_transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+        complementary_data["IK_solution"] = q_target
+        new_transition[TransitionKey.COMPLEMENTARY_DATA] = complementary_data
+        return new_transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        for feat in ["x", "y", "z", "wx", "wy", "wz", "gripper_pos"]:
+            features[PipelineFeatureType.ACTION].pop(f"ee.{feat}", None)
+
+        for name in self.motor_names:
+            features[PipelineFeatureType.ACTION][f"{name}.pos"] = PolicyFeature(
+                type=FeatureType.ACTION, shape=(1,)
+            )
+
+        return features
+
+    def reset(self):
+        """Resets the initial guess for the IK solver."""
+        self.q_curr = None
diff --git a/lerobot/src/lerobot/robots/so_follower/so100.md b/lerobot/src/lerobot/robots/so_follower/so100.md
new file mode 120000
index 0000000000000000000000000000000000000000..ad1154e75a74a496aa74cb1ac1b545238d5174e4
--- /dev/null
+++ b/lerobot/src/lerobot/robots/so_follower/so100.md
@@ -0,0 +1 @@
+../../../../docs/source/so100.mdx
\ No newline at end of file
diff --git a/lerobot/src/lerobot/robots/so_follower/so101.md b/lerobot/src/lerobot/robots/so_follower/so101.md
new file mode 120000
index 0000000000000000000000000000000000000000..27b89266029afbf0aa59be195cc0b4b6ee93ac26
--- /dev/null
+++ b/lerobot/src/lerobot/robots/so_follower/so101.md
@@ -0,0 +1 @@
+../../../../docs/source/so101.mdx
\ No newline at end of file
diff --git a/lerobot/src/lerobot/robots/so_follower/so_follower.py b/lerobot/src/lerobot/robots/so_follower/so_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..ca132d1022a6cc6b61c741f89a65e1c1a8becd40
--- /dev/null
+++ b/lerobot/src/lerobot/robots/so_follower/so_follower.py
@@ -0,0 +1,233 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from functools import cached_property
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.feetech import (
+    FeetechMotorsBus,
+    OperatingMode,
+)
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from ..utils import ensure_safe_goal_position
+from .config_so_follower import SOFollowerRobotConfig
+
+logger = logging.getLogger(__name__)
+
+
+class SOFollower(Robot):
+    """
+    Generic SO follower base implementing common functionality for SO-100/101/10X.
+    Designed to be subclassed with a per-hardware-model `config_class` and `name`.
+    """
+
+    config_class = SOFollowerRobotConfig
+    name = "so_follower"
+
+    def __init__(self, config: SOFollowerRobotConfig):
+        super().__init__(config)
+        self.config = config
+        # choose normalization mode depending on config if available
+        norm_mode_body = MotorNormMode.DEGREES if config.use_degrees else MotorNormMode.RANGE_M100_100
+        self.bus = FeetechMotorsBus(
+            port=self.config.port,
+            motors={
+                "shoulder_pan": Motor(1, "sts3215", norm_mode_body),
+                "shoulder_lift": Motor(2, "sts3215", norm_mode_body),
+                "elbow_flex": Motor(3, "sts3215", norm_mode_body),
+                "wrist_flex": Motor(4, "sts3215", norm_mode_body),
+                "wrist_roll": Motor(5, "sts3215", norm_mode_body),
+                "gripper": Motor(6, "sts3215", MotorNormMode.RANGE_0_100),
+            },
+            calibration=self.calibration,
+        )
+        self.cameras = make_cameras_from_configs(config.cameras)
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.bus.motors}
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected and all(cam.is_connected for cam in self.cameras.values())
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        """
+        We assume that at connection time, arm is in a rest position,
+        and torque can be safely disabled to run calibration.
+        """
+
+        self.bus.connect()
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        for cam in self.cameras.values():
+            cam.connect()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        if self.calibration:
+            # Calibration file exists, ask user whether to use it or run new calibration
+            user_input = input(
+                f"Press ENTER to use provided calibration file associated with the id {self.id}, or type 'c' and press ENTER to run calibration: "
+            )
+            if user_input.strip().lower() != "c":
+                logger.info(f"Writing calibration file associated with the id {self.id} to the motors")
+                self.bus.write_calibration(self.calibration)
+                return
+
+        logger.info(f"\nRunning calibration of {self}")
+        self.bus.disable_torque()
+        for motor in self.bus.motors:
+            self.bus.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+
+        input(f"Move {self} to the middle of its range of motion and press ENTER....")
+        homing_offsets = self.bus.set_half_turn_homings()
+
+        # Attempt to call record_ranges_of_motion with a reduced motor set when appropriate.
+        full_turn_motor = "wrist_roll"
+        unknown_range_motors = [motor for motor in self.bus.motors if motor != full_turn_motor]
+        print(
+            f"Move all joints except '{full_turn_motor}' sequentially through their "
+            "entire ranges of motion.\nRecording positions. Press ENTER to stop..."
+        )
+        range_mins, range_maxes = self.bus.record_ranges_of_motion(unknown_range_motors)
+        range_mins[full_turn_motor] = 0
+        range_maxes[full_turn_motor] = 4095
+
+        self.calibration = {}
+        for motor, m in self.bus.motors.items():
+            self.calibration[motor] = MotorCalibration(
+                id=m.id,
+                drive_mode=0,
+                homing_offset=homing_offsets[motor],
+                range_min=range_mins[motor],
+                range_max=range_maxes[motor],
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        print("Calibration saved to", self.calibration_fpath)
+
+    def configure(self) -> None:
+        with self.bus.torque_disabled():
+            self.bus.configure_motors()
+            for motor in self.bus.motors:
+                self.bus.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+                # Set P_Coefficient to lower value to avoid shakiness (Default is 32)
+                self.bus.write("P_Coefficient", motor, 16)
+                # Set I_Coefficient and D_Coefficient to default value 0 and 32
+                self.bus.write("I_Coefficient", motor, 0)
+                self.bus.write("D_Coefficient", motor, 32)
+
+                if motor == "gripper":
+                    self.bus.write("Max_Torque_Limit", motor, 500)  # 50% of max torque to avoid burnout
+                    self.bus.write("Protection_Current", motor, 250)  # 50% of max current to avoid burnout
+                    self.bus.write("Overload_Torque", motor, 25)  # 25% torque when overloaded
+
+    def setup_motors(self) -> None:
+        for motor in reversed(self.bus.motors):
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        # Read arm position
+        start = time.perf_counter()
+        obs_dict = self.bus.sync_read("Present_Position")
+        obs_dict = {f"{motor}.pos": val for motor, val in obs_dict.items()}
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read state: {dt_ms:.1f}ms")
+
+        # Capture images from cameras
+        for cam_key, cam in self.cameras.items():
+            start = time.perf_counter()
+            obs_dict[cam_key] = cam.read_latest()
+            dt_ms = (time.perf_counter() - start) * 1e3
+            logger.debug(f"{self} read {cam_key}: {dt_ms:.1f}ms")
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        """Command arm to move to a target joint configuration.
+
+        The relative action magnitude may be clipped depending on the configuration parameter
+        `max_relative_target`. In this case, the action sent differs from original action.
+        Thus, this function always returns the action actually sent.
+
+        Raises:
+            RobotDeviceNotConnectedError: if robot is not connected.
+
+        Returns:
+            RobotAction: the action sent to the motors, potentially clipped.
+        """
+
+        goal_pos = {key.removesuffix(".pos"): val for key, val in action.items() if key.endswith(".pos")}
+
+        # Cap goal position when too far away from present position.
+        # /!\ Slower fps expected due to reading from the follower.
+        if self.config.max_relative_target is not None:
+            present_pos = self.bus.sync_read("Present_Position")
+            goal_present_pos = {key: (g_pos, present_pos[key]) for key, g_pos in goal_pos.items()}
+            goal_pos = ensure_safe_goal_position(goal_present_pos, self.config.max_relative_target)
+
+        # Send goal position to the arm
+        self.bus.sync_write("Goal_Position", goal_pos)
+        return {f"{motor}.pos": val for motor, val in goal_pos.items()}
+
+    @check_if_not_connected
+    def disconnect(self):
+        self.bus.disconnect(self.config.disable_torque_on_disconnect)
+        for cam in self.cameras.values():
+            cam.disconnect()
+
+        logger.info(f"{self} disconnected.")
+
+
+SO100Follower = SOFollower
+SO101Follower = SOFollower
diff --git a/lerobot/src/lerobot/robots/unitree_g1/__init__.py b/lerobot/src/lerobot/robots/unitree_g1/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..ef3a9d05ec3b91a6553298ec2bbd0957248cb06a
--- /dev/null
+++ b/lerobot/src/lerobot/robots/unitree_g1/__init__.py
@@ -0,0 +1,20 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_unitree_g1 import UnitreeG1Config
+from .unitree_g1 import UnitreeG1
+
+__all__ = ["UnitreeG1", "UnitreeG1Config"]
diff --git a/lerobot/src/lerobot/robots/unitree_g1/config_unitree_g1.py b/lerobot/src/lerobot/robots/unitree_g1/config_unitree_g1.py
new file mode 100644
index 0000000000000000000000000000000000000000..b786c2a337cb347e8733eae7a4ecf23eff97dd9e
--- /dev/null
+++ b/lerobot/src/lerobot/robots/unitree_g1/config_unitree_g1.py
@@ -0,0 +1,73 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.cameras import CameraConfig
+
+from ..config import RobotConfig
+
+_GAINS: dict[str, dict[str, list[float]]] = {
+    "left_leg": {
+        "kp": [150, 150, 150, 300, 40, 40],
+        "kd": [2, 2, 2, 4, 2, 2],
+    },  # pitch, roll, yaw, knee, ankle_pitch, ankle_roll
+    "right_leg": {"kp": [150, 150, 150, 300, 40, 40], "kd": [2, 2, 2, 4, 2, 2]},
+    "waist": {"kp": [250, 250, 250], "kd": [5, 5, 5]},  # yaw, roll, pitch
+    "left_arm": {"kp": [50, 50, 80, 80], "kd": [3, 3, 3, 3]},  # shoulder_pitch/roll/yaw, elbow
+    "left_wrist": {"kp": [40, 40, 40], "kd": [1.5, 1.5, 1.5]},  # roll, pitch, yaw
+    "right_arm": {"kp": [50, 50, 80, 80], "kd": [3, 3, 3, 3]},
+    "right_wrist": {"kp": [40, 40, 40], "kd": [1.5, 1.5, 1.5]},
+}
+
+
+def _build_gains() -> tuple[list[float], list[float]]:
+    """Build kp and kd lists from body-part groupings."""
+    kp = [v for g in _GAINS.values() for v in g["kp"]]
+    kd = [v for g in _GAINS.values() for v in g["kd"]]
+    return kp, kd
+
+
+_DEFAULT_KP, _DEFAULT_KD = _build_gains()
+
+
+@RobotConfig.register_subclass("unitree_g1")
+@dataclass
+class UnitreeG1Config(RobotConfig):
+    kp: list[float] = field(default_factory=lambda: _DEFAULT_KP.copy())
+    kd: list[float] = field(default_factory=lambda: _DEFAULT_KD.copy())
+
+    # Default joint positions
+    default_positions: list[float] = field(default_factory=lambda: [0.0] * 29)
+
+    # Control loop timestep
+    control_dt: float = 1.0 / 250.0  # 250Hz
+
+    # Launch mujoco simulation
+    is_simulation: bool = True
+
+    # Socket config for ZMQ bridge
+    robot_ip: str = "192.168.123.164"  # default G1 IP
+
+    # Cameras (ZMQ-based remote cameras)
+    cameras: dict[str, CameraConfig] = field(default_factory=dict)
+
+    # Compensates for gravity on the unitree's arms using the arm ik solver
+    gravity_compensation: bool = False
+
+    # Lower-body controller class name, e.g. "GrootLocomotionController" or
+    # "HolosomaLocomotionController". None disables it.
+    controller: str | None = None
diff --git a/lerobot/src/lerobot/robots/unitree_g1/g1_kinematics.py b/lerobot/src/lerobot/robots/unitree_g1/g1_kinematics.py
new file mode 100644
index 0000000000000000000000000000000000000000..f57320a1173481b5f901a10a44a1013d1d482091
--- /dev/null
+++ b/lerobot/src/lerobot/robots/unitree_g1/g1_kinematics.py
@@ -0,0 +1,287 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import os
+from collections import deque
+
+import numpy as np
+
+logger = logging.getLogger(__name__)
+
+
+class WeightedMovingFilter:
+    def __init__(self, weights, data_size=14):
+        self._window_size = len(weights)
+        self._weights = np.array(weights)
+        self._data_size = data_size
+        self._filtered_data = np.zeros(self._data_size)
+        self._data_queue = deque(maxlen=self._window_size)
+
+    def _apply_filter(self):
+        if len(self._data_queue) < self._window_size:
+            return self._data_queue[-1]
+
+        data_array = np.array(self._data_queue)
+        return data_array.T @ self._weights
+
+    def add_data(self, new_data):
+        assert len(new_data) == self._data_size
+
+        if len(self._data_queue) > 0 and np.array_equal(
+            new_data, self._data_queue[-1]
+        ):  # skip duplicate data
+            return
+
+        self._data_queue.append(new_data)
+        self._filtered_data = self._apply_filter()
+
+    @property
+    def filtered_data(self):
+        return self._filtered_data
+
+
+class G1_29_ArmIK:  # noqa: N801
+    def __init__(self, unit_test=False):
+        import casadi
+        import pinocchio as pin
+        from huggingface_hub import snapshot_download
+        from pinocchio import casadi as cpin
+
+        self._pin = pin
+        self.unit_test = unit_test
+
+        self.repo_path = snapshot_download("lerobot/unitree-g1-mujoco")
+        urdf_path = os.path.join(self.repo_path, "assets", "g1_body29_hand14.urdf")
+        mesh_dir = os.path.join(self.repo_path, "assets")
+
+        self.robot = self._pin.RobotWrapper.BuildFromURDF(urdf_path, mesh_dir)
+
+        self.mixed_jointsToLockIDs = [
+            "left_hip_pitch_joint",
+            "left_hip_roll_joint",
+            "left_hip_yaw_joint",
+            "left_knee_joint",
+            "left_ankle_pitch_joint",
+            "left_ankle_roll_joint",
+            "right_hip_pitch_joint",
+            "right_hip_roll_joint",
+            "right_hip_yaw_joint",
+            "right_knee_joint",
+            "right_ankle_pitch_joint",
+            "right_ankle_roll_joint",
+            "waist_yaw_joint",
+            "waist_roll_joint",
+            "waist_pitch_joint",
+            "left_hand_thumb_0_joint",
+            "left_hand_thumb_1_joint",
+            "left_hand_thumb_2_joint",
+            "left_hand_middle_0_joint",
+            "left_hand_middle_1_joint",
+            "left_hand_index_0_joint",
+            "left_hand_index_1_joint",
+            "right_hand_thumb_0_joint",
+            "right_hand_thumb_1_joint",
+            "right_hand_thumb_2_joint",
+            "right_hand_index_0_joint",
+            "right_hand_index_1_joint",
+            "right_hand_middle_0_joint",
+            "right_hand_middle_1_joint",
+        ]
+
+        self.reduced_robot = self.robot.buildReducedRobot(
+            list_of_joints_to_lock=self.mixed_jointsToLockIDs,
+            reference_configuration=np.array([0.0] * self.robot.model.nq),
+        )
+
+        # Arm joint names in G1 motor order (G1_29_JointArmIndex)
+        self._arm_joint_names_g1 = [
+            "left_shoulder_pitch_joint",
+            "left_shoulder_roll_joint",
+            "left_shoulder_yaw_joint",
+            "left_elbow_joint",
+            "left_wrist_roll_joint",
+            "left_wrist_pitch_joint",
+            "left_wrist_yaw_joint",
+            "right_shoulder_pitch_joint",
+            "right_shoulder_roll_joint",
+            "right_shoulder_yaw_joint",
+            "right_elbow_joint",
+            "right_wrist_roll_joint",
+            "right_wrist_pitch_joint",
+            "right_wrist_yaw_joint",
+        ]
+        # Pinocchio uses its own joint order in q; build index mapping.
+        self._arm_joint_names_pin = sorted(
+            self._arm_joint_names_g1,
+            key=lambda name: self.reduced_robot.model.idx_qs[self.reduced_robot.model.getJointId(name)],
+        )
+        logger.info(f"Pinocchio arm joint order: {self._arm_joint_names_pin}")
+        self._arm_reorder_g1_to_pin = [
+            self._arm_joint_names_g1.index(name) for name in self._arm_joint_names_pin
+        ]
+        # Inverse mapping to return tau in G1 motor order.
+        self._arm_reorder_pin_to_g1 = np.argsort(self._arm_reorder_g1_to_pin)
+
+        self.reduced_robot.model.addFrame(
+            self._pin.Frame(
+                "L_ee",
+                self.reduced_robot.model.getJointId("left_wrist_yaw_joint"),
+                self._pin.SE3(np.eye(3), np.array([0.05, 0, 0]).T),
+                self._pin.FrameType.OP_FRAME,
+            )
+        )
+
+        self.reduced_robot.model.addFrame(
+            self._pin.Frame(
+                "R_ee",
+                self.reduced_robot.model.getJointId("right_wrist_yaw_joint"),
+                self._pin.SE3(np.eye(3), np.array([0.05, 0, 0]).T),
+                self._pin.FrameType.OP_FRAME,
+            )
+        )
+
+        # Creating Casadi models and data for symbolic computing
+        self.cmodel = cpin.Model(self.reduced_robot.model)
+        self.cdata = self.cmodel.createData()
+
+        # Creating symbolic variables
+        self.cq = casadi.SX.sym("q", self.reduced_robot.model.nq, 1)
+        self.cTf_l = casadi.SX.sym("tf_l", 4, 4)
+        self.cTf_r = casadi.SX.sym("tf_r", 4, 4)
+        cpin.framesForwardKinematics(self.cmodel, self.cdata, self.cq)
+
+        # Get the hand joint ID and define the error function
+        self.L_hand_id = self.reduced_robot.model.getFrameId("L_ee")
+        self.R_hand_id = self.reduced_robot.model.getFrameId("R_ee")
+
+        self.translational_error = casadi.Function(
+            "translational_error",
+            [self.cq, self.cTf_l, self.cTf_r],
+            [
+                casadi.vertcat(
+                    self.cdata.oMf[self.L_hand_id].translation - self.cTf_l[:3, 3],
+                    self.cdata.oMf[self.R_hand_id].translation - self.cTf_r[:3, 3],
+                )
+            ],
+        )
+        self.rotational_error = casadi.Function(
+            "rotational_error",
+            [self.cq, self.cTf_l, self.cTf_r],
+            [
+                casadi.vertcat(
+                    cpin.log3(self.cdata.oMf[self.L_hand_id].rotation @ self.cTf_l[:3, :3].T),
+                    cpin.log3(self.cdata.oMf[self.R_hand_id].rotation @ self.cTf_r[:3, :3].T),
+                )
+            ],
+        )
+
+        # Defining the optimization problem
+        self.opti = casadi.Opti()
+        self.var_q = self.opti.variable(self.reduced_robot.model.nq)
+        self.var_q_last = self.opti.parameter(self.reduced_robot.model.nq)  # for smooth
+        self.param_tf_l = self.opti.parameter(4, 4)
+        self.param_tf_r = self.opti.parameter(4, 4)
+        self.translational_cost = casadi.sumsqr(
+            self.translational_error(self.var_q, self.param_tf_l, self.param_tf_r)
+        )
+        self.rotation_cost = casadi.sumsqr(
+            self.rotational_error(self.var_q, self.param_tf_l, self.param_tf_r)
+        )
+        self.regularization_cost = casadi.sumsqr(self.var_q)
+        self.smooth_cost = casadi.sumsqr(self.var_q - self.var_q_last)
+
+        # Setting optimization constraints and goals
+        self.opti.subject_to(
+            self.opti.bounded(
+                self.reduced_robot.model.lowerPositionLimit,
+                self.var_q,
+                self.reduced_robot.model.upperPositionLimit,
+            )
+        )
+        self.opti.minimize(
+            50 * self.translational_cost
+            + self.rotation_cost
+            + 0.02 * self.regularization_cost
+            + 0.1 * self.smooth_cost
+        )
+
+        opts = {
+            "ipopt": {"print_level": 0, "max_iter": 50, "tol": 1e-6},
+            "print_time": False,  # print or not
+            "calc_lam_p": False,  # https://github.com/casadi/casadi/wiki/FAQ:-Why-am-I-getting-%22NaN-detected%22in-my-optimization%3F
+        }
+        self.opti.solver("ipopt", opts)
+
+        self.init_data = np.zeros(self.reduced_robot.model.nq)
+        self.smooth_filter = WeightedMovingFilter(np.array([0.4, 0.3, 0.2, 0.1]), 14)
+
+    def solve_ik(self, left_wrist, right_wrist, current_lr_arm_motor_q=None, current_lr_arm_motor_dq=None):
+        if current_lr_arm_motor_q is not None:
+            self.init_data = current_lr_arm_motor_q
+        self.opti.set_initial(self.var_q, self.init_data)
+
+        self.opti.set_value(self.param_tf_l, left_wrist)
+        self.opti.set_value(self.param_tf_r, right_wrist)
+        self.opti.set_value(self.var_q_last, self.init_data)  # for smooth
+
+        converged = True
+        try:
+            self.opti.solve()
+            sol_q = self.opti.value(self.var_q)
+        except Exception as e:
+            converged = False
+            logger.error(f"IK convergence error: {e}")
+            sol_q = self.opti.debug.value(self.var_q)
+
+        self.smooth_filter.add_data(sol_q)
+        sol_q = self.smooth_filter.filtered_data
+        self.init_data = sol_q
+
+        if not converged:
+            logger.error(
+                f"sol_q:{sol_q} \nmotorstate: \n{current_lr_arm_motor_q} \nleft_pose: \n{left_wrist} \nright_pose: \n{right_wrist}"
+            )
+            return current_lr_arm_motor_q, np.zeros(self.reduced_robot.model.nv)
+
+        sol_tauff = self._pin.rnea(
+            self.reduced_robot.model,
+            self.reduced_robot.data,
+            sol_q,
+            np.zeros(self.reduced_robot.model.nv),
+            np.zeros(self.reduced_robot.model.nv),
+        )
+
+        return sol_q, sol_tauff
+
+    def solve_tau(self, current_lr_arm_motor_q=None, current_lr_arm_motor_dq=None):
+        try:
+            q_g1 = np.array(current_lr_arm_motor_q, dtype=float)
+            if q_g1.shape[0] != len(self._arm_joint_names_g1):
+                raise ValueError(f"Expected {len(self._arm_joint_names_g1)} arm joints, got {q_g1.shape[0]}")
+            q_pin = q_g1[self._arm_reorder_g1_to_pin]
+            sol_tauff = self._pin.rnea(
+                self.reduced_robot.model,
+                self.reduced_robot.data,
+                q_pin,
+                np.zeros(self.reduced_robot.model.nv),
+                np.zeros(self.reduced_robot.model.nv),
+            )
+            return sol_tauff[self._arm_reorder_pin_to_g1]
+
+        except Exception as e:
+            logger.error(f"ERROR in convergence, plotting debug info.{e}")
+            return np.zeros(self.reduced_robot.model.nv)
diff --git a/lerobot/src/lerobot/robots/unitree_g1/g1_utils.py b/lerobot/src/lerobot/robots/unitree_g1/g1_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..91f009b26016901f8e07d941b946286e4de36424
--- /dev/null
+++ b/lerobot/src/lerobot/robots/unitree_g1/g1_utils.py
@@ -0,0 +1,118 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import importlib
+from enum import IntEnum
+
+import numpy as np
+
+# ruff: noqa: N801, N815
+
+NUM_MOTORS = 29
+
+REMOTE_AXES = ("remote.lx", "remote.ly", "remote.rx", "remote.ry")
+REMOTE_BUTTONS = tuple(f"remote.button.{i}" for i in range(16))
+REMOTE_KEYS = REMOTE_AXES + REMOTE_BUTTONS
+
+
+def default_remote_input() -> dict[str, float]:
+    """Return a zeroed-out remote input dict (axes + buttons)."""
+    return dict.fromkeys(REMOTE_KEYS, 0.0)
+
+
+def get_gravity_orientation(quaternion: list[float] | np.ndarray) -> np.ndarray:
+    """Get gravity orientation from quaternion [w, x, y, z]."""
+    qw, qx, qy, qz = quaternion
+    gravity_orientation = np.zeros(3, dtype=np.float32)
+    gravity_orientation[0] = 2 * (-qz * qx + qw * qy)
+    gravity_orientation[1] = -2 * (qz * qy + qw * qx)
+    gravity_orientation[2] = 1 - 2 * (qw * qw + qz * qz)
+    return gravity_orientation
+
+
+class G1_29_JointArmIndex(IntEnum):
+    # Left arm
+    kLeftShoulderPitch = 15
+    kLeftShoulderRoll = 16
+    kLeftShoulderYaw = 17
+    kLeftElbow = 18
+    kLeftWristRoll = 19
+    kLeftWristPitch = 20
+    kLeftWristYaw = 21
+
+    # Right arm
+    kRightShoulderPitch = 22
+    kRightShoulderRoll = 23
+    kRightShoulderYaw = 24
+    kRightElbow = 25
+    kRightWristRoll = 26
+    kRightWristPitch = 27
+    kRightWristYaw = 28
+
+
+def make_locomotion_controller(name: str | None):
+    """Instantiate a locomotion controller by class name. Returns None if name is None."""
+    if name is None:
+        return None
+    controllers = {
+        "GrootLocomotionController": "lerobot.robots.unitree_g1.gr00t_locomotion",
+        "HolosomaLocomotionController": "lerobot.robots.unitree_g1.holosoma_locomotion",
+    }
+    module_path = controllers.get(name)
+    if module_path is None:
+        raise ValueError(f"Unknown controller: {name!r}. Available: {list(controllers)}")
+    module = importlib.import_module(module_path)
+    return getattr(module, name)()
+
+
+class G1_29_JointIndex(IntEnum):
+    # Left leg
+    kLeftHipPitch = 0
+    kLeftHipRoll = 1
+    kLeftHipYaw = 2
+    kLeftKnee = 3
+    kLeftAnklePitch = 4
+    kLeftAnkleRoll = 5
+
+    # Right leg
+    kRightHipPitch = 6
+    kRightHipRoll = 7
+    kRightHipYaw = 8
+    kRightKnee = 9
+    kRightAnklePitch = 10
+    kRightAnkleRoll = 11
+
+    kWaistYaw = 12
+    kWaistRoll = 13
+    kWaistPitch = 14
+
+    # Left arm
+    kLeftShoulderPitch = 15
+    kLeftShoulderRoll = 16
+    kLeftShoulderYaw = 17
+    kLeftElbow = 18
+    kLeftWristRoll = 19
+    kLeftWristPitch = 20
+    kLeftWristYaw = 21
+
+    # Right arm
+    kRightShoulderPitch = 22
+    kRightShoulderRoll = 23
+    kRightShoulderYaw = 24
+    kRightElbow = 25
+    kRightWristRoll = 26
+    kRightWristPitch = 27
+    kRightWristYaw = 28
diff --git a/lerobot/src/lerobot/robots/unitree_g1/gr00t_locomotion.py b/lerobot/src/lerobot/robots/unitree_g1/gr00t_locomotion.py
new file mode 100644
index 0000000000000000000000000000000000000000..31166e123c4999bcddc2772747fe0b8cd163deb7
--- /dev/null
+++ b/lerobot/src/lerobot/robots/unitree_g1/gr00t_locomotion.py
@@ -0,0 +1,205 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+from collections import deque
+
+import numpy as np
+import onnxruntime as ort
+from huggingface_hub import hf_hub_download
+
+from lerobot.robots.unitree_g1.g1_utils import (
+    REMOTE_AXES,
+    REMOTE_BUTTONS,
+    G1_29_JointIndex,
+    get_gravity_orientation,
+)
+
+logger = logging.getLogger(__name__)
+
+
+GROOT_DEFAULT_ANGLES = np.zeros(29, dtype=np.float32)
+GROOT_DEFAULT_ANGLES[[0, 6]] = -0.1  # Hip pitch
+GROOT_DEFAULT_ANGLES[[3, 9]] = 0.3  # Knee
+GROOT_DEFAULT_ANGLES[[4, 10]] = -0.2  # Ankle pitch
+
+# Control parameters
+ACTION_SCALE = 0.25
+CONTROL_DT = 0.02  # 50Hz
+ANG_VEL_SCALE: float = 0.25
+DOF_POS_SCALE: float = 1.0
+DOF_VEL_SCALE: float = 0.05
+CMD_SCALE: list[float] = [2.0, 2.0, 0.25]
+
+
+DEFAULT_GROOT_REPO_ID = "nepyope/GR00T-WholeBodyControl_g1"
+
+
+def load_groot_policies(
+    repo_id: str = DEFAULT_GROOT_REPO_ID,
+) -> tuple[ort.InferenceSession, ort.InferenceSession]:
+    """Load GR00T dual-policy system (Balance + Walk) from the hub.
+
+    Args:
+        repo_id: Hugging Face Hub repository ID containing the ONNX policies.
+    """
+    logger.info(f"Loading GR00T dual-policy system from the hub ({repo_id})...")
+
+    # Download ONNX policies from Hugging Face Hub
+    balance_path = hf_hub_download(
+        repo_id=repo_id,
+        filename="GR00T-WholeBodyControl-Balance.onnx",
+    )
+    walk_path = hf_hub_download(
+        repo_id=repo_id,
+        filename="GR00T-WholeBodyControl-Walk.onnx",
+    )
+
+    # Load ONNX policies
+    policy_balance = ort.InferenceSession(balance_path)
+    policy_walk = ort.InferenceSession(walk_path)
+
+    logger.info("GR00T policies loaded successfully")
+
+    return policy_balance, policy_walk
+
+
+class GrootLocomotionController:
+    """GR00T lower-body locomotion controller for the Unitree G1."""
+
+    control_dt = CONTROL_DT  # Expose for unitree_g1.py
+
+    def __init__(self):
+        # Load policies
+        self.policy_balance, self.policy_walk = load_groot_policies()
+
+        self.cmd = np.array([0.0, 0.0, 0.0], dtype=np.float32)  # vx, vy, theta_dot
+
+        # Robot state
+        self.groot_qj_all = np.zeros(29, dtype=np.float32)
+        self.groot_dqj_all = np.zeros(29, dtype=np.float32)
+        self.groot_action = np.zeros(15, dtype=np.float32)
+        self.groot_obs_single = np.zeros(86, dtype=np.float32)
+        self.groot_obs_history = deque(maxlen=6)
+        self.groot_obs_stacked = np.zeros(516, dtype=np.float32)
+        self.groot_height_cmd = 0.74  # Default base height
+        self.groot_orientation_cmd = np.array([0.0, 0.0, 0.0], dtype=np.float32)
+
+        # Input to GR00T is 6 frames (6*86D=516)
+        for _ in range(6):
+            self.groot_obs_history.append(np.zeros(86, dtype=np.float32))
+
+        logger.info("GrootLocomotionController initialized")
+
+    def reset(self) -> None:
+        """Reset internal state for a new episode."""
+        self.cmd[:] = 0.0
+        self.groot_qj_all[:] = 0.0
+        self.groot_dqj_all[:] = 0.0
+        self.groot_action[:] = 0.0
+        self.groot_obs_single[:] = 0.0
+        self.groot_obs_stacked[:] = 0.0
+        self.groot_height_cmd = 0.74
+        self.groot_orientation_cmd[:] = 0.0
+        self.groot_obs_history.clear()
+        for _ in range(6):
+            self.groot_obs_history.append(np.zeros(86, dtype=np.float32))
+
+    def run_step(self, action: dict, lowstate) -> dict:
+        """Run one step of the locomotion controller.
+
+        Args:
+            action: Action dict containing remote.lx/ly/rx/ry and buttons
+            lowstate: Robot lowstate containing motor positions/velocities and IMU
+
+        Returns:
+            Action dict for lower body joints (0-14)
+        """
+        if lowstate is None:
+            return {}
+
+        buttons = [int(action.get(k, 0)) for k in REMOTE_BUTTONS]
+        if buttons[0]:  # R1 - raise waist
+            self.groot_height_cmd += 0.001
+            self.groot_height_cmd = np.clip(self.groot_height_cmd, 0.50, 1.00)
+        if buttons[4]:  # R2 - lower waist
+            self.groot_height_cmd -= 0.001
+            self.groot_height_cmd = np.clip(self.groot_height_cmd, 0.50, 1.00)
+
+        lx, ly, rx, _ry = (action.get(k, 0.0) for k in REMOTE_AXES)
+        self.cmd[0] = ly  # Forward/backward
+        self.cmd[1] = -lx  # Left/right (negated)
+        self.cmd[2] = -rx  # Rotation rate (negated)
+
+        # Get joint positions and velocities from lowstate
+        for motor in G1_29_JointIndex:
+            idx = motor.value
+            self.groot_qj_all[idx] = lowstate.motor_state[idx].q
+            self.groot_dqj_all[idx] = lowstate.motor_state[idx].dq
+
+        # Scale joint positions and velocities
+        qj_obs = self.groot_qj_all.copy()
+        dqj_obs = self.groot_dqj_all.copy()
+
+        # Express IMU data in gravity frame of reference
+        quat = lowstate.imu_state.quaternion
+        ang_vel = np.array(lowstate.imu_state.gyroscope, dtype=np.float32)
+        gravity_orientation = get_gravity_orientation(quat)
+
+        # Scale joint positions and velocities before policy inference
+        qj_obs = (qj_obs - GROOT_DEFAULT_ANGLES) * DOF_POS_SCALE
+        dqj_obs = dqj_obs * DOF_VEL_SCALE
+        ang_vel_scaled = ang_vel * ANG_VEL_SCALE
+
+        # Build single frame observation
+        self.groot_obs_single[:3] = self.cmd * np.array(CMD_SCALE)
+        self.groot_obs_single[3] = self.groot_height_cmd
+        self.groot_obs_single[4:7] = self.groot_orientation_cmd
+        self.groot_obs_single[7:10] = ang_vel_scaled
+        self.groot_obs_single[10:13] = gravity_orientation
+        self.groot_obs_single[13:42] = qj_obs
+        self.groot_obs_single[42:71] = dqj_obs
+        self.groot_obs_single[71:86] = self.groot_action  # 15D previous actions
+
+        # Add to history and stack observations (6 frames × 86D = 516D)
+        self.groot_obs_history.append(self.groot_obs_single.copy())
+
+        # Stack all 6 frames into 516D vector
+        for i, obs_frame in enumerate(self.groot_obs_history):
+            start_idx = i * 86
+            end_idx = start_idx + 86
+            self.groot_obs_stacked[start_idx:end_idx] = obs_frame
+
+        cmd_magnitude = np.linalg.norm(self.cmd)
+        selected_policy = (
+            self.policy_balance if cmd_magnitude < 0.05 else self.policy_walk
+        )  # Balance/standing policy for small commands, walking policy for movement commands
+
+        # Run policy inference
+        ort_inputs = {selected_policy.get_inputs()[0].name: np.expand_dims(self.groot_obs_stacked, axis=0)}
+        ort_outs = selected_policy.run(None, ort_inputs)
+        self.groot_action = ort_outs[0].squeeze()
+
+        # Transform action back to target joint positions
+        target_dof_pos_15 = GROOT_DEFAULT_ANGLES[:15] + self.groot_action * ACTION_SCALE
+
+        # Build action dict
+        action_dict = {}
+        for i in range(15):
+            motor_name = G1_29_JointIndex(i).name
+            action_dict[f"{motor_name}.q"] = float(target_dof_pos_15[i])
+
+        return action_dict
diff --git a/lerobot/src/lerobot/robots/unitree_g1/holosoma_locomotion.py b/lerobot/src/lerobot/robots/unitree_g1/holosoma_locomotion.py
new file mode 100644
index 0000000000000000000000000000000000000000..857bb97bcb4807cf24970e3c064280fb16484d37
--- /dev/null
+++ b/lerobot/src/lerobot/robots/unitree_g1/holosoma_locomotion.py
@@ -0,0 +1,214 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import json
+import logging
+
+import numpy as np
+import onnx
+import onnxruntime as ort
+from huggingface_hub import hf_hub_download
+
+from lerobot.robots.unitree_g1.g1_utils import (
+    REMOTE_AXES,
+    G1_29_JointArmIndex,
+    G1_29_JointIndex,
+    get_gravity_orientation,
+)
+
+logger = logging.getLogger(__name__)
+
+DEFAULT_ANGLES = np.zeros(29, dtype=np.float32)
+DEFAULT_ANGLES[[0, 6]] = -0.312  # Hip pitch
+DEFAULT_ANGLES[[3, 9]] = 0.669  # Knee
+DEFAULT_ANGLES[[4, 10]] = -0.363  # Ankle pitch
+DEFAULT_ANGLES[[15, 22]] = 0.2  # Shoulder pitch
+DEFAULT_ANGLES[16] = 0.2  # Left shoulder roll
+DEFAULT_ANGLES[23] = -0.2  # Right shoulder roll
+DEFAULT_ANGLES[[18, 25]] = 0.6  # Elbow
+
+# Control parameters
+ACTION_SCALE = 0.25
+CONTROL_DT = 0.005  # 200Hz
+ANG_VEL_SCALE = 0.25
+DOF_POS_SCALE = 1.0
+DOF_VEL_SCALE = 0.05
+GAIT_PERIOD = 0.5
+
+
+DEFAULT_HOLOSOMA_REPO_ID = "nepyope/holosoma_locomotion"
+
+# Policy filename mapping
+POLICY_FILES = {
+    "fastsac": "fastsac_g1_29dof.onnx",
+    "ppo": "ppo_g1_29dof.onnx",
+}
+
+
+def load_policy(
+    repo_id: str = DEFAULT_HOLOSOMA_REPO_ID,
+    policy_type: str = "fastsac",
+) -> tuple[ort.InferenceSession, np.ndarray, np.ndarray]:
+    """Load Holosoma locomotion policy and extract KP/KD from metadata.
+
+    Args:
+        repo_id: Hugging Face Hub repo ID
+        policy_type: Either "fastsac" (default) or "ppo"
+
+    Returns:
+        (policy, kp, kd) tuple
+    """
+    if policy_type not in POLICY_FILES:
+        raise ValueError(f"Unknown policy type: {policy_type}. Choose from: {list(POLICY_FILES.keys())}")
+
+    filename = POLICY_FILES[policy_type]
+    logger.info(f"Loading {policy_type.upper()} policy from: {repo_id}/{filename}")
+    policy_path = hf_hub_download(repo_id=repo_id, filename=filename)
+
+    policy = ort.InferenceSession(policy_path)
+    logger.info(f"Policy loaded: {policy.get_inputs()[0].shape} → {policy.get_outputs()[0].shape}")
+
+    # Extract KP/KD from ONNX metadata
+    model = onnx.load(policy_path, load_external_data=False)
+    metadata = {prop.key: prop.value for prop in model.metadata_props}
+
+    if "kp" not in metadata or "kd" not in metadata:
+        raise ValueError("ONNX model must contain 'kp' and 'kd' in metadata")
+
+    kp = np.array(json.loads(metadata["kp"]), dtype=np.float32)
+    kd = np.array(json.loads(metadata["kd"]), dtype=np.float32)
+    logger.info(f"Loaded KP/KD from ONNX ({len(kp)} joints)")
+
+    return policy, kp, kd
+
+
+class HolosomaLocomotionController:
+    """Holosoma lower-body locomotion controller for Unitree G1."""
+
+    control_dt = CONTROL_DT  # Expose for unitree_g1.py
+
+    def __init__(self):
+        # Load policy and gains
+        self.policy, self.kp, self.kd = load_policy()
+
+        self.cmd = np.zeros(3, dtype=np.float32)
+
+        # Robot state
+        self.qj = np.zeros(29, dtype=np.float32)
+        self.dqj = np.zeros(29, dtype=np.float32)
+        self.obs = np.zeros(100, dtype=np.float32)
+        self.last_action = np.zeros(29, dtype=np.float32)
+
+        # Gait phase
+        self.phase = np.array([[0.0, np.pi]], dtype=np.float32)
+        self.phase_dt = 2 * np.pi / ((1.0 / CONTROL_DT) * GAIT_PERIOD)
+        self.is_standing = True
+
+        logger.info("HolosomaLocomotionController initialized")
+
+    def reset(self) -> None:
+        """Reset internal state for a new episode."""
+        self.cmd[:] = 0.0
+        self.qj[:] = 0.0
+        self.dqj[:] = 0.0
+        self.obs[:] = 0.0
+        self.last_action[:] = 0.0
+        self.phase = np.array([[0.0, np.pi]], dtype=np.float32)
+        self.is_standing = True
+
+    def run_step(self, action: dict, lowstate) -> dict:
+        """Run one step of the locomotion controller.
+
+        Args:
+            action: Action dict containing remote.lx/ly/rx/ry
+            lowstate: Robot lowstate containing motor positions/velocities and IMU
+
+        Returns:
+            Action dict for lower body joints (0-14)
+        """
+        if lowstate is None:
+            return {}
+
+        lx, ly, rx, _ry = (action.get(k, 0.0) for k in REMOTE_AXES)
+        ly = ly if abs(ly) > 0.1 else 0.0
+        lx = lx if abs(lx) > 0.1 else 0.0
+        rx = rx if abs(rx) > 0.1 else 0.0
+        ly = np.clip(ly, -0.3, 0.3)
+        lx = np.clip(lx, -0.3, 0.3)
+        self.cmd[:] = [ly, -lx, -rx]
+
+        # Get joint positions and velocities from lowstate
+        for motor in G1_29_JointIndex:
+            idx = motor.value
+            self.qj[idx] = lowstate.motor_state[idx].q
+            self.dqj[idx] = lowstate.motor_state[idx].dq
+
+        # Hide arm positions from policy (show DEFAULT_ANGLES instead)
+        # This prevents policy from reacting to teleop arm movements
+        for arm_joint in G1_29_JointArmIndex:
+            self.qj[arm_joint.value] = DEFAULT_ANGLES[arm_joint.value]
+            self.dqj[arm_joint.value] = 0.0
+
+        # Express IMU data in gravity frame of reference
+        quat = lowstate.imu_state.quaternion
+        ang_vel = np.array(lowstate.imu_state.gyroscope, dtype=np.float32)
+        gravity = get_gravity_orientation(quat)
+
+        # Scale joint positions and velocities before policy inference
+        qj_obs = (self.qj - DEFAULT_ANGLES) * DOF_POS_SCALE
+        dqj_obs = self.dqj * DOF_VEL_SCALE
+        ang_vel_s = ang_vel * ANG_VEL_SCALE
+
+        # Update gait phase
+        if np.linalg.norm(self.cmd[:2]) < 0.01 and abs(self.cmd[2]) < 0.01:
+            self.phase[0, :] = np.pi
+            self.is_standing = True
+        elif self.is_standing:
+            self.phase = np.array([[0.0, np.pi]], dtype=np.float32)
+            self.is_standing = False
+        else:
+            self.phase = np.fmod(self.phase + self.phase_dt + np.pi, 2 * np.pi) - np.pi
+
+        sin_ph = np.sin(self.phase[0])
+        cos_ph = np.cos(self.phase[0])
+
+        # Build observations
+        self.obs[0:29] = self.last_action
+        self.obs[29:32] = ang_vel_s
+        self.obs[32] = self.cmd[2]
+        self.obs[33:35] = self.cmd[:2]
+        self.obs[35:37] = cos_ph
+        self.obs[37:66] = qj_obs
+        self.obs[66:95] = dqj_obs
+        self.obs[95:98] = gravity
+        self.obs[98:100] = sin_ph
+
+        # Run policy inference
+        ort_in = {self.policy.get_inputs()[0].name: self.obs.reshape(1, -1).astype(np.float32)}
+        raw_action = self.policy.run(None, ort_in)[0].squeeze()
+        policy_action = np.clip(raw_action, -100.0, 100.0)
+        self.last_action = policy_action.copy()
+
+        # Transform action back to target joint positions
+        target = DEFAULT_ANGLES + policy_action * ACTION_SCALE
+
+        # Build action dict (first 15 joints only)
+        action_dict = {}
+        for i in range(15):
+            motor_name = G1_29_JointIndex(i).name
+            action_dict[f"{motor_name}.q"] = float(target[i])
+
+        return action_dict
diff --git a/lerobot/src/lerobot/robots/unitree_g1/run_g1_server.py b/lerobot/src/lerobot/robots/unitree_g1/run_g1_server.py
new file mode 100644
index 0000000000000000000000000000000000000000..b5bd0baf8494e9219e66776467fa6dde4ab055f4
--- /dev/null
+++ b/lerobot/src/lerobot/robots/unitree_g1/run_g1_server.py
@@ -0,0 +1,243 @@
+#!/usr/bin/env python3
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+DDS-to-ZMQ bridge server for Unitree G1 robot.
+
+This server runs on the robot and forwards:
+- Robot state (LowState) from DDS to ZMQ (for remote clients)
+- Robot commands (LowCmd) from ZMQ to DDS (from remote clients)
+
+Uses JSON for secure serialization instead of pickle.
+"""
+
+import argparse
+import base64
+import contextlib
+import json
+import threading
+import time
+from typing import Any
+
+import zmq
+from unitree_sdk2py.comm.motion_switcher.motion_switcher_client import MotionSwitcherClient
+from unitree_sdk2py.core.channel import ChannelFactoryInitialize, ChannelPublisher, ChannelSubscriber
+from unitree_sdk2py.idl.default import unitree_hg_msg_dds__LowCmd_
+from unitree_sdk2py.idl.unitree_hg.msg.dds_ import LowCmd_ as hg_LowCmd, LowState_ as hg_LowState
+from unitree_sdk2py.utils.crc import CRC
+
+from lerobot.cameras.zmq.image_server import ImageServer
+
+# DDS topic names follow Unitree SDK naming conventions
+# ruff: noqa: N816
+kTopicLowCommand_Debug = "rt/lowcmd"  # action to robot
+kTopicLowState = "rt/lowstate"  # observation from robot
+
+LOWCMD_PORT = 6000
+LOWSTATE_PORT = 6001
+NUM_MOTORS = 35
+
+
+def lowstate_to_dict(msg: hg_LowState) -> dict[str, Any]:
+    """Convert LowState SDK message to a JSON-serializable dictionary."""
+    motor_states = []
+    for i in range(NUM_MOTORS):
+        temp = msg.motor_state[i].temperature
+        avg_temp = float(sum(temp) / len(temp)) if isinstance(temp, list) else float(temp)
+        motor_states.append(
+            {
+                "q": float(msg.motor_state[i].q),
+                "dq": float(msg.motor_state[i].dq),
+                "tau_est": float(msg.motor_state[i].tau_est),
+                "temperature": avg_temp,
+            }
+        )
+
+    return {
+        "motor_state": motor_states,
+        "imu_state": {
+            "quaternion": [float(x) for x in msg.imu_state.quaternion],
+            "gyroscope": [float(x) for x in msg.imu_state.gyroscope],
+            "accelerometer": [float(x) for x in msg.imu_state.accelerometer],
+            "rpy": [float(x) for x in msg.imu_state.rpy],
+            "temperature": float(msg.imu_state.temperature),
+        },
+        # Encode bytes as base64 for JSON compatibility
+        "wireless_remote": base64.b64encode(bytes(msg.wireless_remote)).decode("ascii"),
+        "mode_machine": int(msg.mode_machine),
+    }
+
+
+def dict_to_lowcmd(data: dict[str, Any]) -> hg_LowCmd:
+    """Convert dictionary back to LowCmd SDK message."""
+    cmd = unitree_hg_msg_dds__LowCmd_()
+    cmd.mode_pr = data.get("mode_pr", 0)
+    cmd.mode_machine = data.get("mode_machine", 0)
+
+    for i, motor_data in enumerate(data.get("motor_cmd", [])):
+        cmd.motor_cmd[i].mode = motor_data.get("mode", 0)
+        cmd.motor_cmd[i].q = motor_data.get("q", 0.0)
+        cmd.motor_cmd[i].dq = motor_data.get("dq", 0.0)
+        cmd.motor_cmd[i].kp = motor_data.get("kp", 0.0)
+        cmd.motor_cmd[i].kd = motor_data.get("kd", 0.0)
+        cmd.motor_cmd[i].tau = motor_data.get("tau", 0.0)
+
+    return cmd
+
+
+def state_forward_loop(
+    lowstate_sub: ChannelSubscriber,
+    lowstate_sock: zmq.Socket,
+    state_period: float,
+    shutdown_event: threading.Event,
+) -> None:
+    """Read observation from DDS and forward to ZMQ clients."""
+    last_state_time = 0.0
+
+    while not shutdown_event.is_set():
+        # read from DDS
+        msg = lowstate_sub.Read()
+        if msg is None:
+            continue
+
+        now = time.time()
+        # optional downsampling (if robot dds rate > state_period)
+        if now - last_state_time >= state_period:
+            # Convert to dict and serialize with JSON
+            state_dict = lowstate_to_dict(msg)
+            payload = json.dumps({"topic": kTopicLowState, "data": state_dict}).encode("utf-8")
+            # if no subscribers / tx buffer full, just drop
+            with contextlib.suppress(zmq.Again):
+                lowstate_sock.send(payload, zmq.NOBLOCK)
+            last_state_time = now
+
+
+def cmd_forward_loop(
+    lowcmd_sock: zmq.Socket,
+    lowcmd_pub_debug: ChannelPublisher,
+    crc: CRC,
+) -> None:
+    """Receive commands from ZMQ and forward to DDS."""
+    while True:
+        try:
+            payload = lowcmd_sock.recv()
+        except zmq.ContextTerminated:
+            break
+        msg_dict = json.loads(payload.decode("utf-8"))
+
+        topic = msg_dict.get("topic", "")
+        cmd_data = msg_dict.get("data", {})
+
+        # Reconstruct LowCmd object from dict
+        cmd = dict_to_lowcmd(cmd_data)
+
+        # recompute crc
+        cmd.crc = crc.Crc(cmd)
+
+        if topic == kTopicLowCommand_Debug:
+            lowcmd_pub_debug.Write(cmd)
+
+
+def main() -> None:
+    """Main entry point for the robot server bridge."""
+    parser = argparse.ArgumentParser(description="DDS-to-ZMQ bridge server for Unitree G1")
+    parser.add_argument("--camera", action="store_true", help="Also launch camera server")
+    parser.add_argument("--camera-device", type=int, default=4, help="Camera device ID (default: 4)")
+    parser.add_argument("--camera-fps", type=int, default=30, help="Camera FPS (default: 30)")
+    parser.add_argument("--camera-width", type=int, default=640, help="Camera width (default: 640)")
+    parser.add_argument("--camera-height", type=int, default=480, help="Camera height (default: 480)")
+    parser.add_argument("--camera-port", type=int, default=5555, help="Camera ZMQ port (default: 5555)")
+    args = parser.parse_args()
+
+    # Optionally start camera server in background thread
+    camera_thread = None
+    if args.camera:
+        camera_config = {
+            "fps": args.camera_fps,
+            "cameras": {
+                "head_camera": {
+                    "device_id": args.camera_device,
+                    "shape": [args.camera_height, args.camera_width],
+                }
+            },
+        }
+        camera_server = ImageServer(camera_config, port=args.camera_port)
+        camera_thread = threading.Thread(target=camera_server.run, daemon=True)
+        camera_thread.start()
+        print(f"Camera server started on port {args.camera_port} (device {args.camera_device})")
+
+    # initialize DDS
+    ChannelFactoryInitialize(0)
+
+    # stop all active publishers on the robot
+    msc = MotionSwitcherClient()
+    msc.SetTimeout(5.0)
+    msc.Init()
+
+    status, result = msc.CheckMode()
+    while result is not None and "name" in result and result["name"]:
+        msc.ReleaseMode()
+        status, result = msc.CheckMode()
+        time.sleep(1.0)
+
+    crc = CRC()
+
+    # initialize DDS publisher
+    lowcmd_pub_debug = ChannelPublisher(kTopicLowCommand_Debug, hg_LowCmd)
+    lowcmd_pub_debug.Init()
+
+    # initialize DDS subscriber
+    lowstate_sub = ChannelSubscriber(kTopicLowState, hg_LowState)
+    lowstate_sub.Init()
+
+    # initialize ZMQ
+    ctx = zmq.Context.instance()
+
+    # receive commands from remote client
+    lowcmd_sock = ctx.socket(zmq.PULL)
+    lowcmd_sock.bind(f"tcp://0.0.0.0:{LOWCMD_PORT}")
+
+    # publish state to remote clients
+    lowstate_sock = ctx.socket(zmq.PUB)
+    lowstate_sock.bind(f"tcp://0.0.0.0:{LOWSTATE_PORT}")
+
+    state_period = 0.002  # ~500 hz
+    shutdown_event = threading.Event()
+
+    # start observation forwarding in background thread
+    t_state = threading.Thread(
+        target=state_forward_loop,
+        args=(lowstate_sub, lowstate_sock, state_period, shutdown_event),
+    )
+    t_state.start()
+
+    print("bridge running (lowstate -> zmq, lowcmd -> dds)")
+
+    # run command forwarding in main thread
+    try:
+        cmd_forward_loop(lowcmd_sock, lowcmd_pub_debug, crc)
+    except KeyboardInterrupt:
+        print("shutting down bridge...")
+    finally:
+        shutdown_event.set()
+        ctx.term()  # terminates blocking zmq.recv() calls
+        t_state.join(timeout=2.0)
+        if camera_thread is not None:
+            camera_thread.join(timeout=2.0)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/robots/unitree_g1/unitree_g1.py b/lerobot/src/lerobot/robots/unitree_g1/unitree_g1.py
new file mode 100644
index 0000000000000000000000000000000000000000..9e373c05fe286f830236f589cef44b9043245f16
--- /dev/null
+++ b/lerobot/src/lerobot/robots/unitree_g1/unitree_g1.py
@@ -0,0 +1,570 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from __future__ import annotations
+
+import logging
+import threading
+import time
+from dataclasses import dataclass, field
+from functools import cached_property
+from typing import TYPE_CHECKING, Protocol, runtime_checkable
+
+import numpy as np
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.robots.unitree_g1.g1_kinematics import G1_29_ArmIK
+from lerobot.robots.unitree_g1.g1_utils import (
+    REMOTE_AXES,
+    REMOTE_KEYS,
+    G1_29_JointArmIndex,
+    G1_29_JointIndex,
+    default_remote_input,
+    make_locomotion_controller,
+)
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.import_utils import _unitree_sdk_available
+
+from ..robot import Robot
+from .config_unitree_g1 import UnitreeG1Config
+
+if TYPE_CHECKING or _unitree_sdk_available:
+    from unitree_sdk2py.core.channel import (
+        ChannelFactoryInitialize as _SDKChannelFactoryInitialize,
+        ChannelPublisher as _SDKChannelPublisher,
+        ChannelSubscriber as _SDKChannelSubscriber,
+    )
+    from unitree_sdk2py.idl.default import unitree_hg_msg_dds__LowCmd_
+    from unitree_sdk2py.idl.unitree_hg.msg.dds_ import (
+        LowCmd_ as hg_LowCmd,
+        LowState_ as hg_LowState,
+    )
+    from unitree_sdk2py.utils.crc import CRC
+else:
+    _SDKChannelFactoryInitialize = None
+    _SDKChannelPublisher = None
+    _SDKChannelSubscriber = None
+    unitree_hg_msg_dds__LowCmd_ = None
+    hg_LowCmd = None
+    hg_LowState = None
+    CRC = None
+
+logger = logging.getLogger(__name__)
+
+
+@runtime_checkable
+class LocomotionController(Protocol):
+    control_dt: float
+
+    def run_step(self, action: dict, lowstate) -> dict: ...
+
+    def reset(self) -> None: ...
+
+
+# DDS topic names follow Unitree SDK naming conventions
+# ruff: noqa: N816
+kTopicLowCommand_Debug = "rt/lowcmd"
+kTopicLowState = "rt/lowstate"
+
+
+@dataclass
+class MotorState:
+    q: float | None = None  # position
+    dq: float | None = None  # velocity
+    tau_est: float | None = None  # estimated torque
+    temperature: float | None = None  # motor temperature
+
+
+@dataclass
+class IMUState:
+    quaternion: np.ndarray | None = None  # [w, x, y, z]
+    gyroscope: np.ndarray | None = None  # [x, y, z] angular velocity (rad/s)
+    accelerometer: np.ndarray | None = None  # [x, y, z] linear acceleration (m/s²)
+    rpy: np.ndarray | None = None  # [roll, pitch, yaw] (rad)
+    temperature: float | None = None  # IMU temperature
+
+
+# g1 observation class
+@dataclass
+class G1_29_LowState:  # noqa: N801
+    motor_state: list[MotorState] = field(default_factory=lambda: [MotorState() for _ in G1_29_JointIndex])
+    imu_state: IMUState = field(default_factory=IMUState)
+    wireless_remote: bytes | None = None  # Raw wireless remote data
+    mode_machine: int = 0  # Robot mode
+
+
+class UnitreeG1(Robot):
+    config_class = UnitreeG1Config
+    name = "unitree_g1"
+
+    def __init__(self, config: UnitreeG1Config):
+        super().__init__(config)
+
+        logger.info("Initialize UnitreeG1...")
+
+        self.config = config
+        self.control_dt = config.control_dt
+
+        # Initialize cameras config (ZMQ-based) - actual connection in connect()
+        self._cameras = make_cameras_from_configs(config.cameras)
+
+        # Import channel classes based on mode
+        if config.is_simulation:
+            self._ChannelFactoryInitialize = _SDKChannelFactoryInitialize
+            self._ChannelPublisher = _SDKChannelPublisher
+            self._ChannelSubscriber = _SDKChannelSubscriber
+        else:
+            from lerobot.robots.unitree_g1.unitree_sdk2_socket import (
+                ChannelFactoryInitialize,
+                ChannelPublisher,
+                ChannelSubscriber,
+            )
+
+            self._ChannelFactoryInitialize = ChannelFactoryInitialize
+            self._ChannelPublisher = ChannelPublisher
+            self._ChannelSubscriber = ChannelSubscriber
+
+        # Initialize state variables
+        self.sim_env = None
+        self._env_wrapper = None
+        self._lowstate = None
+        self._lowstate_lock = threading.Lock()
+        self._shutdown_event = threading.Event()
+        self.subscribe_thread = None
+
+        self.arm_ik = G1_29_ArmIK() if config.gravity_compensation else None
+
+        # Lower-body controller loaded dynamically
+        self.controller: LocomotionController | None = make_locomotion_controller(config.controller)
+
+        # Controller thread state
+        self._controller_thread = None
+        self._controller_action_lock = threading.Lock()
+        self.controller_input = default_remote_input()
+        self.controller_output = {}
+
+    def _subscribe_lowstate(self):  # polls robot state @ 250Hz
+        while not self._shutdown_event.is_set():
+            start_time = time.time()
+
+            # Step simulation if in simulation mode
+            if self.config.is_simulation and self.sim_env is not None:
+                self.sim_env.step()
+
+            msg = self.lowstate_subscriber.Read()
+            if msg is not None:
+                lowstate = G1_29_LowState()
+
+                # Capture motor states using jointindex
+                for joint in G1_29_JointIndex:
+                    lowstate.motor_state[joint].q = msg.motor_state[joint].q
+                    lowstate.motor_state[joint].dq = msg.motor_state[joint].dq
+                    lowstate.motor_state[joint].tau_est = msg.motor_state[joint].tau_est
+                    lowstate.motor_state[joint].temperature = msg.motor_state[joint].temperature
+
+                # Capture IMU state
+                lowstate.imu_state.quaternion = list(msg.imu_state.quaternion)
+                lowstate.imu_state.gyroscope = list(msg.imu_state.gyroscope)
+                lowstate.imu_state.accelerometer = list(msg.imu_state.accelerometer)
+                lowstate.imu_state.rpy = list(msg.imu_state.rpy)
+                lowstate.imu_state.temperature = msg.imu_state.temperature
+
+                # Capture wireless remote data
+                lowstate.wireless_remote = msg.wireless_remote
+
+                # Capture mode_machine
+                lowstate.mode_machine = msg.mode_machine
+
+                with self._lowstate_lock:
+                    self._lowstate = lowstate
+
+            current_time = time.time()
+            all_t_elapsed = current_time - start_time
+            sleep_time = max(0, (self.control_dt - all_t_elapsed))  # maintain constant control dt
+            time.sleep(sleep_time)
+
+    def publish_lowcmd(
+        self,
+        action: RobotAction,
+        kp: np.ndarray | list[float] | None = None,
+        kd: np.ndarray | list[float] | None = None,
+        tau: np.ndarray | list[float] | None = None,
+    ) -> None:  # writes robot command whenever requested
+        for motor in G1_29_JointIndex:
+            key = f"{motor.name}.q"
+            if key in action:
+                self.msg.motor_cmd[motor.value].q = action[key]
+                self.msg.motor_cmd[motor.value].qd = 0
+                self.msg.motor_cmd[motor.value].kp = (
+                    kp[motor.value] if kp is not None else self.kp[motor.value]
+                )
+                self.msg.motor_cmd[motor.value].kd = (
+                    kd[motor.value] if kd is not None else self.kd[motor.value]
+                )
+                self.msg.motor_cmd[motor.value].tau = tau[motor.value] if tau is not None else 0.0
+
+        self.msg.crc = self.crc.Crc(self.msg)
+        self.lowcmd_publisher.Write(self.msg)
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        if self.controller is None:
+            return {f"{G1_29_JointIndex(motor).name}.q": float for motor in G1_29_JointIndex}
+
+        arm_features = {f"{G1_29_JointArmIndex(motor).name}.q": float for motor in G1_29_JointArmIndex}
+        remote_features = dict.fromkeys(REMOTE_AXES, float)
+        return {**arm_features, **remote_features}
+
+    def _controller_loop(self):
+        """Background thread that runs controller at policy's control_dt."""
+        control_dt = self.controller.control_dt
+        logger.info(f"Controller loop starting with control_dt={control_dt} ({1.0 / control_dt:.1f}Hz)")
+
+        loop_count = 0
+        last_log_time = time.time()
+
+        while not self._shutdown_event.is_set():
+            start_time = time.time()
+
+            with self._lowstate_lock:
+                lowstate = self._lowstate
+
+            if lowstate is not None and self.controller is not None:
+                loop_count += 1
+                if time.time() - last_log_time >= 5.0:  # Log every 5 seconds
+                    actual_hz = loop_count / (time.time() - last_log_time)
+                    logger.info(
+                        f"Controller actual rate: {actual_hz:.1f}Hz (target: {1.0 / control_dt:.1f}Hz)"
+                    )
+                    loop_count = 0
+                    last_log_time = time.time()
+                # Read controller input snapshot
+                with self._controller_action_lock:
+                    controller_input = dict(self.controller_input)
+
+                # Run controller step
+                controller_action = self.controller.run_step(controller_input, lowstate)
+
+                # Write controller output snapshot
+                with self._controller_action_lock:
+                    self.controller_output = dict(controller_action)
+
+                ctrl_kp = self.controller.kp if hasattr(self.controller, "kp") else None
+                ctrl_kd = self.controller.kd if hasattr(self.controller, "kd") else None
+                self.publish_lowcmd(controller_action, kp=ctrl_kp, kd=ctrl_kd)
+
+            elapsed = time.time() - start_time
+            sleep_time = max(0, control_dt - elapsed)
+            time.sleep(sleep_time)
+
+    def calibrate(self) -> None:
+        # TODO: implement g1_29 calibration
+        pass
+
+    def configure(self) -> None:
+        pass
+
+    def connect(self, calibrate: bool = True) -> None:  # connect to DDS
+        # Initialize DDS channel and simulation environment
+        if self.config.is_simulation:
+            from lerobot.envs.factory import make_env
+
+            self._ChannelFactoryInitialize(0, "lo")
+            self._env_wrapper = make_env("lerobot/unitree-g1-mujoco", trust_remote_code=True)
+            # Extract the actual gym env from the dict structure
+            self.sim_env = self._env_wrapper["hub_env"][0].envs[0]
+        else:
+            self._ChannelFactoryInitialize(0, config=self.config)
+
+        # Initialize direct motor control interface
+        self.lowcmd_publisher = self._ChannelPublisher(kTopicLowCommand_Debug, hg_LowCmd)
+        self.lowcmd_publisher.Init()
+        self.lowstate_subscriber = self._ChannelSubscriber(kTopicLowState, hg_LowState)
+        self.lowstate_subscriber.Init()
+
+        # Start subscribe thread to read robot state
+        self.subscribe_thread = threading.Thread(target=self._subscribe_lowstate)
+        self.subscribe_thread.start()
+
+        # Connect cameras
+        for cam in self._cameras.values():
+            if not cam.is_connected:
+                cam.connect()
+
+        logger.info(f"Connected {len(self._cameras)} camera(s).")
+
+        # Initialize lowcmd message
+        self.crc = CRC()
+        self.msg = unitree_hg_msg_dds__LowCmd_()
+        self.msg.mode_pr = 0
+
+        # Wait for first state message to arrive
+        lowstate = None
+        deadline = time.time() + 10.0
+        while lowstate is None:
+            with self._lowstate_lock:
+                lowstate = self._lowstate
+            if lowstate is None:
+                if time.time() > deadline:
+                    raise TimeoutError("Timed out waiting for robot state (10s)")
+                logger.warning("[UnitreeG1] Waiting for robot state...")
+                time.sleep(0.01)
+        logger.info("[UnitreeG1] Connected to robot.")
+        self.msg.mode_machine = lowstate.mode_machine
+
+        self.kp = np.array(self.config.kp, dtype=np.float32)
+        self.kd = np.array(self.config.kd, dtype=np.float32)
+
+        for joint in G1_29_JointIndex:
+            self.msg.motor_cmd[joint].mode = 1
+            self.msg.motor_cmd[joint].kp = self.kp[joint.value]
+            self.msg.motor_cmd[joint].kd = self.kd[joint.value]
+            self.msg.motor_cmd[joint].q = lowstate.motor_state[joint.value].q
+
+        # Start controller thread if enabled
+        if self.controller is not None:
+            self._controller_thread = threading.Thread(target=self._controller_loop, daemon=True)
+            self._controller_thread.start()
+            fps = int(1.0 / self.controller.control_dt)
+            logger.info(f"Controller thread started ({fps}Hz)")
+
+    def _send_zero_torque(self) -> None:
+        """Send a zero-gain command to make joints passive before shutting down."""
+        try:
+            with self._lowstate_lock:
+                lowstate = self._lowstate
+            if lowstate is None:
+                return
+            action = {f"{motor.name}.q": lowstate.motor_state[motor.value].q for motor in G1_29_JointIndex}
+            zero_gains = np.zeros(29, dtype=np.float32)
+            self.publish_lowcmd(action, kp=zero_gains, kd=zero_gains, tau=zero_gains)
+            logger.info("Sent zero-torque command for safe shutdown")
+        except Exception as e:
+            logger.warning(f"Failed to send zero-torque on disconnect: {e}")
+
+    def disconnect(self):
+        # Put robot in passive mode before stopping threads
+        if not self.config.is_simulation:
+            self._send_zero_torque()
+
+        # Signal thread to stop and unblock any waits
+        self._shutdown_event.set()
+
+        # Wait for subscribe thread to finish
+        if self.subscribe_thread is not None:
+            self.subscribe_thread.join(timeout=2.0)
+            if self.subscribe_thread.is_alive():
+                logger.warning("Subscribe thread did not stop cleanly")
+
+        # Wait for controller thread to finish
+        if self._controller_thread is not None:
+            self._controller_thread.join(timeout=2.0)
+            if self._controller_thread.is_alive():
+                logger.warning("Controller thread did not stop cleanly")
+
+        # Close simulation environment
+        if self.config.is_simulation and self.sim_env is not None:
+            try:
+                # Force-kill the image publish subprocess first to avoid long waits
+                if hasattr(self.sim_env, "simulator") and hasattr(self.sim_env.simulator, "sim_env"):
+                    sim_env_inner = self.sim_env.simulator.sim_env
+                    if hasattr(sim_env_inner, "image_publish_process"):
+                        proc = sim_env_inner.image_publish_process
+                        if proc.process and proc.process.is_alive():
+                            logger.info("Force-terminating image publish subprocess...")
+                            proc.stop_event.set()
+                            proc.process.terminate()
+                            proc.process.join(timeout=1)
+                            if proc.process.is_alive():
+                                proc.process.kill()
+                self.sim_env.close()
+            except Exception as e:
+                logger.warning(f"Error closing sim_env: {e}")
+            self.sim_env = None
+            self._env_wrapper = None
+
+        # Disconnect cameras
+        for cam in self._cameras.values():
+            cam.disconnect()
+
+    def get_observation(self) -> RobotObservation:
+        with self._lowstate_lock:
+            lowstate = self._lowstate
+        if lowstate is None:
+            return {}
+
+        obs = {}
+
+        # Motors - q, dq, tau for all joints
+        for motor in G1_29_JointIndex:
+            name = motor.name
+            idx = motor.value
+            obs[f"{name}.q"] = lowstate.motor_state[idx].q
+            obs[f"{name}.dq"] = lowstate.motor_state[idx].dq
+            obs[f"{name}.tau"] = lowstate.motor_state[idx].tau_est
+
+        # IMU - gyroscope
+        if lowstate.imu_state.gyroscope:
+            obs["imu.gyro.x"] = lowstate.imu_state.gyroscope[0]
+            obs["imu.gyro.y"] = lowstate.imu_state.gyroscope[1]
+            obs["imu.gyro.z"] = lowstate.imu_state.gyroscope[2]
+
+        # IMU - accelerometer
+        if lowstate.imu_state.accelerometer:
+            obs["imu.accel.x"] = lowstate.imu_state.accelerometer[0]
+            obs["imu.accel.y"] = lowstate.imu_state.accelerometer[1]
+            obs["imu.accel.z"] = lowstate.imu_state.accelerometer[2]
+
+        # IMU - quaternion
+        if lowstate.imu_state.quaternion:
+            obs["imu.quat.w"] = lowstate.imu_state.quaternion[0]
+            obs["imu.quat.x"] = lowstate.imu_state.quaternion[1]
+            obs["imu.quat.y"] = lowstate.imu_state.quaternion[2]
+            obs["imu.quat.z"] = lowstate.imu_state.quaternion[3]
+
+        # IMU - rpy
+        if lowstate.imu_state.rpy:
+            obs["imu.rpy.roll"] = lowstate.imu_state.rpy[0]
+            obs["imu.rpy.pitch"] = lowstate.imu_state.rpy[1]
+            obs["imu.rpy.yaw"] = lowstate.imu_state.rpy[2]
+
+        # Wireless remote (raw bytes for teleoperator)
+        if lowstate.wireless_remote:
+            obs["wireless_remote"] = lowstate.wireless_remote
+
+        # Cameras - read images from ZMQ cameras
+        for cam_name, cam in self._cameras.items():
+            obs[cam_name] = cam.read_latest()
+
+        return obs
+
+    def send_action(self, action: RobotAction) -> RobotAction:
+        action_to_publish = action
+        if self.controller is not None:
+            # Controller thread owns legs/waist. Here we only update joystick inputs
+            # and publish arm targets from the teleoperator.
+            self._update_controller_action(action)
+            arm_prefixes = tuple(j.name for j in G1_29_JointArmIndex)
+            action_to_publish = {
+                key: value
+                for key, value in action.items()
+                if key.endswith(".q") and key.startswith(arm_prefixes)
+            }
+
+        tau = None
+        if self.config.gravity_compensation and self.arm_ik is not None:
+            tau = np.zeros(29, dtype=np.float32)
+            action_np = np.array(
+                [
+                    action_to_publish.get(f"{joint.name}.q", self.msg.motor_cmd[joint.value].q)
+                    for joint in G1_29_JointArmIndex
+                ],
+                dtype=np.float32,
+            )
+            arm_tau = self.arm_ik.solve_tau(action_np)
+            arm_start_idx = G1_29_JointArmIndex.kLeftShoulderPitch.value
+            for joint in G1_29_JointArmIndex:
+                local_idx = joint.value - arm_start_idx
+                tau[joint.value] = arm_tau[local_idx]
+
+        self.publish_lowcmd(action_to_publish, tau=tau)
+        return action
+
+    def _update_controller_action(self, action: RobotAction) -> None:
+        """Update controller input state from incoming teleop action."""
+        with self._controller_action_lock:
+            for key in REMOTE_KEYS:
+                if key in action:
+                    self.controller_input[key] = action[key]
+
+    @property
+    def is_calibrated(self) -> bool:
+        return True
+
+    @property
+    def is_connected(self) -> bool:
+        with self._lowstate_lock:
+            return self._lowstate is not None
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        """Joint positions for all 29 joints."""
+        return {f"{G1_29_JointIndex(motor).name}.q": float for motor in G1_29_JointIndex}
+
+    @property
+    def cameras(self) -> dict:
+        return self._cameras
+
+    def reset(
+        self,
+        control_dt: float | None = None,
+        default_positions: list[float] | None = None,
+    ) -> None:  # move robot to default position
+        if control_dt is None:
+            control_dt = self.config.control_dt
+        if default_positions is None:
+            default_positions = np.array(self.config.default_positions, dtype=np.float32)
+
+        if self.config.is_simulation and self.sim_env is not None:
+            self.sim_env.reset()
+            self.publish_lowcmd(
+                {f"{motor.name}.q": float(default_positions[motor.value]) for motor in G1_29_JointIndex}
+            )
+        else:
+            total_time = 3.0
+            num_steps = int(total_time / control_dt)
+
+            # get current state
+            obs = self.get_observation()
+
+            # record current positions
+            init_dof_pos = np.zeros(29, dtype=np.float32)
+            for motor in G1_29_JointIndex:
+                init_dof_pos[motor.value] = obs[f"{motor.name}.q"]
+
+            # Interpolate to default position
+            for step in range(num_steps):
+                start_time = time.time()
+
+                alpha = step / num_steps
+                action_dict = {}
+                for motor in G1_29_JointIndex:
+                    target_pos = default_positions[motor.value]
+                    interp_pos = init_dof_pos[motor.value] * (1 - alpha) + target_pos * alpha
+                    action_dict[f"{motor.name}.q"] = float(interp_pos)
+
+                self.send_action(action_dict)
+
+                # Maintain constant control rate
+                elapsed = time.time() - start_time
+                sleep_time = max(0, control_dt - elapsed)
+                time.sleep(sleep_time)
+
+        # Reset controller internal state (gait phase, obs history, etc.)
+        if self.controller is not None and hasattr(self.controller, "reset"):
+            self.controller.reset()
+
+        logger.info("Reached default position")
diff --git a/lerobot/src/lerobot/robots/unitree_g1/unitree_sdk2_socket.py b/lerobot/src/lerobot/robots/unitree_g1/unitree_sdk2_socket.py
new file mode 100644
index 0000000000000000000000000000000000000000..0f1f8f8d68acd47095d96724590f4b67ca193347
--- /dev/null
+++ b/lerobot/src/lerobot/robots/unitree_g1/unitree_sdk2_socket.py
@@ -0,0 +1,175 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import base64
+import json
+from typing import Any
+
+import zmq
+
+from lerobot.robots.unitree_g1.config_unitree_g1 import UnitreeG1Config
+
+# Module-level ZMQ state mirrors the Unitree SDK's global ChannelFactory Singleton.
+# Only one robot connection per process is supported.
+_ctx: zmq.Context | None = None
+_lowcmd_sock: zmq.Socket | None = None
+_lowstate_sock: zmq.Socket | None = None
+
+LOWCMD_PORT = 6000
+LOWSTATE_PORT = 6001
+
+# DDS topic names follow Unitree SDK naming conventions
+# ruff: noqa: N816
+kTopicLowCommand_Debug = "rt/lowcmd"
+
+
+class LowStateMsg:
+    """
+    Wrapper class that mimics the Unitree SDK LowState_ message structure.
+
+    Reconstructs the message from deserialized JSON data to maintain
+    compatibility with existing code that expects SDK message objects.
+    """
+
+    class MotorState:
+        """Motor state data for a single joint."""
+
+        def __init__(self, data: dict[str, Any]) -> None:
+            self.q: float = data.get("q", 0.0)
+            self.dq: float = data.get("dq", 0.0)
+            self.tau_est: float = data.get("tau_est", 0.0)
+            self.temperature: float = data.get("temperature", 0.0)
+
+    class IMUState:
+        """IMU sensor data."""
+
+        def __init__(self, data: dict[str, Any]) -> None:
+            self.quaternion: list[float] = data.get("quaternion", [1.0, 0.0, 0.0, 0.0])
+            self.gyroscope: list[float] = data.get("gyroscope", [0.0, 0.0, 0.0])
+            self.accelerometer: list[float] = data.get("accelerometer", [0.0, 0.0, 0.0])
+            self.rpy: list[float] = data.get("rpy", [0.0, 0.0, 0.0])
+            self.temperature: float = data.get("temperature", 0.0)
+
+    def __init__(self, data: dict[str, Any]) -> None:
+        """Initialize from deserialized JSON data."""
+        self.motor_state = [self.MotorState(m) for m in data.get("motor_state", [])]
+        self.imu_state = self.IMUState(data.get("imu_state", {}))
+        # Decode base64-encoded wireless_remote bytes
+        wireless_b64 = data.get("wireless_remote", "")
+        self.wireless_remote: bytes = base64.b64decode(wireless_b64) if wireless_b64 else b""
+        self.mode_machine: int = data.get("mode_machine", 0)
+
+
+def lowcmd_to_dict(topic: str, msg: Any) -> dict[str, Any]:
+    """Convert LowCmd message to a JSON-serializable dictionary."""
+    motor_cmds = []
+    # Iterate over all motor commands in the message
+    for i in range(len(msg.motor_cmd)):
+        motor_cmds.append(
+            {
+                "mode": int(msg.motor_cmd[i].mode),
+                "q": float(msg.motor_cmd[i].q),
+                "dq": float(msg.motor_cmd[i].dq),
+                "kp": float(msg.motor_cmd[i].kp),
+                "kd": float(msg.motor_cmd[i].kd),
+                "tau": float(msg.motor_cmd[i].tau),
+            }
+        )
+
+    return {
+        "topic": topic,
+        "data": {
+            "mode_pr": int(msg.mode_pr),
+            "mode_machine": int(msg.mode_machine),
+            "motor_cmd": motor_cmds,
+        },
+    }
+
+
+def ChannelFactoryInitialize(domain_id: int = 0, config: Any = None) -> None:  # noqa: N802
+    """
+    Initialize ZMQ sockets for robot communication.
+
+    This function mimics the Unitree SDK's ChannelFactoryInitialize but uses
+    ZMQ sockets to connect to the robot server bridge instead of DDS.
+
+    Args:
+        domain_id: Ignored (for API compatibility with Unitree SDK)
+        config: UnitreeG1Config instance with robot_ip
+    """
+    global _ctx, _lowcmd_sock, _lowstate_sock
+
+    # read socket config
+    if config is None:
+        config = UnitreeG1Config()
+    robot_ip = config.robot_ip
+
+    ctx = zmq.Context.instance()
+    _ctx = ctx
+
+    # lowcmd: send robot commands
+    lowcmd_sock = ctx.socket(zmq.PUSH)
+    lowcmd_sock.setsockopt(zmq.CONFLATE, 1)  # keep only last message
+    lowcmd_sock.connect(f"tcp://{robot_ip}:{LOWCMD_PORT}")
+    _lowcmd_sock = lowcmd_sock
+
+    # lowstate: receive robot observations
+    lowstate_sock = ctx.socket(zmq.SUB)
+    lowstate_sock.setsockopt(zmq.CONFLATE, 1)  # keep only last message
+    lowstate_sock.connect(f"tcp://{robot_ip}:{LOWSTATE_PORT}")
+    lowstate_sock.setsockopt_string(zmq.SUBSCRIBE, "")
+    _lowstate_sock = lowstate_sock
+
+
+class ChannelPublisher:
+    """ZMQ-based publisher that sends commands to the robot server."""
+
+    def __init__(self, topic: str, msg_type: type) -> None:
+        self.topic = topic
+        self.msg_type = msg_type
+
+    def Init(self) -> None:  # noqa: N802
+        """Initialize the publisher (no-op for ZMQ)."""
+        pass
+
+    def Write(self, msg: Any) -> None:  # noqa: N802
+        """Serialize and send a command message to the robot."""
+        if _lowcmd_sock is None:
+            raise RuntimeError("ChannelFactoryInitialize must be called first")
+
+        payload = json.dumps(lowcmd_to_dict(self.topic, msg)).encode("utf-8")
+        _lowcmd_sock.send(payload)
+
+
+class ChannelSubscriber:
+    """ZMQ-based subscriber that receives state from the robot server."""
+
+    def __init__(self, topic: str, msg_type: type) -> None:
+        self.topic = topic
+        self.msg_type = msg_type
+
+    def Init(self) -> None:  # noqa: N802
+        """Initialize the subscriber (no-op for ZMQ)."""
+        pass
+
+    def Read(self) -> LowStateMsg:  # noqa: N802
+        """Receive and deserialize a state message from the robot."""
+        if _lowstate_sock is None:
+            raise RuntimeError("ChannelFactoryInitialize must be called first")
+
+        payload = _lowstate_sock.recv()
+        msg_dict = json.loads(payload.decode("utf-8"))
+        return LowStateMsg(msg_dict.get("data", {}))
diff --git a/lerobot/src/lerobot/robots/utils.py b/lerobot/src/lerobot/robots/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..92da597f13c7df750862159efbacee9812f362a6
--- /dev/null
+++ b/lerobot/src/lerobot/robots/utils.py
@@ -0,0 +1,118 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+from pprint import pformat
+from typing import cast
+
+from lerobot.utils.import_utils import make_device_from_device_class
+
+from .config import RobotConfig
+from .robot import Robot
+
+
+def make_robot_from_config(config: RobotConfig) -> Robot:
+    # TODO(Steven): Consider just using the make_device_from_device_class for all types
+    if config.type == "koch_follower":
+        from .koch_follower import KochFollower
+
+        return KochFollower(config)
+    elif config.type == "omx_follower":
+        from .omx_follower import OmxFollower
+
+        return OmxFollower(config)
+    elif config.type == "so100_follower":
+        from .so_follower import SO100Follower
+
+        return SO100Follower(config)
+    elif config.type == "so101_follower":
+        from .so_follower import SO101Follower
+
+        return SO101Follower(config)
+    elif config.type == "lekiwi":
+        from .lekiwi import LeKiwi
+
+        return LeKiwi(config)
+    elif config.type == "hope_jr_hand":
+        from .hope_jr import HopeJrHand
+
+        return HopeJrHand(config)
+    elif config.type == "hope_jr_arm":
+        from .hope_jr import HopeJrArm
+
+        return HopeJrArm(config)
+    elif config.type == "bi_so_follower":
+        from .bi_so_follower import BiSOFollower
+
+        return BiSOFollower(config)
+    elif config.type == "reachy2":
+        from .reachy2 import Reachy2Robot
+
+        return Reachy2Robot(config)
+    elif config.type == "openarm_follower":
+        from .openarm_follower import OpenArmFollower
+
+        return OpenArmFollower(config)
+    elif config.type == "bi_openarm_follower":
+        from .bi_openarm_follower import BiOpenArmFollower
+
+        return BiOpenArmFollower(config)
+    elif config.type == "mock_robot":
+        from tests.mocks.mock_robot import MockRobot
+
+        return MockRobot(config)
+    else:
+        try:
+            return cast(Robot, make_device_from_device_class(config))
+        except Exception as e:
+            raise ValueError(f"Error creating robot with config {config}: {e}") from e
+
+
+# TODO(pepijn): Move to pipeline step to make sure we don't have to do this in the robot code and send action to robot is clean for use in dataset
+def ensure_safe_goal_position(
+    goal_present_pos: dict[str, tuple[float, float]], max_relative_target: float | dict[str, float]
+) -> dict[str, float]:
+    """Caps relative action target magnitude for safety."""
+
+    if isinstance(max_relative_target, float):
+        diff_cap = dict.fromkeys(goal_present_pos, max_relative_target)
+    elif isinstance(max_relative_target, dict):
+        if not set(goal_present_pos) == set(max_relative_target):
+            raise ValueError("max_relative_target keys must match those of goal_present_pos.")
+        diff_cap = max_relative_target
+    else:
+        raise TypeError(max_relative_target)
+
+    warnings_dict = {}
+    safe_goal_positions = {}
+    for key, (goal_pos, present_pos) in goal_present_pos.items():
+        diff = goal_pos - present_pos
+        max_diff = diff_cap[key]
+        safe_diff = min(diff, max_diff)
+        safe_diff = max(safe_diff, -max_diff)
+        safe_goal_pos = present_pos + safe_diff
+        safe_goal_positions[key] = safe_goal_pos
+        if abs(safe_goal_pos - goal_pos) > 1e-4:
+            warnings_dict[key] = {
+                "original goal_pos": goal_pos,
+                "safe goal_pos": safe_goal_pos,
+            }
+
+    if warnings_dict:
+        logging.warning(
+            "Relative goal position magnitude had to be clamped to be safe.\n"
+            f"{pformat(warnings_dict, indent=4)}"
+        )
+
+    return safe_goal_positions
diff --git a/lerobot/src/lerobot/scripts/augment_dataset_quantile_stats.py b/lerobot/src/lerobot/scripts/augment_dataset_quantile_stats.py
new file mode 100644
index 0000000000000000000000000000000000000000..4d80c93328f7e477fbe6448dbb20bbcc71225727
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/augment_dataset_quantile_stats.py
@@ -0,0 +1,261 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+This script augments existing LeRobot datasets with quantile statistics.
+
+Most datasets created before the quantile feature was added do not contain
+quantile statistics (q01, q10, q50, q90, q99) in their metadata. This script:
+
+1. Loads an existing LeRobot dataset in v3.0 format
+2. Checks if it already contains quantile statistics
+3. If missing, computes quantile statistics for all features
+4. Updates the dataset metadata with the new quantile statistics
+
+Usage:
+
+```bash
+python src/lerobot/scripts/augment_dataset_quantile_stats.py \
+    --repo-id=lerobot/pusht \
+```
+"""
+
+import argparse
+import concurrent.futures
+import logging
+from pathlib import Path
+
+import numpy as np
+import torch
+from huggingface_hub import HfApi
+from requests import HTTPError
+from tqdm import tqdm
+
+from lerobot.datasets.compute_stats import DEFAULT_QUANTILES, aggregate_stats, get_feature_stats
+from lerobot.datasets.dataset_metadata import CODEBASE_VERSION
+from lerobot.datasets.io_utils import write_stats
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.utils.utils import init_logging
+
+
+def has_quantile_stats(stats: dict[str, dict] | None, quantile_list_keys: list[str] | None = None) -> bool:
+    """Check if dataset statistics already contain quantile information.
+
+    Args:
+        stats: Dataset statistics dictionary
+
+    Returns:
+        True if quantile statistics are present, False otherwise
+    """
+    if quantile_list_keys is None:
+        quantile_list_keys = [f"q{int(q * 100):02d}" for q in DEFAULT_QUANTILES]
+
+    if stats is None:
+        return False
+
+    for feature_stats in stats.values():
+        if any(q_key in feature_stats for q_key in quantile_list_keys):
+            return True
+
+    return False
+
+
+def process_single_episode(dataset: LeRobotDataset, episode_idx: int) -> dict:
+    """Process a single episode and return its statistics.
+
+    Args:
+        dataset: The LeRobot dataset
+        episode_idx: Index of the episode to process
+
+    Returns:
+        Dictionary containing episode statistics
+    """
+    logging.info(f"Computing stats for episode {episode_idx}")
+
+    start_idx = dataset.meta.episodes[episode_idx]["dataset_from_index"]
+    end_idx = dataset.meta.episodes[episode_idx]["dataset_to_index"]
+
+    collected_data: dict[str, list] = {}
+    for idx in range(start_idx, end_idx):
+        item = dataset[idx]
+        for key, value in item.items():
+            if key not in dataset.features:
+                continue
+
+            if key not in collected_data:
+                collected_data[key] = []
+            collected_data[key].append(value)
+
+    ep_stats = {}
+    for key, data_list in collected_data.items():
+        if dataset.features[key]["dtype"] == "string":
+            continue
+
+        data = torch.stack(data_list).cpu().numpy()
+        if dataset.features[key]["dtype"] in ["image", "video"]:
+            if data.dtype == np.uint8:
+                data = data.astype(np.float32) / 255.0
+
+            axes_to_reduce = (0, 2, 3)
+            keepdims = True
+        else:
+            axes_to_reduce = 0
+            keepdims = data.ndim == 1
+
+        ep_stats[key] = get_feature_stats(
+            data, axis=axes_to_reduce, keepdims=keepdims, quantile_list=DEFAULT_QUANTILES
+        )
+
+        if dataset.features[key]["dtype"] in ["image", "video"]:
+            ep_stats[key] = {
+                k: v if k == "count" else np.squeeze(v, axis=0) for k, v in ep_stats[key].items()
+            }
+
+    return ep_stats
+
+
+def compute_quantile_stats_for_dataset(dataset: LeRobotDataset) -> dict[str, dict]:
+    """Compute quantile statistics for all episodes in the dataset.
+
+    Args:
+        dataset: The LeRobot dataset to compute statistics for
+
+    Returns:
+        Dictionary containing aggregated statistics with quantiles
+
+    Note:
+        Video decoding operations are not thread-safe, so we process episodes sequentially
+        when video keys are present. For datasets without videos, we use parallel processing
+        with ThreadPoolExecutor for better performance.
+    """
+    logging.info(f"Computing quantile statistics for dataset with {dataset.num_episodes} episodes")
+
+    episode_stats_list = []
+    has_videos = len(dataset.meta.video_keys) > 0
+
+    if has_videos:
+        logging.info("Dataset contains video keys - using sequential processing for thread safety")
+        for episode_idx in tqdm(range(dataset.num_episodes), desc="Processing episodes"):
+            ep_stats = process_single_episode(dataset, episode_idx)
+            episode_stats_list.append(ep_stats)
+    else:
+        logging.info("Dataset has no video keys - using parallel processing for better performance")
+        max_workers = min(dataset.num_episodes, 16)
+
+        with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
+            future_to_episode = {
+                executor.submit(process_single_episode, dataset, episode_idx): episode_idx
+                for episode_idx in range(dataset.num_episodes)
+            }
+
+            episode_results = {}
+            with tqdm(total=dataset.num_episodes, desc="Processing episodes") as pbar:
+                for future in concurrent.futures.as_completed(future_to_episode):
+                    episode_idx = future_to_episode[future]
+                    ep_stats = future.result()
+                    episode_results[episode_idx] = ep_stats
+                    pbar.update(1)
+
+        for episode_idx in range(dataset.num_episodes):
+            if episode_idx in episode_results:
+                episode_stats_list.append(episode_results[episode_idx])
+
+    if not episode_stats_list:
+        raise ValueError("No episode data found for computing statistics")
+
+    logging.info(f"Aggregating statistics from {len(episode_stats_list)} episodes")
+    return aggregate_stats(episode_stats_list)
+
+
+def augment_dataset_with_quantile_stats(
+    repo_id: str,
+    root: str | Path | None = None,
+    overwrite: bool = False,
+) -> None:
+    """Augment a dataset with quantile statistics if they are missing.
+
+    Args:
+        repo_id: Repository ID of the dataset
+        root: Local root directory for the dataset
+        overwrite: Overwrite existing quantile statistics if they already exist
+    """
+    logging.info(f"Loading dataset: {repo_id}")
+    dataset = LeRobotDataset(
+        repo_id=repo_id,
+        root=root,
+    )
+
+    if not overwrite and has_quantile_stats(dataset.meta.stats):
+        logging.info("Dataset already contains quantile statistics. No action needed.")
+        return
+
+    logging.info("Dataset does not contain quantile statistics. Computing them now...")
+
+    new_stats = compute_quantile_stats_for_dataset(dataset)
+
+    logging.info("Updating dataset metadata with new quantile statistics")
+    dataset.meta.stats = new_stats
+
+    write_stats(new_stats, dataset.meta.root)
+
+    logging.info("Successfully updated dataset with quantile statistics")
+    dataset.push_to_hub()
+
+    hub_api = HfApi()
+    try:
+        hub_api.delete_tag(repo_id, tag=CODEBASE_VERSION, repo_type="dataset")
+    except HTTPError as e:
+        logging.info(f"tag={CODEBASE_VERSION} probably doesn't exist. Skipping exception ({e})")
+        pass
+    hub_api.create_tag(repo_id, tag=CODEBASE_VERSION, revision=None, repo_type="dataset")
+
+
+def main():
+    """Main function to run the augmentation script."""
+    parser = argparse.ArgumentParser(description="Augment LeRobot dataset with quantile statistics")
+
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        required=True,
+        help="Repository ID of the dataset (e.g., 'lerobot/pusht')",
+    )
+
+    parser.add_argument(
+        "--root",
+        type=str,
+        help="Local root directory for the dataset",
+    )
+    parser.add_argument(
+        "--overwrite",
+        action="store_true",
+        help="Overwrite existing quantile statistics if they already exist",
+    )
+
+    args = parser.parse_args()
+    root = Path(args.root) if args.root else None
+
+    init_logging()
+
+    augment_dataset_with_quantile_stats(
+        repo_id=args.repo_id,
+        root=root,
+        overwrite=args.overwrite,
+    )
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/convert_dataset_v21_to_v30.py b/lerobot/src/lerobot/scripts/convert_dataset_v21_to_v30.py
new file mode 100644
index 0000000000000000000000000000000000000000..2b6dcf7324d44695f75dc1d614070e5491821676
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/convert_dataset_v21_to_v30.py
@@ -0,0 +1,578 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+This script will help you convert any LeRobot dataset already pushed to the hub from codebase version 2.1 to
+3.0. It will:
+
+- Generate per-episodes stats and writes them in `episodes_stats.jsonl`
+- Check consistency between these new stats and the old ones.
+- Remove the deprecated `stats.json`.
+- Update codebase_version in `info.json`.
+- Push this new version to the hub on the 'main' branch and tags it with "v3.0".
+
+Usage:
+
+Convert a dataset from the hub:
+```bash
+python src/lerobot/scripts/convert_dataset_v21_to_v30.py \
+    --repo-id=lerobot/pusht
+```
+
+Convert a local dataset (works in place):
+```bash
+python src/lerobot/scripts/convert_dataset_v21_to_v30.py \
+    --repo-id=lerobot/pusht \
+    --root=/path/to/local/dataset/directory \
+    --push-to-hub=false
+
+N.B. Path semantics (v2): --root is the exact dataset folder containing
+meta/, data/, videos/. When omitted, defaults to $HF_LEROBOT_HOME/{repo_id}.
+```
+
+"""
+
+import argparse
+import logging
+import shutil
+from pathlib import Path
+from typing import Any
+
+import jsonlines
+import pandas as pd
+import pyarrow as pa
+import tqdm
+from datasets import Dataset, Features, Image
+from huggingface_hub import HfApi, snapshot_download
+from requests import HTTPError
+
+from lerobot.datasets.compute_stats import aggregate_stats
+from lerobot.datasets.dataset_metadata import CODEBASE_VERSION
+from lerobot.datasets.io_utils import (
+    cast_stats_to_numpy,
+    get_file_size_in_mb,
+    get_parquet_file_size_in_mb,
+    get_parquet_num_frames,
+    load_info,
+    write_episodes,
+    write_info,
+    write_stats,
+    write_tasks,
+)
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.utils import (
+    DEFAULT_CHUNK_SIZE,
+    DEFAULT_DATA_FILE_SIZE_IN_MB,
+    DEFAULT_DATA_PATH,
+    DEFAULT_VIDEO_FILE_SIZE_IN_MB,
+    DEFAULT_VIDEO_PATH,
+    LEGACY_EPISODES_PATH,
+    LEGACY_EPISODES_STATS_PATH,
+    LEGACY_TASKS_PATH,
+    flatten_dict,
+    update_chunk_file_indices,
+)
+from lerobot.datasets.video_utils import concatenate_video_files, get_video_duration_in_s
+from lerobot.utils.constants import HF_LEROBOT_HOME
+from lerobot.utils.utils import init_logging
+
+V21 = "v2.1"
+V30 = "v3.0"
+
+"""
+-------------------------
+OLD
+data/chunk-000/episode_000000.parquet
+
+NEW
+data/chunk-000/file_000.parquet
+-------------------------
+OLD
+videos/chunk-000/CAMERA/episode_000000.mp4
+
+NEW
+videos/CAMERA/chunk-000/file_000.mp4
+-------------------------
+OLD
+episodes.jsonl
+{"episode_index": 1, "tasks": ["Put the blue block in the green bowl"], "length": 266}
+
+NEW
+meta/episodes/chunk-000/file_000.parquet
+episode_index | video_chunk_index | video_file_index | data_chunk_index | data_file_index | tasks | length
+-------------------------
+OLD
+tasks.jsonl
+{"task_index": 1, "task": "Put the blue block in the green bowl"}
+
+NEW
+meta/tasks.parquet
+task_index | task
+-------------------------
+OLD
+episodes_stats.jsonl
+{"episode_index": 1, "stats": {"feature_name": {"min": ..., "max": ..., "mean": ..., "std": ..., "count": ...}}}
+
+NEW
+meta/episodes/chunk-000/file_000.parquet
+episode_index | feature_name/min | feature_name/max | feature_name/mean | feature_name/std | feature_name/count
+-------------------------
+UPDATE
+meta/info.json
+-------------------------
+"""
+
+
+def load_jsonlines(fpath: Path) -> list[Any]:
+    with jsonlines.open(fpath, "r") as reader:
+        return list(reader)
+
+
+def legacy_load_episodes(local_dir: Path) -> dict:
+    episodes = load_jsonlines(local_dir / LEGACY_EPISODES_PATH)
+    return {item["episode_index"]: item for item in sorted(episodes, key=lambda x: x["episode_index"])}
+
+
+def legacy_load_episodes_stats(local_dir: Path) -> dict:
+    episodes_stats = load_jsonlines(local_dir / LEGACY_EPISODES_STATS_PATH)
+    return {
+        item["episode_index"]: cast_stats_to_numpy(item["stats"])
+        for item in sorted(episodes_stats, key=lambda x: x["episode_index"])
+    }
+
+
+def legacy_load_tasks(local_dir: Path) -> tuple[dict, dict]:
+    tasks = load_jsonlines(local_dir / LEGACY_TASKS_PATH)
+    tasks = {item["task_index"]: item["task"] for item in sorted(tasks, key=lambda x: x["task_index"])}
+    task_to_task_index = {task: task_index for task_index, task in tasks.items()}
+    return tasks, task_to_task_index
+
+
+def validate_local_dataset_version(local_path: Path) -> None:
+    """Validate that the local dataset has the expected v2.1 version."""
+    info = load_info(local_path)
+    dataset_version = info.get("codebase_version", "unknown")
+    if dataset_version != V21:
+        raise ValueError(
+            f"Local dataset has codebase version '{dataset_version}', expected '{V21}'. "
+            f"This script is specifically for converting v2.1 datasets to v3.0."
+        )
+
+
+def convert_tasks(root, new_root):
+    logging.info(f"Converting tasks from {root} to {new_root}")
+    tasks, _ = legacy_load_tasks(root)
+    task_indices = tasks.keys()
+    task_strings = tasks.values()
+    df_tasks = pd.DataFrame({"task_index": task_indices}, index=pd.Index(task_strings, name="task"))
+    write_tasks(df_tasks, new_root)
+
+
+def concat_data_files(paths_to_cat, new_root, chunk_idx, file_idx, image_keys):
+    # TODO(rcadene): to save RAM use Dataset.from_parquet(file) and concatenate_datasets
+    dataframes = [pd.read_parquet(file) for file in paths_to_cat]
+    # Concatenate all DataFrames along rows
+    concatenated_df = pd.concat(dataframes, ignore_index=True)
+
+    path = new_root / DEFAULT_DATA_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
+    path.parent.mkdir(parents=True, exist_ok=True)
+
+    if len(image_keys) > 0:
+        schema = pa.Schema.from_pandas(concatenated_df)
+        features = Features.from_arrow_schema(schema)
+        for key in image_keys:
+            features[key] = Image()
+        schema = features.arrow_schema
+    else:
+        schema = None
+
+    concatenated_df.to_parquet(path, index=False, schema=schema)
+
+
+def convert_data(root: Path, new_root: Path, data_file_size_in_mb: int):
+    data_dir = root / "data"
+    ep_paths = sorted(data_dir.glob("*/*.parquet"))
+
+    image_keys = get_image_keys(root)
+
+    chunk_idx = 0
+    file_idx = 0
+    size_in_mb = 0
+    num_frames = 0
+    paths_to_cat = []
+    episodes_metadata = []
+
+    logging.info(f"Converting data files from {len(ep_paths)} episodes")
+
+    for ep_idx, ep_path in enumerate(tqdm.tqdm(ep_paths, desc="convert data files")):
+        ep_size_in_mb = get_parquet_file_size_in_mb(ep_path)
+        ep_num_frames = get_parquet_num_frames(ep_path)
+
+        # Check if we need to start a new file BEFORE creating metadata
+        if size_in_mb + ep_size_in_mb >= data_file_size_in_mb and len(paths_to_cat) > 0:
+            # Write the accumulated data files
+            concat_data_files(paths_to_cat, new_root, chunk_idx, file_idx, image_keys)
+
+            # Move to next file
+            chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, DEFAULT_CHUNK_SIZE)
+
+            # Reset for the next file
+            size_in_mb = 0
+            paths_to_cat = []
+
+        # Now create metadata with correct chunk/file indices
+        ep_metadata = {
+            "episode_index": ep_idx,
+            "data/chunk_index": chunk_idx,
+            "data/file_index": file_idx,
+            "dataset_from_index": num_frames,
+            "dataset_to_index": num_frames + ep_num_frames,
+        }
+        size_in_mb += ep_size_in_mb
+        num_frames += ep_num_frames
+        episodes_metadata.append(ep_metadata)
+        paths_to_cat.append(ep_path)
+
+    # Write remaining data if any
+    if paths_to_cat:
+        concat_data_files(paths_to_cat, new_root, chunk_idx, file_idx, image_keys)
+
+    return episodes_metadata
+
+
+def get_video_keys(root):
+    info = load_info(root)
+    features = info["features"]
+    video_keys = [key for key, ft in features.items() if ft["dtype"] == "video"]
+    return video_keys
+
+
+def get_image_keys(root):
+    info = load_info(root)
+    features = info["features"]
+    image_keys = [key for key, ft in features.items() if ft["dtype"] == "image"]
+    return image_keys
+
+
+def convert_videos(root: Path, new_root: Path, video_file_size_in_mb: int):
+    logging.info(f"Converting videos from {root} to {new_root}")
+
+    video_keys = get_video_keys(root)
+    if len(video_keys) == 0:
+        return None
+
+    video_keys = sorted(video_keys)
+
+    eps_metadata_per_cam = []
+    for camera in video_keys:
+        eps_metadata = convert_videos_of_camera(root, new_root, camera, video_file_size_in_mb)
+        eps_metadata_per_cam.append(eps_metadata)
+
+    num_eps_per_cam = [len(eps_cam_map) for eps_cam_map in eps_metadata_per_cam]
+    if len(set(num_eps_per_cam)) != 1:
+        raise ValueError(f"All cams dont have same number of episodes ({num_eps_per_cam}).")
+
+    episods_metadata = []
+    num_cameras = len(video_keys)
+    num_episodes = num_eps_per_cam[0]
+    for ep_idx in tqdm.tqdm(range(num_episodes), desc="convert videos"):
+        # Sanity check
+        ep_ids = [eps_metadata_per_cam[cam_idx][ep_idx]["episode_index"] for cam_idx in range(num_cameras)]
+        ep_ids += [ep_idx]
+        if len(set(ep_ids)) != 1:
+            raise ValueError(f"All episode indices need to match ({ep_ids}).")
+
+        ep_dict = {}
+        for cam_idx in range(num_cameras):
+            ep_dict.update(eps_metadata_per_cam[cam_idx][ep_idx])
+        episods_metadata.append(ep_dict)
+
+    return episods_metadata
+
+
+def convert_videos_of_camera(root: Path, new_root: Path, video_key: str, video_file_size_in_mb: int):
+    # Access old paths to mp4
+    videos_dir = root / "videos"
+    ep_paths = sorted(videos_dir.glob(f"*/{video_key}/*.mp4"))
+
+    ep_idx = 0
+    chunk_idx = 0
+    file_idx = 0
+    size_in_mb = 0
+    duration_in_s = 0.0
+    paths_to_cat = []
+    episodes_metadata = []
+
+    for ep_path in tqdm.tqdm(ep_paths, desc=f"convert videos of {video_key}"):
+        ep_size_in_mb = get_file_size_in_mb(ep_path)
+        ep_duration_in_s = get_video_duration_in_s(ep_path)
+
+        # Check if adding this episode would exceed the limit
+        if size_in_mb + ep_size_in_mb >= video_file_size_in_mb and len(paths_to_cat) > 0:
+            # Size limit would be exceeded, save current accumulation WITHOUT this episode
+            concatenate_video_files(
+                paths_to_cat,
+                new_root
+                / DEFAULT_VIDEO_PATH.format(video_key=video_key, chunk_index=chunk_idx, file_index=file_idx),
+            )
+
+            # Update episodes metadata for the file we just saved
+            for i, _ in enumerate(paths_to_cat):
+                past_ep_idx = ep_idx - len(paths_to_cat) + i
+                episodes_metadata[past_ep_idx][f"videos/{video_key}/chunk_index"] = chunk_idx
+                episodes_metadata[past_ep_idx][f"videos/{video_key}/file_index"] = file_idx
+
+            # Move to next file and start fresh with current episode
+            chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, DEFAULT_CHUNK_SIZE)
+            size_in_mb = 0
+            duration_in_s = 0.0
+            paths_to_cat = []
+
+        # Add current episode metadata
+        ep_metadata = {
+            "episode_index": ep_idx,
+            f"videos/{video_key}/chunk_index": chunk_idx,  # Will be updated when file is saved
+            f"videos/{video_key}/file_index": file_idx,  # Will be updated when file is saved
+            f"videos/{video_key}/from_timestamp": duration_in_s,
+            f"videos/{video_key}/to_timestamp": duration_in_s + ep_duration_in_s,
+        }
+        episodes_metadata.append(ep_metadata)
+
+        # Add current episode to accumulation
+        paths_to_cat.append(ep_path)
+        size_in_mb += ep_size_in_mb
+        duration_in_s += ep_duration_in_s
+        ep_idx += 1
+
+    # Write remaining videos if any
+    if paths_to_cat:
+        concatenate_video_files(
+            paths_to_cat,
+            new_root
+            / DEFAULT_VIDEO_PATH.format(video_key=video_key, chunk_index=chunk_idx, file_index=file_idx),
+        )
+
+        # Update episodes metadata for the final file
+        for i, _ in enumerate(paths_to_cat):
+            past_ep_idx = ep_idx - len(paths_to_cat) + i
+            episodes_metadata[past_ep_idx][f"videos/{video_key}/chunk_index"] = chunk_idx
+            episodes_metadata[past_ep_idx][f"videos/{video_key}/file_index"] = file_idx
+
+    return episodes_metadata
+
+
+def generate_episode_metadata_dict(
+    episodes_legacy_metadata, episodes_metadata, episodes_stats, episodes_videos=None
+):
+    num_episodes = len(episodes_metadata)
+    episodes_legacy_metadata_vals = list(episodes_legacy_metadata.values())
+    episodes_stats_vals = list(episodes_stats.values())
+    episodes_stats_keys = list(episodes_stats.keys())
+
+    for i in range(num_episodes):
+        ep_legacy_metadata = episodes_legacy_metadata_vals[i]
+        ep_metadata = episodes_metadata[i]
+        ep_stats = episodes_stats_vals[i]
+
+        ep_ids_set = {
+            ep_legacy_metadata["episode_index"],
+            ep_metadata["episode_index"],
+            episodes_stats_keys[i],
+        }
+
+        if episodes_videos is None:
+            ep_video = {}
+        else:
+            ep_video = episodes_videos[i]
+            ep_ids_set.add(ep_video["episode_index"])
+
+        if len(ep_ids_set) != 1:
+            raise ValueError(f"Number of episodes is not the same ({ep_ids_set}).")
+
+        ep_dict = {**ep_metadata, **ep_video, **ep_legacy_metadata, **flatten_dict({"stats": ep_stats})}
+        ep_dict["meta/episodes/chunk_index"] = 0
+        ep_dict["meta/episodes/file_index"] = 0
+        yield ep_dict
+
+
+def convert_episodes_metadata(root, new_root, episodes_metadata, episodes_video_metadata=None):
+    logging.info(f"Converting episodes metadata from {root} to {new_root}")
+
+    episodes_legacy_metadata = legacy_load_episodes(root)
+    episodes_stats = legacy_load_episodes_stats(root)
+
+    num_eps_set = {len(episodes_legacy_metadata), len(episodes_metadata)}
+    if episodes_video_metadata is not None:
+        num_eps_set.add(len(episodes_video_metadata))
+
+    if len(num_eps_set) != 1:
+        raise ValueError(f"Number of episodes is not the same ({num_eps_set}).")
+
+    ds_episodes = Dataset.from_generator(
+        lambda: generate_episode_metadata_dict(
+            episodes_legacy_metadata, episodes_metadata, episodes_stats, episodes_video_metadata
+        )
+    )
+    write_episodes(ds_episodes, new_root)
+
+    stats = aggregate_stats(list(episodes_stats.values()))
+    write_stats(stats, new_root)
+
+
+def convert_info(root, new_root, data_file_size_in_mb, video_file_size_in_mb):
+    info = load_info(root)
+    info["codebase_version"] = V30
+    del info["total_chunks"]
+    del info["total_videos"]
+    info["data_files_size_in_mb"] = data_file_size_in_mb
+    info["video_files_size_in_mb"] = video_file_size_in_mb
+    info["data_path"] = DEFAULT_DATA_PATH
+    info["video_path"] = DEFAULT_VIDEO_PATH if info["video_path"] is not None else None
+    info["fps"] = int(info["fps"])
+    logging.info(f"Converting info from {root} to {new_root}")
+    for key in info["features"]:
+        if info["features"][key]["dtype"] == "video":
+            # already has fps in video_info
+            continue
+        info["features"][key]["fps"] = info["fps"]
+    write_info(info, new_root)
+
+
+def convert_dataset(
+    repo_id: str,
+    branch: str | None = None,
+    data_file_size_in_mb: int | None = None,
+    video_file_size_in_mb: int | None = None,
+    root: str | Path | None = None,
+    push_to_hub: bool = True,
+    force_conversion: bool = False,
+):
+    if data_file_size_in_mb is None:
+        data_file_size_in_mb = DEFAULT_DATA_FILE_SIZE_IN_MB
+    if video_file_size_in_mb is None:
+        video_file_size_in_mb = DEFAULT_VIDEO_FILE_SIZE_IN_MB
+
+    # First check if the dataset already has a v3.0 version
+    if root is None and not force_conversion:
+        try:
+            print("Trying to download v3.0 version of the dataset from the hub...")
+            snapshot_download(repo_id, repo_type="dataset", revision=V30, local_dir=HF_LEROBOT_HOME / repo_id)
+            return
+        except Exception:
+            print("Dataset does not have an uploaded v3.0 version. Continuing with conversion.")
+
+    # Set root based on whether local dataset path is provided
+    use_local_dataset = False
+    root = HF_LEROBOT_HOME / repo_id if root is None else Path(root)
+    if root.exists():
+        validate_local_dataset_version(root)
+        use_local_dataset = True
+        print(f"Using local dataset at {root}")
+
+    old_root = root.parent / f"{root.name}_old"
+    new_root = root.parent / f"{root.name}_v30"
+
+    # Handle old_root cleanup if both old_root and root exist
+    if old_root.is_dir() and root.is_dir():
+        shutil.rmtree(str(root))
+        shutil.move(str(old_root), str(root))
+
+    if new_root.is_dir():
+        shutil.rmtree(new_root)
+
+    if not use_local_dataset:
+        snapshot_download(
+            repo_id,
+            repo_type="dataset",
+            revision=V21,
+            local_dir=root,
+        )
+
+    convert_info(root, new_root, data_file_size_in_mb, video_file_size_in_mb)
+    convert_tasks(root, new_root)
+    episodes_metadata = convert_data(root, new_root, data_file_size_in_mb)
+    episodes_videos_metadata = convert_videos(root, new_root, video_file_size_in_mb)
+    convert_episodes_metadata(root, new_root, episodes_metadata, episodes_videos_metadata)
+
+    shutil.move(str(root), str(old_root))
+    shutil.move(str(new_root), str(root))
+
+    if push_to_hub:
+        hub_api = HfApi()
+        try:
+            hub_api.delete_tag(repo_id, tag=CODEBASE_VERSION, repo_type="dataset")
+        except HTTPError as e:
+            print(f"tag={CODEBASE_VERSION} probably doesn't exist. Skipping exception ({e})")
+            pass
+        hub_api.delete_files(
+            delete_patterns=["data/chunk*/episode_*", "meta/*.jsonl", "videos/chunk*"],
+            repo_id=repo_id,
+            revision=branch,
+            repo_type="dataset",
+        )
+        hub_api.create_tag(repo_id, tag=CODEBASE_VERSION, revision=branch, repo_type="dataset")
+
+        LeRobotDataset(repo_id).push_to_hub()
+
+
+if __name__ == "__main__":
+    init_logging()
+    parser = argparse.ArgumentParser()
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        required=True,
+        help="Repository identifier on Hugging Face: a community or a user name `/` the name of the dataset "
+        "(e.g. `lerobot/pusht`, `<USER>/aloha_sim_insertion_human`).",
+    )
+    parser.add_argument(
+        "--branch",
+        type=str,
+        default=None,
+        help="Repo branch to push your dataset. Defaults to the main branch.",
+    )
+    parser.add_argument(
+        "--data-file-size-in-mb",
+        type=int,
+        default=None,
+        help="File size in MB. Defaults to 100 for data and 500 for videos.",
+    )
+    parser.add_argument(
+        "--video-file-size-in-mb",
+        type=int,
+        default=None,
+        help="File size in MB. Defaults to 100 for data and 500 for videos.",
+    )
+    parser.add_argument(
+        "--root",
+        type=str,
+        default=None,
+        help="Local directory to use for downloading/writing the dataset. Defaults to $HF_LEROBOT_HOME/repo_id.",
+    )
+    parser.add_argument(
+        "--push-to-hub",
+        type=lambda input: input.lower() == "true",
+        default=True,
+        help="Push the converted dataset to the hub.",
+    )
+    parser.add_argument(
+        "--force-conversion",
+        action="store_true",
+        help="Force conversion even if the dataset already has a v3.0 version.",
+    )
+
+    args = parser.parse_args()
+    convert_dataset(**vars(args))
diff --git a/lerobot/src/lerobot/scripts/lerobot_calibrate.py b/lerobot/src/lerobot/scripts/lerobot_calibrate.py
new file mode 100644
index 0000000000000000000000000000000000000000..242067978a75f9b686859817220a770b19199581
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_calibrate.py
@@ -0,0 +1,103 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Helper to recalibrate your device (robot or teleoperator).
+
+Example:
+
+```shell
+lerobot-calibrate \
+    --teleop.type=so100_leader \
+    --teleop.port=/dev/tty.usbmodem58760431551 \
+    --teleop.id=blue
+```
+"""
+
+import logging
+from dataclasses import asdict, dataclass
+from pprint import pformat
+
+import draccus
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig  # noqa: F401
+from lerobot.cameras.realsense.configuration_realsense import RealSenseCameraConfig  # noqa: F401
+from lerobot.robots import (  # noqa: F401
+    Robot,
+    RobotConfig,
+    bi_openarm_follower,
+    bi_so_follower,
+    hope_jr,
+    koch_follower,
+    lekiwi,
+    make_robot_from_config,
+    omx_follower,
+    openarm_follower,
+    so_follower,
+)
+from lerobot.teleoperators import (  # noqa: F401
+    Teleoperator,
+    TeleoperatorConfig,
+    bi_openarm_leader,
+    bi_so_leader,
+    homunculus,
+    koch_leader,
+    make_teleoperator_from_config,
+    omx_leader,
+    openarm_leader,
+    openarm_mini,
+    so_leader,
+    unitree_g1,
+)
+from lerobot.utils.import_utils import register_third_party_plugins
+from lerobot.utils.utils import init_logging
+
+
+@dataclass
+class CalibrateConfig:
+    teleop: TeleoperatorConfig | None = None
+    robot: RobotConfig | None = None
+
+    def __post_init__(self):
+        if bool(self.teleop) == bool(self.robot):
+            raise ValueError("Choose either a teleop or a robot.")
+
+        self.device = self.robot if self.robot else self.teleop
+
+
+@draccus.wrap()
+def calibrate(cfg: CalibrateConfig):
+    init_logging()
+    logging.info(pformat(asdict(cfg)))
+
+    if isinstance(cfg.device, RobotConfig):
+        device = make_robot_from_config(cfg.device)
+    elif isinstance(cfg.device, TeleoperatorConfig):
+        device = make_teleoperator_from_config(cfg.device)
+
+    device.connect(calibrate=False)
+
+    try:
+        device.calibrate()
+    finally:
+        device.disconnect()
+
+
+def main():
+    register_third_party_plugins()
+    calibrate()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_dataset_viz.py b/lerobot/src/lerobot/scripts/lerobot_dataset_viz.py
new file mode 100644
index 0000000000000000000000000000000000000000..c4b676c67f655952c055834095b66b1492f769b6
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_dataset_viz.py
@@ -0,0 +1,303 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+""" Visualize data of **all** frames of any episode of a dataset of type LeRobotDataset.
+
+Note: The last frame of the episode doesn't always correspond to a final state.
+That's because our datasets are composed of transition from state to state up to
+the antepenultimate state associated to the ultimate action to arrive in the final state.
+However, there might not be a transition from a final state to another state.
+
+Note: This script aims to visualize the data used to train the neural networks.
+~What you see is what you get~. When visualizing image modality, it is often expected to observe
+lossy compression artifacts since these images have been decoded from compressed mp4 videos to
+save disk space. The compression factor applied has been tuned to not affect success rate.
+
+Examples:
+
+- Visualize data stored on a local machine:
+```
+local$ lerobot-dataset-viz \
+    --repo-id lerobot/pusht \
+    --episode-index 0
+```
+
+- Visualize data stored on a distant machine with a local viewer:
+```
+distant$ lerobot-dataset-viz \
+    --repo-id lerobot/pusht \
+    --episode-index 0 \
+    --save 1 \
+    --output-dir path/to/directory
+
+local$ scp distant:path/to/directory/lerobot_pusht_episode_0.rrd .
+local$ rerun lerobot_pusht_episode_0.rrd
+```
+
+- Visualize data stored on a distant machine through streaming:
+```
+distant$ lerobot-dataset-viz \
+    --repo-id lerobot/pusht \
+    --episode-index 0 \
+    --mode distant \
+    --grpc-port 9876
+
+local$ rerun rerun+http://IP:GRPC_PORT/proxy
+```
+
+"""
+
+import argparse
+import gc
+import logging
+import time
+from pathlib import Path
+
+import numpy as np
+import rerun as rr
+import torch
+import torch.utils.data
+import tqdm
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.utils.constants import ACTION, DONE, OBS_STATE, REWARD
+from lerobot.utils.utils import init_logging
+
+
+def to_hwc_uint8_numpy(chw_float32_torch: torch.Tensor) -> np.ndarray:
+    assert chw_float32_torch.dtype == torch.float32
+    assert chw_float32_torch.ndim == 3
+    c, h, w = chw_float32_torch.shape
+    assert c < h and c < w, f"expect channel first images, but instead {chw_float32_torch.shape}"
+    hwc_uint8_numpy = (chw_float32_torch * 255).type(torch.uint8).permute(1, 2, 0).numpy()
+    return hwc_uint8_numpy
+
+
+def visualize_dataset(
+    dataset: LeRobotDataset,
+    episode_index: int,
+    batch_size: int = 32,
+    num_workers: int = 0,
+    mode: str = "local",
+    web_port: int = 9090,
+    grpc_port: int = 9876,
+    save: bool = False,
+    output_dir: Path | None = None,
+    display_compressed_images: bool = False,
+    **kwargs,
+) -> Path | None:
+    if save:
+        assert output_dir is not None, (
+            "Set an output directory where to write .rrd files with `--output-dir path/to/directory`."
+        )
+
+    repo_id = dataset.repo_id
+
+    logging.info("Loading dataloader")
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        num_workers=num_workers,
+        batch_size=batch_size,
+    )
+
+    logging.info("Starting Rerun")
+
+    if mode not in ["local", "distant"]:
+        raise ValueError(mode)
+
+    spawn_local_viewer = mode == "local" and not save
+    rr.init(f"{repo_id}/episode_{episode_index}", spawn=spawn_local_viewer)
+
+    # Manually call python garbage collector after `rr.init` to avoid hanging in a blocking flush
+    # when iterating on a dataloader with `num_workers` > 0
+    # TODO(rcadene): remove `gc.collect` when rerun version 0.16 is out, which includes a fix
+    gc.collect()
+
+    if mode == "distant":
+        server_uri = rr.serve_grpc(grpc_port=grpc_port)
+        logging.info(f"Connect to a Rerun Server: rerun rerun+http://IP:{grpc_port}/proxy")
+        rr.serve_web_viewer(open_browser=False, web_port=web_port, connect_to=server_uri)
+
+    logging.info("Logging to Rerun")
+
+    first_index = None
+    for batch in tqdm.tqdm(dataloader, total=len(dataloader)):
+        if first_index is None:
+            first_index = batch["index"][0].item()
+        # iterate over the batch
+        for i in range(len(batch["index"])):
+            rr.set_time("frame_index", sequence=batch["index"][i].item() - first_index)
+            rr.set_time("timestamp", timestamp=batch["timestamp"][i].item())
+
+            # display each camera image
+            for key in dataset.meta.camera_keys:
+                img = to_hwc_uint8_numpy(batch[key][i])
+                img_entity = rr.Image(img).compress() if display_compressed_images else rr.Image(img)
+                rr.log(key, entity=img_entity)
+
+            # display each dimension of action space (e.g. actuators command)
+            if ACTION in batch:
+                for dim_idx, val in enumerate(batch[ACTION][i]):
+                    rr.log(f"{ACTION}/{dim_idx}", rr.Scalars(val.item()))
+
+            # display each dimension of observed state space (e.g. agent position in joint space)
+            if OBS_STATE in batch:
+                for dim_idx, val in enumerate(batch[OBS_STATE][i]):
+                    rr.log(f"state/{dim_idx}", rr.Scalars(val.item()))
+
+            if DONE in batch:
+                rr.log(DONE, rr.Scalars(batch[DONE][i].item()))
+
+            if REWARD in batch:
+                rr.log(REWARD, rr.Scalars(batch[REWARD][i].item()))
+
+            if "next.success" in batch:
+                rr.log("next.success", rr.Scalars(batch["next.success"][i].item()))
+
+    if mode == "local" and save:
+        # save .rrd locally
+        output_dir = Path(output_dir)
+        output_dir.mkdir(parents=True, exist_ok=True)
+        repo_id_str = repo_id.replace("/", "_")
+        rrd_path = output_dir / f"{repo_id_str}_episode_{episode_index}.rrd"
+        rr.save(rrd_path)
+        return rrd_path
+
+    elif mode == "distant":
+        # stop the process from exiting since it is serving the websocket connection
+        try:
+            while True:
+                time.sleep(1)
+        except KeyboardInterrupt:
+            print("Ctrl-C received. Exiting.")
+
+
+def main():
+    parser = argparse.ArgumentParser()
+
+    parser.add_argument(
+        "--repo-id",
+        type=str,
+        required=True,
+        help="Name of hugging face repository containing a LeRobotDataset dataset (e.g. `lerobot/pusht`).",
+    )
+    parser.add_argument(
+        "--episode-index",
+        type=int,
+        required=True,
+        help="Episode to visualize.",
+    )
+    parser.add_argument(
+        "--root",
+        type=Path,
+        default=None,
+        help="Root directory for the dataset stored locally (e.g. `--root data`). By default, the dataset will be loaded from hugging face cache folder, or downloaded from the hub if available.",
+    )
+    parser.add_argument(
+        "--output-dir",
+        type=Path,
+        default=None,
+        help="Directory path to write a .rrd file when `--save 1` is set.",
+    )
+    parser.add_argument(
+        "--batch-size",
+        type=int,
+        default=32,
+        help="Batch size loaded by DataLoader.",
+    )
+    parser.add_argument(
+        "--num-workers",
+        type=int,
+        default=4,
+        help="Number of processes of Dataloader for loading the data.",
+    )
+    parser.add_argument(
+        "--mode",
+        type=str,
+        default="local",
+        help=(
+            "Mode of viewing between 'local' or 'distant'. "
+            "'local' requires data to be on a local machine. It spawns a viewer to visualize the data locally. "
+            "'distant' creates a server on the distant machine where the data is stored. "
+            "Visualize the data by connecting to the server with `rerun rerun+http://IP:GRPC_PORT/proxy` on the local machine."
+        ),
+    )
+    parser.add_argument(
+        "--web-port",
+        type=int,
+        default=9090,
+        help="Web port for rerun.io when `--mode distant` is set.",
+    )
+    parser.add_argument(
+        "--ws-port",
+        type=int,
+        help="deprecated, please use --grpc-port instead.",
+    )
+    parser.add_argument(
+        "--grpc-port",
+        type=int,
+        default=9876,
+        help="gRPC port for rerun.io when `--mode distant` is set.",
+    )
+    parser.add_argument(
+        "--save",
+        type=int,
+        default=0,
+        help=(
+            "Save a .rrd file in the directory provided by `--output-dir`. "
+            "It also deactivates the spawning of a viewer. "
+            "Visualize the data by running `rerun path/to/file.rrd` on your local machine."
+        ),
+    )
+
+    parser.add_argument(
+        "--tolerance-s",
+        type=float,
+        default=1e-4,
+        help=(
+            "Tolerance in seconds used to ensure data timestamps respect the dataset fps value"
+            "This is argument passed to the constructor of LeRobotDataset and maps to its tolerance_s constructor argument"
+            "If not given, defaults to 1e-4."
+        ),
+    )
+
+    parser.add_argument(
+        "--display-compressed-images",
+        action="store_true",
+        help="If set, display compressed images in Rerun instead of uncompressed ones.",
+    )
+
+    args = parser.parse_args()
+    kwargs = vars(args)
+    repo_id = kwargs.pop("repo_id")
+    root = kwargs.pop("root")
+    tolerance_s = kwargs.pop("tolerance_s")
+
+    if kwargs["ws_port"] is not None:
+        logging.warning(
+            "--ws-port is deprecated and will be removed in future versions. Please use --grpc-port instead."
+        )
+        logging.warning("Setting grpc_port to ws_port value.")
+        kwargs["grpc_port"] = kwargs.pop("ws_port")
+
+    init_logging()
+    logging.info("Loading dataset")
+    dataset = LeRobotDataset(repo_id, episodes=[args.episode_index], root=root, tolerance_s=tolerance_s)
+
+    visualize_dataset(dataset, **vars(args))
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_edit_dataset.py b/lerobot/src/lerobot/scripts/lerobot_edit_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..49825317db3dfb175e46bb2d9e14346ae5e55be7
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_edit_dataset.py
@@ -0,0 +1,612 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Edit LeRobot datasets using various transformation tools.
+
+This script allows you to delete episodes, split datasets, merge datasets,
+remove features, modify tasks, and convert image datasets to video format.
+When new_repo_id is specified, creates a new dataset.
+
+Path semantics (v2): --root and --new_root are exact dataset folders containing
+meta/, data/, videos/. When omitted, defaults to $HF_LEROBOT_HOME/{repo_id}.
+
+Usage Examples:
+
+Delete episodes 0, 2, and 5 from a dataset:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --operation.type delete_episodes \
+        --operation.episode_indices "[0, 2, 5]"
+
+Delete episodes from a local dataset at a specific path:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --root /path/to/pusht \
+        --operation.type delete_episodes \
+        --operation.episode_indices "[0, 2, 5]"
+
+Delete episodes and save to a new dataset at a specific path and with a new repo_id:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --new_repo_id lerobot/pusht_filtered \
+        --new_root /path/to/pusht_filtered \
+        --operation.type delete_episodes \
+        --operation.episode_indices "[0, 2, 5]"
+
+Split dataset by fractions (pusht_train, pusht_val):
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --operation.type split \
+        --operation.splits '{"train": 0.8, "val": 0.2}'
+
+Split dataset by fractions and save split datasets to a specific folder (base_folder/train, base_folder/val):
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --new_root /path/to/base_folder \
+        --operation.type split \
+        --operation.splits '{"train": 0.8, "val": 0.2}'
+
+Split dataset by episode indices:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --operation.type split \
+        --operation.splits '{"train": [0, 1, 2, 3], "val": [4, 5]}'
+
+Split into more than two splits:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --operation.type split \
+        --operation.splits '{"train": 0.6, "val": 0.2, "test": 0.2}'
+
+Merge multiple datasets:
+    lerobot-edit-dataset \
+        --new_repo_id lerobot/pusht_merged \
+        --operation.type merge \
+        --operation.repo_ids "['lerobot/pusht_train', 'lerobot/pusht_val']"
+
+Merge multiple datasets to a specific output path:
+    lerobot-edit-dataset \
+        --new_repo_id lerobot/pusht_merged \
+        --new_root /path/to/pusht_merged \
+        --operation.type merge \
+        --operation.repo_ids "['lerobot/pusht_train', 'lerobot/pusht_val']"
+
+Merge multiple datasets from a list of local dataset paths:
+    lerobot-edit-dataset \
+        --new_repo_id lerobot/pusht_merged \
+        --operation.type merge \
+        --operation.repo_ids "['pusht_train', 'pusht_val']" \
+        --operation.roots "['/path/to/pusht_train', '/path/to/pusht_val']"
+
+Remove camera feature:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --operation.type remove_feature \
+        --operation.feature_names "['observation.image']"
+
+Modify tasks - set a single task for all episodes (WARNING: modifies in-place):
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --operation.type modify_tasks \
+        --operation.new_task "Pick up the cube and place it"
+
+Modify tasks - set different tasks for specific episodes (WARNING: modifies in-place):
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --operation.type modify_tasks \
+        --operation.episode_tasks '{"0": "Task A", "1": "Task B", "2": "Task A"}'
+
+Modify tasks - set default task with overrides for specific episodes (WARNING: modifies in-place):
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht \
+        --operation.type modify_tasks \
+        --operation.new_task "Default task" \
+        --operation.episode_tasks '{"5": "Special task for episode 5"}'
+
+Convert image dataset to video format and save locally:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht_image \
+        --new_root /path/to/output/pusht_video \
+        --operation.type convert_image_to_video
+
+Convert image dataset to video format and save with new repo_id:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht_image \
+        --new_repo_id lerobot/pusht_video \
+        --operation.type convert_image_to_video
+
+Convert image dataset to video format and push to hub:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht_image \
+        --new_repo_id lerobot/pusht_video \
+        --operation.type convert_image_to_video \
+        --push_to_hub true
+
+Show dataset information:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht_image \
+        --operation.type info \
+        --operation.show_features true
+
+Show dataset information without feature details:
+    lerobot-edit-dataset \
+        --repo_id lerobot/pusht_image \
+        --operation.type info \
+        --operation.show_features false
+
+Using JSON config file:
+    lerobot-edit-dataset \
+        --config_path path/to/edit_config.json
+"""
+
+import abc
+import logging
+import shutil
+import sys
+from dataclasses import dataclass
+from pathlib import Path
+
+import draccus
+
+from lerobot.configs import parser
+from lerobot.datasets.dataset_tools import (
+    convert_image_to_video_dataset,
+    delete_episodes,
+    merge_datasets,
+    modify_tasks,
+    remove_feature,
+    split_dataset,
+)
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.utils.constants import HF_LEROBOT_HOME
+from lerobot.utils.utils import init_logging
+
+
+@dataclass
+class OperationConfig(draccus.ChoiceRegistry, abc.ABC):
+    @property
+    def type(self) -> str:
+        return self.get_choice_name(self.__class__)
+
+
+@OperationConfig.register_subclass("delete_episodes")
+@dataclass
+class DeleteEpisodesConfig(OperationConfig):
+    episode_indices: list[int] | None = None
+
+
+@OperationConfig.register_subclass("split")
+@dataclass
+class SplitConfig(OperationConfig):
+    splits: dict[str, float | list[int]] | None = None
+
+
+@OperationConfig.register_subclass("merge")
+@dataclass
+class MergeConfig(OperationConfig):
+    repo_ids: list[str] | None = None
+    roots: list[str] | None = None
+
+
+@OperationConfig.register_subclass("remove_feature")
+@dataclass
+class RemoveFeatureConfig(OperationConfig):
+    feature_names: list[str] | None = None
+
+
+@OperationConfig.register_subclass("modify_tasks")
+@dataclass
+class ModifyTasksConfig(OperationConfig):
+    new_task: str | None = None
+    episode_tasks: dict[str, str] | None = None
+
+
+@OperationConfig.register_subclass("convert_image_to_video")
+@dataclass
+class ConvertImageToVideoConfig(OperationConfig):
+    output_dir: str | None = None
+    vcodec: str = "libsvtav1"
+    pix_fmt: str = "yuv420p"
+    g: int = 2
+    crf: int = 30
+    fast_decode: int = 0
+    episode_indices: list[int] | None = None
+    num_workers: int = 4
+    max_episodes_per_batch: int | None = None
+    max_frames_per_batch: int | None = None
+
+
+@OperationConfig.register_subclass("info")
+@dataclass
+class InfoConfig(OperationConfig):
+    show_features: bool = False
+
+
+@dataclass
+class EditDatasetConfig:
+    # Operation configuration.
+    operation: OperationConfig
+    # Input dataset identifier. Always required unless for Merge operation.
+    repo_id: str | None = None
+    # Root directory where the input dataset is stored. If not specified, defaults to $HF_LEROBOT_HOME/repo_id.
+    root: str | None = None
+    # Edited dataset identifier. When both new_repo_id (resp. new_root) and repo_id (resp. root) are identical, modifications are applied in-place and a backup of the original dataset is created. Required for Merge operation.
+    new_repo_id: str | None = None
+    # Root directory where the edited dataset will be stored. If not specified, defaults to $HF_LEROBOT_HOME/new_repo_id. For Split operation, this is the base directory for the split datasets.
+    new_root: str | None = None
+    # Upload dataset to Hugging Face hub.
+    push_to_hub: bool = False
+
+
+def get_output_path(
+    repo_id: str,
+    new_repo_id: str | None,
+    root: Path | str | None,
+    new_root: Path | str | None,
+) -> tuple[str, Path]:
+    input_path = Path(root) if root else HF_LEROBOT_HOME / repo_id
+
+    output_repo_id = new_repo_id if new_repo_id else repo_id
+    output_path = Path(new_root) if new_root else HF_LEROBOT_HOME / output_repo_id
+
+    # In case of in-place modification, create a backup of the original dataset (if it exists)
+    if output_path == input_path:
+        backup_path = input_path.with_name(input_path.name + "_old")
+
+        if input_path.exists():
+            if backup_path.exists():
+                shutil.rmtree(backup_path)
+            shutil.move(input_path, backup_path)
+
+    return output_repo_id, output_path
+
+
+def handle_delete_episodes(cfg: EditDatasetConfig) -> None:
+    if not isinstance(cfg.operation, DeleteEpisodesConfig):
+        raise ValueError("Operation config must be DeleteEpisodesConfig")
+
+    if not cfg.operation.episode_indices:
+        raise ValueError("episode_indices must be specified for delete_episodes operation")
+
+    dataset = LeRobotDataset(cfg.repo_id, root=cfg.root)
+    output_repo_id, output_dir = get_output_path(
+        cfg.repo_id,
+        new_repo_id=cfg.new_repo_id,
+        root=cfg.root,
+        new_root=cfg.new_root,
+    )
+
+    # In case of in-place modification, make the dataset point to the backup directory
+    if output_dir == dataset.root:
+        dataset.root = dataset.root.with_name(dataset.root.name + "_old")
+
+    logging.info(f"Deleting episodes {cfg.operation.episode_indices} from {cfg.repo_id}")
+    new_dataset = delete_episodes(
+        dataset,
+        episode_indices=cfg.operation.episode_indices,
+        output_dir=output_dir,
+        repo_id=output_repo_id,
+    )
+
+    logging.info(f"Dataset saved to {output_dir}")
+    logging.info(f"Episodes: {new_dataset.meta.total_episodes}, Frames: {new_dataset.meta.total_frames}")
+
+    if cfg.push_to_hub:
+        logging.info(f"Pushing to hub as {output_repo_id}")
+        LeRobotDataset(output_repo_id, root=output_dir).push_to_hub()
+
+
+def handle_split(cfg: EditDatasetConfig) -> None:
+    if not isinstance(cfg.operation, SplitConfig):
+        raise ValueError("Operation config must be SplitConfig")
+
+    if not cfg.operation.splits:
+        raise ValueError(
+            "splits dict must be specified with split names as keys and fractions/episode lists as values"
+        )
+
+    if cfg.new_repo_id is not None:
+        logging.warning(
+            "split uses the original dataset identifier --repo_id to generate split names. The --new_repo_id parameter is ignored."
+        )
+
+    dataset = LeRobotDataset(cfg.repo_id, root=cfg.root)
+
+    logging.info(f"Splitting dataset {cfg.repo_id} with splits: {cfg.operation.splits}")
+    split_datasets = split_dataset(
+        dataset,
+        splits=cfg.operation.splits,
+        output_dir=cfg.new_root,
+    )
+
+    for split_name, split_ds in split_datasets.items():
+        logging.info(
+            f"{split_name}: {split_ds.meta.total_episodes} episodes, {split_ds.meta.total_frames} frames"
+        )
+
+        if cfg.push_to_hub:
+            logging.info(f"Pushing {split_name} split to hub as {split_ds.repo_id}")
+            LeRobotDataset(split_ds.repo_id, root=split_ds.root).push_to_hub()
+
+
+def handle_merge(cfg: EditDatasetConfig) -> None:
+    if not isinstance(cfg.operation, MergeConfig):
+        raise ValueError("Operation config must be MergeConfig")
+
+    if not cfg.operation.repo_ids:
+        raise ValueError("repo_ids must be specified for merge operation")
+
+    if cfg.repo_id is not None or cfg.root is not None:
+        logging.warning(
+            "merge uses --new_repo_id and --new_root for the merged dataset. The --repo_id and --root parameters are ignored."
+        )
+
+    if cfg.operation.roots:
+        if len(cfg.operation.roots) != len(cfg.operation.repo_ids):
+            raise ValueError("repo_ids and roots must have the same length for merge operation")
+        logging.info(f"Loading {len(cfg.operation.roots)} datasets to merge")
+        datasets = [
+            LeRobotDataset(repo_id=repo_id, root=root)
+            for repo_id, root in zip(cfg.operation.repo_ids, cfg.operation.roots, strict=True)
+        ]
+    else:
+        logging.info(f"Loading {len(cfg.operation.repo_ids)} datasets to merge")
+        datasets = [LeRobotDataset(repo_id) for repo_id in cfg.operation.repo_ids]
+
+    output_dir = Path(cfg.new_root) if cfg.new_root else HF_LEROBOT_HOME / cfg.new_repo_id
+
+    logging.info(f"Merging datasets into {cfg.new_repo_id}")
+    merged_dataset = merge_datasets(
+        datasets,
+        output_repo_id=cfg.new_repo_id,
+        output_dir=output_dir,
+    )
+
+    logging.info(f"Merged dataset saved to {output_dir}")
+    logging.info(
+        f"Episodes: {merged_dataset.meta.total_episodes}, Frames: {merged_dataset.meta.total_frames}"
+    )
+
+    if cfg.push_to_hub:
+        logging.info(f"Pushing to hub as {cfg.new_repo_id}")
+        LeRobotDataset(merged_dataset.repo_id, root=output_dir).push_to_hub()
+
+
+def handle_remove_feature(cfg: EditDatasetConfig) -> None:
+    if not isinstance(cfg.operation, RemoveFeatureConfig):
+        raise ValueError("Operation config must be RemoveFeatureConfig")
+
+    if not cfg.operation.feature_names:
+        raise ValueError("feature_names must be specified for remove_feature operation")
+
+    dataset = LeRobotDataset(cfg.repo_id, root=cfg.root)
+    output_repo_id, output_dir = get_output_path(
+        cfg.repo_id,
+        new_repo_id=cfg.new_repo_id,
+        root=cfg.root,
+        new_root=cfg.new_root,
+    )
+
+    # In case of in-place modification, make the dataset point to the backup directory
+    if output_dir == dataset.root:
+        dataset.root = dataset.root.with_name(dataset.root.name + "_old")
+
+    logging.info(f"Removing features {cfg.operation.feature_names} from {cfg.repo_id}")
+    new_dataset = remove_feature(
+        dataset,
+        feature_names=cfg.operation.feature_names,
+        output_dir=output_dir,
+        repo_id=output_repo_id,
+    )
+
+    logging.info(f"Dataset saved to {output_dir}")
+    logging.info(f"Remaining features: {list(new_dataset.meta.features.keys())}")
+
+    if cfg.push_to_hub:
+        logging.info(f"Pushing to hub as {output_repo_id}")
+        LeRobotDataset(output_repo_id, root=output_dir).push_to_hub()
+
+
+def handle_modify_tasks(cfg: EditDatasetConfig) -> None:
+    if not isinstance(cfg.operation, ModifyTasksConfig):
+        raise ValueError("Operation config must be ModifyTasksConfig")
+
+    new_task = cfg.operation.new_task
+    episode_tasks_raw = cfg.operation.episode_tasks
+
+    if new_task is None and episode_tasks_raw is None:
+        raise ValueError("Must specify at least one of new_task or episode_tasks for modify_tasks operation")
+
+    if cfg.new_repo_id is not None or cfg.new_root is not None:
+        logging.warning(
+            "modify_tasks modifies datasets in-place. The --new_repo_id and --new_root parameters are ignored."
+        )
+
+    dataset = LeRobotDataset(cfg.repo_id, root=cfg.root)
+    logging.warning(f"Modifying dataset in-place at {dataset.root}. Original data will be overwritten.")
+
+    # Convert episode_tasks keys from string to int if needed (CLI passes strings)
+    episode_tasks: dict[int, str] | None = None
+    if episode_tasks_raw is not None:
+        episode_tasks = {int(k): v for k, v in episode_tasks_raw.items()}
+
+    logging.info(f"Modifying tasks in {cfg.repo_id}")
+    if new_task:
+        logging.info(f"  Default task: '{new_task}'")
+    if episode_tasks:
+        logging.info(f"  Episode-specific tasks: {episode_tasks}")
+
+    modified_dataset = modify_tasks(
+        dataset,
+        new_task=new_task,
+        episode_tasks=episode_tasks,
+    )
+
+    logging.info(f"Dataset modified at {dataset.root}")
+    logging.info(f"Tasks: {list(modified_dataset.meta.tasks.index)}")
+
+    if cfg.push_to_hub:
+        logging.info(f"Pushing to hub as {cfg.repo_id}")
+        modified_dataset.push_to_hub()
+
+
+def handle_convert_image_to_video(cfg: EditDatasetConfig) -> None:
+    # Note: Parser may create any config type with the right fields, so we access fields directly
+    # instead of checking isinstance()
+    dataset = LeRobotDataset(cfg.repo_id, root=cfg.root)
+
+    # Determine output directory and repo_id
+    # Priority: 1) new_root, 2) new_repo_id, 3) operation.output_dir, 4) auto-generated name
+    output_dir_config = getattr(cfg.operation, "output_dir", None)
+    if output_dir_config:
+        logging.warning(
+            "--operation.output_dir is deprecated and will be removed in future versions. "
+            "Please use --new_root instead."
+        )
+
+    if cfg.new_root:
+        output_dir = Path(cfg.new_root)
+        output_repo_id = cfg.new_repo_id or f"{cfg.repo_id}_video"
+        logging.info(f"Saving to new_root: {output_dir} as {output_repo_id}")
+    elif cfg.new_repo_id:
+        output_repo_id = cfg.new_repo_id
+        output_dir = HF_LEROBOT_HOME / cfg.new_repo_id
+        logging.info(f"Saving to new dataset: {cfg.new_repo_id} at {output_dir}")
+    elif output_dir_config:
+        output_dir = Path(output_dir_config)
+        output_repo_id = output_dir.name
+        logging.info(f"Saving to local directory: {output_dir} as {output_repo_id}")
+    else:
+        output_repo_id = f"{cfg.repo_id}_video"
+        output_dir = HF_LEROBOT_HOME / output_repo_id
+        logging.info(f"Saving to auto-generated location: {output_dir} as {output_repo_id}")
+
+    logging.info(f"Converting dataset {cfg.repo_id} to video format")
+
+    new_dataset = convert_image_to_video_dataset(
+        dataset=dataset,
+        output_dir=output_dir,
+        repo_id=output_repo_id,
+        vcodec=getattr(cfg.operation, "vcodec", "libsvtav1"),
+        pix_fmt=getattr(cfg.operation, "pix_fmt", "yuv420p"),
+        g=getattr(cfg.operation, "g", 2),
+        crf=getattr(cfg.operation, "crf", 30),
+        fast_decode=getattr(cfg.operation, "fast_decode", 0),
+        episode_indices=getattr(cfg.operation, "episode_indices", None),
+        num_workers=getattr(cfg.operation, "num_workers", 4),
+        max_episodes_per_batch=getattr(cfg.operation, "max_episodes_per_batch", None),
+        max_frames_per_batch=getattr(cfg.operation, "max_frames_per_batch", None),
+    )
+
+    logging.info("Video dataset created successfully!")
+    logging.info(f"Location: {output_dir}")
+    logging.info(f"Episodes: {new_dataset.meta.total_episodes}")
+    logging.info(f"Frames: {new_dataset.meta.total_frames}")
+
+    if cfg.push_to_hub:
+        logging.info(f"Pushing to hub as {output_repo_id}...")
+        new_dataset.push_to_hub()
+        logging.info("✓ Successfully pushed to hub!")
+    else:
+        logging.info("Dataset saved locally (not pushed to hub)")
+
+
+def _get_dataset_size(repo_path):
+    import os
+
+    total = 0
+    with os.scandir(repo_path) as it:
+        for entry in it:
+            if entry.is_file():
+                total += entry.stat().st_size
+            elif entry.is_dir():
+                total += _get_dataset_size(entry.path)
+    return total
+
+
+def handle_info(cfg: EditDatasetConfig):
+    if not isinstance(cfg.operation, InfoConfig):
+        raise ValueError("Operation config must be InfoConfig")
+
+    dataset = LeRobotDataset(cfg.repo_id, root=cfg.root)
+    sys.stdout.write(f"======Info {dataset.meta.repo_id}\n")
+    sys.stdout.write(f"Repository ID: {dataset.meta.repo_id} \n")
+    sys.stdout.write(f"Total episode: {dataset.meta.total_episodes} \n")
+    sys.stdout.write(f"Total task: {dataset.meta.total_tasks} \n")
+    sys.stdout.write(f"Total frame(Actual Count): {dataset.meta.total_frames}({len(dataset)}) \n")
+    sys.stdout.write(
+        f"Average frame per episode: {dataset.meta.total_frames / dataset.meta.total_episodes:.1f}\n"
+    )
+    sys.stdout.write(
+        f"Average episode time(sec): {(dataset.meta.total_frames / dataset.meta.total_episodes) / dataset.meta.fps:.1f}\n"
+    )
+    sys.stdout.write(f"FPS: {dataset.meta.fps}\n")
+
+    total_file_size = _get_dataset_size(dataset.root)
+    sys.stdout.write(f"Size: {total_file_size / (1024 * 1024):.1f} MB\n")
+    if cfg.operation.show_features:
+        import json
+
+        feature_dump_str = json.dumps(
+            dataset.meta.features, ensure_ascii=False, indent=4, sort_keys=True, separators=(",", ": ")
+        )
+        sys.stdout.write("Features:\n")
+        sys.stdout.write(f"{feature_dump_str}\n")
+
+
+def _validate_config(cfg: EditDatasetConfig) -> None:
+    if isinstance(cfg.operation, MergeConfig):
+        if not cfg.new_repo_id:
+            raise ValueError("--new_repo_id is required for merge operation (the merged dataset identifier)")
+    else:
+        if not cfg.repo_id:
+            raise ValueError(
+                f"--repo_id is required for {cfg.operation.type} operation (the input dataset identifier)"
+            )
+
+
+@parser.wrap()
+def edit_dataset(cfg: EditDatasetConfig) -> None:
+    _validate_config(cfg)
+    operation_type = cfg.operation.type
+
+    if operation_type == "delete_episodes":
+        handle_delete_episodes(cfg)
+    elif operation_type == "split":
+        handle_split(cfg)
+    elif operation_type == "merge":
+        handle_merge(cfg)
+    elif operation_type == "remove_feature":
+        handle_remove_feature(cfg)
+    elif operation_type == "modify_tasks":
+        handle_modify_tasks(cfg)
+    elif operation_type == "convert_image_to_video":
+        handle_convert_image_to_video(cfg)
+    elif operation_type == "info":
+        handle_info(cfg)
+    else:
+        available = ", ".join(OperationConfig.get_known_choices())
+        raise ValueError(f"Unknown operation: {operation_type}\nAvailable operations: {available}")
+
+
+def main() -> None:
+    init_logging()
+    edit_dataset()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_eval.py b/lerobot/src/lerobot/scripts/lerobot_eval.py
new file mode 100644
index 0000000000000000000000000000000000000000..6d814f498577f7dcd5ed7d425cb5f84045f06908
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_eval.py
@@ -0,0 +1,814 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Evaluate a policy on an environment by running rollouts and computing metrics.
+
+Usage examples:
+
+You want to evaluate a model from the hub (eg: https://huggingface.co/lerobot/diffusion_pusht)
+for 10 episodes.
+
+```
+lerobot-eval \
+    --policy.path=lerobot/diffusion_pusht \
+    --env.type=pusht \
+    --eval.batch_size=10 \
+    --eval.n_episodes=10 \
+    --policy.use_amp=false \
+    --policy.device=cuda
+```
+
+OR, you want to evaluate a model checkpoint from the LeRobot training script for 10 episodes.
+```
+lerobot-eval \
+    --policy.path=outputs/train/diffusion_pusht/checkpoints/005000/pretrained_model \
+    --env.type=pusht \
+    --eval.batch_size=10 \
+    --eval.n_episodes=10 \
+    --policy.use_amp=false \
+    --policy.device=cuda
+```
+
+Note that in both examples, the repo/folder should contain at least `config.json` and `model.safetensors` files.
+
+You can learn about the CLI options for this script in the `EvalPipelineConfig` in lerobot/configs/eval.py
+"""
+
+import concurrent.futures as cf
+import json
+import logging
+import threading
+import time
+from collections import defaultdict
+from collections.abc import Callable
+from contextlib import nullcontext
+from copy import deepcopy
+from dataclasses import asdict
+from functools import partial
+from pathlib import Path
+from pprint import pformat
+from typing import Any, TypedDict
+
+import einops
+import gymnasium as gym
+import numpy as np
+import torch
+from termcolor import colored
+from torch import Tensor, nn
+from tqdm import trange
+
+from lerobot.configs import parser
+from lerobot.configs.eval import EvalPipelineConfig
+from lerobot.envs.factory import make_env, make_env_pre_post_processors
+from lerobot.envs.utils import (
+    add_envs_task,
+    check_env_attributes_and_types,
+    close_envs,
+    preprocess_observation,
+)
+from lerobot.policies.factory import make_policy, make_pre_post_processors
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.processor import PolicyProcessorPipeline
+from lerobot.types import PolicyAction
+from lerobot.utils.constants import ACTION, DONE, OBS_STR, REWARD
+from lerobot.utils.device_utils import get_safe_torch_device
+from lerobot.utils.import_utils import register_third_party_plugins
+from lerobot.utils.io_utils import write_video
+from lerobot.utils.random_utils import set_seed
+from lerobot.utils.utils import (
+    init_logging,
+    inside_slurm,
+)
+
+
+def rollout(
+    env: gym.vector.VectorEnv,
+    policy: PreTrainedPolicy,
+    env_preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    env_postprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    postprocessor: PolicyProcessorPipeline[PolicyAction, PolicyAction],
+    seeds: list[int] | None = None,
+    return_observations: bool = False,
+    render_callback: Callable[[gym.vector.VectorEnv], None] | None = None,
+) -> dict:
+    """Run a batched policy rollout once through a batch of environments.
+
+    Note that all environments in the batch are run until the last environment is done. This means some
+    data will probably need to be discarded (for environments that aren't the first one to be done).
+
+    The return dictionary contains:
+        (optional) "observation": A dictionary of (batch, sequence + 1, *) tensors mapped to observation
+            keys. NOTE that this has an extra sequence element relative to the other keys in the
+            dictionary. This is because an extra observation is included for after the environment is
+            terminated or truncated.
+        "action": A (batch, sequence, action_dim) tensor of actions applied based on the observations (not
+            including the last observations).
+        "reward": A (batch, sequence) tensor of rewards received for applying the actions.
+        "success": A (batch, sequence) tensor of success conditions (the only time this can be True is upon
+            environment termination/truncation).
+        "done": A (batch, sequence) tensor of **cumulative** done conditions. For any given batch element,
+            the first True is followed by True's all the way till the end. This can be used for masking
+            extraneous elements from the sequences above.
+
+    Args:
+        env: The batch of environments.
+        policy: The policy. Must be a PyTorch nn module.
+        seeds: The environments are seeded once at the start of the rollout. If provided, this argument
+            specifies the seeds for each of the environments.
+        return_observations: Whether to include all observations in the returned rollout data. Observations
+            are returned optionally because they typically take more memory to cache. Defaults to False.
+        render_callback: Optional rendering callback to be used after the environments are reset, and after
+            every step.
+    Returns:
+        The dictionary described above.
+    """
+    assert isinstance(policy, nn.Module), "Policy must be a PyTorch nn module."
+
+    # Reset the policy and environments.
+    policy.reset()
+    observation, info = env.reset(seed=seeds)
+    if render_callback is not None:
+        render_callback(env)
+
+    all_observations = []
+    all_actions = []
+    all_rewards = []
+    all_successes = []
+    all_dones = []
+
+    step = 0
+    # Keep track of which environments are done.
+    done = np.array([False] * env.num_envs)
+    max_steps = env.call("_max_episode_steps")[0]
+    progbar = trange(
+        max_steps,
+        desc=f"Running rollout with at most {max_steps} steps",
+        disable=inside_slurm(),  # we dont want progress bar when we use slurm, since it clutters the logs
+        leave=False,
+    )
+    check_env_attributes_and_types(env)
+    while not np.all(done) and step < max_steps:
+        # Numpy array to tensor and changing dictionary keys to LeRobot policy format.
+        observation = preprocess_observation(observation)
+        if return_observations:
+            all_observations.append(deepcopy(observation))
+
+        # Infer "task" from attributes of environments.
+        # TODO: works with SyncVectorEnv but not AsyncVectorEnv
+        observation = add_envs_task(env, observation)
+
+        # Apply environment-specific preprocessing (e.g., LiberoProcessorStep for LIBERO)
+        observation = env_preprocessor(observation)
+
+        observation = preprocessor(observation)
+        with torch.inference_mode():
+            action = policy.select_action(observation)
+        action = postprocessor(action)
+
+        action_transition = {ACTION: action}
+        action_transition = env_postprocessor(action_transition)
+        action = action_transition[ACTION]
+
+        # Convert to CPU / numpy.
+        action_numpy: np.ndarray = action.to("cpu").numpy()
+        assert action_numpy.ndim == 2, "Action dimensions should be (batch, action_dim)"
+
+        # Apply the next action.
+        observation, reward, terminated, truncated, info = env.step(action_numpy)
+        if render_callback is not None:
+            render_callback(env)
+
+        # VectorEnv stores is_success in `info["final_info"][env_index]["is_success"]`. "final_info" isn't
+        # available if none of the envs finished.
+        if "final_info" in info:
+            final_info = info["final_info"]
+            if not isinstance(final_info, dict):
+                raise RuntimeError(
+                    "Unsupported `final_info` format: expected dict (Gymnasium >= 1.0). "
+                    "You're likely using an older version of gymnasium (< 1.0). Please upgrade."
+                )
+            successes = final_info["is_success"].tolist()
+        else:
+            successes = [False] * env.num_envs
+
+        # Keep track of which environments are done so far.
+        # Mark the episode as done if we reach the maximum step limit.
+        # This ensures that the rollout always terminates cleanly at `max_steps`,
+        # and allows logging/saving (e.g., videos) to be triggered consistently.
+        done = terminated | truncated | done
+        if step + 1 == max_steps:
+            done = np.ones_like(done, dtype=bool)
+
+        all_actions.append(torch.from_numpy(action_numpy))
+        all_rewards.append(torch.from_numpy(reward))
+        all_dones.append(torch.from_numpy(done))
+        all_successes.append(torch.tensor(successes))
+
+        step += 1
+        running_success_rate = (
+            einops.reduce(torch.stack(all_successes, dim=1), "b n -> b", "any").numpy().mean()
+        )
+        progbar.set_postfix({"running_success_rate": f"{running_success_rate.item() * 100:.1f}%"})
+        progbar.update()
+
+    # Track the final observation.
+    if return_observations:
+        observation = preprocess_observation(observation)
+        all_observations.append(deepcopy(observation))
+
+    # Stack the sequence along the first dimension so that we have (batch, sequence, *) tensors.
+    ret = {
+        ACTION: torch.stack(all_actions, dim=1),
+        "reward": torch.stack(all_rewards, dim=1),
+        "success": torch.stack(all_successes, dim=1),
+        "done": torch.stack(all_dones, dim=1),
+    }
+    if return_observations:
+        stacked_observations = {}
+        for key in all_observations[0]:
+            stacked_observations[key] = torch.stack([obs[key] for obs in all_observations], dim=1)
+        ret[OBS_STR] = stacked_observations
+
+    if hasattr(policy, "use_original_modules"):
+        policy.use_original_modules()
+
+    return ret
+
+
+def eval_policy(
+    env: gym.vector.VectorEnv,
+    policy: PreTrainedPolicy,
+    env_preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    env_postprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    postprocessor: PolicyProcessorPipeline[PolicyAction, PolicyAction],
+    n_episodes: int,
+    max_episodes_rendered: int = 0,
+    videos_dir: Path | None = None,
+    return_episode_data: bool = False,
+    start_seed: int | None = None,
+) -> dict:
+    """
+    Args:
+        env: The batch of environments.
+        policy: The policy.
+        n_episodes: The number of episodes to evaluate.
+        max_episodes_rendered: Maximum number of episodes to render into videos.
+        videos_dir: Where to save rendered videos.
+        return_episode_data: Whether to return episode data for online training. Incorporates the data into
+            the "episodes" key of the returned dictionary.
+        start_seed: The first seed to use for the first individual rollout. For all subsequent rollouts the
+            seed is incremented by 1. If not provided, the environments are not manually seeded.
+    Returns:
+        Dictionary with metrics and data regarding the rollouts.
+    """
+    if max_episodes_rendered > 0 and not videos_dir:
+        raise ValueError("If max_episodes_rendered > 0, videos_dir must be provided.")
+
+    if not isinstance(policy, PreTrainedPolicy):
+        exc = ValueError(
+            f"Policy of type 'PreTrainedPolicy' is expected, but type '{type(policy)}' was provided."
+        )
+        try:
+            from peft import PeftModel
+
+            if not isinstance(policy, PeftModel):
+                raise exc
+        except ImportError:
+            raise exc from None
+
+    start = time.time()
+    policy.eval()
+
+    # Determine how many batched rollouts we need to get n_episodes. Note that if n_episodes is not evenly
+    # divisible by env.num_envs we end up discarding some data in the last batch.
+    n_batches = n_episodes // env.num_envs + int((n_episodes % env.num_envs) != 0)
+
+    # Keep track of some metrics.
+    sum_rewards = []
+    max_rewards = []
+    all_successes = []
+    all_seeds = []
+    threads = []  # for video saving threads
+    n_episodes_rendered = 0  # for saving the correct number of videos
+
+    # Callback for visualization.
+    def render_frame(env: gym.vector.VectorEnv):
+        # noqa: B023
+        if n_episodes_rendered >= max_episodes_rendered:
+            return
+        n_to_render_now = min(max_episodes_rendered - n_episodes_rendered, env.num_envs)
+        if isinstance(env, gym.vector.SyncVectorEnv):
+            ep_frames.append(np.stack([env.envs[i].render() for i in range(n_to_render_now)]))  # noqa: B023
+        elif isinstance(env, gym.vector.AsyncVectorEnv):
+            # Here we must render all frames and discard any we don't need.
+            ep_frames.append(np.stack(env.call("render")[:n_to_render_now]))
+
+    if max_episodes_rendered > 0:
+        video_paths: list[str] = []
+
+    if return_episode_data:
+        episode_data: dict | None = None
+
+    # we dont want progress bar when we use slurm, since it clutters the logs
+    progbar = trange(n_batches, desc="Stepping through eval batches", disable=inside_slurm())
+    for batch_ix in progbar:
+        # Cache frames for rendering videos. Each item will be (b, h, w, c), and the list indexes the rollout
+        # step.
+        if max_episodes_rendered > 0:
+            ep_frames: list[np.ndarray] = []
+
+        if start_seed is None:
+            seeds = None
+        else:
+            seeds = range(
+                start_seed + (batch_ix * env.num_envs), start_seed + ((batch_ix + 1) * env.num_envs)
+            )
+        rollout_data = rollout(
+            env=env,
+            policy=policy,
+            env_preprocessor=env_preprocessor,
+            env_postprocessor=env_postprocessor,
+            preprocessor=preprocessor,
+            postprocessor=postprocessor,
+            seeds=list(seeds) if seeds else None,
+            return_observations=return_episode_data,
+            render_callback=render_frame if max_episodes_rendered > 0 else None,
+        )
+
+        # Figure out where in each rollout sequence the first done condition was encountered (results after
+        # this won't be included).
+        n_steps = rollout_data["done"].shape[1]
+        # Note: this relies on a property of argmax: that it returns the first occurrence as a tiebreaker.
+        done_indices = torch.argmax(rollout_data["done"].to(int), dim=1)
+
+        # Make a mask with shape (batch, n_steps) to mask out rollout data after the first done
+        # (batch-element-wise). Note the `done_indices + 1` to make sure to keep the data from the done step.
+        mask = (torch.arange(n_steps) <= einops.repeat(done_indices + 1, "b -> b s", s=n_steps)).int()
+        # Extend metrics.
+        batch_sum_rewards = einops.reduce((rollout_data["reward"] * mask), "b n -> b", "sum")
+        sum_rewards.extend(batch_sum_rewards.tolist())
+        batch_max_rewards = einops.reduce((rollout_data["reward"] * mask), "b n -> b", "max")
+        max_rewards.extend(batch_max_rewards.tolist())
+        batch_successes = einops.reduce((rollout_data["success"] * mask), "b n -> b", "any")
+        all_successes.extend(batch_successes.tolist())
+        if seeds:
+            all_seeds.extend(seeds)
+        else:
+            all_seeds.append(None)
+
+        # FIXME: episode_data is either None or it doesn't exist
+        if return_episode_data:
+            this_episode_data = _compile_episode_data(
+                rollout_data,
+                done_indices,
+                start_episode_index=batch_ix * env.num_envs,
+                start_data_index=(0 if episode_data is None else (episode_data["index"][-1].item() + 1)),
+                fps=env.unwrapped.metadata["render_fps"],
+            )
+            if episode_data is None:
+                episode_data = this_episode_data
+            else:
+                # Some sanity checks to make sure we are correctly compiling the data.
+                assert episode_data["episode_index"][-1] + 1 == this_episode_data["episode_index"][0]
+                assert episode_data["index"][-1] + 1 == this_episode_data["index"][0]
+                # Concatenate the episode data.
+                episode_data = {k: torch.cat([episode_data[k], this_episode_data[k]]) for k in episode_data}
+
+        # Maybe render video for visualization.
+        if max_episodes_rendered > 0 and len(ep_frames) > 0:
+            batch_stacked_frames = np.stack(ep_frames, axis=1)  # (b, t, *)
+            for stacked_frames, done_index in zip(
+                batch_stacked_frames, done_indices.flatten().tolist(), strict=False
+            ):
+                if n_episodes_rendered >= max_episodes_rendered:
+                    break
+
+                videos_dir.mkdir(parents=True, exist_ok=True)
+                video_path = videos_dir / f"eval_episode_{n_episodes_rendered}.mp4"
+                video_paths.append(str(video_path))
+                thread = threading.Thread(
+                    target=write_video,
+                    args=(
+                        str(video_path),
+                        stacked_frames[: done_index + 1],  # + 1 to capture the last observation
+                        env.unwrapped.metadata["render_fps"],
+                    ),
+                )
+                thread.start()
+                threads.append(thread)
+                n_episodes_rendered += 1
+
+        progbar.set_postfix(
+            {"running_success_rate": f"{np.mean(all_successes[:n_episodes]).item() * 100:.1f}%"}
+        )
+
+    # Wait till all video rendering threads are done.
+    for thread in threads:
+        thread.join()
+
+    # Compile eval info.
+    info = {
+        "per_episode": [
+            {
+                "episode_ix": i,
+                "sum_reward": sum_reward,
+                "max_reward": max_reward,
+                "success": success,
+                "seed": seed,
+            }
+            for i, (sum_reward, max_reward, success, seed) in enumerate(
+                zip(
+                    sum_rewards[:n_episodes],
+                    max_rewards[:n_episodes],
+                    all_successes[:n_episodes],
+                    all_seeds[:n_episodes],
+                    strict=True,
+                )
+            )
+        ],
+        "aggregated": {
+            "avg_sum_reward": float(np.nanmean(sum_rewards[:n_episodes])),
+            "avg_max_reward": float(np.nanmean(max_rewards[:n_episodes])),
+            "pc_success": float(np.nanmean(all_successes[:n_episodes]) * 100),
+            "eval_s": time.time() - start,
+            "eval_ep_s": (time.time() - start) / n_episodes,
+        },
+    }
+
+    if return_episode_data:
+        info["episodes"] = episode_data
+
+    if max_episodes_rendered > 0:
+        info["video_paths"] = video_paths
+
+    return info
+
+
+def _compile_episode_data(
+    rollout_data: dict, done_indices: Tensor, start_episode_index: int, start_data_index: int, fps: float
+) -> dict:
+    """Convenience function for `eval_policy(return_episode_data=True)`
+
+    Compiles all the rollout data into a Hugging Face dataset.
+
+    Similar logic is implemented when datasets are pushed to hub (see: `push_to_hub`).
+    """
+    ep_dicts = []
+    total_frames = 0
+    for ep_ix in range(rollout_data[ACTION].shape[0]):
+        # + 2 to include the first done frame and the last observation frame.
+        num_frames = done_indices[ep_ix].item() + 2
+        total_frames += num_frames
+
+        # Here we do `num_frames - 1` as we don't want to include the last observation frame just yet.
+        ep_dict = {
+            ACTION: rollout_data[ACTION][ep_ix, : num_frames - 1],
+            "episode_index": torch.tensor([start_episode_index + ep_ix] * (num_frames - 1)),
+            "frame_index": torch.arange(0, num_frames - 1, 1),
+            "timestamp": torch.arange(0, num_frames - 1, 1) / fps,
+            DONE: rollout_data["done"][ep_ix, : num_frames - 1],
+            "next.success": rollout_data["success"][ep_ix, : num_frames - 1],
+            REWARD: rollout_data["reward"][ep_ix, : num_frames - 1].type(torch.float32),
+        }
+
+        # For the last observation frame, all other keys will just be copy padded.
+        for k in ep_dict:
+            ep_dict[k] = torch.cat([ep_dict[k], ep_dict[k][-1:]])
+
+        for key in rollout_data[OBS_STR]:
+            ep_dict[key] = rollout_data[OBS_STR][key][ep_ix, :num_frames]
+
+        ep_dicts.append(ep_dict)
+
+    data_dict = {}
+    for key in ep_dicts[0]:
+        data_dict[key] = torch.cat([x[key] for x in ep_dicts])
+
+    data_dict["index"] = torch.arange(start_data_index, start_data_index + total_frames, 1)
+
+    return data_dict
+
+
+@parser.wrap()
+def eval_main(cfg: EvalPipelineConfig):
+    logging.info(pformat(asdict(cfg)))
+
+    # Check device is available
+    device = get_safe_torch_device(cfg.policy.device, log=True)
+
+    torch.backends.cudnn.benchmark = True
+    torch.backends.cuda.matmul.allow_tf32 = True
+    set_seed(cfg.seed)
+
+    logging.info(colored("Output dir:", "yellow", attrs=["bold"]) + f" {cfg.output_dir}")
+
+    logging.info("Making environment.")
+    envs = make_env(
+        cfg.env,
+        n_envs=cfg.eval.batch_size,
+        use_async_envs=cfg.eval.use_async_envs,
+        trust_remote_code=cfg.trust_remote_code,
+    )
+
+    logging.info("Making policy.")
+
+    policy = make_policy(
+        cfg=cfg.policy,
+        env_cfg=cfg.env,
+        rename_map=cfg.rename_map,
+    )
+
+    policy.eval()
+
+    # The inference device is automatically set to match the detected hardware, overriding any previous device settings from training to ensure compatibility.
+    preprocessor_overrides = {
+        "device_processor": {"device": str(policy.config.device)},
+        "rename_observations_processor": {"rename_map": cfg.rename_map},
+    }
+
+    preprocessor, postprocessor = make_pre_post_processors(
+        policy_cfg=cfg.policy,
+        pretrained_path=cfg.policy.pretrained_path,
+        preprocessor_overrides=preprocessor_overrides,
+    )
+
+    # Create environment-specific preprocessor and postprocessor (e.g., for LIBERO environments)
+    env_preprocessor, env_postprocessor = make_env_pre_post_processors(env_cfg=cfg.env, policy_cfg=cfg.policy)
+
+    with torch.no_grad(), torch.autocast(device_type=device.type) if cfg.policy.use_amp else nullcontext():
+        info = eval_policy_all(
+            envs=envs,
+            policy=policy,
+            env_preprocessor=env_preprocessor,
+            env_postprocessor=env_postprocessor,
+            preprocessor=preprocessor,
+            postprocessor=postprocessor,
+            n_episodes=cfg.eval.n_episodes,
+            max_episodes_rendered=10,
+            videos_dir=Path(cfg.output_dir) / "videos",
+            start_seed=cfg.seed,
+            max_parallel_tasks=cfg.env.max_parallel_tasks,
+        )
+        print("Overall Aggregated Metrics:")
+        print(info["overall"])
+
+        # Print per-suite stats
+        for task_group, task_group_info in info.items():
+            print(f"\nAggregated Metrics for {task_group}:")
+            print(task_group_info)
+    # Close all vec envs
+    close_envs(envs)
+
+    # Save info
+    with open(Path(cfg.output_dir) / "eval_info.json", "w") as f:
+        json.dump(info, f, indent=2)
+
+    logging.info("End of eval")
+
+
+# ---- typed payload returned by one task eval ----
+class TaskMetrics(TypedDict):
+    sum_rewards: list[float]
+    max_rewards: list[float]
+    successes: list[bool]
+    video_paths: list[str]
+
+
+ACC_KEYS = ("sum_rewards", "max_rewards", "successes", "video_paths")
+
+
+def eval_one(
+    env: gym.vector.VectorEnv,
+    *,
+    policy: PreTrainedPolicy,
+    env_preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    env_postprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    postprocessor: PolicyProcessorPipeline[PolicyAction, PolicyAction],
+    n_episodes: int,
+    max_episodes_rendered: int,
+    videos_dir: Path | None,
+    return_episode_data: bool,
+    start_seed: int | None,
+) -> TaskMetrics:
+    """Evaluates one task_id of one suite using the provided vec env."""
+
+    task_videos_dir = videos_dir
+
+    task_result = eval_policy(
+        env=env,
+        policy=policy,
+        env_preprocessor=env_preprocessor,
+        env_postprocessor=env_postprocessor,
+        preprocessor=preprocessor,
+        postprocessor=postprocessor,
+        n_episodes=n_episodes,
+        max_episodes_rendered=max_episodes_rendered,
+        videos_dir=task_videos_dir,
+        return_episode_data=return_episode_data,
+        start_seed=start_seed,
+    )
+
+    per_episode = task_result["per_episode"]
+    return TaskMetrics(
+        sum_rewards=[ep["sum_reward"] for ep in per_episode],
+        max_rewards=[ep["max_reward"] for ep in per_episode],
+        successes=[ep["success"] for ep in per_episode],
+        video_paths=task_result.get("video_paths", []),
+    )
+
+
+def run_one(
+    task_group: str,
+    task_id: int,
+    env,
+    *,
+    policy,
+    env_preprocessor,
+    env_postprocessor,
+    preprocessor,
+    postprocessor,
+    n_episodes: int,
+    max_episodes_rendered: int,
+    videos_dir: Path | None,
+    return_episode_data: bool,
+    start_seed: int | None,
+):
+    """
+    Run eval_one for a single (task_group, task_id, env).
+    Returns (task_group, task_id, task_metrics_dict).
+    This function is intentionally module-level to make it easy to test.
+    """
+    task_videos_dir = None
+    if videos_dir is not None:
+        task_videos_dir = videos_dir / f"{task_group}_{task_id}"
+        task_videos_dir.mkdir(parents=True, exist_ok=True)
+
+    # Call the existing eval_one (assumed to return TaskMetrics-like dict)
+    metrics = eval_one(
+        env,
+        policy=policy,
+        env_preprocessor=env_preprocessor,
+        env_postprocessor=env_postprocessor,
+        preprocessor=preprocessor,
+        postprocessor=postprocessor,
+        n_episodes=n_episodes,
+        max_episodes_rendered=max_episodes_rendered,
+        videos_dir=task_videos_dir,
+        return_episode_data=return_episode_data,
+        start_seed=start_seed,
+    )
+    # ensure we always provide video_paths key to simplify accumulation
+    if max_episodes_rendered > 0:
+        metrics.setdefault("video_paths", [])
+    return task_group, task_id, metrics
+
+
+def eval_policy_all(
+    envs: dict[str, dict[int, gym.vector.VectorEnv]],
+    policy,
+    env_preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    env_postprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    postprocessor: PolicyProcessorPipeline[PolicyAction, PolicyAction],
+    n_episodes: int,
+    *,
+    max_episodes_rendered: int = 0,
+    videos_dir: Path | None = None,
+    return_episode_data: bool = False,
+    start_seed: int | None = None,
+    max_parallel_tasks: int = 1,
+) -> dict:
+    """
+    Evaluate a nested `envs` dict: {task_group: {task_id: vec_env}}.
+    This implementation flattens tasks, runs them sequentially or via ThreadPoolExecutor,
+    accumulates per-group and overall statistics, and returns the same aggregate metrics
+    schema as the single-env evaluator (avg_sum_reward / avg_max_reward / pc_success / timings)
+    plus per-task infos.
+    """
+    start_t = time.time()
+
+    # Flatten envs into list of (task_group, task_id, env)
+    tasks = [(tg, tid, vec) for tg, group in envs.items() for tid, vec in group.items()]
+
+    # accumulators: track metrics at both per-group level and across all groups
+    group_acc: dict[str, dict[str, list]] = defaultdict(lambda: {k: [] for k in ACC_KEYS})
+    overall: dict[str, list] = {k: [] for k in ACC_KEYS}
+    per_task_infos: list[dict] = []
+
+    # small inline helper to accumulate one task's metrics into accumulators
+    def _accumulate_to(group: str, metrics: dict):
+        # metrics expected to contain 'sum_rewards', 'max_rewards', 'successes', optionally 'video_paths'
+        # but eval_one may store per-episode lists; we assume metrics uses scalars averaged per task as before.
+        # To be robust, accept scalars or lists.
+        def _append(key, value):
+            if value is None:
+                return
+            if isinstance(value, list):
+                group_acc[group][key].extend(value)
+                overall[key].extend(value)
+            else:
+                group_acc[group][key].append(value)
+                overall[key].append(value)
+
+        _append("sum_rewards", metrics.get("sum_rewards"))
+        _append("max_rewards", metrics.get("max_rewards"))
+        _append("successes", metrics.get("successes"))
+        # video_paths is list-like
+        paths = metrics.get("video_paths", [])
+        if paths:
+            group_acc[group]["video_paths"].extend(paths)
+            overall["video_paths"].extend(paths)
+
+    # Choose runner (sequential vs threaded)
+    task_runner = partial(
+        run_one,
+        policy=policy,
+        env_preprocessor=env_preprocessor,
+        env_postprocessor=env_postprocessor,
+        preprocessor=preprocessor,
+        postprocessor=postprocessor,
+        n_episodes=n_episodes,
+        max_episodes_rendered=max_episodes_rendered,
+        videos_dir=videos_dir,
+        return_episode_data=return_episode_data,
+        start_seed=start_seed,
+    )
+
+    if max_parallel_tasks <= 1:
+        # sequential path (single accumulator path on the main thread)
+        # NOTE: keeping a single-threaded accumulator avoids concurrent list appends or locks
+        for task_group, task_id, env in tasks:
+            tg, tid, metrics = task_runner(task_group, task_id, env)
+            _accumulate_to(tg, metrics)
+            per_task_infos.append({"task_group": tg, "task_id": tid, "metrics": metrics})
+    else:
+        # threaded path: submit all tasks, consume completions on main thread and accumulate there
+        with cf.ThreadPoolExecutor(max_workers=max_parallel_tasks) as executor:
+            fut2meta = {}
+            for task_group, task_id, env in tasks:
+                fut = executor.submit(task_runner, task_group, task_id, env)
+                fut2meta[fut] = (task_group, task_id)
+            for fut in cf.as_completed(fut2meta):
+                tg, tid, metrics = fut.result()
+                _accumulate_to(tg, metrics)
+                per_task_infos.append({"task_group": tg, "task_id": tid, "metrics": metrics})
+
+    # compute aggregated metrics helper (robust to lists/scalars)
+    def _agg_from_list(xs):
+        if not xs:
+            return float("nan")
+        arr = np.array(xs, dtype=float)
+        return float(np.nanmean(arr))
+
+    # compute per-group aggregates
+    groups_aggregated = {}
+    for group, acc in group_acc.items():
+        groups_aggregated[group] = {
+            "avg_sum_reward": _agg_from_list(acc["sum_rewards"]),
+            "avg_max_reward": _agg_from_list(acc["max_rewards"]),
+            "pc_success": _agg_from_list(acc["successes"]) * 100 if acc["successes"] else float("nan"),
+            "n_episodes": len(acc["sum_rewards"]),
+            "video_paths": list(acc["video_paths"]),
+        }
+
+    # overall aggregates
+    overall_agg = {
+        "avg_sum_reward": _agg_from_list(overall["sum_rewards"]),
+        "avg_max_reward": _agg_from_list(overall["max_rewards"]),
+        "pc_success": _agg_from_list(overall["successes"]) * 100 if overall["successes"] else float("nan"),
+        "n_episodes": len(overall["sum_rewards"]),
+        "eval_s": time.time() - start_t,
+        "eval_ep_s": (time.time() - start_t) / max(1, len(overall["sum_rewards"])),
+        "video_paths": list(overall["video_paths"]),
+    }
+
+    return {
+        "per_task": per_task_infos,
+        "per_group": groups_aggregated,
+        "overall": overall_agg,
+    }
+
+
+def main():
+    init_logging()
+    register_third_party_plugins()
+    eval_main()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_find_cameras.py b/lerobot/src/lerobot/scripts/lerobot_find_cameras.py
new file mode 100644
index 0000000000000000000000000000000000000000..0248a276854ee16e8ddd5f7b4e78bec58d7d95e5
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_find_cameras.py
@@ -0,0 +1,319 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Helper to find the camera devices available in your system.
+
+Example:
+
+```shell
+lerobot-find-cameras
+```
+"""
+
+# NOTE(Steven): RealSense can also be identified/opened as OpenCV cameras. If you know the camera is a RealSense, use the `lerobot-find-cameras realsense` flag to avoid confusion.
+# NOTE(Steven): macOS cameras sometimes report different FPS at init time, not an issue here as we don't specify FPS when opening the cameras, but the information displayed might not be truthful.
+
+import argparse
+import concurrent.futures
+import logging
+import time
+from pathlib import Path
+from typing import Any
+
+import numpy as np
+from PIL import Image
+
+from lerobot.cameras.configs import ColorMode
+from lerobot.cameras.opencv.camera_opencv import OpenCVCamera
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig
+from lerobot.cameras.realsense.camera_realsense import RealSenseCamera
+from lerobot.cameras.realsense.configuration_realsense import RealSenseCameraConfig
+
+logger = logging.getLogger(__name__)
+
+
+def find_all_opencv_cameras() -> list[dict[str, Any]]:
+    """
+    Finds all available OpenCV cameras plugged into the system.
+
+    Returns:
+        A list of all available OpenCV cameras with their metadata.
+    """
+    all_opencv_cameras_info: list[dict[str, Any]] = []
+    logger.info("Searching for OpenCV cameras...")
+    try:
+        opencv_cameras = OpenCVCamera.find_cameras()
+        for cam_info in opencv_cameras:
+            all_opencv_cameras_info.append(cam_info)
+        logger.info(f"Found {len(opencv_cameras)} OpenCV cameras.")
+    except Exception as e:
+        logger.error(f"Error finding OpenCV cameras: {e}")
+
+    return all_opencv_cameras_info
+
+
+def find_all_realsense_cameras() -> list[dict[str, Any]]:
+    """
+    Finds all available RealSense cameras plugged into the system.
+
+    Returns:
+        A list of all available RealSense cameras with their metadata.
+    """
+    all_realsense_cameras_info: list[dict[str, Any]] = []
+    logger.info("Searching for RealSense cameras...")
+    try:
+        realsense_cameras = RealSenseCamera.find_cameras()
+        for cam_info in realsense_cameras:
+            all_realsense_cameras_info.append(cam_info)
+        logger.info(f"Found {len(realsense_cameras)} RealSense cameras.")
+    except ImportError:
+        logger.warning("Skipping RealSense camera search: pyrealsense2 library not found or not importable.")
+    except Exception as e:
+        logger.error(f"Error finding RealSense cameras: {e}")
+
+    return all_realsense_cameras_info
+
+
+def find_and_print_cameras(camera_type_filter: str | None = None) -> list[dict[str, Any]]:
+    """
+    Finds available cameras based on an optional filter and prints their information.
+
+    Args:
+        camera_type_filter: Optional string to filter cameras ("realsense" or "opencv").
+                            If None, lists all cameras.
+
+    Returns:
+        A list of all available cameras matching the filter, with their metadata.
+    """
+    all_cameras_info: list[dict[str, Any]] = []
+
+    if camera_type_filter:
+        camera_type_filter = camera_type_filter.lower()
+
+    if camera_type_filter is None or camera_type_filter == "opencv":
+        all_cameras_info.extend(find_all_opencv_cameras())
+    if camera_type_filter is None or camera_type_filter == "realsense":
+        all_cameras_info.extend(find_all_realsense_cameras())
+
+    if not all_cameras_info:
+        if camera_type_filter:
+            logger.warning(f"No {camera_type_filter} cameras were detected.")
+        else:
+            logger.warning("No cameras (OpenCV or RealSense) were detected.")
+    else:
+        print("\n--- Detected Cameras ---")
+        for i, cam_info in enumerate(all_cameras_info):
+            print(f"Camera #{i}:")
+            for key, value in cam_info.items():
+                if key == "default_stream_profile" and isinstance(value, dict):
+                    print(f"  {key.replace('_', ' ').capitalize()}:")
+                    for sub_key, sub_value in value.items():
+                        print(f"    {sub_key.capitalize()}: {sub_value}")
+                else:
+                    print(f"  {key.replace('_', ' ').capitalize()}: {value}")
+            print("-" * 20)
+    return all_cameras_info
+
+
+def save_image(
+    img_array: np.ndarray,
+    camera_identifier: str | int,
+    images_dir: Path,
+    camera_type: str,
+):
+    """
+    Saves a single image to disk using Pillow. Handles color conversion if necessary.
+    """
+    try:
+        img = Image.fromarray(img_array, mode="RGB")
+
+        safe_identifier = str(camera_identifier).replace("/", "_").replace("\\", "_")
+        filename_prefix = f"{camera_type.lower()}_{safe_identifier}"
+        filename = f"{filename_prefix}.png"
+
+        path = images_dir / filename
+        path.parent.mkdir(parents=True, exist_ok=True)
+        img.save(str(path))
+        logger.info(f"Saved image: {path}")
+    except Exception as e:
+        logger.error(f"Failed to save image for camera {camera_identifier} (type {camera_type}): {e}")
+
+
+def create_camera_instance(cam_meta: dict[str, Any]) -> dict[str, Any] | None:
+    """Create and connect to a camera instance based on metadata."""
+    cam_type = cam_meta.get("type")
+    cam_id = cam_meta.get("id")
+    instance = None
+
+    logger.info(f"Preparing {cam_type} ID {cam_id} with default profile")
+
+    try:
+        if cam_type == "OpenCV":
+            cv_config = OpenCVCameraConfig(
+                index_or_path=cam_id,
+                color_mode=ColorMode.RGB,
+            )
+            instance = OpenCVCamera(cv_config)
+        elif cam_type == "RealSense":
+            rs_config = RealSenseCameraConfig(
+                serial_number_or_name=cam_id,
+                color_mode=ColorMode.RGB,
+            )
+            instance = RealSenseCamera(rs_config)
+        else:
+            logger.warning(f"Unknown camera type: {cam_type} for ID {cam_id}. Skipping.")
+            return None
+
+        if instance:
+            logger.info(f"Connecting to {cam_type} camera: {cam_id}...")
+            instance.connect(warmup=True)
+            return {"instance": instance, "meta": cam_meta}
+    except Exception as e:
+        logger.error(f"Failed to connect or configure {cam_type} camera {cam_id}: {e}")
+        if instance and instance.is_connected:
+            instance.disconnect()
+        return None
+
+
+def process_camera_image(
+    cam_dict: dict[str, Any], output_dir: Path, current_time: float
+) -> concurrent.futures.Future | None:
+    """Capture and process an image from a single camera."""
+    cam = cam_dict["instance"]
+    meta = cam_dict["meta"]
+    cam_type_str = str(meta.get("type", "unknown"))
+    cam_id_str = str(meta.get("id", "unknown"))
+
+    try:
+        image_data = cam.read()
+
+        return save_image(
+            image_data,
+            cam_id_str,
+            output_dir,
+            cam_type_str,
+        )
+    except TimeoutError:
+        logger.warning(
+            f"Timeout reading from {cam_type_str} camera {cam_id_str} at time {current_time:.2f}s."
+        )
+    except Exception as e:
+        logger.error(f"Error reading from {cam_type_str} camera {cam_id_str}: {e}")
+    return None
+
+
+def cleanup_cameras(cameras_to_use: list[dict[str, Any]]):
+    """Disconnect all cameras."""
+    logger.info(f"Disconnecting {len(cameras_to_use)} cameras...")
+    for cam_dict in cameras_to_use:
+        try:
+            if cam_dict["instance"] and cam_dict["instance"].is_connected:
+                cam_dict["instance"].disconnect()
+        except Exception as e:
+            logger.error(f"Error disconnecting camera {cam_dict['meta'].get('id')}: {e}")
+
+
+def save_images_from_all_cameras(
+    output_dir: Path,
+    record_time_s: float = 2.0,
+    camera_type: str | None = None,
+):
+    """
+    Connects to detected cameras (optionally filtered by type) and saves images from each.
+    Uses default stream profiles for width, height, and FPS.
+
+    Args:
+        output_dir: Directory to save images.
+        record_time_s: Duration in seconds to record images.
+        camera_type: Optional string to filter cameras ("realsense" or "opencv").
+                            If None, uses all detected cameras.
+    """
+    output_dir.mkdir(parents=True, exist_ok=True)
+    logger.info(f"Saving images to {output_dir}")
+    all_camera_metadata = find_and_print_cameras(camera_type_filter=camera_type)
+
+    if not all_camera_metadata:
+        logger.warning("No cameras detected matching the criteria. Cannot save images.")
+        return
+
+    cameras_to_use = []
+    for cam_meta in all_camera_metadata:
+        camera_instance = create_camera_instance(cam_meta)
+        if camera_instance:
+            cameras_to_use.append(camera_instance)
+
+    if not cameras_to_use:
+        logger.warning("No cameras could be connected. Aborting image save.")
+        return
+
+    logger.info(f"Starting image capture for {record_time_s} seconds from {len(cameras_to_use)} cameras.")
+    start_time = time.perf_counter()
+
+    with concurrent.futures.ThreadPoolExecutor(max_workers=len(cameras_to_use) * 2) as executor:
+        try:
+            while time.perf_counter() - start_time < record_time_s:
+                futures = []
+                current_capture_time = time.perf_counter()
+
+                for cam_dict in cameras_to_use:
+                    future = process_camera_image(cam_dict, output_dir, current_capture_time)
+                    if future:
+                        futures.append(future)
+
+                if futures:
+                    concurrent.futures.wait(futures)
+
+        except KeyboardInterrupt:
+            logger.info("Capture interrupted by user.")
+        finally:
+            print("\nFinalizing image saving...")
+            executor.shutdown(wait=True)
+            cleanup_cameras(cameras_to_use)
+            print(f"Image capture finished. Images saved to {output_dir}")
+
+
+def main():
+    parser = argparse.ArgumentParser(
+        description="Unified camera utility script for listing cameras and capturing images."
+    )
+
+    parser.add_argument(
+        "camera_type",
+        type=str,
+        nargs="?",
+        default=None,
+        choices=["realsense", "opencv"],
+        help="Specify camera type to capture from (e.g., 'realsense', 'opencv'). Captures from all if omitted.",
+    )
+    parser.add_argument(
+        "--output-dir",
+        type=Path,
+        default="outputs/captured_images",
+        help="Directory to save images. Default: outputs/captured_images",
+    )
+    parser.add_argument(
+        "--record-time-s",
+        type=float,
+        default=6.0,
+        help="Time duration to attempt capturing frames. Default: 6 seconds.",
+    )
+    args = parser.parse_args()
+    save_images_from_all_cameras(**vars(args))
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_find_joint_limits.py b/lerobot/src/lerobot/scripts/lerobot_find_joint_limits.py
new file mode 100644
index 0000000000000000000000000000000000000000..bcb93ba129cc170c691ad6f5a85e06274a981911
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_find_joint_limits.py
@@ -0,0 +1,222 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Script to find joint limits and end-effector bounds via teleoperation.
+
+Example:
+
+```shell
+lerobot-find-joint-limits \
+  --robot.type=so100_follower \
+  --robot.port=/dev/tty.usbmodem58760432981 \
+  --robot.id=black \
+  --teleop.type=so100_leader \
+  --teleop.port=/dev/tty.usbmodem58760434471 \
+  --teleop.id=blue \
+  --urdf_path=<user>/SO-ARM100-main/Simulation/SO101/so101_new_calib.urdf \
+  --target_frame_name=gripper \
+  --teleop_time_s=30 \
+  --warmup_time_s=5 \
+  --control_loop_fps=30
+```
+"""
+
+import time
+from dataclasses import dataclass
+
+import draccus
+import numpy as np
+
+from lerobot.model.kinematics import RobotKinematics
+from lerobot.robots import (  # noqa: F401
+    RobotConfig,
+    bi_openarm_follower,
+    bi_so_follower,
+    koch_follower,
+    make_robot_from_config,
+    omx_follower,
+    openarm_follower,
+    so_follower,
+)
+from lerobot.teleoperators import (  # noqa: F401
+    TeleoperatorConfig,
+    bi_openarm_leader,
+    bi_so_leader,
+    gamepad,
+    koch_leader,
+    make_teleoperator_from_config,
+    omx_leader,
+    openarm_leader,
+    openarm_mini,
+    so_leader,
+)
+from lerobot.utils.robot_utils import precise_sleep
+
+
+@dataclass
+class FindJointLimitsConfig:
+    teleop: TeleoperatorConfig
+    robot: RobotConfig
+
+    # Path to URDF file for kinematics
+    # NOTE: It is highly recommended to use the urdf in the SO-ARM100 repo:
+    # https://github.com/TheRobotStudio/SO-ARM100/blob/main/Simulation/SO101/so101_new_calib.urdf
+    urdf_path: str
+    target_frame_name: str = "gripper"
+
+    # Duration of the recording phase in seconds
+    teleop_time_s: float = 30
+    # Duration of the warmup phase in seconds
+    warmup_time_s: float = 5
+    # Control loop frequency
+    control_loop_fps: int = 30
+
+
+@draccus.wrap()
+def find_joint_and_ee_bounds(cfg: FindJointLimitsConfig):
+    teleop = make_teleoperator_from_config(cfg.teleop)
+    robot = make_robot_from_config(cfg.robot)
+
+    print(f"Connecting to robot: {cfg.robot.type}...")
+    teleop.connect()
+    robot.connect()
+    print("Devices connected.")
+
+    # Initialize Kinematics
+    try:
+        kinematics = RobotKinematics(cfg.urdf_path, cfg.target_frame_name)
+    except Exception as e:
+        print(f"Error initializing kinematics: {e}")
+        print("Ensure URDF path and target frame name are correct.")
+        robot.disconnect()
+        teleop.disconnect()
+        return
+
+    # Initialize variables
+    max_pos = None
+    min_pos = None
+    max_ee = None
+    min_ee = None
+
+    start_t = time.perf_counter()
+    warmup_done = False
+
+    print("\n" + "=" * 40)
+    print(f"  WARMUP PHASE ({cfg.warmup_time_s}s)")
+    print("  Move the robot freely to ensure control works.")
+    print("  Data is NOT being recorded yet.")
+    print("=" * 40 + "\n")
+
+    try:
+        while True:
+            t0 = time.perf_counter()
+
+            # 1. Teleoperation Control Loop
+            action = teleop.get_action()
+            robot.send_action(action)
+
+            # 2. Read Observations
+            observation = robot.get_observation()
+            joint_positions = np.array([observation[f"{key}.pos"] for key in robot.bus.motors])
+
+            # 3. Calculate Kinematics
+            # Forward kinematics to get (x, y, z) translation
+            ee_pos = kinematics.forward_kinematics(joint_positions)[:3, 3]
+
+            current_time = time.perf_counter()
+            elapsed = current_time - start_t
+
+            # 4. Handle Phases
+            if elapsed < cfg.warmup_time_s:
+                # Still in warmup
+                pass
+
+            else:
+                # Phase Transition: Warmup -> Recording
+                if not warmup_done:
+                    print("\n" + "=" * 40)
+                    print("  RECORDING STARTED")
+                    print("  Move robot to ALL joint limits.")
+                    print("  Press Ctrl+C to stop early and save results.")
+                    print("=" * 40 + "\n")
+
+                    # Initialize limits with current position at start of recording
+                    max_pos = joint_positions.copy()
+                    min_pos = joint_positions.copy()
+                    max_ee = ee_pos.copy()
+                    min_ee = ee_pos.copy()
+                    warmup_done = True
+
+                # Update Limits
+                max_ee = np.maximum(max_ee, ee_pos)
+                min_ee = np.minimum(min_ee, ee_pos)
+                max_pos = np.maximum(max_pos, joint_positions)
+                min_pos = np.minimum(min_pos, joint_positions)
+
+                # Time check
+                recording_time = elapsed - cfg.warmup_time_s
+                remaining = cfg.teleop_time_s - recording_time
+
+                # Simple throttle for print statements (every ~1 sec)
+                if int(recording_time * 100) % 100 == 0:
+                    print(f"Time remaining: {remaining:.1f}s", end="\r")
+
+                if recording_time > cfg.teleop_time_s:
+                    print("\nTime limit reached.")
+                    break
+
+            precise_sleep(max(1.0 / cfg.control_loop_fps - (time.perf_counter() - t0), 0.0))
+
+    except KeyboardInterrupt:
+        print("\n\nInterrupted by user. Stopping safely...")
+
+    finally:
+        # Safety: Disconnect devices
+        print("\nDisconnecting devices...")
+        robot.disconnect()
+        teleop.disconnect()
+
+    # Results Output
+    if max_pos is not None:
+        print("\n" + "=" * 40)
+        print("FINAL RESULTS")
+        print("=" * 40)
+
+        # Rounding for readability
+        r_max_ee = np.round(max_ee, 4).tolist()
+        r_min_ee = np.round(min_ee, 4).tolist()
+        r_max_pos = np.round(max_pos, 4).tolist()
+        r_min_pos = np.round(min_pos, 4).tolist()
+
+        print("\n# End Effector Bounds (x, y, z):")
+        print(f"max_ee = {r_max_ee}")
+        print(f"min_ee = {r_min_ee}")
+
+        print("\n# Joint Position Limits (radians):")
+        print(f"max_pos = {r_max_pos}")
+        print(f"min_pos = {r_min_pos}")
+
+    else:
+        print("No data recorded (exited during warmup).")
+
+
+def main():
+    find_joint_and_ee_bounds()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_find_port.py b/lerobot/src/lerobot/scripts/lerobot_find_port.py
new file mode 100644
index 0000000000000000000000000000000000000000..e32b9cb993648964297c1ad1f2755aade3623105
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_find_port.py
@@ -0,0 +1,69 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Helper to find the USB port associated with your MotorsBus.
+
+Example:
+
+```shell
+lerobot-find-port
+```
+"""
+
+import platform
+import time
+from pathlib import Path
+
+
+def find_available_ports():
+    from serial.tools import list_ports  # Part of pyserial library
+
+    if platform.system() == "Windows":
+        # List COM ports using pyserial
+        ports = [port.device for port in list_ports.comports()]
+    else:  # Linux/macOS
+        # List /dev/tty* ports for Unix-based systems
+        ports = [str(path) for path in Path("/dev").glob("tty*")]
+    return ports
+
+
+def find_port():
+    print("Finding all available ports for the MotorsBus.")
+    ports_before = find_available_ports()
+    print("Ports before disconnecting:", ports_before)
+
+    print("Remove the USB cable from your MotorsBus and press Enter when done.")
+    input()  # Wait for user to disconnect the device
+
+    time.sleep(0.5)  # Allow some time for port to be released
+    ports_after = find_available_ports()
+    ports_diff = list(set(ports_before) - set(ports_after))
+
+    if len(ports_diff) == 1:
+        port = ports_diff[0]
+        print(f"The port of this MotorsBus is '{port}'")
+        print("Reconnect the USB cable.")
+    elif len(ports_diff) == 0:
+        raise OSError(f"Could not detect the port. No difference was found ({ports_diff}).")
+    else:
+        raise OSError(f"Could not detect the port. More than one port was found ({ports_diff}).")
+
+
+def main():
+    find_port()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_imgtransform_viz.py b/lerobot/src/lerobot/scripts/lerobot_imgtransform_viz.py
new file mode 100644
index 0000000000000000000000000000000000000000..bc13f05086ab19e2c494a40a0b07ef87be912102
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_imgtransform_viz.py
@@ -0,0 +1,134 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+""" Visualize effects of image transforms for a given configuration.
+
+This script will generate examples of transformed images as they are output by LeRobot dataset.
+Additionally, each individual transform can be visualized separately as well as examples of combined transforms
+
+Example:
+```bash
+lerobot-imgtransform-viz \
+  --repo_id=lerobot/pusht \
+  --episodes='[0]' \
+  --image_transforms.enable=True
+```
+"""
+
+import logging
+from copy import deepcopy
+from dataclasses import replace
+from pathlib import Path
+
+import draccus
+from torchvision.transforms import ToPILImage
+
+from lerobot.configs.default import DatasetConfig
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.transforms import (
+    ImageTransforms,
+    ImageTransformsConfig,
+    make_transform_from_config,
+)
+
+OUTPUT_DIR = Path("outputs/image_transforms")
+to_pil = ToPILImage()
+
+
+def save_all_transforms(cfg: ImageTransformsConfig, original_frame, output_dir, n_examples):
+    output_dir_all = output_dir / "all"
+    output_dir_all.mkdir(parents=True, exist_ok=True)
+
+    tfs = ImageTransforms(cfg)
+    for i in range(1, n_examples + 1):
+        transformed_frame = tfs(original_frame)
+        to_pil(transformed_frame).save(output_dir_all / f"{i}.png", quality=100)
+
+    print("Combined transforms examples saved to:")
+    print(f"    {output_dir_all}")
+
+
+def save_each_transform(cfg: ImageTransformsConfig, original_frame, output_dir, n_examples):
+    if not cfg.enable:
+        logging.warning(
+            "No single transforms will be saved, because `image_transforms.enable=False`. To enable, set `enable` to True in `ImageTransformsConfig` or in the command line with `--image_transforms.enable=True`."
+        )
+        return
+
+    print("Individual transforms examples saved to:")
+    for tf_name, tf_cfg in cfg.tfs.items():
+        # Apply a few transformation with random value in min_max range
+        output_dir_single = output_dir / tf_name
+        output_dir_single.mkdir(parents=True, exist_ok=True)
+
+        tf = make_transform_from_config(tf_cfg)
+        for i in range(1, n_examples + 1):
+            transformed_frame = tf(original_frame)
+            to_pil(transformed_frame).save(output_dir_single / f"{i}.png", quality=100)
+
+        # Apply min, max, average transformations
+        tf_cfg_kwgs_min = deepcopy(tf_cfg.kwargs)
+        tf_cfg_kwgs_max = deepcopy(tf_cfg.kwargs)
+        tf_cfg_kwgs_avg = deepcopy(tf_cfg.kwargs)
+
+        for key, (min_, max_) in tf_cfg.kwargs.items():
+            avg = (min_ + max_) / 2
+            tf_cfg_kwgs_min[key] = [min_, min_]
+            tf_cfg_kwgs_max[key] = [max_, max_]
+            tf_cfg_kwgs_avg[key] = [avg, avg]
+
+        tf_min = make_transform_from_config(replace(tf_cfg, **{"kwargs": tf_cfg_kwgs_min}))
+        tf_max = make_transform_from_config(replace(tf_cfg, **{"kwargs": tf_cfg_kwgs_max}))
+        tf_avg = make_transform_from_config(replace(tf_cfg, **{"kwargs": tf_cfg_kwgs_avg}))
+
+        tf_frame_min = tf_min(original_frame)
+        tf_frame_max = tf_max(original_frame)
+        tf_frame_avg = tf_avg(original_frame)
+
+        to_pil(tf_frame_min).save(output_dir_single / "min.png", quality=100)
+        to_pil(tf_frame_max).save(output_dir_single / "max.png", quality=100)
+        to_pil(tf_frame_avg).save(output_dir_single / "mean.png", quality=100)
+
+        print(f"    {output_dir_single}")
+
+
+@draccus.wrap()
+def visualize_image_transforms(cfg: DatasetConfig, output_dir: Path = OUTPUT_DIR, n_examples: int = 5):
+    dataset = LeRobotDataset(
+        repo_id=cfg.repo_id,
+        episodes=cfg.episodes,
+        revision=cfg.revision,
+        video_backend=cfg.video_backend,
+    )
+
+    output_dir = output_dir / cfg.repo_id.split("/")[-1]
+    output_dir.mkdir(parents=True, exist_ok=True)
+
+    # Get 1st frame from 1st camera of 1st episode
+    original_frame = dataset[0][dataset.meta.camera_keys[0]]
+    to_pil(original_frame).save(output_dir / "original_frame.png", quality=100)
+    print("\nOriginal frame saved to:")
+    print(f"    {output_dir / 'original_frame.png'}.")
+
+    save_all_transforms(cfg.image_transforms, original_frame, output_dir, n_examples)
+    save_each_transform(cfg.image_transforms, original_frame, output_dir, n_examples)
+
+
+def main():
+    visualize_image_transforms()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_info.py b/lerobot/src/lerobot/scripts/lerobot_info.py
new file mode 100644
index 0000000000000000000000000000000000000000..879d392be96d680a79a9d45c0c6b460f51cc5c14
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_info.py
@@ -0,0 +1,126 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Use this script to get a quick summary of your system config.
+It should be able to run without any of LeRobot's dependencies or LeRobot itself installed.
+
+Example:
+
+```shell
+lerobot-info
+```
+"""
+
+import importlib
+import platform
+import shutil
+import subprocess
+from importlib.metadata import PackageNotFoundError, distribution
+
+PACKAGE_NAME = "lerobot"
+
+
+def get_ffmpeg_version() -> str:
+    """Get the ffmpeg version if installed, otherwise return 'N/A'."""
+    command_path = shutil.which("ffmpeg")
+    if command_path is None:
+        return "N/A"
+    try:
+        result = subprocess.run([command_path, "-version"], capture_output=True, text=True, check=True)
+        first_line = result.stdout.splitlines()[0]
+        version_info = first_line.split(" ")[2]
+        return version_info
+    except (subprocess.SubprocessError, IndexError):
+        return "Installed (version parsing failed)"
+
+
+def get_package_version(package_name: str) -> str:
+    """Get the version of a package if it exists, otherwise return 'N/A'."""
+    try:
+        module = importlib.import_module(package_name)
+        return getattr(module, "__version__", "Installed (version not found)")
+    except ImportError:
+        return "N/A"
+
+
+def get_sys_info() -> dict[str, str]:
+    """Run this to get basic system info to help for tracking issues & bugs."""
+    # General package versions
+    info = {
+        "LeRobot version": get_package_version(PACKAGE_NAME),
+        "Platform": platform.platform(),
+        "Python version": platform.python_version(),
+        "Huggingface Hub version": get_package_version("huggingface_hub"),
+        "Datasets version": get_package_version("datasets"),
+        "Numpy version": get_package_version("numpy"),
+        "FFmpeg version": get_ffmpeg_version(),
+    }
+
+    # PyTorch and GPU specific information
+    torch_version = "N/A"
+    torch_cuda_available = "N/A"
+    cuda_version = "N/A"
+    gpu_model = "N/A"
+    try:
+        import torch
+
+        torch_version = str(torch.__version__)
+        torch_cuda_available = torch.cuda.is_available()
+        if torch_cuda_available:
+            cuda_version = str(torch.version.cuda)
+            # Gets the name of the first available GPU
+            gpu_model = torch.cuda.get_device_name(0)
+    except ImportError:
+        # If torch is not installed, the default "N/A" values will be used.
+        pass
+
+    info.update(
+        {
+            "PyTorch version": torch_version,
+            "Is PyTorch built with CUDA support?": str(torch_cuda_available),
+            "Cuda version": cuda_version,
+            "GPU model": gpu_model,
+            "Using GPU in script?": "<fill in>",
+        }
+    )
+    scripts = "N/A"
+    try:
+        dist = distribution(PACKAGE_NAME)
+        scripts = [ep.name for ep in dist.entry_points if ep.group == "console_scripts"]
+    except PackageNotFoundError:
+        pass
+
+    info.update({f"{PACKAGE_NAME} scripts": str(scripts)})
+
+    return info
+
+
+def format_dict_for_markdown(d: dict[str, str]) -> str:
+    """Formats a dictionary into a markdown-friendly bulleted list."""
+    return "\n".join([f"- {prop}: {val}" for prop, val in d.items()])
+
+
+def main():
+    """
+    Main function to print system info in markdown format.
+    """
+    system_info = get_sys_info()
+    print(format_dict_for_markdown(system_info))
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_record.py b/lerobot/src/lerobot/scripts/lerobot_record.py
new file mode 100644
index 0000000000000000000000000000000000000000..819634ba2a371a6f8791a6dffb11e1b2ac97a4b1
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_record.py
@@ -0,0 +1,610 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Records a dataset. Actions for the robot can be either generated by teleoperation or by a policy.
+
+Example:
+
+```shell
+lerobot-record \
+    --robot.type=so100_follower \
+    --robot.port=/dev/tty.usbmodem58760431541 \
+    --robot.cameras="{laptop: {type: opencv, index_or_path: 0, width: 640, height: 480, fps: 30}}" \
+    --robot.id=black \
+    --dataset.repo_id=<my_username>/<my_dataset_name> \
+    --dataset.num_episodes=2 \
+    --dataset.single_task="Grab the cube" \
+    --dataset.streaming_encoding=true \
+    --dataset.encoder_threads=2 \
+    --display_data=true
+    # <- Optional: specify video codec (auto, h264, hevc, libsvtav1). Default is libsvtav1. \
+    # --dataset.vcodec=h264 \
+    # <- Teleop optional if you want to teleoperate to record or in between episodes with a policy \
+    # --teleop.type=so100_leader \
+    # --teleop.port=/dev/tty.usbmodem58760431551 \
+    # --teleop.id=blue \
+    # <- Policy optional if you want to record with a policy \
+    # --policy.path=${HF_USER}/my_policy \
+```
+
+Example recording with bimanual so100:
+```shell
+lerobot-record \
+  --robot.type=bi_so_follower \
+  --robot.left_arm_config.port=/dev/tty.usbmodem5A460822851 \
+  --robot.right_arm_config.port=/dev/tty.usbmodem5A460814411 \
+  --robot.id=bimanual_follower \
+  --robot.left_arm_config.cameras='{
+    wrist: {"type": "opencv", "index_or_path": 1, "width": 640, "height": 480, "fps": 30},
+    top: {"type": "opencv", "index_or_path": 3, "width": 640, "height": 480, "fps": 30},
+  }' --robot.right_arm_config.cameras='{
+    wrist: {"type": "opencv", "index_or_path": 2, "width": 640, "height": 480, "fps": 30},
+    front: {"type": "opencv", "index_or_path": 4, "width": 640, "height": 480, "fps": 30},
+  }' \
+  --teleop.type=bi_so_leader \
+  --teleop.left_arm_config.port=/dev/tty.usbmodem5A460852721 \
+  --teleop.right_arm_config.port=/dev/tty.usbmodem5A460819811 \
+  --teleop.id=bimanual_leader \
+  --display_data=true \
+  --dataset.repo_id=${HF_USER}/bimanual-so-handover-cube \
+  --dataset.num_episodes=25 \
+  --dataset.single_task="Grab and handover the red cube to the other arm" \
+  --dataset.streaming_encoding=true \
+  # --dataset.vcodec=auto \
+  --dataset.encoder_threads=2
+```
+"""
+
+import logging
+import time
+from dataclasses import asdict, dataclass, field
+from pathlib import Path
+from pprint import pformat
+from typing import Any
+
+from lerobot.cameras import (  # noqa: F401
+    CameraConfig,  # noqa: F401
+)
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig  # noqa: F401
+from lerobot.cameras.reachy2_camera.configuration_reachy2_camera import Reachy2CameraConfig  # noqa: F401
+from lerobot.cameras.realsense.configuration_realsense import RealSenseCameraConfig  # noqa: F401
+from lerobot.cameras.zmq.configuration_zmq import ZMQCameraConfig  # noqa: F401
+from lerobot.configs import parser
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.datasets.feature_utils import build_dataset_frame, combine_feature_dicts
+from lerobot.datasets.image_writer import safe_stop_image_writer
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.datasets.video_utils import VideoEncodingManager
+from lerobot.policies.factory import make_policy, make_pre_post_processors
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.utils import make_robot_action
+from lerobot.processor import (
+    PolicyAction,
+    PolicyProcessorPipeline,
+    RobotAction,
+    RobotObservation,
+    RobotProcessorPipeline,
+    make_default_processors,
+)
+from lerobot.processor.rename_processor import rename_stats
+from lerobot.robots import (  # noqa: F401
+    Robot,
+    RobotConfig,
+    bi_openarm_follower,
+    bi_so_follower,
+    earthrover_mini_plus,
+    hope_jr,
+    koch_follower,
+    make_robot_from_config,
+    omx_follower,
+    openarm_follower,
+    reachy2,
+    so_follower,
+    unitree_g1 as unitree_g1_robot,
+)
+from lerobot.teleoperators import (  # noqa: F401
+    Teleoperator,
+    TeleoperatorConfig,
+    bi_openarm_leader,
+    bi_so_leader,
+    homunculus,
+    koch_leader,
+    make_teleoperator_from_config,
+    omx_leader,
+    openarm_leader,
+    openarm_mini,
+    reachy2_teleoperator,
+    so_leader,
+    unitree_g1,
+)
+from lerobot.teleoperators.keyboard.teleop_keyboard import KeyboardTeleop
+from lerobot.utils.constants import ACTION, OBS_STR
+from lerobot.utils.control_utils import (
+    init_keyboard_listener,
+    is_headless,
+    predict_action,
+    sanity_check_dataset_name,
+    sanity_check_dataset_robot_compatibility,
+)
+from lerobot.utils.device_utils import get_safe_torch_device
+from lerobot.utils.import_utils import register_third_party_plugins
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import (
+    init_logging,
+    log_say,
+)
+from lerobot.utils.visualization_utils import init_rerun, log_rerun_data
+
+
+@dataclass
+class DatasetRecordConfig:
+    # Dataset identifier. By convention it should match '{hf_username}/{dataset_name}' (e.g. `lerobot/test`).
+    repo_id: str
+    # A short but accurate description of the task performed during the recording (e.g. "Pick the Lego block and drop it in the box on the right.")
+    single_task: str
+    # Root directory where the dataset will be stored (e.g. 'dataset/path'). If None, defaults to $HF_LEROBOT_HOME/repo_id.
+    root: str | Path | None = None
+    # Limit the frames per second.
+    fps: int = 30
+    # Number of seconds for data recording for each episode.
+    episode_time_s: int | float = 60
+    # Number of seconds for resetting the environment after each episode.
+    reset_time_s: int | float = 60
+    # Number of episodes to record.
+    num_episodes: int = 50
+    # Encode frames in the dataset into video
+    video: bool = True
+    # Upload dataset to Hugging Face hub.
+    push_to_hub: bool = True
+    # Upload on private repository on the Hugging Face hub.
+    private: bool = False
+    # Add tags to your dataset on the hub.
+    tags: list[str] | None = None
+    # Number of subprocesses handling the saving of frames as PNG. Set to 0 to use threads only;
+    # set to ≥1 to use subprocesses, each using threads to write images. The best number of processes
+    # and threads depends on your system. We recommend 4 threads per camera with 0 processes.
+    # If fps is unstable, adjust the thread count. If still unstable, try using 1 or more subprocesses.
+    num_image_writer_processes: int = 0
+    # Number of threads writing the frames as png images on disk, per camera.
+    # Too many threads might cause unstable teleoperation fps due to main thread being blocked.
+    # Not enough threads might cause low camera fps.
+    num_image_writer_threads_per_camera: int = 4
+    # Number of episodes to record before batch encoding videos
+    # Set to 1 for immediate encoding (default behavior), or higher for batched encoding
+    video_encoding_batch_size: int = 1
+    # Video codec for encoding videos. Options: 'h264', 'hevc', 'libsvtav1', 'auto',
+    # or hardware-specific: 'h264_videotoolbox', 'h264_nvenc', 'h264_vaapi', 'h264_qsv'.
+    # Use 'auto' to auto-detect the best available hardware encoder.
+    vcodec: str = "libsvtav1"
+    # Enable streaming video encoding: encode frames in real-time during capture instead
+    # of writing PNG images first. Makes save_episode() near-instant. More info in the documentation: https://huggingface.co/docs/lerobot/streaming_video_encoding
+    streaming_encoding: bool = False
+    # Maximum number of frames to buffer per camera when using streaming encoding.
+    # ~1s buffer at 30fps. Provides backpressure if the encoder can't keep up.
+    encoder_queue_maxsize: int = 30
+    # Number of threads per encoder instance. None = auto (codec default).
+    # Lower values reduce CPU usage, maps to 'lp' (via svtav1-params) for libsvtav1 and 'threads' for h264/hevc..
+    encoder_threads: int | None = None
+    # Rename map for the observation to override the image and state keys
+    rename_map: dict[str, str] = field(default_factory=dict)
+
+    def __post_init__(self):
+        if self.single_task is None:
+            raise ValueError("You need to provide a task as argument in `single_task`.")
+
+
+@dataclass
+class RecordConfig:
+    robot: RobotConfig
+    dataset: DatasetRecordConfig
+    # Whether to control the robot with a teleoperator
+    teleop: TeleoperatorConfig | None = None
+    # Whether to control the robot with a policy
+    policy: PreTrainedConfig | None = None
+    # Display all cameras on screen
+    display_data: bool = False
+    # Display data on a remote Rerun server
+    display_ip: str | None = None
+    # Port of the remote Rerun server
+    display_port: int | None = None
+    # Whether to  display compressed images in Rerun
+    display_compressed_images: bool = False
+    # Use vocal synthesis to read events.
+    play_sounds: bool = True
+    # Resume recording on an existing dataset.
+    resume: bool = False
+
+    def __post_init__(self):
+        # HACK: We parse again the cli args here to get the pretrained path if there was one.
+        policy_path = parser.get_path_arg("policy")
+
+        if policy_path:
+            cli_overrides = parser.get_cli_overrides("policy")
+
+            self.policy = PreTrainedConfig.from_pretrained(policy_path, cli_overrides=cli_overrides)
+            self.policy.pretrained_path = policy_path
+
+        if self.teleop is None and self.policy is None:
+            raise ValueError("Choose a policy, a teleoperator or both to control the robot")
+
+    @classmethod
+    def __get_path_fields__(cls) -> list[str]:
+        """This enables the parser to load config from the policy using `--policy.path=local/dir`"""
+        return ["policy"]
+
+
+""" --------------- record_loop() data flow --------------------------
+       [ Robot ]
+           V
+     [ robot.get_observation() ] ---> raw_obs
+           V
+     [ robot_observation_processor ] ---> processed_obs
+           V
+     .-----( ACTION LOGIC )------------------.
+     V                                       V
+     [ From Teleoperator ]                   [ From Policy ]
+     |                                       |
+     |  [teleop.get_action] -> raw_action    |   [predict_action]
+     |          |                            |          |
+     |          V                            |          V
+     | [teleop_action_processor]             |          |
+     |          |                            |          |
+     '---> processed_teleop_action           '---> processed_policy_action
+     |                                       |
+     '-------------------------.-------------'
+                               V
+                  [ robot_action_processor ] --> robot_action_to_send
+                               V
+                    [ robot.send_action() ] -- (Robot Executes)
+                               V
+                    ( Save to Dataset )
+                               V
+                  ( Rerun Log / Loop Wait )
+"""
+
+
+@safe_stop_image_writer
+def record_loop(
+    robot: Robot,
+    events: dict,
+    fps: int,
+    teleop_action_processor: RobotProcessorPipeline[
+        tuple[RobotAction, RobotObservation], RobotAction
+    ],  # runs after teleop
+    robot_action_processor: RobotProcessorPipeline[
+        tuple[RobotAction, RobotObservation], RobotAction
+    ],  # runs before robot
+    robot_observation_processor: RobotProcessorPipeline[
+        RobotObservation, RobotObservation
+    ],  # runs after robot
+    dataset: LeRobotDataset | None = None,
+    teleop: Teleoperator | list[Teleoperator] | None = None,
+    policy: PreTrainedPolicy | None = None,
+    preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]] | None = None,
+    postprocessor: PolicyProcessorPipeline[PolicyAction, PolicyAction] | None = None,
+    control_time_s: int | None = None,
+    single_task: str | None = None,
+    display_data: bool = False,
+    display_compressed_images: bool = False,
+):
+    if dataset is not None and dataset.fps != fps:
+        raise ValueError(f"The dataset fps should be equal to requested fps ({dataset.fps} != {fps}).")
+
+    teleop_arm = teleop_keyboard = None
+    if isinstance(teleop, list):
+        teleop_keyboard = next((t for t in teleop if isinstance(t, KeyboardTeleop)), None)
+        teleop_arm = next(
+            (
+                t
+                for t in teleop
+                if isinstance(
+                    t,
+                    (
+                        so_leader.SO100Leader
+                        | so_leader.SO101Leader
+                        | koch_leader.KochLeader
+                        | omx_leader.OmxLeader
+                    ),
+                )
+            ),
+            None,
+        )
+
+        if not (teleop_arm and teleop_keyboard and len(teleop) == 2 and robot.name == "lekiwi_client"):
+            raise ValueError(
+                "For multi-teleop, the list must contain exactly one KeyboardTeleop and one arm teleoperator. Currently only supported for LeKiwi robot."
+            )
+
+    # Reset policy and processor if they are provided
+    if policy is not None and preprocessor is not None and postprocessor is not None:
+        policy.reset()
+        preprocessor.reset()
+        postprocessor.reset()
+
+    no_action_count = 0
+    timestamp = 0
+    start_episode_t = time.perf_counter()
+    while timestamp < control_time_s:
+        start_loop_t = time.perf_counter()
+
+        if events["exit_early"]:
+            events["exit_early"] = False
+            break
+
+        # Get robot observation
+        obs = robot.get_observation()
+
+        # Applies a pipeline to the raw robot observation, default is IdentityProcessor
+        obs_processed = robot_observation_processor(obs)
+
+        if policy is not None or dataset is not None:
+            observation_frame = build_dataset_frame(dataset.features, obs_processed, prefix=OBS_STR)
+
+        # Get action from either policy or teleop
+        if policy is not None and preprocessor is not None and postprocessor is not None:
+            action_values = predict_action(
+                observation=observation_frame,
+                policy=policy,
+                device=get_safe_torch_device(policy.config.device),
+                preprocessor=preprocessor,
+                postprocessor=postprocessor,
+                use_amp=policy.config.use_amp,
+                task=single_task,
+                robot_type=robot.robot_type,
+            )
+
+            act_processed_policy: RobotAction = make_robot_action(action_values, dataset.features)
+
+        elif policy is None and isinstance(teleop, Teleoperator):
+            if robot.name == "unitree_g1":
+                teleop.send_feedback(obs)
+            act = teleop.get_action()
+
+            # Applies a pipeline to the raw teleop action, default is IdentityProcessor
+            act_processed_teleop = teleop_action_processor((act, obs))
+
+        elif policy is None and isinstance(teleop, list):
+            arm_action = teleop_arm.get_action()
+            arm_action = {f"arm_{k}": v for k, v in arm_action.items()}
+            keyboard_action = teleop_keyboard.get_action()
+            base_action = robot._from_keyboard_to_base_action(keyboard_action)
+            act = {**arm_action, **base_action} if len(base_action) > 0 else arm_action
+            act_processed_teleop = teleop_action_processor((act, obs))
+        else:
+            no_action_count += 1
+            if no_action_count == 1 or no_action_count % 10 == 0:
+                logging.warning(
+                    "No policy or teleoperator provided, skipping action generation. "
+                    "This is likely to happen when resetting the environment without a teleop device. "
+                    "The robot won't be at its rest position at the start of the next episode."
+                )
+            continue
+
+        # Applies a pipeline to the action, default is IdentityProcessor
+        if policy is not None and act_processed_policy is not None:
+            action_values = act_processed_policy
+            robot_action_to_send = robot_action_processor((act_processed_policy, obs))
+        else:
+            action_values = act_processed_teleop
+            robot_action_to_send = robot_action_processor((act_processed_teleop, obs))
+
+        # Send action to robot
+        # Action can eventually be clipped using `max_relative_target`,
+        # so action actually sent is saved in the dataset. action = postprocessor.process(action)
+        # TODO(steven, pepijn, adil): we should use a pipeline step to clip the action, so the sent action is the action that we input to the robot.
+        _sent_action = robot.send_action(robot_action_to_send)
+
+        # Write to dataset
+        if dataset is not None:
+            action_frame = build_dataset_frame(dataset.features, action_values, prefix=ACTION)
+            frame = {**observation_frame, **action_frame, "task": single_task}
+            dataset.add_frame(frame)
+
+        if display_data:
+            log_rerun_data(
+                observation=obs_processed, action=action_values, compress_images=display_compressed_images
+            )
+
+        dt_s = time.perf_counter() - start_loop_t
+
+        sleep_time_s: float = 1 / fps - dt_s
+        if sleep_time_s < 0:
+            logging.warning(
+                f"Record loop is running slower ({1 / dt_s:.1f} Hz) than the target FPS ({fps} Hz). Dataset frames might be dropped and robot control might be unstable. Common causes are: 1) Camera FPS not keeping up 2) Policy inference taking too long 3) CPU starvation"
+            )
+
+        precise_sleep(max(sleep_time_s, 0.0))
+
+        timestamp = time.perf_counter() - start_episode_t
+
+
+@parser.wrap()
+def record(cfg: RecordConfig) -> LeRobotDataset:
+    init_logging()
+    logging.info(pformat(asdict(cfg)))
+    if cfg.display_data:
+        init_rerun(session_name="recording", ip=cfg.display_ip, port=cfg.display_port)
+    display_compressed_images = (
+        True
+        if (cfg.display_data and cfg.display_ip is not None and cfg.display_port is not None)
+        else cfg.display_compressed_images
+    )
+
+    robot = make_robot_from_config(cfg.robot)
+    teleop = make_teleoperator_from_config(cfg.teleop) if cfg.teleop is not None else None
+
+    teleop_action_processor, robot_action_processor, robot_observation_processor = make_default_processors()
+
+    dataset_features = combine_feature_dicts(
+        aggregate_pipeline_dataset_features(
+            pipeline=teleop_action_processor,
+            initial_features=create_initial_features(
+                action=robot.action_features
+            ),  # TODO(steven, pepijn): in future this should be come from teleop or policy
+            use_videos=cfg.dataset.video,
+        ),
+        aggregate_pipeline_dataset_features(
+            pipeline=robot_observation_processor,
+            initial_features=create_initial_features(observation=robot.observation_features),
+            use_videos=cfg.dataset.video,
+        ),
+    )
+
+    dataset = None
+    listener = None
+
+    try:
+        if cfg.resume:
+            dataset = LeRobotDataset(
+                cfg.dataset.repo_id,
+                root=cfg.dataset.root,
+                batch_encoding_size=cfg.dataset.video_encoding_batch_size,
+                vcodec=cfg.dataset.vcodec,
+                streaming_encoding=cfg.dataset.streaming_encoding,
+                encoder_queue_maxsize=cfg.dataset.encoder_queue_maxsize,
+                encoder_threads=cfg.dataset.encoder_threads,
+            )
+
+            if hasattr(robot, "cameras") and len(robot.cameras) > 0:
+                dataset.start_image_writer(
+                    num_processes=cfg.dataset.num_image_writer_processes,
+                    num_threads=cfg.dataset.num_image_writer_threads_per_camera * len(robot.cameras),
+                )
+            sanity_check_dataset_robot_compatibility(dataset, robot, cfg.dataset.fps, dataset_features)
+        else:
+            # Create empty dataset or load existing saved episodes
+            sanity_check_dataset_name(cfg.dataset.repo_id, cfg.policy)
+            dataset = LeRobotDataset.create(
+                cfg.dataset.repo_id,
+                cfg.dataset.fps,
+                root=cfg.dataset.root,
+                robot_type=robot.name,
+                features=dataset_features,
+                use_videos=cfg.dataset.video,
+                image_writer_processes=cfg.dataset.num_image_writer_processes,
+                image_writer_threads=cfg.dataset.num_image_writer_threads_per_camera * len(robot.cameras),
+                batch_encoding_size=cfg.dataset.video_encoding_batch_size,
+                vcodec=cfg.dataset.vcodec,
+                streaming_encoding=cfg.dataset.streaming_encoding,
+                encoder_queue_maxsize=cfg.dataset.encoder_queue_maxsize,
+                encoder_threads=cfg.dataset.encoder_threads,
+            )
+
+        # Load pretrained policy
+        policy = None if cfg.policy is None else make_policy(cfg.policy, ds_meta=dataset.meta)
+        preprocessor = None
+        postprocessor = None
+        if cfg.policy is not None:
+            preprocessor, postprocessor = make_pre_post_processors(
+                policy_cfg=cfg.policy,
+                pretrained_path=cfg.policy.pretrained_path,
+                dataset_stats=rename_stats(dataset.meta.stats, cfg.dataset.rename_map),
+                preprocessor_overrides={
+                    "device_processor": {"device": cfg.policy.device},
+                    "rename_observations_processor": {"rename_map": cfg.dataset.rename_map},
+                },
+            )
+
+        robot.connect()
+        if teleop is not None:
+            teleop.connect()
+
+        listener, events = init_keyboard_listener()
+
+        if not cfg.dataset.streaming_encoding:
+            logging.info(
+                "Streaming encoding is disabled. If you have capable hardware, consider enabling it for way faster episode saving. --dataset.streaming_encoding=true --dataset.encoder_threads=2 # --dataset.vcodec=auto. More info in the documentation: https://huggingface.co/docs/lerobot/streaming_video_encoding"
+            )
+
+        with VideoEncodingManager(dataset):
+            recorded_episodes = 0
+            while recorded_episodes < cfg.dataset.num_episodes and not events["stop_recording"]:
+                log_say(f"Recording episode {dataset.num_episodes}", cfg.play_sounds)
+                record_loop(
+                    robot=robot,
+                    events=events,
+                    fps=cfg.dataset.fps,
+                    teleop_action_processor=teleop_action_processor,
+                    robot_action_processor=robot_action_processor,
+                    robot_observation_processor=robot_observation_processor,
+                    teleop=teleop,
+                    policy=policy,
+                    preprocessor=preprocessor,
+                    postprocessor=postprocessor,
+                    dataset=dataset,
+                    control_time_s=cfg.dataset.episode_time_s,
+                    single_task=cfg.dataset.single_task,
+                    display_data=cfg.display_data,
+                    display_compressed_images=display_compressed_images,
+                )
+
+                # Execute a few seconds without recording to give time to manually reset the environment
+                # Skip reset for the last episode to be recorded
+                if not events["stop_recording"] and (
+                    (recorded_episodes < cfg.dataset.num_episodes - 1) or events["rerecord_episode"]
+                ):
+                    log_say("Reset the environment", cfg.play_sounds)
+
+                    record_loop(
+                        robot=robot,
+                        events=events,
+                        fps=cfg.dataset.fps,
+                        teleop_action_processor=teleop_action_processor,
+                        robot_action_processor=robot_action_processor,
+                        robot_observation_processor=robot_observation_processor,
+                        teleop=teleop,
+                        control_time_s=cfg.dataset.reset_time_s,
+                        single_task=cfg.dataset.single_task,
+                        display_data=cfg.display_data,
+                    )
+
+                if events["rerecord_episode"]:
+                    log_say("Re-record episode", cfg.play_sounds)
+                    events["rerecord_episode"] = False
+                    events["exit_early"] = False
+                    dataset.clear_episode_buffer()
+                    continue
+
+                dataset.save_episode()
+                recorded_episodes += 1
+    finally:
+        log_say("Stop recording", cfg.play_sounds, blocking=True)
+
+        if dataset:
+            dataset.finalize()
+
+        if robot.is_connected:
+            robot.disconnect()
+        if teleop and teleop.is_connected:
+            teleop.disconnect()
+
+        if not is_headless() and listener:
+            listener.stop()
+
+        if cfg.dataset.push_to_hub:
+            dataset.push_to_hub(tags=cfg.dataset.tags, private=cfg.dataset.private)
+
+        log_say("Exiting", cfg.play_sounds)
+    return dataset
+
+
+def main():
+    register_third_party_plugins()
+    record()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_replay.py b/lerobot/src/lerobot/scripts/lerobot_replay.py
new file mode 100644
index 0000000000000000000000000000000000000000..7c0b5b96b45355ed6e490e20d8d760a10c482e8c
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_replay.py
@@ -0,0 +1,141 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Replays the actions of an episode from a dataset on a robot.
+
+Examples:
+
+```shell
+lerobot-replay \
+    --robot.type=so100_follower \
+    --robot.port=/dev/tty.usbmodem58760431541 \
+    --robot.id=black \
+    --dataset.repo_id=<USER>/record-test \
+    --dataset.episode=0
+```
+
+Example replay with bimanual so100:
+```shell
+lerobot-replay \
+  --robot.type=bi_so_follower \
+  --robot.left_arm_port=/dev/tty.usbmodem5A460851411 \
+  --robot.right_arm_port=/dev/tty.usbmodem5A460812391 \
+  --robot.id=bimanual_follower \
+  --dataset.repo_id=${HF_USER}/bimanual-so100-handover-cube \
+  --dataset.episode=0
+```
+
+"""
+
+import logging
+import time
+from dataclasses import asdict, dataclass
+from pathlib import Path
+from pprint import pformat
+
+from lerobot.configs import parser
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.processor import (
+    make_default_robot_action_processor,
+)
+from lerobot.robots import (  # noqa: F401
+    Robot,
+    RobotConfig,
+    bi_openarm_follower,
+    bi_so_follower,
+    earthrover_mini_plus,
+    hope_jr,
+    koch_follower,
+    make_robot_from_config,
+    omx_follower,
+    openarm_follower,
+    reachy2,
+    so_follower,
+    unitree_g1,
+)
+from lerobot.utils.constants import ACTION
+from lerobot.utils.import_utils import register_third_party_plugins
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import (
+    init_logging,
+    log_say,
+)
+
+
+@dataclass
+class DatasetReplayConfig:
+    # Dataset identifier. By convention it should match '{hf_username}/{dataset_name}' (e.g. `lerobot/test`).
+    repo_id: str
+    # Episode to replay.
+    episode: int
+    # Root directory where the dataset will be stored (e.g. 'dataset/path'). If None, defaults to $HF_LEROBOT_HOME/repo_id.
+    root: str | Path | None = None
+    # Limit the frames per second. By default, uses the policy fps.
+    fps: int = 30
+
+
+@dataclass
+class ReplayConfig:
+    robot: RobotConfig
+    dataset: DatasetReplayConfig
+    # Use vocal synthesis to read events.
+    play_sounds: bool = True
+
+
+@parser.wrap()
+def replay(cfg: ReplayConfig):
+    init_logging()
+    logging.info(pformat(asdict(cfg)))
+
+    robot_action_processor = make_default_robot_action_processor()
+
+    robot = make_robot_from_config(cfg.robot)
+    dataset = LeRobotDataset(cfg.dataset.repo_id, root=cfg.dataset.root, episodes=[cfg.dataset.episode])
+
+    # Filter dataset to only include frames from the specified episode since episodes are chunked in dataset V3.0
+    episode_frames = dataset.hf_dataset.filter(lambda x: x["episode_index"] == cfg.dataset.episode)
+    actions = episode_frames.select_columns(ACTION)
+
+    robot.connect()
+
+    try:
+        log_say("Replaying episode", cfg.play_sounds, blocking=True)
+        for idx in range(len(episode_frames)):
+            start_episode_t = time.perf_counter()
+
+            action_array = actions[idx][ACTION]
+            action = {}
+            for i, name in enumerate(dataset.features[ACTION]["names"]):
+                action[name] = action_array[i]
+
+            robot_obs = robot.get_observation()
+
+            processed_action = robot_action_processor((action, robot_obs))
+
+            _ = robot.send_action(processed_action)
+
+            dt_s = time.perf_counter() - start_episode_t
+            precise_sleep(max(1 / dataset.fps - dt_s, 0.0))
+    finally:
+        robot.disconnect()
+
+
+def main():
+    register_third_party_plugins()
+    replay()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_setup_can.py b/lerobot/src/lerobot/scripts/lerobot_setup_can.py
new file mode 100644
index 0000000000000000000000000000000000000000..b28fca44d2c03e0e72388ad9303425450c5cbe03
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_setup_can.py
@@ -0,0 +1,361 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Setup and debug CAN interfaces for Damiao motors (e.g., OpenArms).
+
+Examples:
+
+Setup CAN interfaces with CAN FD:
+```shell
+lerobot-setup-can --mode=setup --interfaces=can0,can1,can2,can3
+```
+
+Test motors on a single interface:
+```shell
+lerobot-setup-can --mode=test --interfaces=can0
+```
+
+Test motors on all interfaces:
+```shell
+lerobot-setup-can --mode=test --interfaces=can0,can1,can2,can3
+```
+
+Speed test:
+```shell
+lerobot-setup-can --mode=speed --interfaces=can0
+```
+"""
+
+import subprocess
+import sys
+import time
+from dataclasses import dataclass, field
+
+import draccus
+
+from lerobot.utils.import_utils import _can_available
+
+MOTOR_NAMES = {
+    0x01: "joint_1",
+    0x02: "joint_2",
+    0x03: "joint_3",
+    0x04: "joint_4",
+    0x05: "joint_5",
+    0x06: "joint_6",
+    0x07: "joint_7",
+    0x08: "gripper",
+}
+
+
+@dataclass
+class CANSetupConfig:
+    mode: str = "test"
+    interfaces: str = "can0"  # Comma-separated, e.g. "can0,can1,can2,can3"
+    bitrate: int = 1000000
+    data_bitrate: int = 5000000
+    use_fd: bool = True
+    motor_ids: list[int] = field(default_factory=lambda: list(range(0x01, 0x09)))
+    timeout: float = 1.0
+    speed_iterations: int = 100
+
+    def get_interfaces(self) -> list[str]:
+        return [i.strip() for i in self.interfaces.split(",") if i.strip()]
+
+
+def check_interface_status(interface: str) -> tuple[bool, str, bool]:
+    """Check if CAN interface is UP and configured."""
+    try:
+        result = subprocess.run(["ip", "link", "show", interface], capture_output=True, text=True)  # nosec B607
+        if result.returncode != 0:
+            return False, "Interface not found", False
+
+        output = result.stdout
+        is_up = "UP" in output
+        is_fd = "fd on" in output.lower() or "canfd" in output.lower()
+        status = "UP" if is_up else "DOWN"
+        if is_fd:
+            status += " (CAN FD)"
+
+        return is_up, status, is_fd
+    except FileNotFoundError:
+        return False, "ip command not found", False
+
+
+def setup_interface(interface: str, bitrate: int, data_bitrate: int, use_fd: bool) -> bool:
+    """Configure a CAN interface."""
+    try:
+        subprocess.run(["sudo", "ip", "link", "set", interface, "down"], check=False, capture_output=True)  # nosec B607
+
+        cmd = ["sudo", "ip", "link", "set", interface, "type", "can", "bitrate", str(bitrate)]
+        if use_fd:
+            cmd.extend(["dbitrate", str(data_bitrate), "fd", "on"])
+
+        result = subprocess.run(cmd, capture_output=True, text=True)  # nosec B607
+        if result.returncode != 0:
+            print(f"  ✗ Failed to configure: {result.stderr}")
+            return False
+
+        result = subprocess.run(  # nosec B607
+            ["sudo", "ip", "link", "set", interface, "up"], capture_output=True, text=True
+        )
+        if result.returncode != 0:
+            print(f"  ✗ Failed to bring up: {result.stderr}")
+            return False
+
+        return True
+    except Exception as e:
+        print(f"  ✗ Error: {e}")
+        return False
+
+
+def test_motor(bus, motor_id: int, timeout: float, use_fd: bool):
+    """Test a single motor and return responses."""
+    import can
+
+    enable_msg = can.Message(
+        arbitration_id=motor_id,
+        data=[0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFC],
+        is_extended_id=False,
+        is_fd=use_fd,
+    )
+
+    try:
+        bus.send(enable_msg)
+    except Exception as e:
+        return None, f"Send error: {e}"
+
+    responses = []
+    start_time = time.time()
+
+    while time.time() - start_time < timeout:
+        msg = bus.recv(timeout=0.1)
+        if msg:
+            responses.append((msg.arbitration_id, msg.data.hex(), getattr(msg, "is_fd", False)))
+
+    disable_msg = can.Message(
+        arbitration_id=motor_id,
+        data=[0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFD],
+        is_extended_id=False,
+        is_fd=use_fd,
+    )
+    try:
+        bus.send(disable_msg)
+        bus.recv(timeout=0.1)  # Clear any pending responses
+    except Exception:
+        print(f"Error sending message to motor 0x{motor_id:02X}")
+
+    return responses, None
+
+
+def test_interface(cfg: CANSetupConfig, interface: str):
+    """Test all motors on a CAN interface."""
+    import can
+
+    is_up, status, _ = check_interface_status(interface)
+    print(f"\n{interface}: {status}")
+
+    if not is_up:
+        print(f"  ⚠ Interface is not UP. Run: lerobot-setup-can --mode=setup --interfaces {interface}")
+        return {}
+
+    try:
+        kwargs = {"channel": interface, "interface": "socketcan", "bitrate": cfg.bitrate}
+        if cfg.use_fd:
+            kwargs.update({"data_bitrate": cfg.data_bitrate, "fd": True})
+        bus = can.interface.Bus(**kwargs)
+    except Exception as e:
+        print(f"  ✗ Connection failed: {e}")
+        return {}
+
+    results = {}
+    try:
+        while bus.recv(timeout=0.01):
+            pass
+
+        for motor_id in cfg.motor_ids:
+            motor_name = MOTOR_NAMES.get(motor_id, f"motor_0x{motor_id:02X}")
+            responses, error = test_motor(bus, motor_id, cfg.timeout, cfg.use_fd)
+
+            if error:
+                print(f"  Motor 0x{motor_id:02X} ({motor_name}): ✗ {error}")
+                results[motor_id] = {"found": False, "error": error}
+            elif responses:
+                print(f"  Motor 0x{motor_id:02X} ({motor_name}): ✓ FOUND")
+                for resp_id, data, is_fd in responses:
+                    fd_flag = " [FD]" if is_fd else ""
+                    print(f"    → Response 0x{resp_id:02X}{fd_flag}: {data}")
+                results[motor_id] = {"found": True, "responses": responses}
+            else:
+                print(f"  Motor 0x{motor_id:02X} ({motor_name}): ✗ No response")
+                results[motor_id] = {"found": False}
+
+            time.sleep(0.05)
+    finally:
+        bus.shutdown()
+
+    found = sum(1 for r in results.values() if r.get("found"))
+    print(f"\n  Summary: {found}/{len(cfg.motor_ids)} motors found")
+    return results
+
+
+def speed_test(cfg: CANSetupConfig, interface: str):
+    """Test communication speed with motors."""
+    import can
+
+    is_up, status, _ = check_interface_status(interface)
+    if not is_up:
+        print(f"{interface}: {status} - skipping")
+        return
+
+    print(f"\n{interface}: Running speed test ({cfg.speed_iterations} iterations)...")
+
+    try:
+        kwargs = {"channel": interface, "interface": "socketcan", "bitrate": cfg.bitrate}
+        if cfg.use_fd:
+            kwargs.update({"data_bitrate": cfg.data_bitrate, "fd": True})
+        bus = can.interface.Bus(**kwargs)
+    except Exception as e:
+        print(f"  ✗ Connection failed: {e}")
+        return
+
+    responding_motor = None
+    for motor_id in cfg.motor_ids:
+        responses, _ = test_motor(bus, motor_id, 0.5, cfg.use_fd)
+        if responses:
+            responding_motor = motor_id
+            break
+
+    if not responding_motor:
+        print("  ✗ No responding motors found")
+        bus.shutdown()
+        return
+
+    print(f"  Testing with motor 0x{responding_motor:02X}...")
+    latencies = []
+
+    for _ in range(cfg.speed_iterations):
+        start = time.perf_counter()
+        msg = can.Message(
+            arbitration_id=responding_motor,
+            data=[0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFC],
+            is_extended_id=False,
+            is_fd=cfg.use_fd,
+        )
+        bus.send(msg)
+        resp = bus.recv(timeout=0.1)
+        if resp:
+            latencies.append((time.perf_counter() - start) * 1000)
+
+    bus.shutdown()
+
+    if latencies:
+        avg_latency = sum(latencies) / len(latencies)
+        hz = 1000.0 / avg_latency if avg_latency > 0 else 0
+        print(f"  ✓ Success rate: {len(latencies)}/{cfg.speed_iterations}")
+        print(f"  ✓ Avg latency: {avg_latency:.2f} ms")
+        print(f"  ✓ Max frequency: {hz:.1f} Hz")
+    else:
+        print("  ✗ No successful responses")
+
+
+def run_setup(cfg: CANSetupConfig):
+    """Setup CAN interfaces."""
+    print("=" * 50)
+    print("CAN Interface Setup")
+    print("=" * 50)
+    print(f"Mode: {'CAN FD' if cfg.use_fd else 'CAN 2.0'}")
+    print(f"Bitrate: {cfg.bitrate / 1_000_000:.1f} Mbps")
+    if cfg.use_fd:
+        print(f"Data bitrate: {cfg.data_bitrate / 1_000_000:.1f} Mbps")
+    print()
+
+    interfaces = cfg.get_interfaces()
+    for interface in interfaces:
+        print(f"Configuring {interface}...")
+        if setup_interface(interface, cfg.bitrate, cfg.data_bitrate, cfg.use_fd):
+            is_up, status, _ = check_interface_status(interface)
+            print(f"  ✓ {interface}: {status}")
+        else:
+            print(f"  ✗ {interface}: Failed")
+
+    print("\nSetup complete!")
+    print("\nNext: Test motors with:")
+    print(f"  lerobot-setup-can --mode=test --interfaces {','.join(interfaces)}")
+
+
+def run_test(cfg: CANSetupConfig):
+    """Test motors on CAN interfaces."""
+    print("=" * 50)
+    print("CAN Motor Test")
+    print("=" * 50)
+    print(f"Testing motors 0x{min(cfg.motor_ids):02X}-0x{max(cfg.motor_ids):02X}")
+    print(f"Mode: {'CAN FD' if cfg.use_fd else 'CAN 2.0'}")
+    print()
+
+    interfaces = cfg.get_interfaces()
+    all_results = {}
+    for interface in interfaces:
+        all_results[interface] = test_interface(cfg, interface)
+
+    total_found = sum(sum(1 for r in res.values() if r.get("found")) for res in all_results.values())
+
+    print("\n" + "=" * 50)
+    print("Summary")
+    print("=" * 50)
+    print(f"Total motors found: {total_found}")
+
+    if total_found == 0:
+        print("\n⚠ No motors found! Check:")
+        print("  1. Motors are powered (24V)")
+        print("  2. CAN wiring (CANH, CANL, GND)")
+        print("  3. Motor timeout parameter > 0 (use Damiao tools)")
+        print("  4. 120Ω termination at both cable ends")
+        print(f"  5. Interface configured: lerobot-setup-can --mode=setup --interfaces {interfaces[0]}")
+
+
+def run_speed(cfg: CANSetupConfig):
+    """Run speed tests on CAN interfaces."""
+    print("=" * 50)
+    print("CAN Speed Test")
+    print("=" * 50)
+
+    for interface in cfg.get_interfaces():
+        speed_test(cfg, interface)
+
+
+@draccus.wrap()
+def setup_can(cfg: CANSetupConfig):
+    if not _can_available:
+        print("Error: python-can not installed. Install with: pip install python-can")
+        sys.exit(1)
+
+    if cfg.mode == "setup":
+        run_setup(cfg)
+    elif cfg.mode == "test":
+        run_test(cfg)
+    elif cfg.mode == "speed":
+        run_speed(cfg)
+    else:
+        print(f"Unknown mode: {cfg.mode}")
+        print("Available modes: setup, test, speed")
+        sys.exit(1)
+
+
+def main():
+    setup_can()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_setup_motors.py b/lerobot/src/lerobot/scripts/lerobot_setup_motors.py
new file mode 100644
index 0000000000000000000000000000000000000000..2c962a6e25ce7a5db56d4be4d52d41192b87e802
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_setup_motors.py
@@ -0,0 +1,94 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Helper to set motor ids and baudrate.
+
+Example:
+
+```shell
+lerobot-setup-motors \
+    --teleop.type=so100_leader \
+    --teleop.port=/dev/tty.usbmodem575E0031751
+```
+"""
+
+from dataclasses import dataclass
+
+import draccus
+
+from lerobot.robots import (  # noqa: F401
+    RobotConfig,
+    bi_so_follower,
+    koch_follower,
+    lekiwi,
+    make_robot_from_config,
+    omx_follower,
+    so_follower,
+)
+from lerobot.teleoperators import (  # noqa: F401
+    TeleoperatorConfig,
+    bi_so_leader,
+    koch_leader,
+    make_teleoperator_from_config,
+    omx_leader,
+    openarm_mini,
+    so_leader,
+)
+
+COMPATIBLE_DEVICES = [
+    "koch_follower",
+    "koch_leader",
+    "omx_follower",
+    "omx_leader",
+    "openarm_mini",
+    "so100_follower",
+    "so100_leader",
+    "so101_follower",
+    "so101_leader",
+    "lekiwi",
+]
+
+
+@dataclass
+class SetupConfig:
+    teleop: TeleoperatorConfig | None = None
+    robot: RobotConfig | None = None
+
+    def __post_init__(self):
+        if bool(self.teleop) == bool(self.robot):
+            raise ValueError("Choose either a teleop or a robot.")
+
+        self.device = self.robot if self.robot else self.teleop
+
+
+@draccus.wrap()
+def setup_motors(cfg: SetupConfig):
+    if cfg.device.type not in COMPATIBLE_DEVICES:
+        raise NotImplementedError
+
+    if isinstance(cfg.device, RobotConfig):
+        device = make_robot_from_config(cfg.device)
+    else:
+        device = make_teleoperator_from_config(cfg.device)
+
+    device.setup_motors()
+
+
+def main():
+    setup_motors()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_teleoperate.py b/lerobot/src/lerobot/scripts/lerobot_teleoperate.py
new file mode 100644
index 0000000000000000000000000000000000000000..f050d572a8deb8fee17943d99dcf992417d5a59a
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_teleoperate.py
@@ -0,0 +1,254 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Simple script to control a robot from teleoperation.
+
+Example:
+
+```shell
+lerobot-teleoperate \
+    --robot.type=so101_follower \
+    --robot.port=/dev/tty.usbmodem58760431541 \
+    --robot.cameras="{ front: {type: opencv, index_or_path: 0, width: 1920, height: 1080, fps: 30}}" \
+    --robot.id=black \
+    --teleop.type=so101_leader \
+    --teleop.port=/dev/tty.usbmodem58760431551 \
+    --teleop.id=blue \
+    --display_data=true
+```
+
+Example teleoperation with bimanual so100:
+
+```shell
+lerobot-teleoperate \
+  --robot.type=bi_so_follower \
+  --robot.left_arm_config.port=/dev/tty.usbmodem5A460822851 \
+  --robot.right_arm_config.port=/dev/tty.usbmodem5A460814411 \
+  --robot.id=bimanual_follower \
+  --robot.left_arm_config.cameras='{
+    wrist: {"type": "opencv", "index_or_path": 1, "width": 640, "height": 480, "fps": 30},
+  }' --robot.right_arm_config.cameras='{
+    wrist: {"type": "opencv", "index_or_path": 2, "width": 640, "height": 480, "fps": 30},
+  }' \
+  --teleop.type=bi_so_leader \
+  --teleop.left_arm_config.port=/dev/tty.usbmodem5A460852721 \
+  --teleop.right_arm_config.port=/dev/tty.usbmodem5A460819811 \
+  --teleop.id=bimanual_leader \
+  --display_data=true
+```
+
+"""
+
+import logging
+import time
+from dataclasses import asdict, dataclass
+from pprint import pformat
+
+import rerun as rr
+
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig  # noqa: F401
+from lerobot.cameras.realsense.configuration_realsense import RealSenseCameraConfig  # noqa: F401
+from lerobot.cameras.zmq.configuration_zmq import ZMQCameraConfig  # noqa: F401
+from lerobot.configs import parser
+from lerobot.processor import (
+    RobotAction,
+    RobotObservation,
+    RobotProcessorPipeline,
+    make_default_processors,
+)
+from lerobot.robots import (  # noqa: F401
+    Robot,
+    RobotConfig,
+    bi_openarm_follower,
+    bi_so_follower,
+    earthrover_mini_plus,
+    hope_jr,
+    koch_follower,
+    make_robot_from_config,
+    omx_follower,
+    openarm_follower,
+    reachy2,
+    so_follower,
+    unitree_g1 as unitree_g1_robot,
+)
+from lerobot.teleoperators import (  # noqa: F401
+    Teleoperator,
+    TeleoperatorConfig,
+    bi_openarm_leader,
+    bi_so_leader,
+    gamepad,
+    homunculus,
+    keyboard,
+    koch_leader,
+    make_teleoperator_from_config,
+    omx_leader,
+    openarm_leader,
+    openarm_mini,
+    reachy2_teleoperator,
+    so_leader,
+    unitree_g1,
+)
+from lerobot.utils.import_utils import register_third_party_plugins
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import init_logging, move_cursor_up
+from lerobot.utils.visualization_utils import init_rerun, log_rerun_data
+
+
+@dataclass
+class TeleoperateConfig:
+    # TODO: pepijn, steven: if more robots require multiple teleoperators (like lekiwi) its good to make this possibele in teleop.py and record.py with List[Teleoperator]
+    teleop: TeleoperatorConfig
+    robot: RobotConfig
+    # Limit the maximum frames per second.
+    fps: int = 60
+    teleop_time_s: float | None = None
+    # Display all cameras on screen
+    display_data: bool = False
+    # Display data on a remote Rerun server
+    display_ip: str | None = None
+    # Port of the remote Rerun server
+    display_port: int | None = None
+    # Whether to  display compressed images in Rerun
+    display_compressed_images: bool = False
+
+
+def teleop_loop(
+    teleop: Teleoperator,
+    robot: Robot,
+    fps: int,
+    teleop_action_processor: RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction],
+    robot_action_processor: RobotProcessorPipeline[tuple[RobotAction, RobotObservation], RobotAction],
+    robot_observation_processor: RobotProcessorPipeline[RobotObservation, RobotObservation],
+    display_data: bool = False,
+    duration: float | None = None,
+    display_compressed_images: bool = False,
+):
+    """
+    This function continuously reads actions from a teleoperation device, processes them through optional
+    pipelines, sends them to a robot, and optionally displays the robot's state. The loop runs at a
+    specified frequency until a set duration is reached or it is manually interrupted.
+
+    Args:
+        teleop: The teleoperator device instance providing control actions.
+        robot: The robot instance being controlled.
+        fps: The target frequency for the control loop in frames per second.
+        display_data: If True, fetches robot observations and displays them in the console and Rerun.
+        display_compressed_images: If True, compresses images before sending them to Rerun for display.
+        duration: The maximum duration of the teleoperation loop in seconds. If None, the loop runs indefinitely.
+        teleop_action_processor: An optional pipeline to process raw actions from the teleoperator.
+        robot_action_processor: An optional pipeline to process actions before they are sent to the robot.
+        robot_observation_processor: An optional pipeline to process raw observations from the robot.
+    """
+
+    display_len = max(len(key) for key in robot.action_features)
+    start = time.perf_counter()
+    while True:
+        loop_start = time.perf_counter()
+
+        # Get robot observation
+        # Not really needed for now other than for visualization
+        # teleop_action_processor can take None as an observation
+        # given that it is the identity processor as default
+        obs = robot.get_observation()
+
+        if robot.name == "unitree_g1":
+            teleop.send_feedback(obs)
+
+        # Get teleop action
+        raw_action = teleop.get_action()
+
+        # Process teleop action through pipeline
+        teleop_action = teleop_action_processor((raw_action, obs))
+
+        # Process action for robot through pipeline
+        robot_action_to_send = robot_action_processor((teleop_action, obs))
+
+        # Send processed action to robot (robot_action_processor.to_output should return RobotAction)
+        _ = robot.send_action(robot_action_to_send)
+
+        if display_data:
+            # Process robot observation through pipeline
+            obs_transition = robot_observation_processor(obs)
+
+            log_rerun_data(
+                observation=obs_transition,
+                action=teleop_action,
+                compress_images=display_compressed_images,
+            )
+
+            print("\n" + "-" * (display_len + 10))
+            print(f"{'NAME':<{display_len}} | {'NORM':>7}")
+            # Display the final robot action that was sent
+            for motor, value in robot_action_to_send.items():
+                print(f"{motor:<{display_len}} | {value:>7.2f}")
+            move_cursor_up(len(robot_action_to_send) + 3)
+
+        dt_s = time.perf_counter() - loop_start
+        precise_sleep(max(1 / fps - dt_s, 0.0))
+        loop_s = time.perf_counter() - loop_start
+        print(f"Teleop loop time: {loop_s * 1e3:.2f}ms ({1 / loop_s:.0f} Hz)")
+        move_cursor_up(1)
+
+        if duration is not None and time.perf_counter() - start >= duration:
+            return
+
+
+@parser.wrap()
+def teleoperate(cfg: TeleoperateConfig):
+    init_logging()
+    logging.info(pformat(asdict(cfg)))
+    if cfg.display_data:
+        init_rerun(session_name="teleoperation", ip=cfg.display_ip, port=cfg.display_port)
+    display_compressed_images = (
+        True
+        if (cfg.display_data and cfg.display_ip is not None and cfg.display_port is not None)
+        else cfg.display_compressed_images
+    )
+
+    teleop = make_teleoperator_from_config(cfg.teleop)
+    robot = make_robot_from_config(cfg.robot)
+    teleop_action_processor, robot_action_processor, robot_observation_processor = make_default_processors()
+
+    teleop.connect()
+    robot.connect()
+
+    try:
+        teleop_loop(
+            teleop=teleop,
+            robot=robot,
+            fps=cfg.fps,
+            display_data=cfg.display_data,
+            duration=cfg.teleop_time_s,
+            teleop_action_processor=teleop_action_processor,
+            robot_action_processor=robot_action_processor,
+            robot_observation_processor=robot_observation_processor,
+            display_compressed_images=display_compressed_images,
+        )
+    except KeyboardInterrupt:
+        pass
+    finally:
+        if cfg.display_data:
+            rr.rerun_shutdown()
+        teleop.disconnect()
+        robot.disconnect()
+
+
+def main():
+    register_third_party_plugins()
+    teleoperate()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_train.py b/lerobot/src/lerobot/scripts/lerobot_train.py
new file mode 100644
index 0000000000000000000000000000000000000000..916e2a5e03ac598f296cf9d12c6afad0e4716054
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_train.py
@@ -0,0 +1,575 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import dataclasses
+import logging
+import shutil
+import time
+from contextlib import nullcontext
+from pprint import pformat
+from typing import Any
+
+import torch
+from accelerate import Accelerator
+from termcolor import colored
+from torch.optim import Optimizer
+from tqdm import tqdm
+
+from lerobot.configs import parser
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.datasets.factory import make_dataset
+from lerobot.datasets.sampler import EpisodeAwareSampler
+from lerobot.datasets.utils import cycle
+from lerobot.envs.factory import make_env, make_env_pre_post_processors
+from lerobot.envs.utils import close_envs
+from lerobot.optim.factory import make_optimizer_and_scheduler
+from lerobot.policies.factory import make_policy, make_pre_post_processors
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.rl.wandb_utils import WandBLogger
+from lerobot.scripts.lerobot_eval import eval_policy_all
+from lerobot.utils.import_utils import register_third_party_plugins
+from lerobot.utils.logging_utils import AverageMeter, MetricsTracker
+from lerobot.utils.random_utils import set_seed
+from lerobot.utils.train_utils import (
+    get_step_checkpoint_dir,
+    get_step_identifier,
+    load_training_state,
+    save_checkpoint,
+    update_last_checkpoint,
+)
+from lerobot.utils.utils import (
+    format_big_number,
+    has_method,
+    init_logging,
+    inside_slurm,
+)
+
+
+def update_policy(
+    train_metrics: MetricsTracker,
+    policy: PreTrainedPolicy,
+    batch: Any,
+    optimizer: Optimizer,
+    grad_clip_norm: float,
+    accelerator: Accelerator,
+    lr_scheduler=None,
+    lock=None,
+    rabc_weights_provider=None,
+) -> tuple[MetricsTracker, dict]:
+    """
+    Performs a single training step to update the policy's weights.
+
+    This function executes the forward and backward passes, clips gradients, and steps the optimizer and
+    learning rate scheduler. Accelerator handles mixed-precision training automatically.
+
+    Args:
+        train_metrics: A MetricsTracker instance to record training statistics.
+        policy: The policy model to be trained.
+        batch: A batch of training data.
+        optimizer: The optimizer used to update the policy's parameters.
+        grad_clip_norm: The maximum norm for gradient clipping.
+        accelerator: The Accelerator instance for distributed training and mixed precision.
+        lr_scheduler: An optional learning rate scheduler.
+        lock: An optional lock for thread-safe optimizer updates.
+        rabc_weights_provider: Optional RABCWeights instance for sample weighting.
+
+    Returns:
+        A tuple containing:
+        - The updated MetricsTracker with new statistics for this step.
+        - A dictionary of outputs from the policy's forward pass, for logging purposes.
+    """
+    start_time = time.perf_counter()
+    policy.train()
+
+    # Get RA-BC weights if enabled
+    rabc_batch_weights = None
+    rabc_batch_stats = None
+    if rabc_weights_provider is not None:
+        rabc_batch_weights, rabc_batch_stats = rabc_weights_provider.compute_batch_weights(batch)
+
+    # Let accelerator handle mixed precision
+    with accelerator.autocast():
+        # Use per-sample loss when RA-BC is enabled for proper weighting
+        if rabc_batch_weights is not None:
+            # Get per-sample losses
+            per_sample_loss, output_dict = policy.forward(batch, reduction="none")
+
+            # Apply RA-BC weights: L_RA-BC = Σ(w_i * l_i) / (Σw_i + ε)
+            # rabc_batch_weights is already normalized to sum to batch_size
+            epsilon = 1e-6
+            loss = (per_sample_loss * rabc_batch_weights).sum() / (rabc_batch_weights.sum() + epsilon)
+            # Log raw mean weight (before normalization) - this is the meaningful metric
+            output_dict["rabc_mean_weight"] = rabc_batch_stats["raw_mean_weight"]
+            output_dict["rabc_num_zero_weight"] = rabc_batch_stats["num_zero_weight"]
+            output_dict["rabc_num_full_weight"] = rabc_batch_stats["num_full_weight"]
+        else:
+            loss, output_dict = policy.forward(batch)
+
+        # TODO(rcadene): policy.unnormalize_outputs(out_dict)
+
+    # Use accelerator's backward method
+    accelerator.backward(loss)
+
+    # Clip gradients if specified
+    if grad_clip_norm > 0:
+        grad_norm = accelerator.clip_grad_norm_(policy.parameters(), grad_clip_norm)
+    else:
+        grad_norm = torch.nn.utils.clip_grad_norm_(
+            policy.parameters(), float("inf"), error_if_nonfinite=False
+        )
+
+    # Optimizer step
+    with lock if lock is not None else nullcontext():
+        optimizer.step()
+
+    optimizer.zero_grad()
+
+    # Step through pytorch scheduler at every batch instead of epoch
+    if lr_scheduler is not None:
+        lr_scheduler.step()
+
+    # Update internal buffers if policy has update method
+    if has_method(accelerator.unwrap_model(policy, keep_fp32_wrapper=True), "update"):
+        accelerator.unwrap_model(policy, keep_fp32_wrapper=True).update()
+
+    train_metrics.loss = loss.item()
+    train_metrics.grad_norm = grad_norm.item()
+    train_metrics.lr = optimizer.param_groups[0]["lr"]
+    train_metrics.update_s = time.perf_counter() - start_time
+    return train_metrics, output_dict
+
+
+@parser.wrap()
+def train(cfg: TrainPipelineConfig, accelerator: Accelerator | None = None):
+    """
+    Main function to train a policy.
+
+    This function orchestrates the entire training pipeline, including:
+    - Setting up logging, seeding, and device configuration.
+    - Creating the dataset, evaluation environment (if applicable), policy, and optimizer.
+    - Handling resumption from a checkpoint.
+    - Running the main training loop, which involves fetching data batches and calling `update_policy`.
+    - Periodically logging metrics, saving model checkpoints, and evaluating the policy.
+    - Pushing the final trained model to the Hugging Face Hub if configured.
+
+    Args:
+        cfg: A `TrainPipelineConfig` object containing all training configurations.
+        accelerator: Optional Accelerator instance. If None, one will be created automatically.
+    """
+    cfg.validate()
+
+    # Create Accelerator if not provided
+    # It will automatically detect if running in distributed mode or single-process mode
+    # We set step_scheduler_with_optimizer=False to prevent accelerate from adjusting the lr_scheduler steps based on the num_processes
+    # We set find_unused_parameters=True to handle models with conditional computation
+    if accelerator is None:
+        from accelerate.utils import DistributedDataParallelKwargs
+
+        ddp_kwargs = DistributedDataParallelKwargs(find_unused_parameters=True)
+        # Accelerate auto-detects the device based on the available hardware and ignores the policy.device setting.
+        # Force the device to be CPU when policy.device is set to CPU.
+        force_cpu = cfg.policy.device == "cpu"
+        accelerator = Accelerator(
+            step_scheduler_with_optimizer=False,
+            kwargs_handlers=[ddp_kwargs],
+            cpu=force_cpu,
+        )
+
+    init_logging(accelerator=accelerator)
+
+    # Determine if this is the main process (for logging and checkpointing)
+    # When using accelerate, only the main process should log to avoid duplicate outputs
+    is_main_process = accelerator.is_main_process
+
+    # Only log on main process
+    if is_main_process:
+        logging.info(pformat(cfg.to_dict()))
+
+    # Initialize wandb only on main process
+    if cfg.wandb.enable and cfg.wandb.project and is_main_process:
+        wandb_logger = WandBLogger(cfg)
+    else:
+        wandb_logger = None
+        if is_main_process:
+            logging.info(colored("Logs will be saved locally.", "yellow", attrs=["bold"]))
+
+    if cfg.seed is not None:
+        set_seed(cfg.seed, accelerator=accelerator)
+
+    # Use accelerator's device
+    device = accelerator.device
+    if cfg.cudnn_deterministic:
+        torch.backends.cudnn.deterministic = True
+        torch.backends.cudnn.benchmark = False
+    else:
+        torch.backends.cudnn.benchmark = True
+    torch.backends.cuda.matmul.allow_tf32 = True
+
+    # Dataset loading synchronization: main process downloads first to avoid race conditions
+    if is_main_process:
+        logging.info("Creating dataset")
+        dataset = make_dataset(cfg)
+
+    accelerator.wait_for_everyone()
+
+    # Now all other processes can safely load the dataset
+    if not is_main_process:
+        dataset = make_dataset(cfg)
+
+    # Create environment used for evaluating checkpoints during training on simulation data.
+    # On real-world data, no need to create an environment as evaluations are done outside train.py,
+    # using the eval.py instead, with gym_dora environment and dora-rs.
+    eval_env = None
+    if cfg.eval_freq > 0 and cfg.env is not None and is_main_process:
+        logging.info("Creating env")
+        eval_env = make_env(cfg.env, n_envs=cfg.eval.batch_size, use_async_envs=cfg.eval.use_async_envs)
+
+    if is_main_process:
+        logging.info("Creating policy")
+    policy = make_policy(
+        cfg=cfg.policy,
+        ds_meta=dataset.meta,
+        rename_map=cfg.rename_map,
+    )
+
+    if cfg.peft is not None:
+        logging.info("Using PEFT! Wrapping model.")
+        # Convert CLI peft config to dict for overrides
+        peft_cli_overrides = dataclasses.asdict(cfg.peft)
+        policy = policy.wrap_with_peft(peft_cli_overrides=peft_cli_overrides)
+
+    # Wait for all processes to finish policy creation before continuing
+    accelerator.wait_for_everyone()
+
+    # Create processors - only provide dataset_stats if not resuming from saved processors
+    processor_kwargs = {}
+    postprocessor_kwargs = {}
+    if (cfg.policy.pretrained_path and not cfg.resume) or not cfg.policy.pretrained_path:
+        # Only provide dataset_stats when not resuming from saved processor state
+        processor_kwargs["dataset_stats"] = dataset.meta.stats
+
+    # For SARM, always provide dataset_meta for progress normalization
+    if cfg.policy.type == "sarm":
+        processor_kwargs["dataset_meta"] = dataset.meta
+
+    if cfg.policy.pretrained_path is not None:
+        processor_kwargs["preprocessor_overrides"] = {
+            "device_processor": {"device": device.type},
+            "normalizer_processor": {
+                "stats": dataset.meta.stats,
+                "features": {**policy.config.input_features, **policy.config.output_features},
+                "norm_map": policy.config.normalization_mapping,
+            },
+        }
+        processor_kwargs["preprocessor_overrides"]["rename_observations_processor"] = {
+            "rename_map": cfg.rename_map
+        }
+        postprocessor_kwargs["postprocessor_overrides"] = {
+            "unnormalizer_processor": {
+                "stats": dataset.meta.stats,
+                "features": policy.config.output_features,
+                "norm_map": policy.config.normalization_mapping,
+            },
+        }
+
+    preprocessor, postprocessor = make_pre_post_processors(
+        policy_cfg=cfg.policy,
+        pretrained_path=cfg.policy.pretrained_path,
+        **processor_kwargs,
+        **postprocessor_kwargs,
+    )
+
+    if is_main_process:
+        logging.info("Creating optimizer and scheduler")
+    optimizer, lr_scheduler = make_optimizer_and_scheduler(cfg, policy)
+
+    # Load precomputed SARM progress for RA-BC if enabled
+    # Generate progress using: src/lerobot/policies/sarm/compute_rabc_weights.py
+    rabc_weights = None
+    if cfg.use_rabc:
+        from lerobot.utils.rabc import RABCWeights
+
+        # Get chunk_size from policy config
+        chunk_size = getattr(policy.config, "chunk_size", None)
+        if chunk_size is None:
+            raise ValueError("Chunk size is not found in policy config")
+
+        head_mode = getattr(cfg, "rabc_head_mode", "sparse")
+        logging.info(f"Loading SARM progress for RA-BC from {cfg.rabc_progress_path}")
+        logging.info(f"Using chunk_size={chunk_size} from policy config, head_mode={head_mode}")
+        rabc_weights = RABCWeights(
+            progress_path=cfg.rabc_progress_path,
+            chunk_size=chunk_size,
+            head_mode=head_mode,
+            kappa=getattr(cfg, "rabc_kappa", 0.01),
+            epsilon=getattr(cfg, "rabc_epsilon", 1e-6),
+            device=device,
+        )
+
+    step = 0  # number of policy updates (forward + backward + optim)
+
+    if cfg.resume:
+        step, optimizer, lr_scheduler = load_training_state(cfg.checkpoint_path, optimizer, lr_scheduler)
+
+    num_learnable_params = sum(p.numel() for p in policy.parameters() if p.requires_grad)
+    num_total_params = sum(p.numel() for p in policy.parameters())
+
+    if is_main_process:
+        logging.info(colored("Output dir:", "yellow", attrs=["bold"]) + f" {cfg.output_dir}")
+        if cfg.env is not None:
+            logging.info(f"{cfg.env.task=}")
+            logging.info("Creating environment processors")
+            env_preprocessor, env_postprocessor = make_env_pre_post_processors(
+                env_cfg=cfg.env, policy_cfg=cfg.policy
+            )
+        logging.info(f"{cfg.steps=} ({format_big_number(cfg.steps)})")
+        logging.info(f"{dataset.num_frames=} ({format_big_number(dataset.num_frames)})")
+        logging.info(f"{dataset.num_episodes=}")
+        num_processes = accelerator.num_processes
+        effective_bs = cfg.batch_size * num_processes
+        logging.info(f"Effective batch size: {cfg.batch_size} x {num_processes} = {effective_bs}")
+        logging.info(f"{num_learnable_params=} ({format_big_number(num_learnable_params)})")
+        logging.info(f"{num_total_params=} ({format_big_number(num_total_params)})")
+
+    # create dataloader for offline training
+    if hasattr(cfg.policy, "drop_n_last_frames"):
+        shuffle = False
+        sampler = EpisodeAwareSampler(
+            dataset.meta.episodes["dataset_from_index"],
+            dataset.meta.episodes["dataset_to_index"],
+            episode_indices_to_use=dataset.episodes,
+            drop_n_last_frames=cfg.policy.drop_n_last_frames,
+            shuffle=True,
+        )
+    else:
+        shuffle = True
+        sampler = None
+
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        num_workers=cfg.num_workers,
+        batch_size=cfg.batch_size,
+        shuffle=shuffle and not cfg.dataset.streaming,
+        sampler=sampler,
+        pin_memory=device.type == "cuda",
+        drop_last=False,
+        prefetch_factor=2 if cfg.num_workers > 0 else None,
+    )
+
+    # Prepare everything with accelerator
+    accelerator.wait_for_everyone()
+    policy, optimizer, dataloader, lr_scheduler = accelerator.prepare(
+        policy, optimizer, dataloader, lr_scheduler
+    )
+    dl_iter = cycle(dataloader)
+
+    policy.train()
+
+    train_metrics = {
+        "loss": AverageMeter("loss", ":.3f"),
+        "grad_norm": AverageMeter("grdn", ":.3f"),
+        "lr": AverageMeter("lr", ":0.1e"),
+        "update_s": AverageMeter("updt_s", ":.3f"),
+        "dataloading_s": AverageMeter("data_s", ":.3f"),
+    }
+
+    # Keep global batch size for logging; MetricsTracker handles world size internally.
+    effective_batch_size = cfg.batch_size * accelerator.num_processes
+    train_tracker = MetricsTracker(
+        cfg.batch_size,
+        dataset.num_frames,
+        dataset.num_episodes,
+        train_metrics,
+        initial_step=step,
+        accelerator=accelerator,
+    )
+
+    if is_main_process:
+        progbar = tqdm(
+            total=cfg.steps - step,
+            desc="Training",
+            unit="step",
+            disable=inside_slurm(),
+            position=0,
+            leave=True,
+        )
+        logging.info(
+            f"Start offline training on a fixed dataset, with effective batch size: {effective_batch_size}"
+        )
+
+    prev_checkpoint_dir = None
+    for _ in range(step, cfg.steps):
+        start_time = time.perf_counter()
+        batch = next(dl_iter)
+        batch = preprocessor(batch)
+        train_tracker.dataloading_s = time.perf_counter() - start_time
+
+        train_tracker, output_dict = update_policy(
+            train_tracker,
+            policy,
+            batch,
+            optimizer,
+            cfg.optimizer.grad_clip_norm,
+            accelerator=accelerator,
+            lr_scheduler=lr_scheduler,
+            rabc_weights_provider=rabc_weights,
+        )
+
+        # Note: eval and checkpoint happens *after* the `step`th training update has completed, so we
+        # increment `step` here.
+        step += 1
+        if is_main_process:
+            progbar.update(1)
+        train_tracker.step()
+        is_log_step = cfg.log_freq > 0 and step % cfg.log_freq == 0 and is_main_process
+        is_saving_step = step % cfg.save_freq == 0 or step == cfg.steps
+        is_eval_step = cfg.eval_freq > 0 and step % cfg.eval_freq == 0
+
+        if is_log_step:
+            logging.info(train_tracker)
+            if wandb_logger:
+                wandb_log_dict = train_tracker.to_dict()
+                if output_dict:
+                    wandb_log_dict.update(output_dict)
+                # Log RA-BC statistics if enabled
+                if rabc_weights is not None:
+                    rabc_stats = rabc_weights.get_stats()
+                    wandb_log_dict.update(
+                        {
+                            "rabc_delta_mean": rabc_stats["delta_mean"],
+                            "rabc_delta_std": rabc_stats["delta_std"],
+                            "rabc_num_frames": rabc_stats["num_frames"],
+                        }
+                    )
+                wandb_logger.log_dict(wandb_log_dict, step)
+            train_tracker.reset_averages()
+
+        if cfg.save_checkpoint and is_saving_step:
+            if is_main_process:
+                logging.info(f"Checkpoint policy after step {step}")
+                checkpoint_dir = get_step_checkpoint_dir(cfg.output_dir, cfg.steps, step)
+                save_checkpoint(
+                    checkpoint_dir=checkpoint_dir,
+                    step=step,
+                    cfg=cfg,
+                    policy=accelerator.unwrap_model(policy),
+                    optimizer=optimizer,
+                    scheduler=lr_scheduler,
+                    preprocessor=preprocessor,
+                    postprocessor=postprocessor,
+                )
+                update_last_checkpoint(checkpoint_dir)
+                # Upload checkpoint to HF then delete previous local copy
+                uploaded = False
+                if cfg.policy.push_to_hub:
+                    try:
+                        from huggingface_hub import HfApi
+                        api = HfApi()
+                        api.upload_folder(
+                            folder_path=checkpoint_dir,
+                            repo_id=cfg.policy.repo_id,
+                            path_in_repo=f"checkpoints/step_{step:06d}",
+                        )
+                        logging.info(f"Uploaded checkpoint step {step} to HF")
+                        uploaded = True
+                    except Exception as e:
+                        logging.warning(f"Failed to upload checkpoint step {step}: {e}")
+                if uploaded and prev_checkpoint_dir is not None and Path(prev_checkpoint_dir).exists():
+                    shutil.rmtree(prev_checkpoint_dir)
+                prev_checkpoint_dir = checkpoint_dir
+                if wandb_logger:
+                    wandb_logger.log_policy(checkpoint_dir)
+
+            accelerator.wait_for_everyone()
+
+        if cfg.env and is_eval_step:
+            if is_main_process:
+                step_id = get_step_identifier(step, cfg.steps)
+                logging.info(f"Eval policy at step {step}")
+                with torch.no_grad(), accelerator.autocast():
+                    eval_info = eval_policy_all(
+                        envs=eval_env,  # dict[suite][task_id] -> vec_env
+                        policy=accelerator.unwrap_model(policy),
+                        env_preprocessor=env_preprocessor,
+                        env_postprocessor=env_postprocessor,
+                        preprocessor=preprocessor,
+                        postprocessor=postprocessor,
+                        n_episodes=cfg.eval.n_episodes,
+                        videos_dir=cfg.output_dir / "eval" / f"videos_step_{step_id}",
+                        max_episodes_rendered=4,
+                        start_seed=cfg.seed,
+                        max_parallel_tasks=cfg.env.max_parallel_tasks,
+                    )
+                # overall metrics (suite-agnostic)
+                aggregated = eval_info["overall"]
+
+                # optional: per-suite logging
+                for suite, suite_info in eval_info.items():
+                    logging.info("Suite %s aggregated: %s", suite, suite_info)
+
+                # meters/tracker
+                eval_metrics = {
+                    "avg_sum_reward": AverageMeter("∑rwrd", ":.3f"),
+                    "pc_success": AverageMeter("success", ":.1f"),
+                    "eval_s": AverageMeter("eval_s", ":.3f"),
+                }
+                eval_tracker = MetricsTracker(
+                    cfg.batch_size,
+                    dataset.num_frames,
+                    dataset.num_episodes,
+                    eval_metrics,
+                    initial_step=step,
+                    accelerator=accelerator,
+                )
+                eval_tracker.eval_s = aggregated.pop("eval_s")
+                eval_tracker.avg_sum_reward = aggregated.pop("avg_sum_reward")
+                eval_tracker.pc_success = aggregated.pop("pc_success")
+                if wandb_logger:
+                    wandb_log_dict = {**eval_tracker.to_dict(), **eval_info}
+                    wandb_logger.log_dict(wandb_log_dict, step, mode="eval")
+                    wandb_logger.log_video(eval_info["overall"]["video_paths"][0], step, mode="eval")
+
+            accelerator.wait_for_everyone()
+
+    if is_main_process:
+        progbar.close()
+
+    if eval_env:
+        close_envs(eval_env)
+
+    if is_main_process:
+        logging.info("End of training")
+
+        if cfg.policy.push_to_hub:
+            unwrapped_policy = accelerator.unwrap_model(policy)
+            if cfg.policy.use_peft:
+                unwrapped_policy.push_model_to_hub(cfg, peft_model=unwrapped_policy)
+            else:
+                unwrapped_policy.push_model_to_hub(cfg)
+            preprocessor.push_to_hub(cfg.policy.repo_id)
+            postprocessor.push_to_hub(cfg.policy.repo_id)
+
+    # Properly clean up the distributed process group
+    accelerator.wait_for_everyone()
+    accelerator.end_training()
+
+
+def main():
+    register_third_party_plugins()
+    train()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/scripts/lerobot_train_tokenizer.py b/lerobot/src/lerobot/scripts/lerobot_train_tokenizer.py
new file mode 100644
index 0000000000000000000000000000000000000000..807d483338334c66be8a74830a6d7f1d36605236
--- /dev/null
+++ b/lerobot/src/lerobot/scripts/lerobot_train_tokenizer.py
@@ -0,0 +1,604 @@
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Train FAST tokenizer for action encoding.
+
+This script:
+1. Loads action chunks from LeRobotDataset (with episode sampling)
+2. Optionally applies delta transforms (relative vs absolute actions)
+3. Extracts specified action dimensions for encoding
+4. Applies normalization (MEAN_STD, MIN_MAX, QUANTILES, or other modes)
+5. Trains FAST tokenizer (BPE on DCT coefficients) on the action chunks
+6. Saves tokenizer to output directory
+7. Optionally pushes tokenizer to Hugging Face Hub
+8. Reports compression statistics
+
+Example:
+
+```shell
+lerobot-train-tokenizer \
+    --repo_id=user/dataset_name \
+    --action_horizon=10 \
+    --max_episodes=100 \
+    --sample_fraction=0.1 \
+    --encoded_dims="0:6" \
+    --delta_dims="0,1,2,3,4,5" \
+    --use_delta_transform=true \
+    --state_key="observation.state" \
+    --normalization_mode="QUANTILES" \
+    --vocab_size=1024 \
+    --scale=10.0 \
+    --output_dir="./fast_tokenizer_dataset_name" \
+    --push_to_hub=true \
+    --hub_repo_id="user/fast_tokenizer_dataset_name" \
+    --hub_private=false
+"""
+
+import json
+from dataclasses import dataclass
+from pathlib import Path
+from typing import TYPE_CHECKING
+
+import numpy as np
+import torch
+from huggingface_hub import HfApi
+
+from lerobot.utils.import_utils import _transformers_available
+
+if TYPE_CHECKING or _transformers_available:
+    from transformers import AutoProcessor
+else:
+    AutoProcessor = None
+
+from lerobot.configs import parser
+from lerobot.configs.types import NormalizationMode
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.utils.constants import ACTION, OBS_STATE
+
+
+@dataclass
+class TokenizerTrainingConfig:
+    """Configuration for training FAST tokenizer."""
+
+    # LeRobot dataset repository ID
+    repo_id: str
+    # Root directory for dataset (default: ~/.cache/huggingface/lerobot)
+    root: str | None = None
+    # Number of future actions in each chunk
+    action_horizon: int = 10
+    # Max episodes to use (None = all episodes in dataset)
+    max_episodes: int | None = None
+    # Fraction of chunks to sample per episode
+    sample_fraction: float = 0.1
+    # Comma-separated dimension ranges to encode (e.g., "0:6,7:23")
+    encoded_dims: str = "0:6,7:23"
+    # Comma-separated dimension indices for delta transform (e.g., "0,1,2,3,4,5")
+    delta_dims: str | None = None
+    # Whether to apply delta transform (relative actions vs absolute actions)
+    use_delta_transform: bool = False
+    # Dataset key for state observations (default: "observation.state")
+    state_key: str = OBS_STATE
+    # Normalization mode (MEAN_STD, MIN_MAX, QUANTILES, QUANTILE10, IDENTITY)
+    normalization_mode: str = "QUANTILES"
+    # FAST vocabulary size (BPE vocab size)
+    vocab_size: int = 1024
+    # DCT scaling factor (default: 10.0)
+    scale: float = 10.0
+    # Directory to save tokenizer (default: ./fast_tokenizer_{repo_id})
+    output_dir: str | None = None
+    # Whether to push the tokenizer to Hugging Face Hub
+    push_to_hub: bool = False
+    # Hub repository ID (e.g., "username/tokenizer-name"). If None, uses output_dir name
+    hub_repo_id: str | None = None
+    # Whether to create a private repository on the Hub
+    hub_private: bool = False
+
+
+def apply_delta_transform(state: np.ndarray, actions: np.ndarray, delta_dims: list[int] | None) -> np.ndarray:
+    """Apply delta transform to specified dimensions.
+
+    Args:
+        state: Current state [D]
+        actions: Future actions [D]
+        delta_dims: List of dimension indices to apply delta transform to
+
+    Returns:
+        Transformed actions [D]
+    """
+    if delta_dims is None or len(delta_dims) == 0:
+        return actions
+
+    delta_actions = actions.copy()
+    for dim in delta_dims:
+        delta_actions[dim] = actions[dim] - state[dim]
+
+    return delta_actions
+
+
+def apply_normalization(
+    data: np.ndarray,
+    stats: dict[str, np.ndarray],
+    mode: NormalizationMode,
+    eps: float = 1e-8,
+) -> np.ndarray:
+    """Apply normalization to data based on the specified mode.
+
+    Args:
+        data: Data to normalize [N, H, D] or [D]
+        stats: Dictionary of statistics (mean, std, min, max, q01, q99, q10, q90)
+        mode: Normalization mode to apply
+        eps: Small epsilon for numerical stability
+
+    Returns:
+        Normalized data with the same shape as input
+    """
+    if mode == NormalizationMode.IDENTITY:
+        return data
+
+    if mode == NormalizationMode.MEAN_STD:
+        mean = stats.get("mean")
+        std = stats.get("std")
+        if mean is None or std is None:
+            raise ValueError("MEAN_STD mode requires 'mean' and 'std' in stats")
+        return (data - mean) / np.maximum(std, eps)
+
+    if mode == NormalizationMode.MIN_MAX:
+        min_val = stats.get("min")
+        max_val = stats.get("max")
+        if min_val is None or max_val is None:
+            raise ValueError("MIN_MAX mode requires 'min' and 'max' in stats")
+        denom = np.maximum(max_val - min_val, eps)
+        return 2.0 * (data - min_val) / denom - 1.0
+
+    if mode == NormalizationMode.QUANTILES:
+        q01 = stats.get("q01")
+        q99 = stats.get("q99")
+        if q01 is None or q99 is None:
+            raise ValueError("QUANTILES mode requires 'q01' and 'q99' in stats")
+        denom = np.maximum(q99 - q01, eps)
+        # Clip to quantile range then normalize to [-1, 1]
+        clipped = np.clip(data, q01, q99)
+        return 2.0 * (clipped - q01) / denom - 1.0
+
+    if mode == NormalizationMode.QUANTILE10:
+        q10 = stats.get("q10")
+        q90 = stats.get("q90")
+        if q10 is None or q90 is None:
+            raise ValueError("QUANTILE10 mode requires 'q10' and 'q90' in stats")
+        denom = np.maximum(q90 - q10, eps)
+        # Clip to quantile range then normalize to [-1, 1]
+        clipped = np.clip(data, q10, q90)
+        return 2.0 * (clipped - q10) / denom - 1.0
+
+    raise ValueError(f"Unsupported normalization mode: {mode}")
+
+
+def process_episode(args):
+    """Process single episode and return action chunks."""
+    dataset, ep_idx, action_horizon, delta_dims, sample_fraction, state_key, use_delta_transform = args
+
+    try:
+        # get episode info
+        ep_info = dataset.meta.episodes[ep_idx]
+        from_idx = ep_info["dataset_from_index"]
+        to_idx = ep_info["dataset_to_index"]
+        ep_length = to_idx - from_idx
+
+        if ep_length < action_horizon:
+            return None
+
+        # load all frames in episode
+        # if dataset has episode filtering, we need to use the mapping
+        states = []
+        actions = []
+
+        for abs_idx in range(from_idx, to_idx):
+            # map absolute index to relative index if needed
+            if dataset._absolute_to_relative_idx is not None:
+                if abs_idx not in dataset._absolute_to_relative_idx:
+                    # this episode's frames aren't in the filtered dataset
+                    return None
+                rel_idx = dataset._absolute_to_relative_idx[abs_idx]
+            else:
+                rel_idx = abs_idx
+
+            frame = dataset.hf_dataset[rel_idx]
+
+            # get state (could be from observation.state or other state key)
+            if state_key in frame:
+                state = (
+                    frame[state_key].numpy()
+                    if torch.is_tensor(frame[state_key])
+                    else np.array(frame[state_key])
+                )
+            else:
+                # if no state key, use zeros (no delta transform)
+                state = np.zeros_like(
+                    frame[ACTION].numpy() if torch.is_tensor(frame[ACTION]) else np.array(frame[ACTION])
+                )
+
+            action = frame[ACTION].numpy() if torch.is_tensor(frame[ACTION]) else np.array(frame[ACTION])
+
+            states.append(state)
+            actions.append(action)
+
+        states = np.array(states)
+        actions = np.array(actions)
+
+        # create action chunks (sliding window)
+        # all actions in a chunk are relative to the FIRST state in that chunk
+        action_chunks = []
+
+        for i in range(len(states) - action_horizon + 1):
+            current_state = states[i]  # First state in chunk
+            future_absolute_actions = actions[i : i + action_horizon]
+
+            if use_delta_transform:
+                # relative actions
+                delta_chunk = np.zeros_like(future_absolute_actions)
+                for t in range(action_horizon):
+                    delta_chunk[t] = apply_delta_transform(
+                        current_state,
+                        future_absolute_actions[t],
+                        delta_dims,
+                    )
+                action_chunks.append(delta_chunk)
+            else:
+                # absolute actions (no delta)
+                action_chunks.append(future_absolute_actions)
+
+        if len(action_chunks) == 0:
+            return None
+
+        action_chunks = np.array(action_chunks)
+
+        # sample chunks
+        if sample_fraction < 1.0:
+            n_chunks = len(action_chunks)
+            n_samples = max(1, int(n_chunks * sample_fraction))
+            episode_seed = hash(ep_idx) % (2**31)
+            rng = np.random.RandomState(episode_seed)
+            indices = rng.choice(n_chunks, size=n_samples, replace=False)
+            action_chunks = action_chunks[indices]
+
+        return action_chunks
+
+    except Exception as e:
+        print(f"Error processing episode {ep_idx}: {e}")
+        import traceback
+
+        traceback.print_exc()
+        return None
+
+
+def train_fast_tokenizer(
+    action_chunks: np.ndarray,
+    vocab_size: int = 1024,
+    scale: float = 10.0,
+) -> AutoProcessor:
+    """
+    Train FAST tokenizer (BPE on DCT coefficients) on action chunks.
+
+    Uses the .fit() method to train a new tokenizer on the provided data.
+
+    Args:
+        action_chunks: Array of action chunks [N, H, D] where N=num_chunks, H=horizon, D=action_dim
+        vocab_size: BPE vocabulary size
+        scale: DCT scaling factor for quantization
+
+    Returns:
+        Trained FAST tokenizer
+    """
+    print(f"Training FAST tokenizer on {len(action_chunks)} action chunks...")
+    print(f"Action chunk shape: {action_chunks.shape}")
+    print(f"Vocab size: {vocab_size}")
+    print(f"DCT scale: {scale}")
+
+    # download the tokenizer source code (not pretrained weights)
+    # we'll train a new tokenizer on our own data
+    base_tokenizer = AutoProcessor.from_pretrained("lerobot/fast-action-tokenizer", trust_remote_code=True)
+
+    # convert action_chunks array to list of arrays (expected by .fit())
+    action_data_list = [action_chunks[i] for i in range(len(action_chunks))]
+
+    # train the new tokenizer on our action data using .fit()
+    # this trains the BPE tokenizer on DCT coefficients
+    print("Training new tokenizer (this may take a few minutes)...")
+    tokenizer = base_tokenizer.fit(
+        action_data_list,
+        scale=scale,
+        vocab_size=vocab_size,
+        time_horizon=action_chunks.shape[1],  # action_horizon
+        action_dim=action_chunks.shape[2],  # encoded dimensions
+    )
+    print("✓ Tokenizer training complete!")
+
+    # validate it works
+    sample_chunk = action_chunks[0]
+    encoded = tokenizer(sample_chunk[None])[0]
+    if isinstance(encoded, list):
+        encoded = np.array(encoded)
+    print(f"Sample encoding: {len(encoded)} tokens for chunk shape {sample_chunk.shape}")
+
+    return tokenizer
+
+
+def compute_compression_stats(tokenizer, action_chunks: np.ndarray):
+    """Compute compression statistics."""
+    print("\nComputing compression statistics...")
+
+    # sample for stats (use max 1000 chunks for speed)
+    sample_size = min(1000, len(action_chunks))
+    sample_indices = np.random.RandomState(42).choice(len(action_chunks), size=sample_size, replace=False)
+    sample_chunks = action_chunks[sample_indices]
+
+    token_lengths = []
+    for chunk in sample_chunks:
+        encoded = tokenizer(chunk[None])[0]
+        if isinstance(encoded, list):
+            token_lengths.append(len(encoded))
+        else:
+            token_lengths.append(encoded.shape[0] if hasattr(encoded, "shape") else len(encoded))
+
+    token_lengths = np.array(token_lengths)
+
+    # compression ratio: (H * D) / avg_tokens
+    input_size = action_chunks.shape[1] * action_chunks.shape[2]
+    avg_tokens = np.mean(token_lengths)
+    compression_ratio = input_size / avg_tokens
+
+    stats = {
+        "compression_ratio": float(compression_ratio),
+        "mean_token_length": float(np.mean(token_lengths)),
+        "p99_token_length": float(np.percentile(token_lengths, 99)),
+        "min_token_length": float(np.min(token_lengths)),
+        "max_token_length": float(np.max(token_lengths)),
+    }
+
+    print("Compression Statistics:")
+    print(f"  Average compression ratio: {stats['compression_ratio']:.2f}x")
+    print(f"  Mean token length: {stats['mean_token_length']:.1f}")
+    print(f"  P99 token length: {stats['p99_token_length']:.0f}")
+    print(f"  Min token length: {stats['min_token_length']:.0f}")
+    print(f"  Max token length: {stats['max_token_length']:.0f}")
+
+    return stats
+
+
+@parser.wrap()
+def train_tokenizer(cfg: TokenizerTrainingConfig):
+    """
+    Train FAST tokenizer for action encoding.
+
+    Args:
+        cfg: TokenizerTrainingConfig dataclass with all configuration parameters
+    """
+    # load dataset
+    print(f"Loading dataset: {cfg.repo_id}")
+    dataset = LeRobotDataset(repo_id=cfg.repo_id, root=cfg.root)
+    print(f"Dataset loaded: {dataset.num_episodes} episodes, {dataset.num_frames} frames")
+
+    # parse normalization mode
+    try:
+        norm_mode = NormalizationMode(cfg.normalization_mode)
+    except ValueError as err:
+        raise ValueError(
+            f"Invalid normalization_mode: {cfg.normalization_mode}. "
+            f"Must be one of: {', '.join([m.value for m in NormalizationMode])}"
+        ) from err
+    print(f"Normalization mode: {norm_mode.value}")
+
+    # parse encoded dimensions
+    encoded_dim_ranges = []
+    for range_str in cfg.encoded_dims.split(","):
+        start, end = map(int, range_str.strip().split(":"))
+        encoded_dim_ranges.append((start, end))
+
+    total_encoded_dims = sum(end - start for start, end in encoded_dim_ranges)
+    print(f"Encoding {total_encoded_dims} dimensions: {cfg.encoded_dims}")
+
+    # parse delta dimensions
+    delta_dim_list = None
+    if cfg.delta_dims is not None and cfg.delta_dims.strip():
+        delta_dim_list = [int(d.strip()) for d in cfg.delta_dims.split(",")]
+        print(f"Delta dimensions: {delta_dim_list}")
+    else:
+        print("No delta dimensions specified")
+
+    print(f"Use delta transform: {cfg.use_delta_transform}")
+    if cfg.use_delta_transform and (delta_dim_list is None or len(delta_dim_list) == 0):
+        print("Warning: use_delta_transform=True but no delta_dims specified. No delta will be applied.")
+
+    print(f"Action horizon: {cfg.action_horizon}")
+    print(f"State key: {cfg.state_key}")
+
+    # determine episodes to process
+    num_episodes = dataset.num_episodes
+    if cfg.max_episodes is not None:
+        num_episodes = min(cfg.max_episodes, num_episodes)
+
+    print(f"Processing {num_episodes} episodes...")
+
+    # process episodes sequentially (to avoid pickling issues with dataset)
+    all_chunks = []
+    for ep_idx in range(num_episodes):
+        if ep_idx % 10 == 0:
+            print(f"  Processing episode {ep_idx}/{num_episodes}...")
+
+        chunks = process_episode(
+            (
+                dataset,
+                ep_idx,
+                cfg.action_horizon,
+                delta_dim_list,
+                cfg.sample_fraction,
+                cfg.state_key,
+                cfg.use_delta_transform,
+            )
+        )
+        if chunks is not None:
+            all_chunks.append(chunks)
+
+    # concatenate all chunks
+    all_chunks = np.concatenate(all_chunks, axis=0)
+    print(f"Collected {len(all_chunks)} action chunks")
+
+    # extract only encoded dimensions FIRST (before normalization)
+    encoded_chunks = []
+    for start, end in encoded_dim_ranges:
+        encoded_chunks.append(all_chunks[:, :, start:end])
+    encoded_chunks = np.concatenate(encoded_chunks, axis=-1)  # [N, H, D_encoded]
+    print(f"Extracted {encoded_chunks.shape[-1]} encoded dimensions")
+
+    # apply normalization to encoded dimensions
+    print("\nBefore normalization - overall stats:")
+    print(f"  Min: {np.min(encoded_chunks):.4f}, Max: {np.max(encoded_chunks):.4f}")
+    print(f"  Mean: {np.mean(encoded_chunks):.4f}, Std: {np.std(encoded_chunks):.4f}")
+
+    # get normalization stats from dataset
+    norm_stats = dataset.meta.stats
+    if norm_stats is not None and ACTION in norm_stats:
+        action_stats = norm_stats[ACTION]
+
+        # build encoded dimension indices
+        encoded_dim_indices = []
+        for start, end in encoded_dim_ranges:
+            encoded_dim_indices.extend(range(start, end))
+        encoded_dim_indices = np.array(encoded_dim_indices)
+
+        # extract stats for encoded dimensions only
+        encoded_stats = {}
+        for stat_name, stat_values in action_stats.items():
+            if isinstance(stat_values, (list, np.ndarray)):
+                stat_array = np.array(stat_values)
+                if len(stat_array) > max(encoded_dim_indices):
+                    encoded_stats[stat_name] = stat_array[encoded_dim_indices]
+
+        if encoded_stats:
+            print(f"\nNormalization stats for encoded dimensions (mode: {norm_mode.value}):")
+            for stat_name, stat_values in encoded_stats.items():
+                print(
+                    f"  {stat_name}: shape={stat_values.shape}, "
+                    f"range=[{np.min(stat_values):.4f}, {np.max(stat_values):.4f}]"
+                )
+
+            # apply normalization based on mode
+            try:
+                encoded_chunks = apply_normalization(encoded_chunks, encoded_stats, norm_mode, eps=1e-8)
+                print(f"\nApplied {norm_mode.value} normalization")
+            except ValueError as e:
+                print(f"Warning: {e}. Using raw actions without normalization.")
+
+            print("\nAfter normalization - overall stats:")
+            print(f"  Min: {np.min(encoded_chunks):.4f}, Max: {np.max(encoded_chunks):.4f}")
+            print(f"  Mean: {np.mean(encoded_chunks):.4f}, Std: {np.std(encoded_chunks):.4f}")
+
+            print("\nPer-dimension stats (after normalization):")
+            for d in range(encoded_chunks.shape[-1]):
+                dim_data = encoded_chunks[:, :, d]
+                print(
+                    f"  Dim {d}: min={np.min(dim_data):7.4f}, max={np.max(dim_data):7.4f}, "
+                    f"mean={np.mean(dim_data):7.4f}, std={np.std(dim_data):7.4f}"
+                )
+        else:
+            print("Warning: Could not extract stats for encoded dimensions, using raw actions")
+    else:
+        print("Warning: No normalization stats found in dataset, using raw actions")
+
+    print(f"Encoded chunks shape: {encoded_chunks.shape}")
+
+    # train FAST tokenizer
+    tokenizer = train_fast_tokenizer(
+        encoded_chunks,
+        vocab_size=cfg.vocab_size,
+        scale=cfg.scale,
+    )
+
+    # compute compression statistics
+    compression_stats = compute_compression_stats(tokenizer, encoded_chunks)
+
+    # save tokenizer
+    output_dir = cfg.output_dir
+    if output_dir is None:
+        output_dir = f"fast_tokenizer_{cfg.repo_id.replace('/', '_')}"
+    output_path = Path(output_dir)
+    output_path.mkdir(parents=True, exist_ok=True)
+
+    tokenizer.save_pretrained(output_path)
+
+    # save metadata
+    metadata = {
+        "repo_id": cfg.repo_id,
+        "vocab_size": cfg.vocab_size,
+        "scale": cfg.scale,
+        "encoded_dims": cfg.encoded_dims,
+        "encoded_dim_ranges": encoded_dim_ranges,
+        "total_encoded_dims": total_encoded_dims,
+        "delta_dims": cfg.delta_dims,
+        "delta_dim_list": delta_dim_list,
+        "use_delta_transform": cfg.use_delta_transform,
+        "state_key": cfg.state_key,
+        "normalization_mode": norm_mode.value,
+        "action_horizon": cfg.action_horizon,
+        "num_training_chunks": len(encoded_chunks),
+        "compression_stats": compression_stats,
+    }
+
+    with open(output_path / "metadata.json", "w") as f:
+        json.dump(metadata, f, indent=2)
+
+    print(f"\nSaved FAST tokenizer to {output_path}")
+    print(f"Metadata: {json.dumps(metadata, indent=2)}")
+
+    # push to Hugging Face Hub if requested
+    if cfg.push_to_hub:
+        # determine the hub repository ID
+        hub_repo_id = cfg.hub_repo_id
+        if hub_repo_id is None:
+            hub_repo_id = output_path.name
+            print(f"\nNo hub_repo_id provided, using: {hub_repo_id}")
+
+        print(f"\nPushing tokenizer to Hugging Face Hub: {hub_repo_id}")
+        print(f"   Private: {cfg.hub_private}")
+
+        try:
+            # use the tokenizer's push_to_hub method
+            tokenizer.push_to_hub(
+                repo_id=hub_repo_id,
+                private=cfg.hub_private,
+                commit_message=f"Upload FAST tokenizer trained on {cfg.repo_id}",
+            )
+
+            # also upload the metadata.json file separately
+            api = HfApi()
+            api.upload_file(
+                path_or_fileobj=str(output_path / "metadata.json"),
+                path_in_repo="metadata.json",
+                repo_id=hub_repo_id,
+                repo_type="model",
+                commit_message="Upload tokenizer metadata",
+            )
+
+            print(f"Successfully pushed tokenizer to: https://huggingface.co/{hub_repo_id}")
+        except Exception as e:
+            print(f"Error pushing to hub: {e}")
+            print("   Make sure you're logged in with `huggingface-cli login`")
+
+
+def main():
+    """CLI entry point that parses arguments and runs the tokenizer training."""
+    train_tokenizer()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/src/lerobot/teleoperators/__init__.py b/lerobot/src/lerobot/teleoperators/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..ee508dddb3cc8597e9354e7fd5b0710c92463d61
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/__init__.py
@@ -0,0 +1,19 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config import TeleoperatorConfig
+from .teleoperator import Teleoperator
+from .utils import TeleopEvents, make_teleoperator_from_config
diff --git a/lerobot/src/lerobot/teleoperators/bi_openarm_leader/__init__.py b/lerobot/src/lerobot/teleoperators/bi_openarm_leader/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..fe728b8265fa5978c63e60c062e837bb37499722
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/bi_openarm_leader/__init__.py
@@ -0,0 +1,20 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .bi_openarm_leader import BiOpenArmLeader
+from .config_bi_openarm_leader import BiOpenArmLeaderConfig
+
+__all__ = ["BiOpenArmLeader", "BiOpenArmLeaderConfig"]
diff --git a/lerobot/src/lerobot/teleoperators/bi_openarm_leader/bi_openarm_leader.py b/lerobot/src/lerobot/teleoperators/bi_openarm_leader/bi_openarm_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..b44f1fbea30433240ff446b131b7c924efba4b30
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/bi_openarm_leader/bi_openarm_leader.py
@@ -0,0 +1,135 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+from functools import cached_property
+
+from lerobot.teleoperators.openarm_leader import OpenArmLeaderConfig
+from lerobot.types import RobotAction
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..openarm_leader import OpenArmLeader
+from ..teleoperator import Teleoperator
+from .config_bi_openarm_leader import BiOpenArmLeaderConfig
+
+logger = logging.getLogger(__name__)
+
+
+class BiOpenArmLeader(Teleoperator):
+    """
+    Bimanual OpenArm Leader Arms
+    """
+
+    config_class = BiOpenArmLeaderConfig
+    name = "bi_openarm_leader"
+
+    def __init__(self, config: BiOpenArmLeaderConfig):
+        super().__init__(config)
+        self.config = config
+
+        left_arm_config = OpenArmLeaderConfig(
+            id=f"{config.id}_left" if config.id else None,
+            calibration_dir=config.calibration_dir,
+            port=config.left_arm_config.port,
+            can_interface=config.left_arm_config.can_interface,
+            use_can_fd=config.left_arm_config.use_can_fd,
+            can_bitrate=config.left_arm_config.can_bitrate,
+            can_data_bitrate=config.left_arm_config.can_data_bitrate,
+            motor_config=config.left_arm_config.motor_config,
+            manual_control=config.left_arm_config.manual_control,
+            position_kd=config.left_arm_config.position_kd,
+            position_kp=config.left_arm_config.position_kp,
+        )
+
+        right_arm_config = OpenArmLeaderConfig(
+            id=f"{config.id}_right" if config.id else None,
+            calibration_dir=config.calibration_dir,
+            port=config.right_arm_config.port,
+            can_interface=config.right_arm_config.can_interface,
+            use_can_fd=config.right_arm_config.use_can_fd,
+            can_bitrate=config.right_arm_config.can_bitrate,
+            can_data_bitrate=config.right_arm_config.can_data_bitrate,
+            motor_config=config.right_arm_config.motor_config,
+            manual_control=config.right_arm_config.manual_control,
+            position_kd=config.right_arm_config.position_kd,
+            position_kp=config.right_arm_config.position_kp,
+        )
+
+        self.left_arm = OpenArmLeader(left_arm_config)
+        self.right_arm = OpenArmLeader(right_arm_config)
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        left_arm_features = self.left_arm.action_features
+        right_arm_features = self.right_arm.action_features
+
+        return {
+            **{f"left_{k}": v for k, v in left_arm_features.items()},
+            **{f"right_{k}": v for k, v in right_arm_features.items()},
+        }
+
+    @cached_property
+    def feedback_features(self) -> dict[str, type]:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        return self.left_arm.is_connected and self.right_arm.is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.left_arm.connect(calibrate)
+        self.right_arm.connect(calibrate)
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.left_arm.is_calibrated and self.right_arm.is_calibrated
+
+    def calibrate(self) -> None:
+        self.left_arm.calibrate()
+        self.right_arm.calibrate()
+
+    def configure(self) -> None:
+        self.left_arm.configure()
+        self.right_arm.configure()
+
+    def setup_motors(self) -> None:
+        raise NotImplementedError(
+            "Motor ID configuration is typically done via manufacturer tools for CAN motors."
+        )
+
+    @check_if_not_connected
+    def get_action(self) -> RobotAction:
+        action_dict = {}
+
+        # Add "left_" prefix
+        left_action = self.left_arm.get_action()
+        action_dict.update({f"left_{key}": value for key, value in left_action.items()})
+
+        # Add "right_" prefix
+        right_action = self.right_arm.get_action()
+        action_dict.update({f"right_{key}": value for key, value in right_action.items()})
+
+        return action_dict
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        # TODO: Implement force feedback
+        raise NotImplementedError
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self.left_arm.disconnect()
+        self.right_arm.disconnect()
diff --git a/lerobot/src/lerobot/teleoperators/bi_openarm_leader/config_bi_openarm_leader.py b/lerobot/src/lerobot/teleoperators/bi_openarm_leader/config_bi_openarm_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..39fc90add5766a4edc656733e123ef7a38abed24
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/bi_openarm_leader/config_bi_openarm_leader.py
@@ -0,0 +1,30 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from lerobot.teleoperators.openarm_leader import OpenArmLeaderConfigBase
+
+from ..config import TeleoperatorConfig
+
+
+@TeleoperatorConfig.register_subclass("bi_openarm_leader")
+@dataclass
+class BiOpenArmLeaderConfig(TeleoperatorConfig):
+    """Configuration class for Bi OpenArm Follower robots."""
+
+    left_arm_config: OpenArmLeaderConfigBase
+    right_arm_config: OpenArmLeaderConfigBase
diff --git a/lerobot/src/lerobot/teleoperators/bi_so_leader/__init__.py b/lerobot/src/lerobot/teleoperators/bi_so_leader/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..b902270f94524d564b4aee77aa252fbcf561b702
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/bi_so_leader/__init__.py
@@ -0,0 +1,17 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .bi_so_leader import BiSOLeader, BiSOLeaderConfig
diff --git a/lerobot/src/lerobot/teleoperators/bi_so_leader/bi_so_leader.py b/lerobot/src/lerobot/teleoperators/bi_so_leader/bi_so_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..e84ac6f509a8df06c7197585d721991ac7b5f6c6
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/bi_so_leader/bi_so_leader.py
@@ -0,0 +1,117 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+from functools import cached_property
+
+from lerobot.teleoperators.so_leader import SOLeaderTeleopConfig
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..so_leader import SOLeader
+from ..teleoperator import Teleoperator
+from .config_bi_so_leader import BiSOLeaderConfig
+
+logger = logging.getLogger(__name__)
+
+
+class BiSOLeader(Teleoperator):
+    """
+    [Bimanual SO Leader Arms](https://github.com/TheRobotStudio/SO-ARM100) designed by TheRobotStudio
+    """
+
+    config_class = BiSOLeaderConfig
+    name = "bi_so_leader"
+
+    def __init__(self, config: BiSOLeaderConfig):
+        super().__init__(config)
+        self.config = config
+
+        left_arm_config = SOLeaderTeleopConfig(
+            id=f"{config.id}_left" if config.id else None,
+            calibration_dir=config.calibration_dir,
+            port=config.left_arm_config.port,
+        )
+
+        right_arm_config = SOLeaderTeleopConfig(
+            id=f"{config.id}_right" if config.id else None,
+            calibration_dir=config.calibration_dir,
+            port=config.right_arm_config.port,
+        )
+
+        self.left_arm = SOLeader(left_arm_config)
+        self.right_arm = SOLeader(right_arm_config)
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        left_arm_features = self.left_arm.action_features
+        right_arm_features = self.right_arm.action_features
+
+        return {
+            **{f"left_{k}": v for k, v in left_arm_features.items()},
+            **{f"right_{k}": v for k, v in right_arm_features.items()},
+        }
+
+    @cached_property
+    def feedback_features(self) -> dict[str, type]:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        return self.left_arm.is_connected and self.right_arm.is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.left_arm.connect(calibrate)
+        self.right_arm.connect(calibrate)
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.left_arm.is_calibrated and self.right_arm.is_calibrated
+
+    def calibrate(self) -> None:
+        self.left_arm.calibrate()
+        self.right_arm.calibrate()
+
+    def configure(self) -> None:
+        self.left_arm.configure()
+        self.right_arm.configure()
+
+    def setup_motors(self) -> None:
+        self.left_arm.setup_motors()
+        self.right_arm.setup_motors()
+
+    @check_if_not_connected
+    def get_action(self) -> dict[str, float]:
+        action_dict = {}
+
+        # Add "left_" prefix
+        left_action = self.left_arm.get_action()
+        action_dict.update({f"left_{key}": value for key, value in left_action.items()})
+
+        # Add "right_" prefix
+        right_action = self.right_arm.get_action()
+        action_dict.update({f"right_{key}": value for key, value in right_action.items()})
+
+        return action_dict
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        # TODO: Implement force feedback
+        raise NotImplementedError
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self.left_arm.disconnect()
+        self.right_arm.disconnect()
diff --git a/lerobot/src/lerobot/teleoperators/bi_so_leader/config_bi_so_leader.py b/lerobot/src/lerobot/teleoperators/bi_so_leader/config_bi_so_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..c2f23c617b6efd5f367e6886efdbcba40baa6253
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/bi_so_leader/config_bi_so_leader.py
@@ -0,0 +1,30 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from lerobot.teleoperators.so_leader import SOLeaderConfig
+
+from ..config import TeleoperatorConfig
+
+
+@TeleoperatorConfig.register_subclass("bi_so_leader")
+@dataclass
+class BiSOLeaderConfig(TeleoperatorConfig):
+    """Configuration class for Bi SO Leader teleoperators."""
+
+    left_arm_config: SOLeaderConfig
+    right_arm_config: SOLeaderConfig
diff --git a/lerobot/src/lerobot/teleoperators/config.py b/lerobot/src/lerobot/teleoperators/config.py
new file mode 100644
index 0000000000000000000000000000000000000000..1b42b4edbe036a38d86174e2af881e1189f69020
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/config.py
@@ -0,0 +1,31 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import abc
+from dataclasses import dataclass
+from pathlib import Path
+
+import draccus
+
+
+@dataclass(kw_only=True)
+class TeleoperatorConfig(draccus.ChoiceRegistry, abc.ABC):
+    # Allows to distinguish between different teleoperators of the same type
+    id: str | None = None
+    # Directory to store calibration file
+    calibration_dir: Path | None = None
+
+    @property
+    def type(self) -> str:
+        return self.get_choice_name(self.__class__)
diff --git a/lerobot/src/lerobot/teleoperators/gamepad/__init__.py b/lerobot/src/lerobot/teleoperators/gamepad/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..6f9f7fbd9122d4650c2e5becffa3a13016cbefe1
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/gamepad/__init__.py
@@ -0,0 +1,18 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configuration_gamepad import GamepadTeleopConfig
+from .teleop_gamepad import GamepadTeleop
diff --git a/lerobot/src/lerobot/teleoperators/gamepad/configuration_gamepad.py b/lerobot/src/lerobot/teleoperators/gamepad/configuration_gamepad.py
new file mode 100644
index 0000000000000000000000000000000000000000..b3a565c07202b5514118f038fb7db012132ba1a3
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/gamepad/configuration_gamepad.py
@@ -0,0 +1,25 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..config import TeleoperatorConfig
+
+
+@TeleoperatorConfig.register_subclass("gamepad")
+@dataclass
+class GamepadTeleopConfig(TeleoperatorConfig):
+    use_gripper: bool = True
diff --git a/lerobot/src/lerobot/teleoperators/gamepad/gamepad_utils.py b/lerobot/src/lerobot/teleoperators/gamepad/gamepad_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..9f94b6746a0379f853bf826dd491124b14ce806d
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/gamepad/gamepad_utils.py
@@ -0,0 +1,460 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+
+from ..utils import TeleopEvents
+
+
+class InputController:
+    """Base class for input controllers that generate motion deltas."""
+
+    def __init__(self, x_step_size=1.0, y_step_size=1.0, z_step_size=1.0):
+        """
+        Initialize the controller.
+
+        Args:
+            x_step_size: Base movement step size in meters
+            y_step_size: Base movement step size in meters
+            z_step_size: Base movement step size in meters
+        """
+        self.x_step_size = x_step_size
+        self.y_step_size = y_step_size
+        self.z_step_size = z_step_size
+        self.running = True
+        self.episode_end_status = None  # None, "success", or "failure"
+        self.intervention_flag = False
+        self.open_gripper_command = False
+        self.close_gripper_command = False
+
+    def start(self):
+        """Start the controller and initialize resources."""
+        pass
+
+    def stop(self):
+        """Stop the controller and release resources."""
+        pass
+
+    def get_deltas(self):
+        """Get the current movement deltas (dx, dy, dz) in meters."""
+        return 0.0, 0.0, 0.0
+
+    def update(self):
+        """Update controller state - call this once per frame."""
+        pass
+
+    def __enter__(self):
+        """Support for use in 'with' statements."""
+        self.start()
+        return self
+
+    def __exit__(self, exc_type, exc_val, exc_tb):
+        """Ensure resources are released when exiting 'with' block."""
+        self.stop()
+
+    def get_episode_end_status(self):
+        """
+        Get the current episode end status.
+
+        Returns:
+            None if episode should continue, "success" or "failure" otherwise
+        """
+        status = self.episode_end_status
+        self.episode_end_status = None  # Reset after reading
+        return status
+
+    def should_intervene(self):
+        """Return True if intervention flag was set."""
+        return self.intervention_flag
+
+    def gripper_command(self):
+        """Return the current gripper command."""
+        if self.open_gripper_command == self.close_gripper_command:
+            return "stay"
+        elif self.open_gripper_command:
+            return "open"
+        elif self.close_gripper_command:
+            return "close"
+
+
+class KeyboardController(InputController):
+    """Generate motion deltas from keyboard input."""
+
+    def __init__(self, x_step_size=1.0, y_step_size=1.0, z_step_size=1.0):
+        super().__init__(x_step_size, y_step_size, z_step_size)
+        self.key_states = {
+            "forward_x": False,
+            "backward_x": False,
+            "forward_y": False,
+            "backward_y": False,
+            "forward_z": False,
+            "backward_z": False,
+            "quit": False,
+            "success": False,
+            "failure": False,
+        }
+        self.listener = None
+
+    def start(self):
+        """Start the keyboard listener."""
+        from pynput import keyboard
+
+        def on_press(key):
+            try:
+                if key == keyboard.Key.up:
+                    self.key_states["forward_x"] = True
+                elif key == keyboard.Key.down:
+                    self.key_states["backward_x"] = True
+                elif key == keyboard.Key.left:
+                    self.key_states["forward_y"] = True
+                elif key == keyboard.Key.right:
+                    self.key_states["backward_y"] = True
+                elif key == keyboard.Key.shift:
+                    self.key_states["backward_z"] = True
+                elif key == keyboard.Key.shift_r:
+                    self.key_states["forward_z"] = True
+                elif key == keyboard.Key.esc:
+                    self.key_states["quit"] = True
+                    self.running = False
+                    return False
+                elif key == keyboard.Key.enter:
+                    self.key_states["success"] = True
+                    self.episode_end_status = TeleopEvents.SUCCESS
+                elif key == keyboard.Key.backspace:
+                    self.key_states["failure"] = True
+                    self.episode_end_status = TeleopEvents.FAILURE
+            except AttributeError:
+                pass
+
+        def on_release(key):
+            try:
+                if key == keyboard.Key.up:
+                    self.key_states["forward_x"] = False
+                elif key == keyboard.Key.down:
+                    self.key_states["backward_x"] = False
+                elif key == keyboard.Key.left:
+                    self.key_states["forward_y"] = False
+                elif key == keyboard.Key.right:
+                    self.key_states["backward_y"] = False
+                elif key == keyboard.Key.shift:
+                    self.key_states["backward_z"] = False
+                elif key == keyboard.Key.shift_r:
+                    self.key_states["forward_z"] = False
+                elif key == keyboard.Key.enter:
+                    self.key_states["success"] = False
+                elif key == keyboard.Key.backspace:
+                    self.key_states["failure"] = False
+            except AttributeError:
+                pass
+
+        self.listener = keyboard.Listener(on_press=on_press, on_release=on_release)
+        self.listener.start()
+
+        print("Keyboard controls:")
+        print("  Arrow keys: Move in X-Y plane")
+        print("  Shift and Shift_R: Move in Z axis")
+        print("  Enter: End episode with SUCCESS")
+        print("  Backspace: End episode with FAILURE")
+        print("  ESC: Exit")
+
+    def stop(self):
+        """Stop the keyboard listener."""
+        if self.listener and self.listener.is_alive():
+            self.listener.stop()
+
+    def get_deltas(self):
+        """Get the current movement deltas from keyboard state."""
+        delta_x = delta_y = delta_z = 0.0
+
+        if self.key_states["forward_x"]:
+            delta_x += self.x_step_size
+        if self.key_states["backward_x"]:
+            delta_x -= self.x_step_size
+        if self.key_states["forward_y"]:
+            delta_y += self.y_step_size
+        if self.key_states["backward_y"]:
+            delta_y -= self.y_step_size
+        if self.key_states["forward_z"]:
+            delta_z += self.z_step_size
+        if self.key_states["backward_z"]:
+            delta_z -= self.z_step_size
+
+        return delta_x, delta_y, delta_z
+
+
+class GamepadController(InputController):
+    """Generate motion deltas from gamepad input."""
+
+    def __init__(self, x_step_size=1.0, y_step_size=1.0, z_step_size=1.0, deadzone=0.1):
+        super().__init__(x_step_size, y_step_size, z_step_size)
+        self.deadzone = deadzone
+        self.joystick = None
+        self.intervention_flag = False
+
+    def start(self):
+        """Initialize pygame and the gamepad."""
+        import pygame
+
+        pygame.init()
+        pygame.joystick.init()
+
+        if pygame.joystick.get_count() == 0:
+            logging.error("No gamepad detected. Please connect a gamepad and try again.")
+            self.running = False
+            return
+
+        self.joystick = pygame.joystick.Joystick(0)
+        self.joystick.init()
+        logging.info(f"Initialized gamepad: {self.joystick.get_name()}")
+
+        print("Gamepad controls:")
+        print("  Left analog stick: Move in X-Y plane")
+        print("  Right analog stick (vertical): Move in Z axis")
+        print("  B/Circle button: Exit")
+        print("  Y/Triangle button: End episode with SUCCESS")
+        print("  A/Cross button: End episode with FAILURE")
+        print("  X/Square button: Rerecord episode")
+
+    def stop(self):
+        """Clean up pygame resources."""
+        import pygame
+
+        if pygame.joystick.get_init():
+            if self.joystick:
+                self.joystick.quit()
+            pygame.joystick.quit()
+        pygame.quit()
+
+    def update(self):
+        """Process pygame events to get fresh gamepad readings."""
+        import pygame
+
+        for event in pygame.event.get():
+            if event.type == pygame.JOYBUTTONDOWN:
+                if event.button == 3:
+                    self.episode_end_status = TeleopEvents.SUCCESS
+                # A button (1) for failure
+                elif event.button == 1:
+                    self.episode_end_status = TeleopEvents.FAILURE
+                # X button (0) for rerecord
+                elif event.button == 0:
+                    self.episode_end_status = TeleopEvents.RERECORD_EPISODE
+
+                # RB button (6) for closing gripper
+                elif event.button == 6:
+                    self.close_gripper_command = True
+
+                # LT button (7) for opening gripper
+                elif event.button == 7:
+                    self.open_gripper_command = True
+
+            # Reset episode status on button release
+            elif event.type == pygame.JOYBUTTONUP:
+                if event.button in [0, 2, 3]:
+                    self.episode_end_status = None
+
+                elif event.button == 6:
+                    self.close_gripper_command = False
+
+                elif event.button == 7:
+                    self.open_gripper_command = False
+
+            # Check for RB button (typically button 5) for intervention flag
+            if self.joystick.get_button(5):
+                self.intervention_flag = True
+            else:
+                self.intervention_flag = False
+
+    def get_deltas(self):
+        """Get the current movement deltas from gamepad state."""
+        import pygame
+
+        try:
+            # Read joystick axes
+            # Left stick X and Y (typically axes 0 and 1)
+            y_input = self.joystick.get_axis(0)  # Up/Down (often inverted)
+            x_input = self.joystick.get_axis(1)  # Left/Right
+
+            # Right stick Y (typically axis 3 or 4)
+            z_input = self.joystick.get_axis(3)  # Up/Down for Z
+
+            # Apply deadzone to avoid drift
+            x_input = 0 if abs(x_input) < self.deadzone else x_input
+            y_input = 0 if abs(y_input) < self.deadzone else y_input
+            z_input = 0 if abs(z_input) < self.deadzone else z_input
+
+            # Calculate deltas (note: may need to invert axes depending on controller)
+            delta_x = -x_input * self.x_step_size  # Forward/backward
+            delta_y = -y_input * self.y_step_size  # Left/right
+            delta_z = -z_input * self.z_step_size  # Up/down
+
+            return delta_x, delta_y, delta_z
+
+        except pygame.error:
+            logging.error("Error reading gamepad. Is it still connected?")
+            return 0.0, 0.0, 0.0
+
+
+class GamepadControllerHID(InputController):
+    """Generate motion deltas from gamepad input using HIDAPI."""
+
+    def __init__(
+        self,
+        x_step_size=1.0,
+        y_step_size=1.0,
+        z_step_size=1.0,
+        deadzone=0.1,
+    ):
+        """
+        Initialize the HID gamepad controller.
+
+        Args:
+            step_size: Base movement step size in meters
+            z_scale: Scaling factor for Z-axis movement
+            deadzone: Joystick deadzone to prevent drift
+        """
+        super().__init__(x_step_size, y_step_size, z_step_size)
+        self.deadzone = deadzone
+        self.device = None
+        self.device_info = None
+
+        # Movement values (normalized from -1.0 to 1.0)
+        self.left_x = 0.0
+        self.left_y = 0.0
+        self.right_x = 0.0
+        self.right_y = 0.0
+
+        # Button states
+        self.buttons = {}
+
+    def find_device(self):
+        """Look for the gamepad device by vendor and product ID."""
+        import hid
+
+        devices = hid.enumerate()
+        for device in devices:
+            device_name = device["product_string"]
+            if any(controller in device_name for controller in ["Logitech", "Xbox", "PS4", "PS5"]):
+                return device
+
+        logging.error(
+            "No gamepad found, check the connection and the product string in HID to add your gamepad"
+        )
+        return None
+
+    def start(self):
+        """Connect to the gamepad using HIDAPI."""
+        import hid
+
+        self.device_info = self.find_device()
+        if not self.device_info:
+            self.running = False
+            return
+
+        try:
+            logging.info(f"Connecting to gamepad at path: {self.device_info['path']}")
+            self.device = hid.device()
+            self.device.open_path(self.device_info["path"])
+            self.device.set_nonblocking(1)
+
+            manufacturer = self.device.get_manufacturer_string()
+            product = self.device.get_product_string()
+            logging.info(f"Connected to {manufacturer} {product}")
+
+            logging.info("Gamepad controls (HID mode):")
+            logging.info("  Left analog stick: Move in X-Y plane")
+            logging.info("  Right analog stick: Move in Z axis (vertical)")
+            logging.info("  Button 1/B/Circle: Exit")
+            logging.info("  Button 2/A/Cross: End episode with SUCCESS")
+            logging.info("  Button 3/X/Square: End episode with FAILURE")
+
+        except OSError as e:
+            logging.error(f"Error opening gamepad: {e}")
+            logging.error("You might need to run this with sudo/admin privileges on some systems")
+            self.running = False
+
+    def stop(self):
+        """Close the HID device connection."""
+        if self.device:
+            self.device.close()
+            self.device = None
+
+    def update(self):
+        """
+        Read and process the latest gamepad data.
+        Due to an issue with the HIDAPI, we need to read the read the device several times in order to get a stable reading
+        """
+        for _ in range(10):
+            self._update()
+
+    def _update(self):
+        """Read and process the latest gamepad data."""
+        if not self.device or not self.running:
+            return
+
+        try:
+            # Read data from the gamepad
+            data = self.device.read(64)
+            # Interpret gamepad data - this will vary by controller model
+            # These offsets are for the Logitech RumblePad 2
+            if data and len(data) >= 8:
+                # Normalize joystick values from 0-255 to -1.0-1.0
+                self.left_y = (data[1] - 128) / 128.0
+                self.left_x = (data[2] - 128) / 128.0
+                self.right_x = (data[3] - 128) / 128.0
+                self.right_y = (data[4] - 128) / 128.0
+
+                # Apply deadzone
+                self.left_y = 0 if abs(self.left_y) < self.deadzone else self.left_y
+                self.left_x = 0 if abs(self.left_x) < self.deadzone else self.left_x
+                self.right_x = 0 if abs(self.right_x) < self.deadzone else self.right_x
+                self.right_y = 0 if abs(self.right_y) < self.deadzone else self.right_y
+
+                # Parse button states (byte 5 in the Logitech RumblePad 2)
+                buttons = data[5]
+
+                # Check if RB is pressed then the intervention flag should be set
+                self.intervention_flag = data[6] in [2, 6, 10, 14]
+
+                # Check if RT is pressed
+                self.open_gripper_command = data[6] in [8, 10, 12]
+
+                # Check if LT is pressed
+                self.close_gripper_command = data[6] in [4, 6, 12]
+
+                # Check if Y/Triangle button (bit 7) is pressed for saving
+                # Check if X/Square button (bit 5) is pressed for failure
+                # Check if A/Cross button (bit 4) is pressed for rerecording
+                if buttons & 1 << 7:
+                    self.episode_end_status = TeleopEvents.SUCCESS
+                elif buttons & 1 << 5:
+                    self.episode_end_status = TeleopEvents.FAILURE
+                elif buttons & 1 << 4:
+                    self.episode_end_status = TeleopEvents.RERECORD_EPISODE
+                else:
+                    self.episode_end_status = None
+
+        except OSError as e:
+            logging.error(f"Error reading from gamepad: {e}")
+
+    def get_deltas(self):
+        """Get the current movement deltas from gamepad state."""
+        # Calculate deltas - invert as needed based on controller orientation
+        delta_x = -self.left_x * self.x_step_size  # Forward/backward
+        delta_y = -self.left_y * self.y_step_size  # Left/right
+        delta_z = -self.right_y * self.z_step_size  # Up/down
+
+        return delta_x, delta_y, delta_z
diff --git a/lerobot/src/lerobot/teleoperators/gamepad/teleop_gamepad.py b/lerobot/src/lerobot/teleoperators/gamepad/teleop_gamepad.py
new file mode 100644
index 0000000000000000000000000000000000000000..8c1796e45a38bd631fecd8241b0cae3cb7162ff1
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/gamepad/teleop_gamepad.py
@@ -0,0 +1,186 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import sys
+from enum import IntEnum
+from typing import Any
+
+import numpy as np
+
+from lerobot.types import RobotAction
+from lerobot.utils.decorators import check_if_not_connected
+
+from ..teleoperator import Teleoperator
+from ..utils import TeleopEvents
+from .configuration_gamepad import GamepadTeleopConfig
+
+
+class GripperAction(IntEnum):
+    CLOSE = 0
+    STAY = 1
+    OPEN = 2
+
+
+gripper_action_map = {
+    "close": GripperAction.CLOSE.value,
+    "open": GripperAction.OPEN.value,
+    "stay": GripperAction.STAY.value,
+}
+
+
+class GamepadTeleop(Teleoperator):
+    """
+    Teleop class to use gamepad inputs for control.
+    """
+
+    config_class = GamepadTeleopConfig
+    name = "gamepad"
+
+    def __init__(self, config: GamepadTeleopConfig):
+        super().__init__(config)
+        self.config = config
+        self.robot_type = config.type
+
+        self.gamepad = None
+
+    @property
+    def action_features(self) -> dict:
+        if self.config.use_gripper:
+            return {
+                "dtype": "float32",
+                "shape": (4,),
+                "names": {"delta_x": 0, "delta_y": 1, "delta_z": 2, "gripper": 3},
+            }
+        else:
+            return {
+                "dtype": "float32",
+                "shape": (3,),
+                "names": {"delta_x": 0, "delta_y": 1, "delta_z": 2},
+            }
+
+    @property
+    def feedback_features(self) -> dict:
+        return {}
+
+    def connect(self) -> None:
+        # use HidApi for macos
+        if sys.platform == "darwin":
+            # NOTE: On macOS, pygame doesn’t reliably detect input from some controllers so we fall back to hidapi
+            from .gamepad_utils import GamepadControllerHID as Gamepad
+        else:
+            from .gamepad_utils import GamepadController as Gamepad
+
+        self.gamepad = Gamepad()
+        self.gamepad.start()
+
+    @check_if_not_connected
+    def get_action(self) -> RobotAction:
+        # Update the controller to get fresh inputs
+        self.gamepad.update()
+
+        # Get movement deltas from the controller
+        delta_x, delta_y, delta_z = self.gamepad.get_deltas()
+
+        # Create action from gamepad input
+        gamepad_action = np.array([delta_x, delta_y, delta_z], dtype=np.float32)
+
+        action_dict = {
+            "delta_x": gamepad_action[0],
+            "delta_y": gamepad_action[1],
+            "delta_z": gamepad_action[2],
+        }
+
+        # Default gripper action is to stay
+        gripper_action = GripperAction.STAY.value
+        if self.config.use_gripper:
+            gripper_command = self.gamepad.gripper_command()
+            gripper_action = gripper_action_map[gripper_command]
+            action_dict["gripper"] = gripper_action
+
+        return action_dict
+
+    def get_teleop_events(self) -> dict[str, Any]:
+        """
+        Get extra control events from the gamepad such as intervention status,
+        episode termination, success indicators, etc.
+
+        Returns:
+            Dictionary containing:
+                - is_intervention: bool - Whether human is currently intervening
+                - terminate_episode: bool - Whether to terminate the current episode
+                - success: bool - Whether the episode was successful
+                - rerecord_episode: bool - Whether to rerecord the episode
+        """
+        if self.gamepad is None:
+            return {
+                TeleopEvents.IS_INTERVENTION: False,
+                TeleopEvents.TERMINATE_EPISODE: False,
+                TeleopEvents.SUCCESS: False,
+                TeleopEvents.RERECORD_EPISODE: False,
+            }
+
+        # Update gamepad state to get fresh inputs
+        self.gamepad.update()
+
+        # Check if intervention is active
+        is_intervention = self.gamepad.should_intervene()
+
+        # Get episode end status
+        episode_end_status = self.gamepad.get_episode_end_status()
+        terminate_episode = episode_end_status in [
+            TeleopEvents.RERECORD_EPISODE,
+            TeleopEvents.FAILURE,
+        ]
+        success = episode_end_status == TeleopEvents.SUCCESS
+        rerecord_episode = episode_end_status == TeleopEvents.RERECORD_EPISODE
+
+        return {
+            TeleopEvents.IS_INTERVENTION: is_intervention,
+            TeleopEvents.TERMINATE_EPISODE: terminate_episode,
+            TeleopEvents.SUCCESS: success,
+            TeleopEvents.RERECORD_EPISODE: rerecord_episode,
+        }
+
+    def disconnect(self) -> None:
+        """Disconnect from the gamepad."""
+        if self.gamepad is not None:
+            self.gamepad.stop()
+            self.gamepad = None
+
+    @property
+    def is_connected(self) -> bool:
+        """Check if gamepad is connected."""
+        return self.gamepad is not None
+
+    def calibrate(self) -> None:
+        """Calibrate the gamepad."""
+        # No calibration needed for gamepad
+        pass
+
+    def is_calibrated(self) -> bool:
+        """Check if gamepad is calibrated."""
+        # Gamepad doesn't require calibration
+        return True
+
+    def configure(self) -> None:
+        """Configure the gamepad."""
+        # No additional configuration needed
+        pass
+
+    def send_feedback(self, feedback: dict) -> None:
+        """Send feedback to the gamepad."""
+        # Gamepad doesn't support feedback
+        pass
diff --git a/lerobot/src/lerobot/teleoperators/homunculus/__init__.py b/lerobot/src/lerobot/teleoperators/homunculus/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..b3c6c0bf5cccbe8cd192a8eb9ca8d4b7105e7da8
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/homunculus/__init__.py
@@ -0,0 +1,20 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_homunculus import HomunculusArmConfig, HomunculusGloveConfig
+from .homunculus_arm import HomunculusArm
+from .homunculus_glove import HomunculusGlove
+from .joints_translation import homunculus_glove_to_hope_jr_hand
diff --git a/lerobot/src/lerobot/teleoperators/homunculus/config_homunculus.py b/lerobot/src/lerobot/teleoperators/homunculus/config_homunculus.py
new file mode 100644
index 0000000000000000000000000000000000000000..da465215ab700bf82d665dbe63c2ff48d132d3c8
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/homunculus/config_homunculus.py
@@ -0,0 +1,38 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..config import TeleoperatorConfig
+
+
+@TeleoperatorConfig.register_subclass("homunculus_glove")
+@dataclass
+class HomunculusGloveConfig(TeleoperatorConfig):
+    port: str  # Port to connect to the glove
+    side: str  # "left" / "right"
+    baud_rate: int = 115_200
+
+    def __post_init__(self):
+        if self.side not in ["right", "left"]:
+            raise ValueError(self.side)
+
+
+@TeleoperatorConfig.register_subclass("homunculus_arm")
+@dataclass
+class HomunculusArmConfig(TeleoperatorConfig):
+    port: str  # Port to connect to the arm
+    baud_rate: int = 115_200
diff --git a/lerobot/src/lerobot/teleoperators/homunculus/homunculus_arm.py b/lerobot/src/lerobot/teleoperators/homunculus/homunculus_arm.py
new file mode 100644
index 0000000000000000000000000000000000000000..178eed544ab0832d45adfb0d8fc374d50d3977a7
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/homunculus/homunculus_arm.py
@@ -0,0 +1,313 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import threading
+from collections import deque
+from pprint import pformat
+
+import serial
+
+from lerobot.motors.motors_bus import MotorCalibration, MotorNormMode
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.utils import enter_pressed, move_cursor_up
+
+from ..teleoperator import Teleoperator
+from .config_homunculus import HomunculusArmConfig
+
+logger = logging.getLogger(__name__)
+
+
+class HomunculusArm(Teleoperator):
+    """
+    Homunculus Arm designed by Hugging Face.
+    """
+
+    config_class = HomunculusArmConfig
+    name = "homunculus_arm"
+
+    def __init__(self, config: HomunculusArmConfig):
+        super().__init__(config)
+        self.config = config
+        self.serial = serial.Serial(config.port, config.baud_rate, timeout=1)
+        self.serial_lock = threading.Lock()
+
+        self.joints = {
+            "shoulder_pitch": MotorNormMode.RANGE_M100_100,
+            "shoulder_yaw": MotorNormMode.RANGE_M100_100,
+            "shoulder_roll": MotorNormMode.RANGE_M100_100,
+            "elbow_flex": MotorNormMode.RANGE_M100_100,
+            "wrist_roll": MotorNormMode.RANGE_M100_100,
+            "wrist_yaw": MotorNormMode.RANGE_M100_100,
+            "wrist_pitch": MotorNormMode.RANGE_M100_100,
+        }
+        n = 50
+        # EMA parameters ---------------------------------------------------
+        self.n: int = n
+        self.alpha: float = 2 / (n + 1)
+        # one deque *per joint* so we can inspect raw history if needed
+        self._buffers: dict[str, deque[int]] = {
+            joint: deque(maxlen=n)
+            for joint in (
+                "shoulder_pitch",
+                "shoulder_yaw",
+                "shoulder_roll",
+                "elbow_flex",
+                "wrist_roll",
+                "wrist_yaw",
+                "wrist_pitch",
+            )
+        }
+        # running EMA value per joint – lazily initialised on first read
+        self._ema: dict[str, float | None] = dict.fromkeys(self._buffers)
+
+        self._state: dict[str, float] | None = None
+        self.new_state_event = threading.Event()
+        self.stop_event = threading.Event()
+        self.thread = threading.Thread(target=self._read_loop, daemon=True, name=f"{self} _read_loop")
+        self.state_lock = threading.Lock()
+
+    @property
+    def action_features(self) -> dict:
+        return {f"{joint}.pos": float for joint in self.joints}
+
+    @property
+    def feedback_features(self) -> dict:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        with self.serial_lock:
+            return self.serial.is_open and self.thread.is_alive()
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        if not self.serial.is_open:
+            self.serial.open()
+        self.thread.start()
+
+        # wait for the thread to ramp up & 1st state to be ready
+        if not self.new_state_event.wait(timeout=2):
+            raise TimeoutError(f"{self}: Timed out waiting for state after 2s.")
+
+        if not self.is_calibrated and calibrate:
+            self.calibrate()
+
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.calibration_fpath.is_file()
+
+    def calibrate(self) -> None:
+        print(
+            "\nMove all joints through their entire range of motion."
+            "\nRecording positions. Press ENTER to stop..."
+        )
+        range_mins, range_maxes = self._record_ranges_of_motion()
+
+        self.calibration = {}
+        for id_, joint in enumerate(self.joints):
+            self.calibration[joint] = MotorCalibration(
+                id=id_,
+                drive_mode=0,
+                homing_offset=0,
+                range_min=range_mins[joint],
+                range_max=range_maxes[joint],
+            )
+
+        self._save_calibration()
+        print("Calibration saved to", self.calibration_fpath)
+
+    # TODO(Steven): This function is copy/paste from the `HomunculusGlove` class. Consider moving it to an utility to reduce duplicated code.
+    def _record_ranges_of_motion(
+        self, joints: list[str] | None = None, display_values: bool = True
+    ) -> tuple[dict[str, int], dict[str, int]]:
+        """Interactively record the min/max encoder values of each joint.
+
+        Move the joints while the method streams live positions. Press :kbd:`Enter` to finish.
+
+        Args:
+            joints (list[str] | None, optional):  Joints to record. Defaults to every joint (`None`).
+            display_values (bool, optional): When `True` (default) a live table is printed to the console.
+
+        Raises:
+            TypeError: `joints` is not `None` or a list.
+            ValueError: any joint's recorded min and max are the same.
+
+        Returns:
+            tuple[dict[str, int], dict[str, int]]: Two dictionaries *mins* and *maxes* with the extreme values
+            observed for each joint.
+        """
+        if joints is None:
+            joints = list(self.joints)
+        elif not isinstance(joints, list):
+            raise TypeError(joints)
+
+        display_len = max(len(key) for key in joints)
+
+        start_positions = self._read(joints, normalize=False)
+        mins = start_positions.copy()
+        maxes = start_positions.copy()
+
+        user_pressed_enter = False
+        while not user_pressed_enter:
+            positions = self._read(joints, normalize=False)
+            mins = {joint: int(min(positions[joint], min_)) for joint, min_ in mins.items()}
+            maxes = {joint: int(max(positions[joint], max_)) for joint, max_ in maxes.items()}
+
+            if display_values:
+                print("\n-------------------------------------------")
+                print(f"{'NAME':<{display_len}} | {'MIN':>6} | {'POS':>6} | {'MAX':>6}")
+                for joint in joints:
+                    print(
+                        f"{joint:<{display_len}} | {mins[joint]:>6} | {positions[joint]:>6} | {maxes[joint]:>6}"
+                    )
+
+            if enter_pressed():
+                user_pressed_enter = True
+
+            if display_values and not user_pressed_enter:
+                # Move cursor up to overwrite the previous output
+                move_cursor_up(len(joints) + 3)
+
+        same_min_max = [joint for joint in joints if mins[joint] == maxes[joint]]
+        if same_min_max:
+            raise ValueError(f"Some joints have the same min and max values:\n{pformat(same_min_max)}")
+
+        return mins, maxes
+
+    def configure(self) -> None:
+        pass
+
+    # TODO(Steven): This function is copy/paste from the `HomunculusGlove` class. Consider moving it to an utility to reduce duplicated code.
+    def _normalize(self, values: dict[str, int]) -> dict[str, float]:
+        if not self.calibration:
+            raise RuntimeError(f"{self} has no calibration registered.")
+
+        normalized_values = {}
+        for joint, val in values.items():
+            min_ = self.calibration[joint].range_min
+            max_ = self.calibration[joint].range_max
+            drive_mode = self.calibration[joint].drive_mode
+            bounded_val = min(max_, max(min_, val))
+
+            if self.joints[joint] is MotorNormMode.RANGE_M100_100:
+                norm = (((bounded_val - min_) / (max_ - min_)) * 200) - 100
+                normalized_values[joint] = -norm if drive_mode else norm
+            elif self.joints[joint] is MotorNormMode.RANGE_0_100:
+                norm = ((bounded_val - min_) / (max_ - min_)) * 100
+                normalized_values[joint] = 100 - norm if drive_mode else norm
+
+        return normalized_values
+
+    def _apply_ema(self, raw: dict[str, int]) -> dict[str, float]:
+        """Update buffers & running EMA values; return smoothed dict."""
+        smoothed: dict[str, float] = {}
+        for joint, value in raw.items():
+            # maintain raw history
+            self._buffers[joint].append(value)
+
+            # initialise on first run
+            if self._ema[joint] is None:
+                self._ema[joint] = float(value)
+            else:
+                self._ema[joint] = self.alpha * value + (1 - self.alpha) * self._ema[joint]
+
+            smoothed[joint] = self._ema[joint]
+        return smoothed
+
+    def _read(
+        self, joints: list[str] | None = None, normalize: bool = True, timeout: float = 1
+    ) -> dict[str, int | float]:
+        """
+        Return the most recent (single) values from self.last_d,
+        optionally applying calibration.
+        """
+        if not self.new_state_event.wait(timeout=timeout):
+            raise TimeoutError(f"{self}: Timed out waiting for state after {timeout}s.")
+
+        with self.state_lock:
+            state = self._state
+
+        self.new_state_event.clear()
+
+        if state is None:
+            raise RuntimeError(f"{self} Internal error: Event set but no state available.")
+
+        if joints is not None:
+            state = {k: v for k, v in state.items() if k in joints}
+
+        if normalize:
+            state = self._normalize(state)
+
+        state = self._apply_ema(state)
+
+        return state
+
+    def _read_loop(self):
+        """
+        Continuously read from the serial buffer in its own thread and sends values to the main thread through
+        a queue.
+        """
+        while not self.stop_event.is_set():
+            try:
+                raw_values = None
+                with self.serial_lock:
+                    if self.serial.in_waiting > 0:
+                        lines = []
+                        while self.serial.in_waiting > 0:
+                            line = self.serial.read_until().decode("utf-8").strip()
+                            if line:
+                                lines.append(line.split(" "))
+
+                        if lines:
+                            raw_values = lines[-1]
+
+                if raw_values is None or len(raw_values) != 21:  # 16 raw + 5 angle values
+                    continue
+
+                joint_angles = {
+                    "shoulder_pitch": int(raw_values[19]),
+                    "shoulder_yaw": int(raw_values[18]),
+                    "shoulder_roll": int(raw_values[20]),
+                    "elbow_flex": int(raw_values[17]),
+                    "wrist_roll": int(raw_values[16]),
+                    "wrist_yaw": int(raw_values[1]),
+                    "wrist_pitch": int(raw_values[0]),
+                }
+
+                with self.state_lock:
+                    self._state = joint_angles
+                self.new_state_event.set()
+
+            except Exception as e:
+                logger.debug(f"Error reading frame in background thread for {self}: {e}")
+
+    @check_if_not_connected
+    def get_action(self) -> dict[str, float]:
+        joint_positions = self._read()
+        return {f"{joint}.pos": pos for joint, pos in joint_positions.items()}
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        raise NotImplementedError
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self.stop_event.set()
+        self.thread.join(timeout=1)
+        self.serial.close()
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/teleoperators/homunculus/homunculus_glove.py b/lerobot/src/lerobot/teleoperators/homunculus/homunculus_glove.py
new file mode 100644
index 0000000000000000000000000000000000000000..c4393d66044c89cf90d4c9e758ab71d25a279cc7
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/homunculus/homunculus_glove.py
@@ -0,0 +1,341 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import threading
+from collections import deque
+from pprint import pformat
+
+import serial
+
+from lerobot.motors import MotorCalibration
+from lerobot.motors.motors_bus import MotorNormMode
+from lerobot.teleoperators.homunculus.joints_translation import homunculus_glove_to_hope_jr_hand
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.utils import enter_pressed, move_cursor_up
+
+from ..teleoperator import Teleoperator
+from .config_homunculus import HomunculusGloveConfig
+
+logger = logging.getLogger(__name__)
+
+LEFT_HAND_INVERSIONS = [
+    "thumb_cmc",
+    "index_dip",
+    "middle_mcp_abduction",
+    "middle_dip",
+    "pinky_mcp_abduction",
+    "pinky_dip",
+]
+
+RIGHT_HAND_INVERSIONS = [
+    "thumb_mcp",
+    "thumb_cmc",
+    "thumb_pip",
+    "thumb_dip",
+    "index_mcp_abduction",
+    # "index_dip",
+    "middle_mcp_abduction",
+    # "middle_dip",
+    "ring_mcp_abduction",
+    "ring_mcp_flexion",
+    # "ring_dip",
+    "pinky_mcp_abduction",
+]
+
+
+class HomunculusGlove(Teleoperator):
+    """
+    Homunculus Glove designed by NepYope & Hugging Face.
+    """
+
+    config_class = HomunculusGloveConfig
+    name = "homunculus_glove"
+
+    def __init__(self, config: HomunculusGloveConfig):
+        super().__init__(config)
+        self.config = config
+        self.serial = serial.Serial(config.port, config.baud_rate, timeout=1)
+        self.serial_lock = threading.Lock()
+
+        self.joints = {
+            "thumb_cmc": MotorNormMode.RANGE_0_100,
+            "thumb_mcp": MotorNormMode.RANGE_0_100,
+            "thumb_pip": MotorNormMode.RANGE_0_100,
+            "thumb_dip": MotorNormMode.RANGE_0_100,
+            "index_mcp_abduction": MotorNormMode.RANGE_M100_100,
+            "index_mcp_flexion": MotorNormMode.RANGE_0_100,
+            "index_dip": MotorNormMode.RANGE_0_100,
+            "middle_mcp_abduction": MotorNormMode.RANGE_M100_100,
+            "middle_mcp_flexion": MotorNormMode.RANGE_0_100,
+            "middle_dip": MotorNormMode.RANGE_0_100,
+            "ring_mcp_abduction": MotorNormMode.RANGE_M100_100,
+            "ring_mcp_flexion": MotorNormMode.RANGE_0_100,
+            "ring_dip": MotorNormMode.RANGE_0_100,
+            "pinky_mcp_abduction": MotorNormMode.RANGE_M100_100,
+            "pinky_mcp_flexion": MotorNormMode.RANGE_0_100,
+            "pinky_dip": MotorNormMode.RANGE_0_100,
+        }
+        self.inverted_joints = RIGHT_HAND_INVERSIONS if config.side == "right" else LEFT_HAND_INVERSIONS
+
+        n = 10
+        # EMA parameters ---------------------------------------------------
+        self.n: int = n
+        self.alpha: float = 2 / (n + 1)
+        # one deque *per joint* so we can inspect raw history if needed
+        self._buffers: dict[str, deque[int]] = {joint: deque(maxlen=n) for joint in self.joints}
+        # running EMA value per joint – lazily initialised on first read
+        self._ema: dict[str, float | None] = dict.fromkeys(self._buffers)
+
+        self._state: dict[str, float] | None = None
+        self.new_state_event = threading.Event()
+        self.stop_event = threading.Event()
+        self.thread = threading.Thread(target=self._read_loop, daemon=True, name=f"{self} _read_loop")
+        self.state_lock = threading.Lock()
+
+    @property
+    def action_features(self) -> dict:
+        return {f"{joint}.pos": float for joint in self.joints}
+
+    @property
+    def feedback_features(self) -> dict:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        with self.serial_lock:
+            return self.serial.is_open and self.thread.is_alive()
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        if not self.serial.is_open:
+            self.serial.open()
+        self.thread.start()
+
+        # wait for the thread to ramp up & 1st state to be ready
+        if not self.new_state_event.wait(timeout=2):
+            raise TimeoutError(f"{self}: Timed out waiting for state after 2s.")
+
+        if not self.is_calibrated and calibrate:
+            self.calibrate()
+
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.calibration_fpath.is_file()
+
+    def calibrate(self) -> None:
+        range_mins, range_maxes = {}, {}
+        for finger in ["thumb", "index", "middle", "ring", "pinky"]:
+            print(
+                f"\nMove {finger} through its entire range of motion."
+                "\nRecording positions. Press ENTER to stop..."
+            )
+            finger_joints = [joint for joint in self.joints if joint.startswith(finger)]
+            finger_mins, finger_maxes = self._record_ranges_of_motion(finger_joints)
+            range_mins.update(finger_mins)
+            range_maxes.update(finger_maxes)
+
+        self.calibration = {}
+        for id_, joint in enumerate(self.joints):
+            self.calibration[joint] = MotorCalibration(
+                id=id_,
+                drive_mode=1 if joint in self.inverted_joints else 0,
+                homing_offset=0,
+                range_min=range_mins[joint],
+                range_max=range_maxes[joint],
+            )
+
+        self._save_calibration()
+        print("Calibration saved to", self.calibration_fpath)
+
+    # TODO(Steven): This function is copy/paste from the `HomunculusArm` class. Consider moving it to an utility to reduce duplicated code.
+    def _record_ranges_of_motion(
+        self, joints: list[str] | None = None, display_values: bool = True
+    ) -> tuple[dict[str, int], dict[str, int]]:
+        """Interactively record the min/max encoder values of each joint.
+
+        Move the joints while the method streams live positions. Press :kbd:`Enter` to finish.
+
+        Args:
+            joints (list[str] | None, optional):  Joints to record. Defaults to every joint (`None`).
+            display_values (bool, optional): When `True` (default) a live table is printed to the console.
+
+        Raises:
+            TypeError: `joints` is not `None` or a list.
+            ValueError: any joint's recorded min and max are the same.
+
+        Returns:
+            tuple[dict[str, int], dict[str, int]]: Two dictionaries *mins* and *maxes* with the extreme values
+            observed for each joint.
+        """
+        if joints is None:
+            joints = list(self.joints)
+        elif not isinstance(joints, list):
+            raise TypeError(joints)
+
+        display_len = max(len(key) for key in joints)
+
+        start_positions = self._read(joints, normalize=False)
+        mins = start_positions.copy()
+        maxes = start_positions.copy()
+
+        user_pressed_enter = False
+        while not user_pressed_enter:
+            positions = self._read(joints, normalize=False)
+            mins = {joint: int(min(positions[joint], min_)) for joint, min_ in mins.items()}
+            maxes = {joint: int(max(positions[joint], max_)) for joint, max_ in maxes.items()}
+
+            if display_values:
+                print("\n-------------------------------------------")
+                print(f"{'NAME':<{display_len}} | {'MIN':>6} | {'POS':>6} | {'MAX':>6}")
+                for joint in joints:
+                    print(
+                        f"{joint:<{display_len}} | {mins[joint]:>6} | {positions[joint]:>6} | {maxes[joint]:>6}"
+                    )
+
+            if enter_pressed():
+                user_pressed_enter = True
+
+            if display_values and not user_pressed_enter:
+                # Move cursor up to overwrite the previous output
+                move_cursor_up(len(joints) + 3)
+
+        same_min_max = [joint for joint in joints if mins[joint] == maxes[joint]]
+        if same_min_max:
+            raise ValueError(f"Some joints have the same min and max values:\n{pformat(same_min_max)}")
+
+        return mins, maxes
+
+    def configure(self) -> None:
+        pass
+
+    # TODO(Steven): This function is copy/paste from the `HomunculusArm` class. Consider moving it to an utility to reduce duplicated code.
+    def _normalize(self, values: dict[str, int]) -> dict[str, float]:
+        if not self.calibration:
+            raise RuntimeError(f"{self} has no calibration registered.")
+
+        normalized_values = {}
+        for joint, val in values.items():
+            min_ = self.calibration[joint].range_min
+            max_ = self.calibration[joint].range_max
+            drive_mode = self.calibration[joint].drive_mode
+            bounded_val = min(max_, max(min_, val))
+
+            if self.joints[joint] is MotorNormMode.RANGE_M100_100:
+                norm = (((bounded_val - min_) / (max_ - min_)) * 200) - 100
+                normalized_values[joint] = -norm if drive_mode else norm
+            elif self.joints[joint] is MotorNormMode.RANGE_0_100:
+                norm = ((bounded_val - min_) / (max_ - min_)) * 100
+                normalized_values[joint] = 100 - norm if drive_mode else norm
+
+        return normalized_values
+
+    def _apply_ema(self, raw: dict[str, int]) -> dict[str, int]:
+        """Update buffers & running EMA values; return smoothed dict as integers."""
+        smoothed: dict[str, int] = {}
+        for joint, value in raw.items():
+            # maintain raw history
+            self._buffers[joint].append(value)
+
+            # initialise on first run
+            if self._ema[joint] is None:
+                self._ema[joint] = float(value)
+            else:
+                self._ema[joint] = self.alpha * value + (1 - self.alpha) * self._ema[joint]
+
+            # Convert back to int for compatibility with normalization
+            smoothed[joint] = int(round(self._ema[joint]))
+        return smoothed
+
+    def _read(
+        self, joints: list[str] | None = None, normalize: bool = True, timeout: float = 1
+    ) -> dict[str, int | float]:
+        """
+        Return the most recent (single) values from self.last_d,
+        optionally applying calibration.
+        """
+        if not self.new_state_event.wait(timeout=timeout):
+            raise TimeoutError(f"{self}: Timed out waiting for state after {timeout}s.")
+
+        with self.state_lock:
+            state = self._state
+
+        self.new_state_event.clear()
+
+        if state is None:
+            raise RuntimeError(f"{self} Internal error: Event set but no state available.")
+
+        if joints is not None:
+            state = {k: v for k, v in state.items() if k in joints}
+
+        # Apply EMA smoothing to raw values first
+        state = self._apply_ema(state)
+
+        # Then normalize if requested
+        if normalize:
+            state = self._normalize(state)
+
+        return state
+
+    def _read_loop(self):
+        """
+        Continuously read from the serial buffer in its own thread and sends values to the main thread through
+        a queue.
+        """
+        while not self.stop_event.is_set():
+            try:
+                positions = None
+                with self.serial_lock:
+                    if self.serial.in_waiting > 0:
+                        lines = []
+                        while self.serial.in_waiting > 0:
+                            line = self.serial.read_until().decode("utf-8").strip()
+                            if line:
+                                lines.append(line.split(" "))
+
+                        if lines:
+                            positions = lines[-1]
+
+                if positions is None or len(positions) != len(self.joints):
+                    continue
+
+                joint_positions = {joint: int(pos) for joint, pos in zip(self.joints, positions, strict=True)}
+
+                with self.state_lock:
+                    self._state = joint_positions
+                self.new_state_event.set()
+
+            except Exception as e:
+                logger.debug(f"Error reading frame in background thread for {self}: {e}")
+
+    @check_if_not_connected
+    def get_action(self) -> dict[str, float]:
+        joint_positions = self._read()
+        return homunculus_glove_to_hope_jr_hand(
+            {f"{joint}.pos": pos for joint, pos in joint_positions.items()}
+        )
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        raise NotImplementedError
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self.stop_event.set()
+        self.thread.join(timeout=1)
+        self.serial.close()
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/teleoperators/homunculus/joints_translation.py b/lerobot/src/lerobot/teleoperators/homunculus/joints_translation.py
new file mode 100644
index 0000000000000000000000000000000000000000..f14f7b3ef5b1d4d84bb9f3e6caa411d667775566
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/homunculus/joints_translation.py
@@ -0,0 +1,63 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+INDEX_SPLAY = 0.3
+MIDDLE_SPLAY = 0.3
+RING_SPLAY = 0.3
+PINKY_SPLAY = 0.5
+
+
+def get_ulnar_flexion(flexion: float, abduction: float, splay: float):
+    return -abduction * splay + flexion * (1 - splay)
+
+
+def get_radial_flexion(flexion: float, abduction: float, splay: float):
+    return abduction * splay + flexion * (1 - splay)
+
+
+def homunculus_glove_to_hope_jr_hand(glove_action: dict[str, float]) -> dict[str, float]:
+    return {
+        "thumb_cmc.pos": glove_action["thumb_cmc.pos"],
+        "thumb_mcp.pos": glove_action["thumb_mcp.pos"],
+        "thumb_pip.pos": glove_action["thumb_pip.pos"],
+        "thumb_dip.pos": glove_action["thumb_dip.pos"],
+        "index_radial_flexor.pos": get_radial_flexion(
+            glove_action["index_mcp_flexion.pos"], glove_action["index_mcp_abduction.pos"], INDEX_SPLAY
+        ),
+        "index_ulnar_flexor.pos": get_ulnar_flexion(
+            glove_action["index_mcp_flexion.pos"], glove_action["index_mcp_abduction.pos"], INDEX_SPLAY
+        ),
+        "index_pip_dip.pos": glove_action["index_dip.pos"],
+        "middle_radial_flexor.pos": get_radial_flexion(
+            glove_action["middle_mcp_flexion.pos"], glove_action["middle_mcp_abduction.pos"], MIDDLE_SPLAY
+        ),
+        "middle_ulnar_flexor.pos": get_ulnar_flexion(
+            glove_action["middle_mcp_flexion.pos"], glove_action["middle_mcp_abduction.pos"], MIDDLE_SPLAY
+        ),
+        "middle_pip_dip.pos": glove_action["middle_dip.pos"],
+        "ring_radial_flexor.pos": get_radial_flexion(
+            glove_action["ring_mcp_flexion.pos"], glove_action["ring_mcp_abduction.pos"], RING_SPLAY
+        ),
+        "ring_ulnar_flexor.pos": get_ulnar_flexion(
+            glove_action["ring_mcp_flexion.pos"], glove_action["ring_mcp_abduction.pos"], RING_SPLAY
+        ),
+        "ring_pip_dip.pos": glove_action["ring_dip.pos"],
+        "pinky_radial_flexor.pos": get_radial_flexion(
+            glove_action["pinky_mcp_flexion.pos"], glove_action["pinky_mcp_abduction.pos"], PINKY_SPLAY
+        ),
+        "pinky_ulnar_flexor.pos": get_ulnar_flexion(
+            glove_action["pinky_mcp_flexion.pos"], glove_action["pinky_mcp_abduction.pos"], PINKY_SPLAY
+        ),
+        "pinky_pip_dip.pos": glove_action["pinky_dip.pos"],
+    }
diff --git a/lerobot/src/lerobot/teleoperators/keyboard/__init__.py b/lerobot/src/lerobot/teleoperators/keyboard/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..c6b123b78018a5949cf12250cc313dcad839a71c
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/keyboard/__init__.py
@@ -0,0 +1,31 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .configuration_keyboard import (
+    KeyboardEndEffectorTeleopConfig,
+    KeyboardRoverTeleopConfig,
+    KeyboardTeleopConfig,
+)
+from .teleop_keyboard import KeyboardEndEffectorTeleop, KeyboardRoverTeleop, KeyboardTeleop
+
+__all__ = [
+    "KeyboardTeleopConfig",
+    "KeyboardTeleop",
+    "KeyboardEndEffectorTeleopConfig",
+    "KeyboardEndEffectorTeleop",
+    "KeyboardRoverTeleopConfig",
+    "KeyboardRoverTeleop",
+]
diff --git a/lerobot/src/lerobot/teleoperators/keyboard/configuration_keyboard.py b/lerobot/src/lerobot/teleoperators/keyboard/configuration_keyboard.py
new file mode 100644
index 0000000000000000000000000000000000000000..158717434edec9dabc3061dbf6ca724bee3d23a7
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/keyboard/configuration_keyboard.py
@@ -0,0 +1,68 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Configuration for keyboard teleoperators."""
+
+from dataclasses import dataclass
+
+from ..config import TeleoperatorConfig
+
+
+@TeleoperatorConfig.register_subclass("keyboard")
+@dataclass
+class KeyboardTeleopConfig(TeleoperatorConfig):
+    """KeyboardTeleopConfig"""
+
+    # TODO(Steven): Consider setting in here the keys that we want to capture/listen
+
+
+@TeleoperatorConfig.register_subclass("keyboard_ee")
+@dataclass
+class KeyboardEndEffectorTeleopConfig(KeyboardTeleopConfig):
+    """Configuration for keyboard end-effector teleoperator.
+
+    Used for controlling robot end-effectors with keyboard inputs.
+
+    Attributes:
+        use_gripper: Whether to include gripper control in actions
+    """
+
+    use_gripper: bool = True
+
+
+@TeleoperatorConfig.register_subclass("keyboard_rover")
+@dataclass
+class KeyboardRoverTeleopConfig(TeleoperatorConfig):
+    """Configuration for keyboard rover teleoperator.
+
+    Used for controlling mobile robots like EarthRover Mini Plus with WASD controls.
+
+    Attributes:
+        linear_speed: Default linear velocity magnitude (-1 to 1 range for SDK robots)
+        angular_speed: Default angular velocity magnitude (-1 to 1 range for SDK robots)
+        speed_increment: Amount to increase/decrease speed with +/- keys
+        turn_assist_ratio: Forward motion multiplier when turning with A/D keys (0.0-1.0)
+        angular_speed_ratio: Ratio of angular to linear speed for synchronized adjustments
+        min_linear_speed: Minimum linear speed when decreasing (prevents zero speed)
+        min_angular_speed: Minimum angular speed when decreasing (prevents zero speed)
+    """
+
+    linear_speed: float = 1.0
+    angular_speed: float = 1.0
+    speed_increment: float = 0.1
+    turn_assist_ratio: float = 0.3
+    angular_speed_ratio: float = 0.6
+    min_linear_speed: float = 0.1
+    min_angular_speed: float = 0.05
diff --git a/lerobot/src/lerobot/teleoperators/keyboard/teleop_keyboard.py b/lerobot/src/lerobot/teleoperators/keyboard/teleop_keyboard.py
new file mode 100644
index 0000000000000000000000000000000000000000..090aa7fae7c8609e405f3f8f68ce60aa6888b4e6
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/keyboard/teleop_keyboard.py
@@ -0,0 +1,432 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import os
+import sys
+import time
+from queue import Queue
+from typing import Any
+
+from lerobot.types import RobotAction
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..teleoperator import Teleoperator
+from ..utils import TeleopEvents
+from .configuration_keyboard import (
+    KeyboardEndEffectorTeleopConfig,
+    KeyboardRoverTeleopConfig,
+    KeyboardTeleopConfig,
+)
+
+PYNPUT_AVAILABLE = True
+try:
+    if ("DISPLAY" not in os.environ) and ("linux" in sys.platform):
+        logging.info("No DISPLAY set. Skipping pynput import.")
+        raise ImportError("pynput blocked intentionally due to no display.")
+
+    from pynput import keyboard
+except ImportError:
+    keyboard = None
+    PYNPUT_AVAILABLE = False
+except Exception as e:
+    keyboard = None
+    PYNPUT_AVAILABLE = False
+    logging.info(f"Could not import pynput: {e}")
+
+
+class KeyboardTeleop(Teleoperator):
+    """
+    Teleop class to use keyboard inputs for control.
+    """
+
+    config_class = KeyboardTeleopConfig
+    name = "keyboard"
+
+    def __init__(self, config: KeyboardTeleopConfig):
+        super().__init__(config)
+        self.config = config
+        self.robot_type = config.type
+
+        self.event_queue = Queue()
+        self.current_pressed = {}
+        self.listener = None
+        self.logs = {}
+
+    @property
+    def action_features(self) -> dict:
+        return {
+            "dtype": "float32",
+            "shape": (len(self.arm),),
+            "names": {"motors": list(self.arm.motors)},
+        }
+
+    @property
+    def feedback_features(self) -> dict:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        return PYNPUT_AVAILABLE and isinstance(self.listener, keyboard.Listener) and self.listener.is_alive()
+
+    @property
+    def is_calibrated(self) -> bool:
+        pass
+
+    @check_if_already_connected
+    def connect(self) -> None:
+        if PYNPUT_AVAILABLE:
+            logging.info("pynput is available - enabling local keyboard listener.")
+            self.listener = keyboard.Listener(
+                on_press=self._on_press,
+                on_release=self._on_release,
+            )
+            self.listener.start()
+        else:
+            logging.info("pynput not available - skipping local keyboard listener.")
+            self.listener = None
+
+    def calibrate(self) -> None:
+        pass
+
+    def _on_press(self, key):
+        if hasattr(key, "char"):
+            self.event_queue.put((key.char, True))
+
+    def _on_release(self, key):
+        if hasattr(key, "char"):
+            self.event_queue.put((key.char, False))
+        if key == keyboard.Key.esc:
+            logging.info("ESC pressed, disconnecting.")
+            self.disconnect()
+
+    def _drain_pressed_keys(self):
+        while not self.event_queue.empty():
+            key_char, is_pressed = self.event_queue.get_nowait()
+            self.current_pressed[key_char] = is_pressed
+
+    def configure(self):
+        pass
+
+    @check_if_not_connected
+    def get_action(self) -> RobotAction:
+        before_read_t = time.perf_counter()
+
+        self._drain_pressed_keys()
+
+        # Generate action based on current key states
+        action = {key for key, val in self.current_pressed.items() if val}
+        self.logs["read_pos_dt_s"] = time.perf_counter() - before_read_t
+
+        return dict.fromkeys(action, None)
+
+    def send_feedback(self, feedback: dict[str, Any]) -> None:
+        pass
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        if self.listener is not None:
+            self.listener.stop()
+
+
+class KeyboardEndEffectorTeleop(KeyboardTeleop):
+    """
+    Teleop class to use keyboard inputs for end effector control.
+    Designed to be used with the `So100FollowerEndEffector` robot.
+    """
+
+    config_class = KeyboardEndEffectorTeleopConfig
+    name = "keyboard_ee"
+
+    def __init__(self, config: KeyboardEndEffectorTeleopConfig):
+        super().__init__(config)
+        self.config = config
+        self.misc_keys_queue = Queue()
+
+    @property
+    def action_features(self) -> dict:
+        if self.config.use_gripper:
+            return {
+                "dtype": "float32",
+                "shape": (4,),
+                "names": {"delta_x": 0, "delta_y": 1, "delta_z": 2, "gripper": 3},
+            }
+        else:
+            return {
+                "dtype": "float32",
+                "shape": (3,),
+                "names": {"delta_x": 0, "delta_y": 1, "delta_z": 2},
+            }
+
+    @check_if_not_connected
+    def get_action(self) -> RobotAction:
+        self._drain_pressed_keys()
+        delta_x = 0.0
+        delta_y = 0.0
+        delta_z = 0.0
+        gripper_action = 1.0
+
+        # Generate action based on current key states
+        for key, val in self.current_pressed.items():
+            if key == keyboard.Key.up:
+                delta_y = -int(val)
+            elif key == keyboard.Key.down:
+                delta_y = int(val)
+            elif key == keyboard.Key.left:
+                delta_x = int(val)
+            elif key == keyboard.Key.right:
+                delta_x = -int(val)
+            elif key == keyboard.Key.shift:
+                delta_z = -int(val)
+            elif key == keyboard.Key.shift_r:
+                delta_z = int(val)
+            elif key == keyboard.Key.ctrl_r:
+                # Gripper actions are expected to be between 0 (close), 1 (stay), 2 (open)
+                gripper_action = int(val) + 1
+            elif key == keyboard.Key.ctrl_l:
+                gripper_action = int(val) - 1
+            elif val:
+                # If the key is pressed, add it to the misc_keys_queue
+                # this will record key presses that are not part of the delta_x, delta_y, delta_z
+                # this is useful for retrieving other events like interventions for RL, episode success, etc.
+                self.misc_keys_queue.put(key)
+
+        self.current_pressed.clear()
+
+        action_dict = {
+            "delta_x": delta_x,
+            "delta_y": delta_y,
+            "delta_z": delta_z,
+        }
+
+        if self.config.use_gripper:
+            action_dict["gripper"] = gripper_action
+
+        return action_dict
+
+    def get_teleop_events(self) -> dict[str, Any]:
+        """
+        Get extra control events from the keyboard such as intervention status,
+        episode termination, success indicators, etc.
+
+        Keyboard mappings:
+        - Any movement keys pressed = intervention active
+        - 's' key = success (terminate episode successfully)
+        - 'r' key = rerecord episode (terminate and rerecord)
+        - 'q' key = quit episode (terminate without success)
+
+        Returns:
+            Dictionary containing:
+                - is_intervention: bool - Whether human is currently intervening
+                - terminate_episode: bool - Whether to terminate the current episode
+                - success: bool - Whether the episode was successful
+                - rerecord_episode: bool - Whether to rerecord the episode
+        """
+        if not self.is_connected:
+            return {
+                TeleopEvents.IS_INTERVENTION: False,
+                TeleopEvents.TERMINATE_EPISODE: False,
+                TeleopEvents.SUCCESS: False,
+                TeleopEvents.RERECORD_EPISODE: False,
+            }
+
+        # Check if any movement keys are currently pressed (indicates intervention)
+        movement_keys = [
+            keyboard.Key.up,
+            keyboard.Key.down,
+            keyboard.Key.left,
+            keyboard.Key.right,
+            keyboard.Key.shift,
+            keyboard.Key.shift_r,
+            keyboard.Key.ctrl_r,
+            keyboard.Key.ctrl_l,
+        ]
+        is_intervention = any(self.current_pressed.get(key, False) for key in movement_keys)
+
+        # Check for episode control commands from misc_keys_queue
+        terminate_episode = False
+        success = False
+        rerecord_episode = False
+
+        # Process any pending misc keys
+        while not self.misc_keys_queue.empty():
+            key = self.misc_keys_queue.get_nowait()
+            if key == "s":
+                success = True
+            elif key == "r":
+                terminate_episode = True
+                rerecord_episode = True
+            elif key == "q":
+                terminate_episode = True
+                success = False
+
+        return {
+            TeleopEvents.IS_INTERVENTION: is_intervention,
+            TeleopEvents.TERMINATE_EPISODE: terminate_episode,
+            TeleopEvents.SUCCESS: success,
+            TeleopEvents.RERECORD_EPISODE: rerecord_episode,
+        }
+
+
+class KeyboardRoverTeleop(KeyboardTeleop):
+    """
+    Keyboard teleoperator for mobile robots like EarthRover Mini Plus.
+
+    Provides intuitive WASD-style controls for driving a mobile robot:
+    - Linear movement (forward/backward)
+    - Angular movement (turning/rotation)
+    - Speed adjustment
+    - Emergency stop
+
+    Keyboard Controls:
+        Movement:
+            - W: Move forward
+            - S: Move backward
+            - A: Turn left (with forward motion)
+            - D: Turn right (with forward motion)
+            - Q: Rotate left in place
+            - E: Rotate right in place
+            - X: Emergency stop
+
+        Speed Control:
+            - +/=: Increase speed
+            - -: Decrease speed
+
+        System:
+            - ESC: Disconnect teleoperator
+
+    Attributes:
+        config: Teleoperator configuration
+        current_linear_speed: Current linear velocity magnitude
+        current_angular_speed: Current angular velocity magnitude
+
+    Example:
+        ```python
+        from lerobot.teleoperators.keyboard import KeyboardRoverTeleop, KeyboardRoverTeleopConfig
+
+        teleop = KeyboardRoverTeleop(
+            KeyboardRoverTeleopConfig(linear_speed=1.0, angular_speed=1.0, speed_increment=0.1)
+        )
+        teleop.connect()
+
+        while teleop.is_connected:
+            action = teleop.get_action()
+            robot.send_action(action)
+        ```
+    """
+
+    config_class = KeyboardRoverTeleopConfig
+    name = "keyboard_rover"
+
+    def __init__(self, config: KeyboardRoverTeleopConfig):
+        super().__init__(config)
+        # Add rover-specific speed settings
+        self.current_linear_speed = config.linear_speed
+        self.current_angular_speed = config.angular_speed
+
+    @property
+    def action_features(self) -> dict:
+        """Return action format for rover (linear and angular velocities)."""
+        return {
+            "linear_velocity": float,
+            "angular_velocity": float,
+        }
+
+    @property
+    def is_calibrated(self) -> bool:
+        """Rover teleop doesn't require calibration."""
+        return True
+
+    def _drain_pressed_keys(self):
+        """Update current_pressed state from event queue without clearing held keys"""
+        while not self.event_queue.empty():
+            key_char, is_pressed = self.event_queue.get_nowait()
+            if is_pressed:
+                self.current_pressed[key_char] = True
+            else:
+                # Only remove key if it's being released
+                self.current_pressed.pop(key_char, None)
+
+    @check_if_not_connected
+    def get_action(self) -> RobotAction:
+        """
+        Get the current action based on pressed keys.
+
+        Returns:
+            RobotAction with 'linear_velocity' and 'angular_velocity' keys.
+        """
+        before_read_t = time.perf_counter()
+
+        self._drain_pressed_keys()
+
+        linear_velocity = 0.0
+        angular_velocity = 0.0
+
+        # Check which keys are currently pressed (not released)
+        active_keys = {key for key, is_pressed in self.current_pressed.items() if is_pressed}
+
+        # Linear movement (W/S) - these take priority
+        if "w" in active_keys:
+            linear_velocity = self.current_linear_speed
+        elif "s" in active_keys:
+            linear_velocity = -self.current_linear_speed
+
+        # Turning (A/D/Q/E)
+        if "d" in active_keys:
+            angular_velocity = -self.current_angular_speed
+            if linear_velocity == 0:  # If not moving forward/back, add slight forward motion
+                linear_velocity = self.current_linear_speed * self.config.turn_assist_ratio
+        elif "a" in active_keys:
+            angular_velocity = self.current_angular_speed
+            if linear_velocity == 0:  # If not moving forward/back, add slight forward motion
+                linear_velocity = self.current_linear_speed * self.config.turn_assist_ratio
+        elif "q" in active_keys:
+            angular_velocity = self.current_angular_speed
+            linear_velocity = 0  # Rotate in place
+        elif "e" in active_keys:
+            angular_velocity = -self.current_angular_speed
+            linear_velocity = 0  # Rotate in place
+
+        # Stop (X) - overrides everything
+        if "x" in active_keys:
+            linear_velocity = 0
+            angular_velocity = 0
+
+        # Speed adjustment
+        if "+" in active_keys or "=" in active_keys:
+            self.current_linear_speed += self.config.speed_increment
+            self.current_angular_speed += self.config.speed_increment * self.config.angular_speed_ratio
+            logging.info(
+                f"Speed increased: linear={self.current_linear_speed:.2f}, angular={self.current_angular_speed:.2f}"
+            )
+        if "-" in active_keys:
+            self.current_linear_speed = max(
+                self.config.min_linear_speed, self.current_linear_speed - self.config.speed_increment
+            )
+            self.current_angular_speed = max(
+                self.config.min_angular_speed,
+                self.current_angular_speed - self.config.speed_increment * self.config.angular_speed_ratio,
+            )
+            logging.info(
+                f"Speed decreased: linear={self.current_linear_speed:.2f}, angular={self.current_angular_speed:.2f}"
+            )
+
+        self.logs["read_pos_dt_s"] = time.perf_counter() - before_read_t
+
+        return {
+            "linear_velocity": linear_velocity,
+            "angular_velocity": angular_velocity,
+        }
diff --git a/lerobot/src/lerobot/teleoperators/koch_leader/__init__.py b/lerobot/src/lerobot/teleoperators/koch_leader/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..1bf9d51db6d00cf98786c262e1c9487a4019fab7
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/koch_leader/__init__.py
@@ -0,0 +1,18 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_koch_leader import KochLeaderConfig
+from .koch_leader import KochLeader
diff --git a/lerobot/src/lerobot/teleoperators/koch_leader/config_koch_leader.py b/lerobot/src/lerobot/teleoperators/koch_leader/config_koch_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..64aaae123581f1f7ff310b68ae234cd1ced3e872
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/koch_leader/config_koch_leader.py
@@ -0,0 +1,30 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..config import TeleoperatorConfig
+
+
+@TeleoperatorConfig.register_subclass("koch_leader")
+@dataclass
+class KochLeaderConfig(TeleoperatorConfig):
+    # Port to connect to the arm
+    port: str
+
+    # Sets the arm in torque mode with the gripper motor set to this value. This makes it possible to squeeze
+    # the gripper and have it spring back to an open position on its own.
+    gripper_open_pos: float = 50.0
diff --git a/lerobot/src/lerobot/teleoperators/koch_leader/koch_leader.py b/lerobot/src/lerobot/teleoperators/koch_leader/koch_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..87084b6b9e7972fdac9cf74d3548381add8cd713
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/koch_leader/koch_leader.py
@@ -0,0 +1,178 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.dynamixel import (
+    DriveMode,
+    DynamixelMotorsBus,
+    OperatingMode,
+)
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..teleoperator import Teleoperator
+from .config_koch_leader import KochLeaderConfig
+
+logger = logging.getLogger(__name__)
+
+
+class KochLeader(Teleoperator):
+    """
+    - [Koch v1.0](https://github.com/AlexanderKoch-Koch/low_cost_robot), with and without the wrist-to-elbow
+        expansion, developed by Alexander Koch from [Tau Robotics](https://tau-robotics.com)
+    - [Koch v1.1](https://github.com/jess-moss/koch-v1-1) developed by Jess Moss
+    """
+
+    config_class = KochLeaderConfig
+    name = "koch_leader"
+
+    def __init__(self, config: KochLeaderConfig):
+        super().__init__(config)
+        self.config = config
+        self.bus = DynamixelMotorsBus(
+            port=self.config.port,
+            motors={
+                "shoulder_pan": Motor(1, "xl330-m077", MotorNormMode.RANGE_M100_100),
+                "shoulder_lift": Motor(2, "xl330-m077", MotorNormMode.RANGE_M100_100),
+                "elbow_flex": Motor(3, "xl330-m077", MotorNormMode.RANGE_M100_100),
+                "wrist_flex": Motor(4, "xl330-m077", MotorNormMode.RANGE_M100_100),
+                "wrist_roll": Motor(5, "xl330-m077", MotorNormMode.RANGE_M100_100),
+                "gripper": Motor(6, "xl330-m077", MotorNormMode.RANGE_0_100),
+            },
+            calibration=self.calibration,
+        )
+
+    @property
+    def action_features(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.bus.motors}
+
+    @property
+    def feedback_features(self) -> dict[str, type]:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.bus.connect()
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        self.bus.disable_torque()
+        if self.calibration:
+            # Calibration file exists, ask user whether to use it or run new calibration
+            user_input = input(
+                f"Press ENTER to use provided calibration file associated with the id {self.id}, or type 'c' and press ENTER to run calibration: "
+            )
+            if user_input.strip().lower() != "c":
+                logger.info(f"Writing calibration file associated with the id {self.id} to the motors")
+                self.bus.write_calibration(self.calibration)
+                return
+        logger.info(f"\nRunning calibration of {self}")
+        for motor in self.bus.motors:
+            self.bus.write("Operating_Mode", motor, OperatingMode.EXTENDED_POSITION.value)
+
+        self.bus.write("Drive_Mode", "elbow_flex", DriveMode.INVERTED.value)
+        drive_modes = {motor: 1 if motor == "elbow_flex" else 0 for motor in self.bus.motors}
+
+        input(f"Move {self} to the middle of its range of motion and press ENTER....")
+        homing_offsets = self.bus.set_half_turn_homings()
+
+        full_turn_motors = ["shoulder_pan", "wrist_roll"]
+        unknown_range_motors = [motor for motor in self.bus.motors if motor not in full_turn_motors]
+        print(
+            f"Move all joints except {full_turn_motors} sequentially through their "
+            "entire ranges of motion.\nRecording positions. Press ENTER to stop..."
+        )
+        range_mins, range_maxes = self.bus.record_ranges_of_motion(unknown_range_motors)
+        for motor in full_turn_motors:
+            range_mins[motor] = 0
+            range_maxes[motor] = 4095
+
+        self.calibration = {}
+        for motor, m in self.bus.motors.items():
+            self.calibration[motor] = MotorCalibration(
+                id=m.id,
+                drive_mode=drive_modes[motor],
+                homing_offset=homing_offsets[motor],
+                range_min=range_mins[motor],
+                range_max=range_maxes[motor],
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        logger.info(f"Calibration saved to {self.calibration_fpath}")
+
+    def configure(self) -> None:
+        self.bus.disable_torque()
+        self.bus.configure_motors()
+        for motor in self.bus.motors:
+            if motor != "gripper":
+                # Use 'extended position mode' for all motors except gripper, because in joint mode the servos
+                # can't rotate more than 360 degrees (from 0 to 4095) And some mistake can happen while
+                # assembling the arm, you could end up with a servo with a position 0 or 4095 at a crucial
+                # point
+                self.bus.write("Operating_Mode", motor, OperatingMode.EXTENDED_POSITION.value)
+
+        # Use 'position control current based' for gripper to be limited by the limit of the current.
+        # For the follower gripper, it means it can grasp an object without forcing too much even tho,
+        # its goal position is a complete grasp (both gripper fingers are ordered to join and reach a touch).
+        # For the leader gripper, it means we can use it as a physical trigger, since we can force with our finger
+        # to make it move, and it will move back to its original target position when we release the force.
+        self.bus.write("Operating_Mode", "gripper", OperatingMode.CURRENT_POSITION.value)
+        # Set gripper's goal pos in current position mode so that we can use it as a trigger.
+        self.bus.enable_torque("gripper")
+        if self.is_calibrated:
+            self.bus.write("Goal_Position", "gripper", self.config.gripper_open_pos)
+
+    def setup_motors(self) -> None:
+        for motor in reversed(self.bus.motors):
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_action(self) -> dict[str, float]:
+        start = time.perf_counter()
+        action = self.bus.sync_read("Present_Position")
+        action = {f"{motor}.pos": val for motor, val in action.items()}
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read action: {dt_ms:.1f}ms")
+        return action
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        # TODO(rcadene, aliberts): Implement force feedback
+        raise NotImplementedError
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self.bus.disconnect()
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/teleoperators/omx_leader/__init__.py b/lerobot/src/lerobot/teleoperators/omx_leader/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..04d96d63e010ad55a7b055903a35c897e9880082
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/omx_leader/__init__.py
@@ -0,0 +1,18 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_omx_leader import OmxLeaderConfig
+from .omx_leader import OmxLeader
diff --git a/lerobot/src/lerobot/teleoperators/omx_leader/config_omx_leader.py b/lerobot/src/lerobot/teleoperators/omx_leader/config_omx_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..a0eca38f7471be8a3af26378a259323a721ab3e6
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/omx_leader/config_omx_leader.py
@@ -0,0 +1,30 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..config import TeleoperatorConfig
+
+
+@TeleoperatorConfig.register_subclass("omx_leader")
+@dataclass
+class OmxLeaderConfig(TeleoperatorConfig):
+    # Port to connect to the arm
+    port: str
+
+    # Sets the arm in torque mode with the gripper motor set to this value. This makes it possible to squeeze
+    # the gripper and have it spring back to an open position on its own.
+    gripper_open_pos: float = 60.0
diff --git a/lerobot/src/lerobot/teleoperators/omx_leader/omx_leader.py b/lerobot/src/lerobot/teleoperators/omx_leader/omx_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..4264b048530f90607857bc452b433efdee539ed6
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/omx_leader/omx_leader.py
@@ -0,0 +1,167 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.dynamixel import (
+    DriveMode,
+    DynamixelMotorsBus,
+    OperatingMode,
+)
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..teleoperator import Teleoperator
+from .config_omx_leader import OmxLeaderConfig
+
+logger = logging.getLogger(__name__)
+
+
+class OmxLeader(Teleoperator):
+    """
+    - [OMX](https://github.com/ROBOTIS-GIT/open_manipulator),
+        expansion, developed by Woojin Wie and Junha Cha from [ROBOTIS](https://ai.robotis.com/)
+    """
+
+    config_class = OmxLeaderConfig
+    name = "omx_leader"
+
+    def __init__(self, config: OmxLeaderConfig):
+        super().__init__(config)
+        self.config = config
+        self.bus = DynamixelMotorsBus(
+            port=self.config.port,
+            motors={
+                "shoulder_pan": Motor(1, "xl330-m288", MotorNormMode.RANGE_M100_100),
+                "shoulder_lift": Motor(2, "xl330-m288", MotorNormMode.RANGE_M100_100),
+                "elbow_flex": Motor(3, "xl330-m288", MotorNormMode.RANGE_M100_100),
+                "wrist_flex": Motor(4, "xl330-m288", MotorNormMode.RANGE_M100_100),
+                "wrist_roll": Motor(5, "xl330-m288", MotorNormMode.RANGE_M100_100),
+                "gripper": Motor(6, "xl330-m077", MotorNormMode.RANGE_0_100),
+            },
+            calibration=self.calibration,
+        )
+
+    @property
+    def action_features(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.bus.motors}
+
+    @property
+    def feedback_features(self) -> dict[str, type]:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.bus.connect()
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        self.bus.disable_torque()
+        logger.info(f"\nUsing factory default calibration values for {self}")
+        logger.info(f"\nWriting default configuration of {self} to the motors")
+        for motor in self.bus.motors:
+            self.bus.write("Operating_Mode", motor, OperatingMode.EXTENDED_POSITION.value)
+
+        for motor in self.bus.motors:
+            if motor == "gripper":
+                self.bus.write("Drive_Mode", motor, DriveMode.INVERTED.value)
+            else:
+                self.bus.write("Drive_Mode", motor, DriveMode.NON_INVERTED.value)
+        drive_modes = {motor: 1 if motor == "gripper" else 0 for motor in self.bus.motors}
+
+        self.calibration = {}
+        for motor, m in self.bus.motors.items():
+            self.calibration[motor] = MotorCalibration(
+                id=m.id,
+                drive_mode=drive_modes[motor],
+                homing_offset=0 if motor != "gripper" else 100,
+                range_min=0,
+                range_max=4095,
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        logger.info(f"Calibration saved to {self.calibration_fpath}")
+
+    def configure(self) -> None:
+        self.bus.disable_torque()
+        self.bus.configure_motors()
+        for motor in self.bus.motors:
+            if motor != "gripper":
+                # Use 'extended position mode' for all motors except gripper, because in joint mode the servos
+                # can't rotate more than 360 degrees (from 0 to 4095) And some mistake can happen while
+                # assembling the arm, you could end up with a servo with a position 0 or 4095 at a crucial
+                # point
+                self.bus.write("Operating_Mode", motor, OperatingMode.EXTENDED_POSITION.value)
+
+            if motor == "gripper":
+                self.bus.write("Drive_Mode", motor, DriveMode.INVERTED.value)
+            else:
+                self.bus.write("Drive_Mode", motor, DriveMode.NON_INVERTED.value)
+
+        # Use 'position control current based' for gripper to be limited by the limit of the current.
+        # For the follower gripper, it means it can grasp an object without forcing too much even tho,
+        # its goal position is a complete grasp (both gripper fingers are ordered to join and reach a touch).
+        # For the leader gripper, it means we can use it as a physical trigger, since we can force with our finger
+        # to make it move, and it will move back to its original target position when we release the force.
+        self.bus.write("Operating_Mode", "gripper", OperatingMode.CURRENT_POSITION.value)
+        self.bus.write("Current_Limit", "gripper", 100)
+        self.bus.write("Goal_Current", "gripper", 100)
+        self.bus.write("Homing_Offset", "gripper", 100)
+        # Set gripper's goal pos in current position mode so that we can use it as a trigger.
+        self.bus.enable_torque("gripper")
+        if self.is_calibrated:
+            self.bus.write("Goal_Position", "gripper", self.config.gripper_open_pos)
+
+    def setup_motors(self) -> None:
+        for motor in reversed(self.bus.motors):
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_action(self) -> dict[str, float]:
+        start = time.perf_counter()
+        action = self.bus.sync_read("Present_Position")
+        action = {f"{motor}.pos": val for motor, val in action.items()}
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read action: {dt_ms:.1f}ms")
+        return action
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        # TODO(rcadene, aliberts): Implement force feedback
+        raise NotImplementedError
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self.bus.disconnect()
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/teleoperators/openarm_leader/__init__.py b/lerobot/src/lerobot/teleoperators/openarm_leader/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..172cf82283d8411ac1d609604fab7cd9503058b3
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/openarm_leader/__init__.py
@@ -0,0 +1,20 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_openarm_leader import OpenArmLeaderConfig, OpenArmLeaderConfigBase
+from .openarm_leader import OpenArmLeader
+
+__all__ = ["OpenArmLeader", "OpenArmLeaderConfig", "OpenArmLeaderConfigBase"]
diff --git a/lerobot/src/lerobot/teleoperators/openarm_leader/config_openarm_leader.py b/lerobot/src/lerobot/teleoperators/openarm_leader/config_openarm_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..4b12fe7309d085d7773f261128018b9a8849dae8
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/openarm_leader/config_openarm_leader.py
@@ -0,0 +1,75 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from ..config import TeleoperatorConfig
+
+
+@dataclass
+class OpenArmLeaderConfigBase:
+    """Base configuration for the OpenArms leader/teleoperator with Damiao motors."""
+
+    # CAN interfaces - one per arm
+    # Arm CAN interface (e.g., "can3")
+    # Linux: "can0", "can1", etc.
+    port: str
+
+    # CAN interface type: "socketcan" (Linux), "slcan" (serial), or "auto" (auto-detect)
+    can_interface: str = "socketcan"
+
+    # CAN FD settings (OpenArms uses CAN FD by default)
+    use_can_fd: bool = True
+    can_bitrate: int = 1000000  # Nominal bitrate (1 Mbps)
+    can_data_bitrate: int = 5000000  # Data bitrate for CAN FD (5 Mbps)
+
+    # Motor configuration for OpenArms (7 DOF per arm)
+    # Maps motor names to (send_can_id, recv_can_id, motor_type)
+    # Based on: https://docs.openarm.dev/software/setup/configure-test
+    # OpenArms uses 4 types of motors:
+    # - DM8009 (DM-J8009P-2EC) for shoulders (high torque)
+    # - DM4340P and DM4340 for shoulder rotation and elbow
+    # - DM4310 (DM-J4310-2EC V1.1) for wrist and gripper
+    motor_config: dict[str, tuple[int, int, str]] = field(
+        default_factory=lambda: {
+            "joint_1": (0x01, 0x11, "dm8009"),  # J1 - Shoulder pan (DM8009)
+            "joint_2": (0x02, 0x12, "dm8009"),  # J2 - Shoulder lift (DM8009)
+            "joint_3": (0x03, 0x13, "dm4340"),  # J3 - Shoulder rotation (DM4340)
+            "joint_4": (0x04, 0x14, "dm4340"),  # J4 - Elbow flex (DM4340)
+            "joint_5": (0x05, 0x15, "dm4310"),  # J5 - Wrist roll (DM4310)
+            "joint_6": (0x06, 0x16, "dm4310"),  # J6 - Wrist pitch (DM4310)
+            "joint_7": (0x07, 0x17, "dm4310"),  # J7 - Wrist rotation (DM4310)
+            "gripper": (0x08, 0x18, "dm4310"),  # J8 - Gripper (DM4310)
+        }
+    )
+
+    # Torque mode settings for manual control
+    # When enabled, motors have torque disabled for manual movement
+    manual_control: bool = True
+
+    # TODO(Steven, Pepijn): Not used ... ?
+    # MIT control parameters (used when manual_control=False for torque control)
+    # List of 8 values: [joint_1, joint_2, joint_3, joint_4, joint_5, joint_6, joint_7, gripper]
+    position_kp: list[float] = field(
+        default_factory=lambda: [240.0, 240.0, 240.0, 240.0, 24.0, 31.0, 25.0, 16.0]
+    )
+    position_kd: list[float] = field(default_factory=lambda: [3.0, 3.0, 3.0, 3.0, 0.2, 0.2, 0.2, 0.2])
+
+
+@TeleoperatorConfig.register_subclass("openarm_leader")
+@dataclass
+class OpenArmLeaderConfig(TeleoperatorConfig, OpenArmLeaderConfigBase):
+    pass
diff --git a/lerobot/src/lerobot/teleoperators/openarm_leader/openarm_leader.py b/lerobot/src/lerobot/teleoperators/openarm_leader/openarm_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..65da7416a73227c430394631825beb2ea1e2ae44
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/openarm_leader/openarm_leader.py
@@ -0,0 +1,222 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from typing import Any
+
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.damiao import DamiaoMotorsBus
+from lerobot.types import RobotAction
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..teleoperator import Teleoperator
+from .config_openarm_leader import OpenArmLeaderConfig
+
+logger = logging.getLogger(__name__)
+
+
+class OpenArmLeader(Teleoperator):
+    """
+    OpenArm Leader/Teleoperator Arm with Damiao motors.
+
+    This teleoperator uses CAN bus communication to read positions from
+    Damiao motors that are manually moved (torque disabled).
+    """
+
+    config_class = OpenArmLeaderConfig
+    name = "openarm_leader"
+
+    def __init__(self, config: OpenArmLeaderConfig):
+        super().__init__(config)
+        self.config = config
+
+        # Arm motors
+        motors: dict[str, Motor] = {}
+        for motor_name, (send_id, recv_id, motor_type_str) in config.motor_config.items():
+            motor = Motor(
+                send_id, motor_type_str, MotorNormMode.DEGREES
+            )  # Always use degrees for Damiao motors
+            motor.recv_id = recv_id
+            motor.motor_type_str = motor_type_str
+            motors[motor_name] = motor
+
+        self.bus = DamiaoMotorsBus(
+            port=self.config.port,
+            motors=motors,
+            calibration=self.calibration,
+            can_interface=self.config.can_interface,
+            use_can_fd=self.config.use_can_fd,
+            bitrate=self.config.can_bitrate,
+            data_bitrate=self.config.can_data_bitrate if self.config.use_can_fd else None,
+        )
+
+    @property
+    def action_features(self) -> dict[str, type]:
+        """Features produced by this teleoperator."""
+        features: dict[str, type] = {}
+        for motor in self.bus.motors:
+            features[f"{motor}.pos"] = float
+            features[f"{motor}.vel"] = float
+            features[f"{motor}.torque"] = float
+        return features
+
+    @property
+    def feedback_features(self) -> dict[str, type]:
+        """Feedback features (not implemented for OpenArms)."""
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        """Check if teleoperator is connected."""
+        return self.bus.is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        """
+        Connect to the teleoperator.
+
+        For manual control, we disable torque after connecting so the
+        arm can be moved by hand.
+        """
+
+        # Connect to CAN bus
+        logger.info(f"Connecting arm on {self.config.port}...")
+        self.bus.connect()
+
+        # Run calibration if needed
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        self.configure()
+
+        if self.is_calibrated:
+            self.bus.set_zero_position()
+
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        """Check if teleoperator is calibrated."""
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        """
+        Run calibration procedure for OpenArms leader.
+
+        The calibration procedure:
+        1. Disable torque (if not already disabled)
+        2. Ask user to position arm in zero position (hanging with gripper closed)
+        3. Set this as zero position
+        4. Record range of motion for each joint
+        5. Save calibration
+        """
+        if self.calibration:
+            # Calibration file exists, ask user whether to use it or run new calibration
+            user_input = input(
+                f"Press ENTER to use provided calibration file associated with the id {self.id}, or type 'c' and press ENTER to run calibration: "
+            )
+            if user_input.strip().lower() != "c":
+                logger.info(f"Writing calibration file associated with the id {self.id} to the motors")
+                self.bus.write_calibration(self.calibration)
+                return
+
+        logger.info(f"\nRunning calibration for {self}")
+        self.bus.disable_torque()
+
+        # Step 1: Set zero position
+        input(
+            "\nCalibration: Set Zero Position)\n"
+            "Position the arm in the following configuration:\n"
+            "  - Arm hanging straight down\n"
+            "  - Gripper closed\n"
+            "Press ENTER when ready..."
+        )
+
+        # Set current position as zero for all motors
+        self.bus.set_zero_position()
+        logger.info("Arm zero position set.")
+
+        logger.info("Setting range: -90° to +90° by default for all joints")
+        # TODO(Steven, Pepijn): Check if MotorCalibration is actually needed here given that we only use Degrees
+        for motor_name, motor in self.bus.motors.items():
+            self.calibration[motor_name] = MotorCalibration(
+                id=motor.id,
+                drive_mode=0,
+                homing_offset=0,
+                range_min=-90,
+                range_max=90,
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        print(f"Calibration saved to {self.calibration_fpath}")
+
+    def configure(self) -> None:
+        """
+        Configure motors for manual teleoperation.
+
+        For manual control, we disable torque so the arm can be moved by hand.
+        """
+
+        return self.bus.disable_torque() if self.config.manual_control else self.bus.configure_motors()
+
+    def setup_motors(self) -> None:
+        raise NotImplementedError(
+            "Motor ID configuration is typically done via manufacturer tools for CAN motors."
+        )
+
+    @check_if_not_connected
+    def get_action(self) -> RobotAction:
+        """
+        Get current action from the leader arm.
+
+        This is the main method for teleoperators - it reads the current state
+        of the leader arm and returns it as an action that can be sent to a follower.
+
+        Reads all motor states (pos/vel/torque) in one CAN refresh cycle.
+        """
+        start = time.perf_counter()
+
+        action_dict: dict[str, Any] = {}
+
+        # Use sync_read_all_states to get pos/vel/torque in one go
+        states = self.bus.sync_read_all_states()
+        for motor in self.bus.motors:
+            state = states.get(motor, {})
+            action_dict[f"{motor}.pos"] = state.get("position")
+            action_dict[f"{motor}.vel"] = state.get("velocity")
+            action_dict[f"{motor}.torque"] = state.get("torque")
+
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read state: {dt_ms:.1f}ms")
+
+        return action_dict
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        raise NotImplementedError("Feedback is not yet implemented for OpenArm leader.")
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        """Disconnect from teleoperator."""
+
+        # Disconnect CAN bus
+        # For manual control, ensure torque is disabled before disconnecting
+        self.bus.disconnect(disable_torque=self.config.manual_control)
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/teleoperators/openarm_mini/__init__.py b/lerobot/src/lerobot/teleoperators/openarm_mini/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..8620af1d793d582bb2075ca4be1c9a762a2230a6
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/openarm_mini/__init__.py
@@ -0,0 +1,20 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_openarm_mini import OpenArmMiniConfig
+from .openarm_mini import OpenArmMini
+
+__all__ = ["OpenArmMini", "OpenArmMiniConfig"]
diff --git a/lerobot/src/lerobot/teleoperators/openarm_mini/config_openarm_mini.py b/lerobot/src/lerobot/teleoperators/openarm_mini/config_openarm_mini.py
new file mode 100644
index 0000000000000000000000000000000000000000..7dc3e021257cc8b269a6f8d7927ec7daacab4ce8
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/openarm_mini/config_openarm_mini.py
@@ -0,0 +1,30 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..config import TeleoperatorConfig
+
+
+@TeleoperatorConfig.register_subclass("openarm_mini")
+@dataclass
+class OpenArmMiniConfig(TeleoperatorConfig):
+    """Configuration for OpenArm Mini teleoperator with Feetech motors (dual arms)."""
+
+    port_right: str = "/dev/ttyUSB0"
+    port_left: str = "/dev/ttyUSB1"
+
+    use_degrees: bool = True
diff --git a/lerobot/src/lerobot/teleoperators/openarm_mini/openarm_mini.py b/lerobot/src/lerobot/teleoperators/openarm_mini/openarm_mini.py
new file mode 100644
index 0000000000000000000000000000000000000000..23594caa970c1ff0d5e2307d1ea5e8b1f32bcc1b
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/openarm_mini/openarm_mini.py
@@ -0,0 +1,296 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from typing import Any
+
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.feetech import (
+    FeetechMotorsBus,
+    OperatingMode,
+)
+from lerobot.types import RobotAction
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..teleoperator import Teleoperator
+from .config_openarm_mini import OpenArmMiniConfig
+
+logger = logging.getLogger(__name__)
+
+# Motors whose direction is inverted during readout
+RIGHT_MOTORS_TO_FLIP = ["joint_1", "joint_2", "joint_3", "joint_4", "joint_5"]
+LEFT_MOTORS_TO_FLIP = ["joint_1", "joint_3", "joint_4", "joint_5", "joint_6", "joint_7"]
+
+
+class OpenArmMini(Teleoperator):
+    """
+    OpenArm Mini Teleoperator with dual Feetech-based arms (8 motors per arm).
+
+    Each arm has 7 joints plus a gripper, using Feetech STS3215 servos.
+    """
+
+    config_class = OpenArmMiniConfig
+    name = "openarm_mini"
+
+    def __init__(self, config: OpenArmMiniConfig):
+        super().__init__(config)
+        self.config = config
+
+        norm_mode_body = MotorNormMode.DEGREES
+
+        motors_right = {
+            "joint_1": Motor(1, "sts3215", norm_mode_body),
+            "joint_2": Motor(2, "sts3215", norm_mode_body),
+            "joint_3": Motor(3, "sts3215", norm_mode_body),
+            "joint_4": Motor(4, "sts3215", norm_mode_body),
+            "joint_5": Motor(5, "sts3215", norm_mode_body),
+            "joint_6": Motor(6, "sts3215", norm_mode_body),
+            "joint_7": Motor(7, "sts3215", norm_mode_body),
+            "gripper": Motor(8, "sts3215", MotorNormMode.RANGE_0_100),
+        }
+
+        motors_left = {
+            "joint_1": Motor(1, "sts3215", norm_mode_body),
+            "joint_2": Motor(2, "sts3215", norm_mode_body),
+            "joint_3": Motor(3, "sts3215", norm_mode_body),
+            "joint_4": Motor(4, "sts3215", norm_mode_body),
+            "joint_5": Motor(5, "sts3215", norm_mode_body),
+            "joint_6": Motor(6, "sts3215", norm_mode_body),
+            "joint_7": Motor(7, "sts3215", norm_mode_body),
+            "gripper": Motor(8, "sts3215", MotorNormMode.RANGE_0_100),
+        }
+
+        cal_right = {
+            k.replace("right_", ""): v for k, v in (self.calibration or {}).items() if k.startswith("right_")
+        }
+        cal_left = {
+            k.replace("left_", ""): v for k, v in (self.calibration or {}).items() if k.startswith("left_")
+        }
+
+        self.bus_right = FeetechMotorsBus(
+            port=self.config.port_right,
+            motors=motors_right,
+            calibration=cal_right,
+        )
+
+        self.bus_left = FeetechMotorsBus(
+            port=self.config.port_left,
+            motors=motors_left,
+            calibration=cal_left,
+        )
+
+    @property
+    def action_features(self) -> dict[str, type]:
+        features: dict[str, type] = {}
+        for motor in self.bus_right.motors:
+            features[f"right_{motor}.pos"] = float
+        for motor in self.bus_left.motors:
+            features[f"left_{motor}.pos"] = float
+        return features
+
+    @property
+    def feedback_features(self) -> dict[str, type]:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus_right.is_connected and self.bus_left.is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        logger.info(f"Connecting right arm on {self.config.port_right}...")
+        self.bus_right.connect()
+        logger.info(f"Connecting left arm on {self.config.port_left}...")
+        self.bus_left.connect()
+
+        if calibrate:
+            self.calibrate()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus_right.is_calibrated and self.bus_left.is_calibrated
+
+    def calibrate(self) -> None:
+        """
+        Run calibration procedure for OpenArm Mini.
+
+        1. Disable torque
+        2. Ask user to position arms in hanging position with grippers closed
+        3. Set this as zero position via half-turn homing
+        4. Interactive gripper calibration (open/close positions)
+        5. Save calibration
+        """
+        if self.calibration:
+            user_input = input(
+                f"Press ENTER to use existing calibration for {self.id}, "
+                f"or type 'c' and press ENTER to run new calibration: "
+            )
+            if user_input.strip().lower() != "c":
+                logger.info(f"Using existing calibration for {self.id}")
+                cal_right = {
+                    k.replace("right_", ""): v for k, v in self.calibration.items() if k.startswith("right_")
+                }
+                cal_left = {
+                    k.replace("left_", ""): v for k, v in self.calibration.items() if k.startswith("left_")
+                }
+                self.bus_right.write_calibration(cal_right)
+                self.bus_left.write_calibration(cal_left)
+                return
+
+        logger.info(f"\nRunning calibration for {self}")
+
+        self._calibrate_arm("right", self.bus_right)
+        self._calibrate_arm("left", self.bus_left)
+
+        self._save_calibration()
+        print(f"\nCalibration complete and saved to {self.calibration_fpath}")
+
+    def _calibrate_arm(self, arm_name: str, bus: FeetechMotorsBus) -> None:
+        """Calibrate a single arm with Feetech motors."""
+        logger.info(f"\n=== Calibrating {arm_name.upper()} arm ===")
+
+        bus.disable_torque()
+
+        logger.info(f"Setting Phase to 12 for all motors in {arm_name.upper()} arm...")
+        for motor in bus.motors:
+            bus.write("Phase", motor, 12)
+
+        for motor in bus.motors:
+            bus.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+
+        input(
+            f"\nCalibration: Zero Position ({arm_name.upper()} arm)\n"
+            "Position the arm in the following configuration:\n"
+            "  - Arm hanging straight down\n"
+            "  - Gripper closed\n"
+            "Press ENTER when ready..."
+        )
+
+        homing_offsets = bus.set_half_turn_homings()
+        logger.info(f"{arm_name.capitalize()} arm zero position set.")
+
+        print(f"\nSetting motor ranges for {arm_name.upper()} arm\n")
+
+        if self.calibration is None:
+            self.calibration = {}
+
+        motor_resolution = bus.model_resolution_table[list(bus.motors.values())[0].model]
+        max_res = motor_resolution - 1
+
+        for motor_name, motor in bus.motors.items():
+            prefixed_name = f"{arm_name}_{motor_name}"
+
+            if motor_name == "gripper":
+                input(
+                    f"\nGripper Calibration ({arm_name.upper()} arm)\n"
+                    f"Step 1: CLOSE the gripper fully\n"
+                    f"Press ENTER when gripper is closed..."
+                )
+                closed_pos = bus.read("Present_Position", motor_name, normalize=False)
+                logger.info(f"  Gripper closed position recorded: {closed_pos}")
+
+                input("\nStep 2: OPEN the gripper fully\nPress ENTER when gripper is fully open...")
+                open_pos = bus.read("Present_Position", motor_name, normalize=False)
+                logger.info(f"  Gripper open position recorded: {open_pos}")
+
+                if closed_pos < open_pos:
+                    range_min = int(closed_pos)
+                    range_max = int(open_pos)
+                    drive_mode = 0
+                else:
+                    range_min = int(open_pos)
+                    range_max = int(closed_pos)
+                    drive_mode = 1
+
+                logger.info(
+                    f"  {prefixed_name}: range set to [{range_min}, {range_max}] "
+                    f"(0=closed, 100=open, drive_mode={drive_mode})"
+                )
+            else:
+                range_min = 0
+                range_max = max_res
+                drive_mode = 0
+                logger.info(f"  {prefixed_name}: range set to [0, {max_res}] (full motor range)")
+
+            self.calibration[prefixed_name] = MotorCalibration(
+                id=motor.id,
+                drive_mode=drive_mode,
+                homing_offset=homing_offsets[motor_name],
+                range_min=range_min,
+                range_max=range_max,
+            )
+
+        cal_for_bus = {
+            k.replace(f"{arm_name}_", ""): v
+            for k, v in self.calibration.items()
+            if k.startswith(f"{arm_name}_")
+        }
+        bus.write_calibration(cal_for_bus)
+
+    def configure(self) -> None:
+        self.bus_right.disable_torque()
+        self.bus_right.configure_motors()
+        for motor in self.bus_right.motors:
+            self.bus_right.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+
+        self.bus_left.disable_torque()
+        self.bus_left.configure_motors()
+        for motor in self.bus_left.motors:
+            self.bus_left.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+
+    def setup_motors(self) -> None:
+        print("\nSetting up RIGHT arm motors...")
+        for motor in reversed(self.bus_right.motors):
+            input(f"Connect the controller board to the RIGHT '{motor}' motor only and press enter.")
+            self.bus_right.setup_motor(motor)
+            print(f"RIGHT '{motor}' motor id set to {self.bus_right.motors[motor].id}")
+
+        print("\nSetting up LEFT arm motors...")
+        for motor in reversed(self.bus_left.motors):
+            input(f"Connect the controller board to the LEFT '{motor}' motor only and press enter.")
+            self.bus_left.setup_motor(motor)
+            print(f"LEFT '{motor}' motor id set to {self.bus_left.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_action(self) -> RobotAction:
+        """Get current action from both arms (read positions from all motors)."""
+        start = time.perf_counter()
+
+        right_positions = self.bus_right.sync_read("Present_Position")
+        left_positions = self.bus_left.sync_read("Present_Position")
+
+        action: dict[str, Any] = {}
+        for motor, val in right_positions.items():
+            action[f"right_{motor}.pos"] = -val if motor in RIGHT_MOTORS_TO_FLIP else val
+        for motor, val in left_positions.items():
+            action[f"left_{motor}.pos"] = -val if motor in LEFT_MOTORS_TO_FLIP else val
+
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read action: {dt_ms:.1f}ms")
+        return action
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        raise NotImplementedError("Feedback is not yet implemented for OpenArm Mini.")
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self.bus_right.disconnect()
+        self.bus_left.disconnect()
+        logger.info(f"{self} disconnected.")
diff --git a/lerobot/src/lerobot/teleoperators/phone/__init__.py b/lerobot/src/lerobot/teleoperators/phone/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..2b28c1f974624aba727ff5cd6311d60f5f4519f7
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/phone/__init__.py
@@ -0,0 +1,18 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_phone import PhoneConfig
+from .teleop_phone import Phone
diff --git a/lerobot/src/lerobot/teleoperators/phone/config_phone.py b/lerobot/src/lerobot/teleoperators/phone/config_phone.py
new file mode 100644
index 0000000000000000000000000000000000000000..380d5f5ff0b5ec63ae02892572f89a2ff90fd469
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/phone/config_phone.py
@@ -0,0 +1,36 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+from enum import Enum
+
+import numpy as np
+
+from ..config import TeleoperatorConfig
+
+
+class PhoneOS(Enum):
+    ANDROID = "android"
+    IOS = "ios"
+
+
+@TeleoperatorConfig.register_subclass("phone")
+@dataclass
+class PhoneConfig(TeleoperatorConfig):
+    phone_os: PhoneOS = PhoneOS.IOS
+    camera_offset = np.array(
+        [0.0, -0.02, 0.04]
+    )  # iPhone 14 Pro camera is 2cm off center and 4cm above center
diff --git a/lerobot/src/lerobot/teleoperators/phone/phone_processor.py b/lerobot/src/lerobot/teleoperators/phone/phone_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..c498bed7d21171b7dbdf7d167372138cb7e16a7d
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/phone/phone_processor.py
@@ -0,0 +1,111 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from lerobot.processor import ProcessorStepRegistry, RobotActionProcessorStep
+from lerobot.teleoperators.phone.config_phone import PhoneOS
+from lerobot.types import RobotAction
+
+
+@ProcessorStepRegistry.register("map_phone_action_to_robot_action")
+@dataclass
+class MapPhoneActionToRobotAction(RobotActionProcessorStep):
+    """
+    Maps calibrated phone pose actions to standardized robot action inputs.
+
+    This processor step acts as a bridge between the phone teleoperator's output
+    and the robot's expected action format. It remaps the phone's 6-DoF pose
+    (position and rotation) to the robot's target end-effector pose, applying
+    necessary axis inversions and swaps. It also interprets platform-specific
+    button presses to generate a gripper command.
+
+    Attributes:
+        platform: The operating system of the phone (iOS or Android), used
+            to determine the correct button mappings for the gripper.
+    """
+
+    # TODO(Steven): Gripper vel could be output of phone_teleop directly
+    platform: PhoneOS
+    _enabled_prev: bool = field(default=False, init=False, repr=False)
+
+    def action(self, action: RobotAction) -> RobotAction:
+        """
+        Processes the phone action dictionary to create a robot action dictionary.
+
+        Args:
+            act: The input action dictionary from the phone teleoperator.
+
+        Returns:
+            A new action dictionary formatted for the robot controller.
+
+        Raises:
+            ValueError: If 'pos' or 'rot' keys are missing from the input action.
+        """
+        # Pop them from the action
+        enabled = bool(action.pop("phone.enabled"))
+        pos = action.pop("phone.pos")
+        rot = action.pop("phone.rot")
+        inputs = action.pop("phone.raw_inputs")
+
+        if pos is None or rot is None:
+            raise ValueError("pos and rot must be present in action")
+
+        rotvec = rot.as_rotvec()  # Absolute orientation as rotvec
+
+        # Map certain inputs to certain actions
+        if self.platform == PhoneOS.IOS:
+            gripper_vel = float(inputs.get("a3", 0.0))
+        else:
+            a = float(inputs.get("reservedButtonA", 0.0))
+            b = float(inputs.get("reservedButtonB", 0.0))
+            gripper_vel = (
+                a - b
+            )  # Positive if a is pressed, negative if b is pressed, 0 if both or neither are pressed
+
+        # For some actions we need to invert the axis
+        action["enabled"] = enabled
+        action["target_x"] = -pos[1] if enabled else 0.0
+        action["target_y"] = pos[0] if enabled else 0.0
+        action["target_z"] = pos[2] if enabled else 0.0
+        action["target_wx"] = rotvec[1] if enabled else 0.0
+        action["target_wy"] = rotvec[0] if enabled else 0.0
+        action["target_wz"] = -rotvec[2] if enabled else 0.0
+        action["gripper_vel"] = gripper_vel  # Still send gripper action when disabled
+        return action
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        for feat in ["enabled", "pos", "rot", "raw_inputs"]:
+            features[PipelineFeatureType.ACTION].pop(f"phone.{feat}", None)
+
+        for feat in [
+            "enabled",
+            "target_x",
+            "target_y",
+            "target_z",
+            "target_wx",
+            "target_wy",
+            "target_wz",
+            "gripper_vel",
+        ]:
+            features[PipelineFeatureType.ACTION][f"{feat}"] = PolicyFeature(
+                type=FeatureType.ACTION, shape=(1,)
+            )
+
+        return features
diff --git a/lerobot/src/lerobot/teleoperators/phone/teleop_phone.py b/lerobot/src/lerobot/teleoperators/phone/teleop_phone.py
new file mode 100644
index 0000000000000000000000000000000000000000..221ee808395545d1c49237c962f08f816e6bf020
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/phone/teleop_phone.py
@@ -0,0 +1,415 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# Docs:
+# hebi: https://docs.hebi.us/tools.html#mobile-io
+# teleop: https://github.com/SpesRobotics/teleop
+
+import logging
+import threading
+import time
+
+import hebi
+import numpy as np
+from teleop import Teleop
+
+from lerobot.teleoperators.phone.config_phone import PhoneConfig, PhoneOS
+from lerobot.teleoperators.teleoperator import Teleoperator
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.rotation import Rotation
+
+logger = logging.getLogger(__name__)
+
+
+class BasePhone:
+    _enabled: bool = False
+    _calib_pos: np.ndarray | None = None
+    _calib_rot_inv: Rotation | None = None
+
+    def _reapply_position_calibration(self, pos: np.ndarray) -> None:
+        self._calib_pos = pos.copy()
+
+    @property
+    def is_calibrated(self) -> bool:
+        return (self._calib_pos is not None) and (self._calib_rot_inv is not None)
+
+    @property
+    def action_features(self) -> dict[str, type]:
+        return {
+            "phone.pos": np.ndarray,  # shape (3,)
+            "phone.rot": Rotation,  # scipy.spatial.transform.Rotation
+            "phone.raw_inputs": dict,  # analogs/buttons or webXR meta
+            "phone.enabled": bool,
+        }
+
+    @property
+    def feedback_features(self) -> dict[str, type]:
+        # No haptic or other feedback implemented yet
+        pass
+
+    def configure(self) -> None:
+        # No additional configuration required for phone teleop
+        pass
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        # We could add haptic feedback (vibrations) here, but it's not implemented yet
+        raise NotImplementedError
+
+
+class IOSPhone(BasePhone, Teleoperator):
+    name = "ios_phone"
+
+    def __init__(self, config: PhoneConfig):
+        super().__init__(config)
+        self.config = config
+        self._group = None
+
+    @property
+    def is_connected(self) -> bool:
+        return self._group is not None
+
+    @check_if_already_connected
+    def connect(self) -> None:
+        logger.info("Connecting to IPhone, make sure to open the HEBI Mobile I/O app.")
+        lookup = hebi.Lookup()
+        time.sleep(2.0)
+        group = lookup.get_group_from_names(["HEBI"], ["mobileIO"])
+        if group is None:
+            raise RuntimeError("Mobile I/O not found — check name/family settings in the app.")
+        self._group = group
+        logger.info(f"{self} connected to HEBI group with {group.size} module(s).")
+
+        self.calibrate()
+
+    def calibrate(self) -> None:
+        print(
+            "Hold the phone so that: top edge points forward in same direction as the robot (robot +x) and screen points up (robot +z)"
+        )
+        print("Press and hold B1 in the HEBI Mobile I/O app to capture this pose...\n")
+        position, rotation = self._wait_for_capture_trigger()
+        self._calib_pos = position.copy()
+        self._calib_rot_inv = rotation.inv()
+        self._enabled = False
+        print("Calibration done\n")
+
+    def _wait_for_capture_trigger(self) -> tuple[np.ndarray, Rotation]:
+        """
+        Blocks execution until the calibration trigger is detected from the iOS device.
+
+        This method enters a loop, continuously reading the phone's state. It waits for the user to press
+        and hold the 'B1' button in the HEBI Mobile I/O app. Once B1 is pressed, the loop breaks and
+        returns the phone's pose at that exact moment.
+
+        Returns:
+            A tuple containing the position (np.ndarray) and rotation (Rotation) of the phone at the
+            moment the trigger was activated.
+        """
+        while True:
+            has_pose, position, rotation, fb_pose = self._read_current_pose()
+            if not has_pose:
+                time.sleep(0.01)
+                continue
+
+            io = getattr(fb_pose, "io", None)
+            button_b = getattr(io, "b", None) if io is not None else None
+            button_b1_pressed = False
+            if button_b is not None:
+                button_b1_pressed = bool(button_b.get_int(1))
+            if button_b1_pressed:
+                return position, rotation
+
+            time.sleep(0.01)
+
+    def _read_current_pose(self) -> tuple[bool, np.ndarray | None, Rotation | None, object | None]:
+        """
+        Reads the instantaneous 6-DoF pose from the connected iOS device via the HEBI SDK.
+
+        This method fetches the latest feedback packet from the HEBI group, extracts the ARKit
+        position and orientation, and converts them into a standard format. It also applies a
+        configured camera offset to adjust the pose from the camera's frame to the phone's
+        physical frame.
+
+        Returns:
+            A tuple containing:
+            - A boolean indicating if a valid pose was successfully read.
+            - The 3D position as a NumPy array, or None if not available.
+            - The orientation as a `Rotation` object, or None if not available.
+            - The raw HEBI feedback object for accessing other data like button presses.
+        """
+        fbk = self._group.get_next_feedback()
+        pose = fbk[0]
+        ar_pos = getattr(pose, "ar_position", None)
+        ar_quat = getattr(pose, "ar_orientation", None)
+        if ar_pos is None or ar_quat is None:
+            return False, None, None, None
+        # HEBI provides orientation in w, x, y, z format.
+        # Scipy's Rotation expects x, y, z, w.
+        quat_xyzw = np.concatenate((ar_quat[1:], [ar_quat[0]]))  # wxyz to xyzw
+        rot = Rotation.from_quat(quat_xyzw)
+        pos = ar_pos - rot.apply(self.config.camera_offset)
+        return True, pos, rot, pose
+
+    @check_if_not_connected
+    def get_action(self) -> dict:
+        has_pose, raw_position, raw_rotation, fb_pose = self._read_current_pose()
+        if not has_pose or not self.is_calibrated:
+            return {}
+
+        # Collect raw inputs (B1 / analogs on iOS, move/scale on Android)
+        raw_inputs: dict[str, float | int | bool] = {}
+        io = getattr(fb_pose, "io", None)
+        if io is not None:
+            bank_a, bank_b = io.a, io.b
+            if bank_a:
+                for ch in range(1, 9):
+                    if bank_a.has_float(ch):
+                        raw_inputs[f"a{ch}"] = float(bank_a.get_float(ch))
+            if bank_b:
+                for ch in range(1, 9):
+                    if bank_b.has_int(ch):
+                        raw_inputs[f"b{ch}"] = int(bank_b.get_int(ch))
+                    elif hasattr(bank_b, "has_bool") and bank_b.has_bool(ch):
+                        raw_inputs[f"b{ch}"] = int(bank_b.get_bool(ch))
+
+        enable = bool(raw_inputs.get("b1", 0))
+
+        # Rising edge then re-capture calibration immediately from current raw pose
+        if enable and not self._enabled:
+            self._reapply_position_calibration(raw_position)
+
+        # Apply calibration
+        pos_cal = self._calib_rot_inv.apply(raw_position - self._calib_pos)
+        rot_cal = self._calib_rot_inv * raw_rotation
+
+        self._enabled = enable
+
+        return {
+            "phone.pos": pos_cal,
+            "phone.rot": rot_cal,
+            "phone.raw_inputs": raw_inputs,
+            "phone.enabled": self._enabled,
+        }
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self._group = None
+
+
+class AndroidPhone(BasePhone, Teleoperator):
+    name = "android_phone"
+
+    def __init__(self, config: PhoneConfig):
+        super().__init__(config)
+        self.config = config
+        self._teleop = None
+        self._teleop_thread = None
+        self._latest_pose = None
+        self._latest_message = None
+        self._android_lock = threading.Lock()
+
+    @property
+    def is_connected(self) -> bool:
+        return self._teleop is not None
+
+    @check_if_already_connected
+    def connect(self) -> None:
+        logger.info("Starting teleop stream for Android...")
+        self._teleop = Teleop()
+        self._teleop.subscribe(self._android_callback)
+        self._teleop_thread = threading.Thread(target=self._teleop.run, daemon=True)
+        self._teleop_thread.start()
+        logger.info(f"{self} connected, teleop stream started.")
+
+        self.calibrate()
+
+    def calibrate(self) -> None:
+        print(
+            "Hold the phone so that: top edge points forward in same direction as the robot (robot +x) and screen points up (robot +z)"
+        )
+        print("Touch and move on the WebXR page to capture this pose...\n")
+
+        pos, rot = self._wait_for_capture_trigger()
+        self._calib_pos = pos.copy()
+        self._calib_rot_inv = rot.inv()
+        self._enabled = False
+        print("Calibration done\n")
+
+    def _wait_for_capture_trigger(self) -> tuple[np.ndarray, Rotation]:
+        """
+        Blocks execution until the calibration trigger is detected from the Android device.
+
+        This method enters a loop, continuously checking the latest message received from the WebXR
+        session. It waits for the user to touch and move their finger on the screen, which generates
+        a `move` event. Once this event is detected, the loop breaks and returns the phone's current
+        pose.
+
+        Returns:
+            A tuple containing the position (np.ndarray) and rotation (Rotation) of the phone at the
+            moment the trigger was activated.
+        """
+        while True:
+            with self._android_lock:
+                msg = self._latest_message or {}
+
+            if bool(msg.get("move", False)):
+                ok, pos, rot, _pose = self._read_current_pose()
+                if ok:
+                    return pos, rot
+
+            time.sleep(0.01)
+
+    def _read_current_pose(self) -> tuple[bool, np.ndarray | None, Rotation | None, object | None]:
+        """
+        Reads the latest 6-DoF pose received from the Android device's WebXR session.
+
+        This method accesses the most recent pose data stored by the `_android_callback`. It uses a
+        thread lock to safely read the shared `_latest_pose` variable. The pose, a 4x4 matrix, is
+        then decomposed into position and rotation, and the configured camera offset is applied.
+
+        Returns:
+            A tuple containing:
+            - A boolean indicating if a valid pose was available.
+            - The 3D position as a NumPy array, or None if no pose has been received yet.
+            - The orientation as a `Rotation` object, or None if no pose has been received.
+            - The raw 4x4 pose matrix as received from the teleop stream.
+        """
+        with self._android_lock:
+            if self._latest_pose is None:
+                return False, None, None, None
+            p = self._latest_pose.copy()
+            pose = self._latest_pose
+        rot = Rotation.from_matrix(p[:3, :3])
+        pos = p[:3, 3] - rot.apply(self.config.camera_offset)
+        return True, pos, rot, pose
+
+    def _android_callback(self, pose: np.ndarray, message: dict) -> None:
+        """
+        Callback function to handle incoming data from the Android teleop stream.
+
+        This method is executed by the `teleop` package's subscriber thread whenever a new
+        pose and message are received from the WebXR session on the Android phone. It updates
+        the internal state (`_latest_pose` and `_latest_message`) with the new data.
+        A thread lock is used to ensure that these shared variables are updated atomically,
+        preventing race conditions with the main thread that reads them.
+
+        Args:
+            pose: A 4x4 NumPy array representing the phone's transformation matrix.
+            message: A dictionary containing additional data, such as button presses or touch events.
+        """
+        with self._android_lock:
+            self._latest_pose = pose
+            self._latest_message = message
+
+    @check_if_not_connected
+    def get_action(self) -> dict:
+        ok, raw_pos, raw_rot, pose = self._read_current_pose()
+        if not ok or not self.is_calibrated:
+            return {}
+
+        # Collect raw inputs (B1 / analogs on iOS, move/scale on Android)
+        raw_inputs: dict[str, float | int | bool] = {}
+        msg = self._latest_message or {}
+        raw_inputs["move"] = bool(msg.get("move", False))
+        raw_inputs["scale"] = float(msg.get("scale", 1.0))
+        raw_inputs["reservedButtonA"] = bool(msg.get("reservedButtonA", False))
+        raw_inputs["reservedButtonB"] = bool(msg.get("reservedButtonB", False))
+
+        enable = bool(raw_inputs.get("move", False))
+
+        # Rising edge then re-capture calibration immediately from current raw pose
+        if enable and not self._enabled:
+            self._reapply_position_calibration(raw_pos)
+
+        # Apply calibration
+        pos_cal = self._calib_rot_inv.apply(raw_pos - self._calib_pos)
+        rot_cal = self._calib_rot_inv * raw_rot
+
+        self._enabled = enable
+
+        return {
+            "phone.pos": pos_cal,
+            "phone.rot": rot_cal,
+            "phone.raw_inputs": raw_inputs,
+            "phone.enabled": self._enabled,
+        }
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self._teleop = None
+        if self._teleop_thread and self._teleop_thread.is_alive():
+            self._teleop_thread.join(timeout=1.0)
+            self._teleop_thread = None
+            self._latest_pose = None
+
+
+class Phone(Teleoperator):
+    """
+    Phone-based teleoperator using ARKit (iOS via HEBI Mobile I/O App) or the teleop Python package (Android via WebXR API).
+    For HEBI Mobile I/O we also expose 8 analog (a1-a8) and 8 digital (b1-b8) inputs.
+
+    Press and hold **B1** to enable teleoperation. While enabled, the first B1 press
+    captures a reference pose and rotation, when disabled and pressed again the position is reapplied.
+    """
+
+    config_class = PhoneConfig
+    name = "phone"
+
+    def __init__(self, config: PhoneConfig):
+        super().__init__(config)
+        self.config = config
+
+        self._phone_impl: Teleoperator
+
+        if self.config.phone_os == PhoneOS.IOS:
+            self._phone_impl = IOSPhone(config)
+        elif self.config.phone_os == PhoneOS.ANDROID:
+            self._phone_impl = AndroidPhone(config)
+        else:
+            raise ValueError(f"Invalid config phone_os: {self.config.phone_os}")
+
+    @property
+    def is_connected(self) -> bool:
+        return self._phone_impl.is_connected
+
+    def connect(self) -> None:
+        return self._phone_impl.connect()
+
+    def calibrate(self) -> None:
+        return self._phone_impl.calibrate()
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self._phone_impl.is_calibrated
+
+    @property
+    def action_features(self) -> dict[str, type]:
+        return self._phone_impl.action_features
+
+    @property
+    def feedback_features(self) -> dict[str, type]:
+        return self._phone_impl.feedback_features
+
+    def configure(self) -> None:
+        return self._phone_impl.configure()
+
+    def get_action(self) -> dict:
+        return self._phone_impl.get_action()
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        return self._phone_impl.send_feedback(feedback)
+
+    def disconnect(self) -> None:
+        return self._phone_impl.disconnect()
diff --git a/lerobot/src/lerobot/teleoperators/reachy2_teleoperator/__init__.py b/lerobot/src/lerobot/teleoperators/reachy2_teleoperator/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..a07a4a6cd1483e93e9972aa81bf397a4a938bbbf
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/reachy2_teleoperator/__init__.py
@@ -0,0 +1,25 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_reachy2_teleoperator import Reachy2TeleoperatorConfig
+from .reachy2_teleoperator import (
+    REACHY2_ANTENNAS_JOINTS,
+    REACHY2_L_ARM_JOINTS,
+    REACHY2_NECK_JOINTS,
+    REACHY2_R_ARM_JOINTS,
+    REACHY2_VEL,
+    Reachy2Teleoperator,
+)
diff --git a/lerobot/src/lerobot/teleoperators/reachy2_teleoperator/config_reachy2_teleoperator.py b/lerobot/src/lerobot/teleoperators/reachy2_teleoperator/config_reachy2_teleoperator.py
new file mode 100644
index 0000000000000000000000000000000000000000..4e615d36328b17417b15bcb1c07f3b12c0665202
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/reachy2_teleoperator/config_reachy2_teleoperator.py
@@ -0,0 +1,51 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..config import TeleoperatorConfig
+
+
+@TeleoperatorConfig.register_subclass("reachy2_teleoperator")
+@dataclass
+class Reachy2TeleoperatorConfig(TeleoperatorConfig):
+    # IP address of the Reachy 2 robot used as teleoperator
+    ip_address: str | None = "localhost"
+
+    # Whether to use the present position of the joints as actions
+    # if False, the goal position of the joints will be used
+    use_present_position: bool = False
+
+    # Which parts of the robot to use
+    with_mobile_base: bool = True
+    with_l_arm: bool = True
+    with_r_arm: bool = True
+    with_neck: bool = True
+    with_antennas: bool = True
+
+    def __post_init__(self):
+        if not (
+            self.with_mobile_base
+            or self.with_l_arm
+            or self.with_r_arm
+            or self.with_neck
+            or self.with_antennas
+        ):
+            raise ValueError(
+                "No Reachy2Teleoperator part used.\n"
+                "At least one part of the robot must be set to True "
+                "(with_mobile_base, with_l_arm, with_r_arm, with_neck, with_antennas)"
+            )
diff --git a/lerobot/src/lerobot/teleoperators/reachy2_teleoperator/reachy2_teleoperator.py b/lerobot/src/lerobot/teleoperators/reachy2_teleoperator/reachy2_teleoperator.py
new file mode 100644
index 0000000000000000000000000000000000000000..db076b20fd1e982d1ba39ca6a182ee1c2f51d509
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/reachy2_teleoperator/reachy2_teleoperator.py
@@ -0,0 +1,176 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from __future__ import annotations
+
+import logging
+import time
+from typing import TYPE_CHECKING
+
+from lerobot.utils.import_utils import _reachy2_sdk_available
+
+if TYPE_CHECKING or _reachy2_sdk_available:
+    from reachy2_sdk import ReachySDK
+else:
+    ReachySDK = None
+
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from lerobot.utils.errors import DeviceNotConnectedError
+
+from ..teleoperator import Teleoperator
+from .config_reachy2_teleoperator import Reachy2TeleoperatorConfig
+
+logger = logging.getLogger(__name__)
+
+# {lerobot_keys: reachy2_sdk_keys}
+REACHY2_NECK_JOINTS = {
+    "neck_yaw.pos": "head.neck.yaw",
+    "neck_pitch.pos": "head.neck.pitch",
+    "neck_roll.pos": "head.neck.roll",
+}
+
+REACHY2_ANTENNAS_JOINTS = {
+    "l_antenna.pos": "head.l_antenna",
+    "r_antenna.pos": "head.r_antenna",
+}
+
+REACHY2_R_ARM_JOINTS = {
+    "r_shoulder_pitch.pos": "r_arm.shoulder.pitch",
+    "r_shoulder_roll.pos": "r_arm.shoulder.roll",
+    "r_elbow_yaw.pos": "r_arm.elbow.yaw",
+    "r_elbow_pitch.pos": "r_arm.elbow.pitch",
+    "r_wrist_roll.pos": "r_arm.wrist.roll",
+    "r_wrist_pitch.pos": "r_arm.wrist.pitch",
+    "r_wrist_yaw.pos": "r_arm.wrist.yaw",
+    "r_gripper.pos": "r_arm.gripper",
+}
+
+REACHY2_L_ARM_JOINTS = {
+    "l_shoulder_pitch.pos": "l_arm.shoulder.pitch",
+    "l_shoulder_roll.pos": "l_arm.shoulder.roll",
+    "l_elbow_yaw.pos": "l_arm.elbow.yaw",
+    "l_elbow_pitch.pos": "l_arm.elbow.pitch",
+    "l_wrist_roll.pos": "l_arm.wrist.roll",
+    "l_wrist_pitch.pos": "l_arm.wrist.pitch",
+    "l_wrist_yaw.pos": "l_arm.wrist.yaw",
+    "l_gripper.pos": "l_arm.gripper",
+}
+
+REACHY2_VEL = {
+    "mobile_base.vx": "vx",
+    "mobile_base.vy": "vy",
+    "mobile_base.vtheta": "vtheta",
+}
+
+
+class Reachy2Teleoperator(Teleoperator):
+    """
+    [Reachy 2](https://www.pollen-robotics.com/reachy/), by Pollen Robotics.
+    """
+
+    config_class = Reachy2TeleoperatorConfig
+    name = "reachy2_specific"
+
+    def __init__(self, config: Reachy2TeleoperatorConfig):
+        super().__init__(config)
+
+        self.config = config
+        self.reachy: None | ReachySDK = None
+
+        self.joints_dict: dict[str, str] = self._generate_joints_dict()
+
+    def _generate_joints_dict(self) -> dict[str, str]:
+        joints = {}
+        if self.config.with_neck:
+            joints.update(REACHY2_NECK_JOINTS)
+        if self.config.with_l_arm:
+            joints.update(REACHY2_L_ARM_JOINTS)
+        if self.config.with_r_arm:
+            joints.update(REACHY2_R_ARM_JOINTS)
+        if self.config.with_antennas:
+            joints.update(REACHY2_ANTENNAS_JOINTS)
+        return joints
+
+    @property
+    def action_features(self) -> dict[str, type]:
+        if self.config.with_mobile_base:
+            return {
+                **dict.fromkeys(
+                    self.joints_dict.keys(),
+                    float,
+                ),
+                **dict.fromkeys(
+                    REACHY2_VEL.keys(),
+                    float,
+                ),
+            }
+        else:
+            return dict.fromkeys(self.joints_dict.keys(), float)
+
+    @property
+    def feedback_features(self) -> dict[str, type]:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        return self.reachy.is_connected() if self.reachy is not None else False
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.reachy = ReachySDK(self.config.ip_address)
+
+        if not self.is_connected:
+            raise DeviceNotConnectedError()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return True
+
+    def calibrate(self) -> None:
+        pass
+
+    def configure(self) -> None:
+        pass
+
+    @check_if_not_connected
+    def get_action(self) -> dict[str, float]:
+        start = time.perf_counter()
+
+        joint_action: dict[str, float] = {}
+        vel_action: dict[str, float] = {}
+
+        if self.config.use_present_position:
+            joint_action = {k: self.reachy.joints[v].present_position for k, v in self.joints_dict.items()}
+        else:
+            joint_action = {k: self.reachy.joints[v].goal_position for k, v in self.joints_dict.items()}
+        if not self.config.with_mobile_base:
+            dt_ms = (time.perf_counter() - start) * 1e3
+            logger.debug(f"{self} read action: {dt_ms:.1f}ms")
+            return joint_action
+        if self.config.use_present_position:
+            vel_action = {k: self.reachy.mobile_base.odometry[v] for k, v in REACHY2_VEL.items()}
+        else:
+            vel_action = {k: self.reachy.mobile_base.last_cmd_vel[v] for k, v in REACHY2_VEL.items()}
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read action: {dt_ms:.1f}ms")
+        return {**joint_action, **vel_action}
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        raise NotImplementedError
+
+    def disconnect(self) -> None:
+        if self.is_connected:
+            self.reachy.disconnect()
diff --git a/lerobot/src/lerobot/teleoperators/so_leader/__init__.py b/lerobot/src/lerobot/teleoperators/so_leader/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..e5aaa31b62b25a91b0a39581da5c0a6de61b5ddd
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/so_leader/__init__.py
@@ -0,0 +1,23 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_so_leader import (
+    SO100LeaderConfig,
+    SO101LeaderConfig,
+    SOLeaderConfig,
+    SOLeaderTeleopConfig,
+)
+from .so_leader import SO100Leader, SO101Leader, SOLeader
diff --git a/lerobot/src/lerobot/teleoperators/so_leader/config_so_leader.py b/lerobot/src/lerobot/teleoperators/so_leader/config_so_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..189303088b19a07142acd8d6e8780f47f4eff58d
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/so_leader/config_so_leader.py
@@ -0,0 +1,41 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass
+
+from ..config import TeleoperatorConfig
+
+
+@dataclass
+class SOLeaderConfig:
+    """Base configuration class for SO Leader teleoperators."""
+
+    # Port to connect to the arm
+    port: str
+
+    # Whether to use degrees for angles
+    use_degrees: bool = True
+
+
+@TeleoperatorConfig.register_subclass("so101_leader")
+@TeleoperatorConfig.register_subclass("so100_leader")
+@dataclass
+class SOLeaderTeleopConfig(TeleoperatorConfig, SOLeaderConfig):
+    pass
+
+
+SO100LeaderConfig = SOLeaderTeleopConfig
+SO101LeaderConfig = SOLeaderTeleopConfig
diff --git a/lerobot/src/lerobot/teleoperators/so_leader/so100.md b/lerobot/src/lerobot/teleoperators/so_leader/so100.md
new file mode 120000
index 0000000000000000000000000000000000000000..ad1154e75a74a496aa74cb1ac1b545238d5174e4
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/so_leader/so100.md
@@ -0,0 +1 @@
+../../../../docs/source/so100.mdx
\ No newline at end of file
diff --git a/lerobot/src/lerobot/teleoperators/so_leader/so101.md b/lerobot/src/lerobot/teleoperators/so_leader/so101.md
new file mode 120000
index 0000000000000000000000000000000000000000..27b89266029afbf0aa59be195cc0b4b6ee93ac26
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/so_leader/so101.md
@@ -0,0 +1 @@
+../../../../docs/source/so101.mdx
\ No newline at end of file
diff --git a/lerobot/src/lerobot/teleoperators/so_leader/so_leader.py b/lerobot/src/lerobot/teleoperators/so_leader/so_leader.py
new file mode 100644
index 0000000000000000000000000000000000000000..04ce0f21f8abd6f147d24b4748d6941216996c2a
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/so_leader/so_leader.py
@@ -0,0 +1,159 @@
+# !/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.feetech import (
+    FeetechMotorsBus,
+    OperatingMode,
+)
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..teleoperator import Teleoperator
+from .config_so_leader import SOLeaderTeleopConfig
+
+logger = logging.getLogger(__name__)
+
+
+class SOLeader(Teleoperator):
+    """Generic SO leader base for SO-100/101/10X teleoperators."""
+
+    config_class = SOLeaderTeleopConfig
+    name = "so_leader"
+
+    def __init__(self, config: SOLeaderTeleopConfig):
+        super().__init__(config)
+        self.config = config
+        norm_mode_body = MotorNormMode.DEGREES if config.use_degrees else MotorNormMode.RANGE_M100_100
+        self.bus = FeetechMotorsBus(
+            port=self.config.port,
+            motors={
+                "shoulder_pan": Motor(1, "sts3215", norm_mode_body),
+                "shoulder_lift": Motor(2, "sts3215", norm_mode_body),
+                "elbow_flex": Motor(3, "sts3215", norm_mode_body),
+                "wrist_flex": Motor(4, "sts3215", norm_mode_body),
+                "wrist_roll": Motor(5, "sts3215", norm_mode_body),
+                "gripper": Motor(6, "sts3215", MotorNormMode.RANGE_0_100),
+            },
+            calibration=self.calibration,
+        )
+
+    @property
+    def action_features(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.bus.motors}
+
+    @property
+    def feedback_features(self) -> dict[str, type]:
+        return {}
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self.bus.connect()
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def calibrate(self) -> None:
+        if self.calibration:
+            # Calibration file exists, ask user whether to use it or run new calibration
+            user_input = input(
+                f"Press ENTER to use provided calibration file associated with the id {self.id}, or type 'c' and press ENTER to run calibration: "
+            )
+            if user_input.strip().lower() != "c":
+                logger.info(f"Writing calibration file associated with the id {self.id} to the motors")
+                self.bus.write_calibration(self.calibration)
+                return
+
+        logger.info(f"\nRunning calibration of {self}")
+        self.bus.disable_torque()
+        for motor in self.bus.motors:
+            self.bus.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+
+        input(f"Move {self} to the middle of its range of motion and press ENTER....")
+        homing_offsets = self.bus.set_half_turn_homings()
+
+        full_turn_motor = "wrist_roll"
+        unknown_range_motors = [motor for motor in self.bus.motors if motor != full_turn_motor]
+        print(
+            f"Move all joints except '{full_turn_motor}' sequentially through their "
+            "entire ranges of motion.\nRecording positions. Press ENTER to stop..."
+        )
+        range_mins, range_maxes = self.bus.record_ranges_of_motion(unknown_range_motors)
+        range_mins[full_turn_motor] = 0
+        range_maxes[full_turn_motor] = 4095
+
+        self.calibration = {}
+        for motor, m in self.bus.motors.items():
+            self.calibration[motor] = MotorCalibration(
+                id=m.id,
+                drive_mode=0,
+                homing_offset=homing_offsets[motor],
+                range_min=range_mins[motor],
+                range_max=range_maxes[motor],
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        print(f"Calibration saved to {self.calibration_fpath}")
+
+    def configure(self) -> None:
+        self.bus.disable_torque()
+        self.bus.configure_motors()
+        for motor in self.bus.motors:
+            self.bus.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+
+    def setup_motors(self) -> None:
+        for motor in reversed(self.bus.motors):
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_action(self) -> dict[str, float]:
+        start = time.perf_counter()
+        action = self.bus.sync_read("Present_Position")
+        action = {f"{motor}.pos": val for motor, val in action.items()}
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read action: {dt_ms:.1f}ms")
+        return action
+
+    def send_feedback(self, feedback: dict[str, float]) -> None:
+        # TODO: Implement force feedback
+        raise NotImplementedError
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self.bus.disconnect()
+        logger.info(f"{self} disconnected.")
+
+
+SO100Leader = SOLeader
+SO101Leader = SOLeader
diff --git a/lerobot/src/lerobot/teleoperators/teleoperator.py b/lerobot/src/lerobot/teleoperators/teleoperator.py
new file mode 100644
index 0000000000000000000000000000000000000000..f4790442351131272adf43e4c64c927704b404a3
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/teleoperator.py
@@ -0,0 +1,208 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import abc
+import builtins
+from pathlib import Path
+from typing import Any
+
+import draccus
+
+from lerobot.motors.motors_bus import MotorCalibration
+from lerobot.types import RobotAction
+from lerobot.utils.constants import HF_LEROBOT_CALIBRATION, TELEOPERATORS
+
+from .config import TeleoperatorConfig
+
+
+class Teleoperator(abc.ABC):
+    """
+    The base abstract class for all LeRobot-compatible teleoperation devices.
+
+    This class provides a standardized interface for interacting with physical teleoperators.
+    Subclasses must implement all abstract methods and properties to be usable.
+
+    Attributes:
+        config_class (RobotConfig): The expected configuration class for this teleoperator.
+        name (str): The unique name used to identify this teleoperator type.
+    """
+
+    # Set these in ALL subclasses
+    config_class: builtins.type[TeleoperatorConfig]
+    name: str
+
+    def __init__(self, config: TeleoperatorConfig):
+        self.id = config.id
+        self.calibration_dir = (
+            config.calibration_dir
+            if config.calibration_dir
+            else HF_LEROBOT_CALIBRATION / TELEOPERATORS / self.name
+        )
+        self.calibration_dir.mkdir(parents=True, exist_ok=True)
+        self.calibration_fpath = self.calibration_dir / f"{self.id}.json"
+        self.calibration: dict[str, MotorCalibration] = {}
+        if self.calibration_fpath.is_file():
+            self._load_calibration()
+
+    def __str__(self) -> str:
+        return f"{self.id} {self.__class__.__name__}"
+
+    def __enter__(self):
+        """
+        Context manager entry.
+        Automatically connects to the camera.
+        """
+        self.connect()
+        return self
+
+    def __exit__(self, exc_type, exc_value, traceback) -> None:
+        """
+        Context manager exit.
+        Automatically disconnects, ensuring resources are released even on error.
+        """
+        self.disconnect()
+
+    def __del__(self) -> None:
+        """
+        Destructor safety net.
+        Attempts to disconnect if the object is garbage collected without cleanup.
+        """
+        try:
+            if self.is_connected:
+                self.disconnect()
+        except Exception:  # nosec B110
+            pass
+
+    @property
+    @abc.abstractmethod
+    def action_features(self) -> dict:
+        """
+        A dictionary describing the structure and types of the actions produced by the teleoperator. Its
+        structure (keys) should match the structure of what is returned by :pymeth:`get_action`. Values for
+        the dict should be the type of the value if it's a simple value, e.g. `float` for single
+        proprioceptive value (a joint's goal position/velocity)
+
+        Note: this property should be able to be called regardless of whether the robot is connected or not.
+        """
+        pass
+
+    @property
+    @abc.abstractmethod
+    def feedback_features(self) -> dict:
+        """
+        A dictionary describing the structure and types of the feedback actions expected by the robot. Its
+        structure (keys) should match the structure of what is passed to :pymeth:`send_feedback`. Values for
+        the dict should be the type of the value if it's a simple value, e.g. `float` for single
+        proprioceptive value (a joint's goal position/velocity)
+
+        Note: this property should be able to be called regardless of whether the robot is connected or not.
+        """
+        pass
+
+    @property
+    @abc.abstractmethod
+    def is_connected(self) -> bool:
+        """
+        Whether the teleoperator is currently connected or not. If `False`, calling :pymeth:`get_action`
+        or :pymeth:`send_feedback` should raise an error.
+        """
+        pass
+
+    @abc.abstractmethod
+    def connect(self, calibrate: bool = True) -> None:
+        """
+        Establish communication with the teleoperator.
+
+        Args:
+            calibrate (bool): If True, automatically calibrate the teleoperator after connecting if it's not
+                calibrated or needs calibration (this is hardware-dependant).
+        """
+        pass
+
+    @property
+    @abc.abstractmethod
+    def is_calibrated(self) -> bool:
+        """Whether the teleoperator is currently calibrated or not. Should be always `True` if not applicable"""
+        pass
+
+    @abc.abstractmethod
+    def calibrate(self) -> None:
+        """
+        Calibrate the teleoperator if applicable. If not, this should be a no-op.
+
+        This method should collect any necessary data (e.g., motor offsets) and update the
+        :pyattr:`calibration` dictionary accordingly.
+        """
+        pass
+
+    def _load_calibration(self, fpath: Path | None = None) -> None:
+        """
+        Helper to load calibration data from the specified file.
+
+        Args:
+            fpath (Path | None): Optional path to the calibration file. Defaults to `self.calibration_fpath`.
+        """
+        fpath = self.calibration_fpath if fpath is None else fpath
+        with open(fpath) as f, draccus.config_type("json"):
+            self.calibration = draccus.load(dict[str, MotorCalibration], f)
+
+    def _save_calibration(self, fpath: Path | None = None) -> None:
+        """
+        Helper to save calibration data to the specified file.
+
+        Args:
+            fpath (Path | None): Optional path to save the calibration file. Defaults to `self.calibration_fpath`.
+        """
+        fpath = self.calibration_fpath if fpath is None else fpath
+        with open(fpath, "w") as f, draccus.config_type("json"):
+            draccus.dump(self.calibration, f, indent=4)
+
+    @abc.abstractmethod
+    def configure(self) -> None:
+        """
+        Apply any one-time or runtime configuration to the teleoperator.
+        This may include setting motor parameters, control modes, or initial state.
+        """
+        pass
+
+    @abc.abstractmethod
+    def get_action(self) -> RobotAction:
+        """
+        Retrieve the current action from the teleoperator.
+
+        Returns:
+            RobotAction: A flat dictionary representing the teleoperator's current actions. Its
+                structure should match :pymeth:`observation_features`.
+        """
+        pass
+
+    @abc.abstractmethod
+    def send_feedback(self, feedback: dict[str, Any]) -> None:
+        """
+        Send a feedback action command to the teleoperator.
+
+        Args:
+            feedback (dict[str, Any]): Dictionary representing the desired feedback. Its structure should match
+                :pymeth:`feedback_features`.
+
+        Returns:
+            dict[str, Any]: The action actually sent to the motors potentially clipped or modified, e.g. by
+                safety limits on velocity.
+        """
+        pass
+
+    @abc.abstractmethod
+    def disconnect(self) -> None:
+        """Disconnect from the teleoperator and perform any necessary cleanup."""
+        pass
diff --git a/lerobot/src/lerobot/teleoperators/unitree_g1/__init__.py b/lerobot/src/lerobot/teleoperators/unitree_g1/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..5e67538b8db62e1d40f1b799b1e959a5bb68138d
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/unitree_g1/__init__.py
@@ -0,0 +1,31 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from .config_unitree_g1 import ExoskeletonArmPortConfig, UnitreeG1TeleoperatorConfig
+from .exo_calib import ExoskeletonCalibration, ExoskeletonJointCalibration
+from .exo_ik import ExoskeletonIKHelper
+from .exo_serial import ExoskeletonArm
+from .unitree_g1 import UnitreeG1Teleoperator
+
+__all__ = [
+    "ExoskeletonArmPortConfig",
+    "ExoskeletonCalibration",
+    "ExoskeletonIKHelper",
+    "ExoskeletonJointCalibration",
+    "ExoskeletonArm",
+    "UnitreeG1Teleoperator",
+    "UnitreeG1TeleoperatorConfig",
+]
diff --git a/lerobot/src/lerobot/teleoperators/unitree_g1/config_unitree_g1.py b/lerobot/src/lerobot/teleoperators/unitree_g1/config_unitree_g1.py
new file mode 100644
index 0000000000000000000000000000000000000000..66c4e7f31b1f6364e87f832765cb6cbcf99ac67c
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/unitree_g1/config_unitree_g1.py
@@ -0,0 +1,37 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from dataclasses import dataclass, field
+
+from ..config import TeleoperatorConfig
+
+
+@dataclass
+class ExoskeletonArmPortConfig:
+    """Serial port configuration for individual exoskeleton arm."""
+
+    port: str = ""
+    baud_rate: int = 115200
+
+
+@TeleoperatorConfig.register_subclass("unitree_g1")
+@dataclass
+class UnitreeG1TeleoperatorConfig(TeleoperatorConfig):
+    left_arm_config: ExoskeletonArmPortConfig = field(default_factory=ExoskeletonArmPortConfig)
+    right_arm_config: ExoskeletonArmPortConfig = field(default_factory=ExoskeletonArmPortConfig)
+
+    # Frozen joints (comma-separated joint names that won't be moved by IK)
+    frozen_joints: str = ""
diff --git a/lerobot/src/lerobot/teleoperators/unitree_g1/exo_calib.py b/lerobot/src/lerobot/teleoperators/unitree_g1/exo_calib.py
new file mode 100644
index 0000000000000000000000000000000000000000..b90e8fd7e03a6acbfe137c837d3d7a29b5ad5438
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/unitree_g1/exo_calib.py
@@ -0,0 +1,446 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+This module handles calibration of hall effect sensors used in the exoskeleton.
+Each joint has a pair of ADC channels outputting sin and cos values that trace an ellipse
+as the joint rotates due to imprecision in magnet/sensor placement. We fit this ellipse to a unit circle,
+and calculate arctan2 of the unit circle to get the joint angle.
+We then store the ellipse parameters and the zero offset for each joint to be used at runtime.
+"""
+
+import json
+import logging
+import time
+from collections import deque
+from dataclasses import dataclass, field
+from pathlib import Path
+
+import numpy as np
+import serial
+
+logger = logging.getLogger(__name__)
+
+
+ADC_MAX = 2**12 - 1
+ADC_HALF = ADC_MAX / 2
+
+# exoskeleton joint names -> ADC channel pairs. TODO: add wrist pitch and wrist yaw
+JOINTS = {
+    "shoulder_pitch": (0, 1),
+    "shoulder_yaw": (2, 3),
+    "shoulder_roll": (4, 5),
+    "elbow_flex": (6, 7),
+    "wrist_roll": (14, 15),
+}
+
+
+@dataclass
+class ExoskeletonJointCalibration:
+    name: str  # joint name
+    center_fit: list[float]  # center of the ellipse
+    T: list[list[float]]  # 2x2 transformation matrix
+    zero_offset: float = 0.0  # angle at neutral pose
+
+
+@dataclass
+class ExoskeletonCalibration:
+    """Full calibration data for an exoskeleton arm."""
+
+    version: int = 2
+    side: str = ""
+    adc_max: int = ADC_MAX
+    joints: list[ExoskeletonJointCalibration] = field(default_factory=list)
+
+    def to_dict(self) -> dict:
+        return {
+            "version": self.version,
+            "side": self.side,
+            "adc_max": self.adc_max,
+            "joints": [
+                {
+                    "name": j.name,
+                    "center_fit": j.center_fit,
+                    "T": j.T,
+                    "zero_offset": j.zero_offset,
+                }
+                for j in self.joints
+            ],
+        }
+
+    @classmethod
+    def from_dict(cls, data: dict) -> "ExoskeletonCalibration":
+        joints = [
+            ExoskeletonJointCalibration(
+                name=j["name"],
+                center_fit=j["center_fit"],
+                T=j["T"],
+                zero_offset=j.get("zero_offset", 0.0),
+            )
+            for j in data.get("joints", [])
+        ]
+        return cls(
+            version=data.get("version", 2),
+            side=data.get("side", ""),
+            adc_max=data.get("adc_max", ADC_MAX),
+            joints=joints,
+        )
+
+
+@dataclass(frozen=True)
+class CalibParams:
+    fit_every: float = 0.15
+    min_fit_points: int = 60
+    fit_window: int = 900
+    max_fit_points: int = 300
+    trim_low: float = 0.05
+    trim_high: float = 0.95
+    median_window: int = 5
+    history: int = 3500
+    draw_hz: float = 120.0
+    sample_count: int = 50
+
+
+def normalize_angle(angle: float) -> float:
+    """Normalize angle to [-pi, pi]."""
+    return float(np.arctan2(np.sin(angle), np.cos(angle)))
+
+
+def joint_z_and_angle(raw16: list[int], j: ExoskeletonJointCalibration) -> tuple[np.ndarray, float]:
+    """
+    Applies calibration to each joint: raw → centered → ellipse-to-circle → angle.
+    """
+    pair = JOINTS[j.name]
+    s, c = raw16[pair[0]], raw16[pair[1]]  # get sin and cos
+    p = np.array([float(c) - ADC_HALF, float(s) - ADC_HALF])  # center the raw values
+    z = np.asarray(j.T) @ (
+        p - np.asarray(j.center_fit)
+    )  # center the ellipse and invert the transformation matrix to get unit circle coords
+    ang = float(np.arctan2(z[1], z[0])) - j.zero_offset  # calculate the anvgle and apply the zero offset
+    return z, normalize_angle(-ang)  # ensure range is [-pi, pi]
+
+
+def exo_raw_to_angles(raw16: list[int], calib: ExoskeletonCalibration) -> dict[str, float]:
+    """Convert raw sensor readings to joint angles using calibration."""
+    return {j.name: joint_z_and_angle(raw16, j)[1] for j in calib.joints}
+
+
+def run_exo_calibration(
+    ser: serial.Serial,
+    side: str,
+    save_path: Path,
+    params: CalibParams | None = None,
+) -> ExoskeletonCalibration:
+    """
+    Run interactive calibration for an exoskeleton arm.
+    """
+    try:
+        import cv2
+        import matplotlib.pyplot as plt
+    except ImportError as e:
+        raise ImportError(
+            "Calibration requires matplotlib and opencv-python. "
+            "Install with: pip install matplotlib opencv-python"
+        ) from e
+
+    from .exo_serial import read_raw_from_serial
+
+    params = params or CalibParams()
+    joint_list = list(JOINTS.items())  # Convert dict to list for indexing
+    logger.info(f"Starting calibration for {side} exoskeleton arm")
+
+    def running_median(win: deque) -> float:
+        return float(np.median(np.fromiter(win, dtype=float)))
+
+    def read_joint_point(raw16: list[int], pair: tuple[int, int]):
+        s, c = raw16[pair[0]], raw16[pair[1]]
+        return float(c) - ADC_HALF, float(s) - ADC_HALF, float(s), float(c)
+
+    def select_fit_subset(xs, ys):
+        """Select and filter points for ellipse fitting. Trims outliers by radius and downsamples."""
+        n = min(params.fit_window, len(xs))
+        if n <= 0:
+            return None, None
+        x = np.asarray(list(xs)[-n:], dtype=float)  # most recent n samples
+        y = np.asarray(list(ys)[-n:], dtype=float)
+        r = np.sqrt(x * x + y * y)  # radius from origin
+        if len(r) >= 20:
+            lo, hi = np.quantile(r, params.trim_low), np.quantile(r, params.trim_high)  # outlier bounds
+            keep = (r >= lo) & (r <= hi)
+            x, y = x[keep], y[keep]  # remove outliers
+        if len(x) > params.max_fit_points:
+            idx = np.linspace(0, len(x) - 1, params.max_fit_points).astype(int)  # downsample evenly
+            x, y = x[idx], y[idx]
+        return x, y
+
+    def fit_ellipse_opencv(x, y):
+        """Fit ellipse to (x,y) points using OpenCV. Returns center, axes, rotation matrix, and outline."""
+        x, y = np.asarray(x, dtype=float), np.asarray(y, dtype=float)
+        if len(x) < 5:
+            return None
+        pts = np.stack([x, y], axis=1).astype(np.float32).reshape(-1, 1, 2)
+        try:
+            (xc, yc), (w, h), angle_deg = cv2.fitEllipse(pts)  # returns center, axes, rotation in degrees
+        except cv2.error:
+            return None
+        a, b = float(w) * 0.5, float(h) * 0.5  # get ellipse major and minor semi-axes
+        phi = np.deg2rad(float(angle_deg))  # to rad
+        if b > a:  # ensure major axis is a
+            a, b = b, a
+            phi += np.pi / 2.0
+        if not np.isfinite(a) or not np.isfinite(b) or a <= 1e-6 or b <= 1e-6:
+            return None
+        cp, sp = float(np.cos(phi)), float(np.sin(phi))  #
+        rot = np.array([[cp, -sp], [sp, cp]], dtype=float)  # 2x2 rotation matrix
+        center = np.array([float(xc), float(yc)], dtype=float)  # offset vector
+        tt = np.linspace(0, 2 * np.pi, 360)
+        outline = (rot @ np.stack([a * np.cos(tt), b * np.sin(tt)])).T + center  # for viz
+        return {"center": center, "a": a, "b": b, "R": rot, "ex": outline[:, 0], "ey": outline[:, 1]}
+
+    # Setup matplotlib
+    plt.ion()
+    fig, (ax0, ax1) = plt.subplots(1, 2, figsize=(12, 6))
+    ax0.set_xlabel("cos - center")
+    ax0.set_ylabel("sin - center")
+    ax0.grid(True, alpha=0.25)
+    ax0.set_aspect("equal", adjustable="box")
+    ax1.set_title("Unit circle + angle")
+    ax1.set_xlabel("x")
+    ax1.set_ylabel("y")
+    ax1.grid(True, alpha=0.25)
+    ax1.set_aspect("equal", adjustable="box")
+    tt = np.linspace(0, 2 * np.pi, 360)
+    ax1.plot(np.cos(tt), np.sin(tt), "k-", linewidth=1)
+    ax0.set_xlim(-2200, 2200)
+    ax0.set_ylim(-2200, 2200)
+    ax1.set_xlim(-1.4, 1.4)
+    ax1.set_ylim(-1.4, 1.4)
+
+    sc0 = ax0.scatter([], [], s=6, animated=True)
+    (ell_line,) = ax0.plot([], [], "r-", linewidth=2, animated=True)
+    sc1 = ax1.scatter([], [], s=6, animated=True)
+    (radius_line,) = ax1.plot([], [], "g-", linewidth=2, animated=True)
+    angle_text = ax1.text(
+        0.02, 0.98, "", transform=ax1.transAxes, va="top", ha="left", fontsize=12, animated=True
+    )
+
+    fig.canvas.draw()
+    bg0 = fig.canvas.copy_from_bbox(ax0.bbox)
+    bg1 = fig.canvas.copy_from_bbox(ax1.bbox)
+
+    # State
+    joints_out = []
+    joint_idx = 0
+    phase = "ellipse"
+    advance_requested = False
+    zero_samples = []
+
+    def on_key(event):
+        nonlocal advance_requested
+        if event.key in ("n", "N", "enter", " "):
+            advance_requested = True
+
+    fig.canvas.mpl_connect("key_press_event", on_key)
+
+    def reset_state():
+        return {
+            "xs": deque(maxlen=params.history),
+            "ys": deque(maxlen=params.history),
+            "xu": deque(maxlen=params.history),
+            "yu": deque(maxlen=params.history),
+            "win_s": deque(maxlen=params.median_window),
+            "win_c": deque(maxlen=params.median_window),
+            "ellipse_cache": None,
+            "T": None,
+            "center_fit": None,
+            "have_transform": False,
+            "latest_z": None,
+            "last_fit": 0.0,
+        }
+
+    state = reset_state()
+    last_draw = 0.0
+    name, pair = joint_list[joint_idx]
+    fig.canvas.manager.set_window_title(f"[{joint_idx + 1}/{len(joint_list)}] {name} - ELLIPSE")
+    ax0.set_title(f"{name} raw (filtered)")
+    logger.info(f"[{joint_idx + 1}/{len(joint_list)}] Calibrating {name}")
+    logger.info("Step 1: Move joint around to map ellipse, then press 'n'")
+
+    try:
+        while plt.fignum_exists(fig.number):
+            name, pair = joint_list[joint_idx]
+
+            # Handles calibration GUI state: ellipse → zero_pose → next joint -> ellipse -> ...
+            if phase == "ellipse" and advance_requested and state["have_transform"]:
+                joints_out.append(
+                    {
+                        "name": name,
+                        "center_fit": state["center_fit"].tolist(),
+                        "T": state["T"].tolist(),
+                    }
+                )
+                logger.info(f"  -> Ellipse saved for {name}")
+                phase, zero_samples, advance_requested = "zero_pose", [], False
+                fig.canvas.manager.set_window_title(f"[{joint_idx + 1}/{len(joint_list)}] {name} - ZERO POSE")
+                ax0.set_title(f"{name} - hold zero pose")
+                fig.canvas.draw()
+                bg0, bg1 = fig.canvas.copy_from_bbox(ax0.bbox), fig.canvas.copy_from_bbox(ax1.bbox)
+                logger.info(f"Step 2: Hold {name} in zero position, then press 'n'")
+
+            elif phase == "ellipse" and advance_requested and not state["have_transform"]:
+                logger.info("  (Need valid fit first - keep moving the joint)")
+                advance_requested = False
+
+            elif phase == "zero_pose" and advance_requested:
+                if len(zero_samples) >= params.sample_count:
+                    zero_offset = float(np.mean(zero_samples[-params.sample_count :]))
+                    joints_out[-1]["zero_offset"] = zero_offset
+                    logger.info(f"  -> {name} zero: {zero_offset:+.3f} rad ({np.degrees(zero_offset):+.1f}°)")
+                    joint_idx += 1
+                    advance_requested = False
+
+                    if joint_idx >= len(joint_list):
+                        # All joints done
+                        calib = ExoskeletonCalibration(
+                            version=2,
+                            side=side,
+                            adc_max=ADC_MAX,
+                            joints=[
+                                ExoskeletonJointCalibration(
+                                    name=j["name"],
+                                    center_fit=j["center_fit"],
+                                    T=j["T"],
+                                    zero_offset=j.get("zero_offset", 0.0),
+                                )
+                                for j in joints_out
+                            ],
+                        )
+                        save_path.parent.mkdir(parents=True, exist_ok=True)
+                        with open(save_path, "w") as f:
+                            json.dump(calib.to_dict(), f, indent=2)
+                        logger.info(f"Saved calibration to {save_path}")
+                        logger.info("Calibration complete!")
+                        plt.close(fig)
+                        return calib
+
+                    # Next joint
+                    phase, state = "ellipse", reset_state()
+                    name, pair = joint_list[joint_idx]
+                    fig.canvas.manager.set_window_title(
+                        f"[{joint_idx + 1}/{len(joint_list)}] {name} - ELLIPSE"
+                    )
+                    ax0.set_title(f"{name} raw (filtered)")
+                    fig.canvas.draw()
+                    bg0, bg1 = fig.canvas.copy_from_bbox(ax0.bbox), fig.canvas.copy_from_bbox(ax1.bbox)
+                    logger.info(f"[{joint_idx + 1}/{len(joint_list)}] Calibrating {name}")
+                    logger.info("Step 1: Move joint around to map ellipse, then press 'n'")
+                else:
+                    logger.info(
+                        f"  (Collecting samples: {len(zero_samples)}/{params.sample_count} - hold still)"
+                    )
+                    advance_requested = False
+
+            # Read sensor
+            raw16 = read_raw_from_serial(ser)
+            if raw16 is not None:
+                x_raw, y_raw, s_raw, c_raw = read_joint_point(raw16, pair)
+
+                if phase == "ellipse":
+                    if state["have_transform"]:
+                        z = state["T"] @ (np.array([x_raw, y_raw]) - state["center_fit"])
+                        state["xu"].append(float(z[0]))
+                        state["yu"].append(float(z[1]))
+                        state["latest_z"] = (float(z[0]), float(z[1]))
+                    state["win_s"].append(s_raw)
+                    state["win_c"].append(c_raw)
+                    if len(state["win_s"]) >= max(3, params.median_window):
+                        state["ys"].append(running_median(state["win_s"]) - ADC_HALF)
+                        state["xs"].append(running_median(state["win_c"]) - ADC_HALF)
+                else:
+                    jdata = joints_out[-1]
+                    z = np.array(jdata["T"]) @ (np.array([x_raw, y_raw]) - np.array(jdata["center_fit"]))
+                    zero_samples.append(float(np.arctan2(z[1], z[0])))
+                    state["latest_z"] = (float(z[0]), float(z[1]))
+
+            # Ellipse fitting
+            t = time.time()
+            if (
+                phase == "ellipse"
+                and (t - state["last_fit"]) >= params.fit_every
+                and len(state["xs"]) >= params.min_fit_points
+            ):
+                xfit, yfit = select_fit_subset(state["xs"], state["ys"])
+                if xfit is not None and len(xfit) >= params.min_fit_points:
+                    fit = fit_ellipse_opencv(xfit, yfit)
+                    if fit is not None:
+                        state["center_fit"] = fit["center"]
+                        state["T"] = np.diag([1.0 / fit["a"], 1.0 / fit["b"]]) @ fit["R"].T
+                        state["ellipse_cache"] = (fit["ex"], fit["ey"])
+                        state["have_transform"] = True
+                state["last_fit"] = t
+
+            # Drawing
+            if (t - last_draw) >= 1.0 / params.draw_hz:
+                fig.canvas.restore_region(bg0)
+                fig.canvas.restore_region(bg1)
+
+                if phase == "ellipse":
+                    sc0.set_offsets(np.c_[state["xs"], state["ys"]] if state["xs"] else np.empty((0, 2)))
+                    ax0.draw_artist(sc0)
+                    ell_line.set_data(*state["ellipse_cache"] if state["ellipse_cache"] else ([], []))
+                    ax0.draw_artist(ell_line)
+                    sc1.set_offsets(np.c_[state["xu"], state["yu"]] if state["xu"] else np.empty((0, 2)))
+                    ax1.draw_artist(sc1)
+                    if state["latest_z"]:
+                        zx, zy = state["latest_z"]
+                        radius_line.set_data([0.0, zx], [0.0, zy])
+                        ang = float(np.arctan2(zy, zx))
+                        angle_text.set_text(
+                            f"angle: {ang:+.3f} rad  ({np.degrees(ang):+.1f}°)\nmove {name}, press 'n' to advance"
+                        )
+                    else:
+                        radius_line.set_data([], [])
+                        angle_text.set_text("(waiting for fit)")
+                else:
+                    sc0.set_offsets(np.empty((0, 2)))
+                    ax0.draw_artist(sc0)
+                    ell_line.set_data([], [])
+                    ax0.draw_artist(ell_line)
+                    if state["latest_z"]:
+                        zx, zy = state["latest_z"]
+                        sc1.set_offsets([[zx, zy]])
+                        radius_line.set_data([0.0, zx], [0.0, zy])
+                        ang = float(np.arctan2(zy, zx))
+                        angle_text.set_text(
+                            f"Zero pose for {name}\nangle: {ang:+.3f} rad\nsamples: {len(zero_samples)}/{params.sample_count}\nhold still, press 'n'"
+                        )
+                    else:
+                        sc1.set_offsets(np.empty((0, 2)))
+                        radius_line.set_data([], [])
+                        angle_text.set_text("(waiting for data)")
+                    ax1.draw_artist(sc1)
+
+                ax1.draw_artist(radius_line)
+                ax1.draw_artist(angle_text)
+                fig.canvas.blit(ax0.bbox)
+                fig.canvas.blit(ax1.bbox)
+                fig.canvas.flush_events()
+                last_draw = t
+
+            plt.pause(0.001)
+
+    finally:
+        plt.close(fig)
diff --git a/lerobot/src/lerobot/teleoperators/unitree_g1/exo_ik.py b/lerobot/src/lerobot/teleoperators/unitree_g1/exo_ik.py
new file mode 100644
index 0000000000000000000000000000000000000000..3fd18d2f80da9c9e85b16b2ce759063b9893a90e
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/unitree_g1/exo_ik.py
@@ -0,0 +1,353 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+IK helper for exoskeleton-to-G1 teleoperation. We map Exoskeleton joint angles to end-effector pose in world frame,
+visualizing the result in meshcat after calibration.
+"""
+
+import logging
+import os
+from dataclasses import dataclass
+
+import numpy as np
+
+from lerobot.robots.unitree_g1.g1_kinematics import G1_29_ArmIK
+from lerobot.robots.unitree_g1.g1_utils import G1_29_JointArmIndex
+
+from .exo_calib import JOINTS
+
+logger = logging.getLogger(__name__)
+
+
+def _frame_id(model, name: str) -> int | None:
+    try:
+        fid = model.getFrameId(name)
+        return fid if 0 <= fid < model.nframes else None
+    except Exception:
+        return None
+
+
+@dataclass
+class ArmCfg:
+    side: str  # "left" | "right"
+    urdf: str  # exo_left.urdf / exo_right.urdf
+    root: str  # "exo_left" / "exo_right"
+    g1_ee: str  # "l_ee" / "r_ee"
+    offset: np.ndarray  # world offset for viz + target
+    marker_prefix: str  # "left" / "right"
+
+
+class Markers:
+    """Creates meshcat visualization primitives, showing end-effector frames of exoskeleton and G1"""
+
+    def __init__(self, viewer):
+        self.v = viewer
+
+    def sphere(self, path: str, r: float, rgba: tuple[float, float, float, float]):
+        import meshcat.geometry as mg
+
+        c = (int(rgba[0] * 255) << 16) | (int(rgba[1] * 255) << 8) | int(rgba[2] * 255)
+        self.v[path].set_object(
+            mg.Sphere(r),
+            mg.MeshPhongMaterial(color=c, opacity=rgba[3], transparent=rgba[3] < 1.0),
+        )
+
+    def axes(self, path: str, axis_len: float = 0.1, axis_w: int = 6):
+        import meshcat.geometry as mg
+
+        pts = np.array(
+            [[0, 0, 0], [axis_len, 0, 0], [0, 0, 0], [0, axis_len, 0], [0, 0, 0], [0, 0, axis_len]],
+            dtype=np.float32,
+        ).T
+        cols = np.array(
+            [[1, 0, 0], [1, 0, 0], [0, 1, 0], [0, 1, 0], [0, 0, 1], [0, 0, 1]],
+            dtype=np.float32,
+        ).T
+        self.v[path].set_object(
+            mg.LineSegments(
+                mg.PointsGeometry(position=pts, color=cols),
+                mg.LineBasicMaterial(linewidth=axis_w, vertexColors=True),
+            )
+        )
+
+    def tf(self, path: str, mat: np.ndarray):
+        self.v[path].set_transform(mat)
+
+
+class ExoskeletonIKHelper:
+    """
+    - Loads G1 robot and exoskeleton URDF models via Pinocchio
+    - Computes forward kinematics on exoskeleton to get end-effector poses
+    - Solves inverse kinematics on G1 to match those poses
+    - Provides meshcat visualization showing both robots and targets
+
+    Args:
+        frozen_joints: List of G1 joint names to exclude from IK (kept at neutral).
+    """
+
+    def __init__(self, frozen_joints: list[str] | None = None):
+        try:
+            import pinocchio as pin
+        except ImportError as e:
+            raise ImportError("ik mode needs pinocchio: pip install pin") from e
+
+        self.pin = pin
+        self.frozen_joints = frozen_joints or []
+
+        self.g1_ik = G1_29_ArmIK()
+        self.robot_g1 = self.g1_ik.reduced_robot
+        self.robot_g1.data = self.robot_g1.model.createData()
+        self.q_g1 = pin.neutral(self.robot_g1.model)
+
+        assets_dir = os.path.join(self.g1_ik.repo_path, "assets")
+
+        self.frozen_idx = self._frozen_joint_indices()
+
+        self.arms = [
+            ArmCfg(
+                side="left",
+                urdf=os.path.join(assets_dir, "exo_left.urdf"),
+                root="exo_left",
+                g1_ee="L_ee",
+                offset=np.array([0.6, 0.3, 0.0]),
+                marker_prefix="left",
+            ),
+            ArmCfg(
+                side="right",
+                urdf=os.path.join(assets_dir, "exo_right.urdf"),
+                root="exo_right",
+                g1_ee="R_ee",
+                offset=np.array([0.6, -0.3, 0.0]),
+                marker_prefix="right",
+            ),
+        ]
+
+        self.exo = {}  # side -> pin.RobotWrapper
+        self.q_exo = {}  # side -> q
+        self.ee_id_exo = {}  # side -> frame id
+        self.qmap = {}  # side -> {joint_name: q_idx}
+        self.ee_id_g1 = {}  # side -> frame id
+
+        self._load_exo_models(assets_dir)
+        for a in self.arms:
+            self.ee_id_g1[a.side] = _frame_id(self.robot_g1.model, a.g1_ee)
+
+        self.viewer = None
+        self.markers: Markers | None = None
+        self.viz_g1 = None
+        self.viz_exo = {}  # side -> viz
+
+    def _frozen_joint_indices(self) -> dict[str, int]:
+        out = {}
+        m = self.robot_g1.model
+        for name in self.frozen_joints:
+            if name in m.names:
+                jid = m.getJointId(name)
+                out[name] = m.idx_qs[jid]
+                logger.info(f"freezing joint: {name} (q_idx={out[name]})")
+        return out
+
+    def _find_exo_ee(self, model, ee_name: str = "ee") -> int:
+        ee = _frame_id(model, ee_name)
+        if ee is not None:
+            return ee
+        for fid in reversed(range(model.nframes)):
+            if model.frames[fid].type == self.pin.FrameType.BODY:
+                return fid
+        return 0
+
+    def _build_joint_map(self, robot) -> dict[str, int]:
+        m = robot.model
+        return {n: m.idx_qs[m.getJointId(n)] for n in JOINTS if n in m.names}
+
+    def _load_exo_models(self, assets_dir: str):
+        pin = self.pin
+        for a in self.arms:
+            if not os.path.exists(a.urdf):
+                logger.warning(f"{a.side} exo urdf not found: {a.urdf}")
+                continue
+            r = pin.RobotWrapper.BuildFromURDF(a.urdf, assets_dir)
+            self.exo[a.side] = r
+            self.q_exo[a.side] = pin.neutral(r.model)
+            self.ee_id_exo[a.side] = self._find_exo_ee(r.model)
+            self.qmap[a.side] = self._build_joint_map(r)
+            logger.info(f"loaded {a.side} exo urdf: {a.urdf}")
+
+    def init_visualization(self):
+        """
+        Creates a browser-based visualization of exoskeleton and G1 robot,
+        highlighting end-effector frames and target positions.
+        """
+        try:
+            from pinocchio.visualize import MeshcatVisualizer
+        except ImportError as e:
+            logger.warning(f"meshcat viz unavailable: {e}")
+            return
+
+        # g1
+        self.viz_g1 = MeshcatVisualizer(
+            self.robot_g1.model, self.robot_g1.collision_model, self.robot_g1.visual_model
+        )
+        self.viz_g1.initViewer(open=True)
+        self.viz_g1.loadViewerModel("g1")
+        self.viz_g1.display(self.q_g1)
+
+        self.viewer = self.viz_g1.viewer
+        self.markers = Markers(self.viewer)
+
+        # exos
+        for a in self.arms:
+            if a.side not in self.exo:
+                continue
+            r = self.exo[a.side]
+            v = MeshcatVisualizer(r.model, r.collision_model, r.visual_model)
+            v.initViewer(open=False)
+            v.viewer = self.viewer
+            v.loadViewerModel(a.root)
+            offset_tf = np.eye(4)
+            offset_tf[:3, 3] = a.offset
+            self.viewer[a.root].set_transform(offset_tf)
+            v.display(self.q_exo[a.side])
+            self.viz_exo[a.side] = v
+
+        # markers
+        for a in self.arms:
+            p = a.marker_prefix
+            self.markers.sphere(f"markers/{p}_exo_ee", 0.012, (0.2, 1.0, 0.2, 0.9))
+            self.markers.sphere(f"markers/{p}_g1_ee", 0.015, (1.0, 0.2, 0.2, 0.9))
+            self.markers.sphere(f"markers/{p}_ik_target", 0.015, (0.1, 0.3, 1.0, 0.9))
+            self.markers.axes(f"markers/{p}_exo_axes", 0.06)
+            self.markers.axes(f"markers/{p}_g1_axes", 0.08)
+
+        logger.info(f"meshcat viz initialized: {self.viewer.url()}")
+        print(f"\nmeshcat url: {self.viewer.url()}\n")
+
+    def _fk_target_world(self, side: str, angles: dict[str, float]) -> np.ndarray | None:
+        """returns wrist frame target to be used for G1 IK in 4x4 homogeneous transform. Takes offset into account."""
+        if side not in self.exo or not angles:
+            return None
+
+        pin = self.pin
+        q = self.q_exo[side]
+        qmap = self.qmap[side]
+
+        for name, ang in angles.items():
+            idx = qmap.get(name)
+            if idx is not None:
+                q[idx] = float(ang)
+
+        r = self.exo[side]
+        pin.forwardKinematics(r.model, r.data, q)
+        pin.updateFramePlacements(r.model, r.data)
+
+        ee = r.data.oMf[self.ee_id_exo[side]]
+        target = np.eye(4)
+        target[:3, :3] = ee.rotation
+        # offset gets applied in world space
+        cfg = next(a for a in self.arms if a.side == side)
+        target[:3, 3] = cfg.offset + ee.translation
+        return target
+
+    def update_visualization(self):
+        if self.viewer is None or self.markers is None:
+            return
+
+        pin = self.pin
+
+        # g1
+        if self.viz_g1 is not None:
+            self.viz_g1.display(self.q_g1)
+            pin.forwardKinematics(self.robot_g1.model, self.robot_g1.data, self.q_g1)
+            pin.updateFramePlacements(self.robot_g1.model, self.robot_g1.data)
+
+            for a in self.arms:
+                fid = self.ee_id_g1.get(a.side)
+                if fid is None:
+                    continue
+                ee_tf = self.robot_g1.data.oMf[fid].homogeneous
+                p = a.marker_prefix
+                self.markers.tf(f"markers/{p}_g1_ee", ee_tf)
+                self.markers.tf(f"markers/{p}_g1_axes", ee_tf)
+
+        # exos
+        for a in self.arms:
+            side = a.side
+            v = self.viz_exo.get(side)
+            if v is None:
+                continue
+
+            v.display(self.q_exo[side])
+            r = self.exo[side]
+            pin.forwardKinematics(r.model, r.data, self.q_exo[side])
+            pin.updateFramePlacements(r.model, r.data)
+
+            ee = r.data.oMf[self.ee_id_exo[side]]
+            world_tf = (pin.SE3(np.eye(3), a.offset) * ee).homogeneous
+            p = a.marker_prefix
+            self.markers.tf(f"markers/{p}_exo_ee", world_tf)
+            self.markers.tf(f"markers/{p}_exo_axes", world_tf)
+
+            target_tf = np.eye(4)
+            target_tf[:3, :3] = ee.rotation
+            target_tf[:3, 3] = a.offset + ee.translation
+            self.markers.tf(f"markers/{p}_ik_target", target_tf)
+
+    def compute_g1_joints_from_exo(
+        self,
+        left_angles: dict[str, float],
+        right_angles: dict[str, float],
+    ) -> dict[str, float]:
+        """
+        Performs FK on exoskeleton to get end-effector poses in world frame,
+        after which it solves IK on G1 to return joint angles matching those poses in G1 motor order.
+        """
+        pin = self.pin
+
+        targets = {
+            "left": self._fk_target_world("left", left_angles),
+            "right": self._fk_target_world("right", right_angles),
+        }
+
+        # fallback to current g1 ee pose if missing target
+        pin.forwardKinematics(self.robot_g1.model, self.robot_g1.data, self.q_g1)
+        pin.updateFramePlacements(self.robot_g1.model, self.robot_g1.data)
+
+        for a in self.arms:
+            if targets[a.side] is not None:
+                continue
+            fid = self.ee_id_g1.get(a.side)
+            if fid is not None:
+                targets[a.side] = self.robot_g1.data.oMf[fid].homogeneous
+
+        if targets["left"] is None or targets["right"] is None:
+            logger.warning("missing ik targets, returning current pose")
+            return {}
+
+        frozen_vals = {n: self.q_g1[i] for n, i in self.frozen_idx.items()}
+
+        self.q_g1, _ = self.g1_ik.solve_ik(
+            targets["left"], targets["right"], current_lr_arm_motor_q=self.q_g1
+        )
+
+        for n, i in self.frozen_idx.items():
+            self.q_g1[i] = frozen_vals[n]
+
+        return {
+            f"{j.name}.q": float(self.q_g1[i])
+            for i, j in enumerate(G1_29_JointArmIndex)
+            if i < len(self.q_g1)
+        }
diff --git a/lerobot/src/lerobot/teleoperators/unitree_g1/exo_serial.py b/lerobot/src/lerobot/teleoperators/unitree_g1/exo_serial.py
new file mode 100644
index 0000000000000000000000000000000000000000..4f45997c055775b39b0d12783be59471f1534cdb
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/unitree_g1/exo_serial.py
@@ -0,0 +1,124 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import json
+import logging
+from dataclasses import dataclass
+from pathlib import Path
+
+import serial
+
+from .exo_calib import ExoskeletonCalibration, exo_raw_to_angles, run_exo_calibration
+
+logger = logging.getLogger(__name__)
+
+
+def parse_raw16(line: bytes) -> list[int] | None:
+    try:
+        parts = line.decode("utf-8", errors="ignore").split()
+        if len(parts) < 16:
+            return None
+        return [int(x) for x in parts[:16]]
+    except (ValueError, IndexError):
+        return None
+
+
+def read_raw_from_serial(ser) -> list[int] | None:
+    """Read latest sample from serial; if buffer is backed up, keep only the newest."""
+    try:
+        last = None
+        while ser.in_waiting > 0:
+            b = ser.readline()
+            if not b:
+                break
+            raw16 = parse_raw16(b)
+            if raw16 is not None:
+                last = raw16
+        if last is None:
+            b = ser.readline()
+            if b:
+                last = parse_raw16(b)
+        return last
+    except serial.SerialException as e:
+        logger.warning(f"Serial read error: {e}")
+        return None
+
+
+@dataclass
+class ExoskeletonArm:
+    port: str
+    calibration_fpath: Path
+    side: str
+    baud_rate: int = 115200
+
+    _ser: serial.Serial | None = None
+    calibration: ExoskeletonCalibration | None = None
+
+    def __post_init__(self):
+        if self.calibration_fpath.is_file():
+            self._load_calibration()
+
+    @property
+    def is_connected(self) -> bool:
+        return self._ser is not None and getattr(self._ser, "is_open", False)
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.calibration is not None
+
+    def connect(self, calibrate: bool = True) -> None:
+        if self.is_connected:
+            return
+        try:
+            self._ser = serial.Serial(self.port, self.baud_rate, timeout=0.02)
+            self._ser.reset_input_buffer()
+            logger.info(f"connected: {self.port}")
+        except serial.SerialException as e:
+            raise ConnectionError(f"failed to connect to {self.port}: {e}") from e
+
+        if calibrate and not self.is_calibrated:
+            self.calibrate()
+
+    def disconnect(self) -> None:
+        if self._ser:
+            try:
+                self._ser.close()
+            finally:
+                self._ser = None
+
+    def _load_calibration(self) -> None:
+        try:
+            data = json.loads(self.calibration_fpath.read_text())
+            self.calibration = ExoskeletonCalibration.from_dict(data)
+            logger.info(f"loaded calibration: {self.calibration_fpath}")
+        except Exception as e:
+            logger.warning(f"failed to load calibration: {e}")
+
+    def read_raw(self) -> list[int] | None:
+        if not self._ser:
+            return None
+        return read_raw_from_serial(self._ser)
+
+    def get_angles(self) -> dict[str, float]:
+        if not self.calibration:
+            raise RuntimeError("exoskeleton not calibrated")
+        raw = self.read_raw()
+        return {} if raw is None else exo_raw_to_angles(raw, self.calibration)
+
+    def calibrate(self) -> None:
+        if not self.is_connected:
+            raise RuntimeError("Cannot calibrate: exoskeleton not connected")
+        self.calibration = run_exo_calibration(self._ser, self.side, self.calibration_fpath)
diff --git a/lerobot/src/lerobot/teleoperators/unitree_g1/unitree_g1.py b/lerobot/src/lerobot/teleoperators/unitree_g1/unitree_g1.py
new file mode 100644
index 0000000000000000000000000000000000000000..242613e7e75ea47652030bf0034cdafd5b200b5a
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/unitree_g1/unitree_g1.py
@@ -0,0 +1,332 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from functools import cached_property
+from typing import TYPE_CHECKING, Any
+
+from lerobot.robots.unitree_g1.g1_utils import REMOTE_AXES, G1_29_JointArmIndex
+from lerobot.utils.constants import HF_LEROBOT_CALIBRATION, TELEOPERATORS
+from lerobot.utils.import_utils import _unitree_sdk_available
+
+if TYPE_CHECKING or _unitree_sdk_available:
+    from unitree_sdk2py.utils.joystick import Joystick
+else:
+
+    class Joystick:
+        def __init__(self):
+            raise ImportError(
+                "unitree_sdk2py is required for RemoteController. Install with: pip install unitree_sdk2py"
+            )
+
+
+from ..teleoperator import Teleoperator
+from .config_unitree_g1 import UnitreeG1TeleoperatorConfig
+from .exo_ik import ExoskeletonIKHelper
+from .exo_serial import ExoskeletonArm
+
+logger = logging.getLogger(__name__)
+
+
+class RemoteController:
+    """Unitree remote controller data parser for joystick and button state."""
+
+    # ADC parameters for exoskeleton joystick (12-bit ADC)
+    ADC_MAX = 4095
+    ADC_HALF = ADC_MAX / 2
+    JOYSTICK_X_IDX = 11  # X axis in raw ADC array
+    JOYSTICK_BTN_IDX = 12  # Button in raw ADC array
+    JOYSTICK_Y_IDX = 13  # Y axis in raw ADC array
+
+    # Map SDK named buttons to positional indices matching the wireless_remote
+    # byte layout (little-endian uint16 from bytes 2-3).
+    _BUTTON_MAP: list[str] = [
+        "RB",
+        "LB",
+        "start",
+        "back",
+        "RT",
+        "LT",
+        "",
+        "",
+        "A",
+        "B",
+        "X",
+        "Y",
+        "up",
+        "right",
+        "down",
+        "left",
+    ]
+
+    def __init__(self):
+        self.lx = 0.0
+        self.ly = 0.0
+        self.rx = 0.0
+        self.ry = 0.0
+        self.button = [0] * 16
+        self.remote_action = dict.fromkeys(REMOTE_AXES, 0.0)
+
+        # SDK joystick parser for wireless remote bytes
+        self._joystick = Joystick()
+        # Disable axis smoothing and deadzone to preserve raw values
+        for axis in (self._joystick.lx, self._joystick.ly, self._joystick.rx, self._joystick.ry):
+            axis.smooth = 1.0
+            axis.deadzone = 0.0
+
+        # Joystick center calibration (read at connect time)
+        self.left_center_x = self.ADC_HALF
+        self.left_center_y = self.ADC_HALF
+        self.right_center_x = self.ADC_HALF
+        self.right_center_y = self.ADC_HALF
+
+        # Whether to use exo joystick (detected at connect time)
+        self.use_left_exo_joystick = False
+        self.use_right_exo_joystick = False
+
+    def _sync_remote_action(self) -> None:
+        self.remote_action.update(zip(REMOTE_AXES, (self.lx, self.ly, self.rx, self.ry), strict=True))
+
+    def calibrate_center(self, raw16: list[int] | None, side: str) -> None:
+        if raw16 is None or len(raw16) < 16:
+            logger.info(f"{side.capitalize()} exo joystick: no data available")
+            return
+
+        btn_val = raw16[self.JOYSTICK_BTN_IDX]
+        logger.info(f"{side.capitalize()} exo joystick button ADC: {btn_val} (threshold: {self.ADC_HALF})")
+        if btn_val <= self.ADC_HALF:
+            logger.info(f"{side.capitalize()} exo joystick not detected (button below threshold)")
+            return
+
+        x = raw16[self.JOYSTICK_X_IDX]
+        y = raw16[self.JOYSTICK_Y_IDX]
+        if side == "left":
+            self.use_left_exo_joystick = True
+            self.left_center_x, self.left_center_y = x, y
+        else:
+            self.use_right_exo_joystick = True
+            self.right_center_x, self.right_center_y = x, y
+        logger.info(f"{side.capitalize()} exo joystick enabled, center: x={x}, y={y}")
+
+    def set_from_exo(self, raw16: list[int] | None, side: str) -> None:
+        if raw16 is None or len(raw16) < 16:
+            return
+
+        if side == "left":
+            if not self.use_left_exo_joystick:
+                return
+            self.lx = (raw16[self.JOYSTICK_X_IDX] - self.left_center_x) / self.ADC_HALF
+            self.ly = (raw16[self.JOYSTICK_Y_IDX] - self.left_center_y) / self.ADC_HALF
+            self.button[4] = 1 if raw16[self.JOYSTICK_BTN_IDX] < self.ADC_HALF else 0
+            return
+
+        if not self.use_right_exo_joystick:
+            return
+        self.rx = (raw16[self.JOYSTICK_X_IDX] - self.right_center_x) / self.ADC_HALF
+        self.ry = (raw16[self.JOYSTICK_Y_IDX] - self.right_center_y) / self.ADC_HALF
+        self.button[0] = 1 if raw16[self.JOYSTICK_BTN_IDX] < self.ADC_HALF else 0
+
+    def set_from_wireless(self, wireless_remote: bytes) -> None:
+        """Parse Unitree wireless remote raw bytes into joystick + button state."""
+        if len(wireless_remote) < 24:
+            return
+        self._joystick.extract(wireless_remote)
+
+        self.lx = self._joystick.lx.data
+        self.ly = self._joystick.ly.data
+        self.rx = self._joystick.rx.data
+        self.ry = self._joystick.ry.data
+
+        for i, name in enumerate(self._BUTTON_MAP):
+            if name:
+                self.button[i] = getattr(self._joystick, name).data
+
+
+class UnitreeG1Teleoperator(Teleoperator):
+    """
+    Bimanual exoskeleton arms teleoperator for Unitree G1 arms.
+
+    Uses inverse kinematics: exoskeleton FK computes end-effector pose,
+    G1 IK solves for joint angles.
+    """
+
+    config_class = UnitreeG1TeleoperatorConfig
+    name = "unitree_g1"
+
+    def __init__(self, config: UnitreeG1TeleoperatorConfig):
+        super().__init__(config)
+        self.config = config
+        left_exo_enabled = bool(config.left_arm_config.port.strip())
+        right_exo_enabled = bool(config.right_arm_config.port.strip())
+        if left_exo_enabled != right_exo_enabled:
+            raise ValueError(
+                "Invalid exo config: set both left/right exo ports, or leave both empty for remote-only mode."
+            )
+        self._arm_control_enabled = left_exo_enabled and right_exo_enabled
+
+        # Setup calibration directory
+        self.calibration_dir = (
+            config.calibration_dir
+            if config.calibration_dir
+            else HF_LEROBOT_CALIBRATION / TELEOPERATORS / self.name
+        )
+        self.calibration_dir.mkdir(parents=True, exist_ok=True)
+
+        left_id = f"{config.id}_left" if config.id else "left"
+        right_id = f"{config.id}_right" if config.id else "right"
+
+        # Create exoskeleton arm instances
+        self.left_arm = ExoskeletonArm(
+            port=config.left_arm_config.port,
+            baud_rate=config.left_arm_config.baud_rate,
+            calibration_fpath=self.calibration_dir / f"{left_id}.json",
+            side="left",
+        )
+        self.right_arm = ExoskeletonArm(
+            port=config.right_arm_config.port,
+            baud_rate=config.right_arm_config.baud_rate,
+            calibration_fpath=self.calibration_dir / f"{right_id}.json",
+            side="right",
+        )
+
+        self.ik_helper: ExoskeletonIKHelper | None = None
+        self.remote_controller = RemoteController()
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        remote_features = dict.fromkeys(self.remote_controller.remote_action, float)
+        if not self._arm_control_enabled:
+            return remote_features
+        joint_features = {f"{name}.q": float for name in self._g1_arm_joint_names}
+        return {**joint_features, **remote_features}
+
+    @cached_property
+    def feedback_features(self) -> dict[str, type]:
+        return {"wireless_remote": bytes}
+
+    @property
+    def is_connected(self) -> bool:
+        if not self._arm_control_enabled:
+            return True
+        return self.left_arm.is_connected and self.right_arm.is_connected
+
+    @property
+    def is_calibrated(self) -> bool:
+        if not self._arm_control_enabled:
+            return True
+        return self.left_arm.is_calibrated and self.right_arm.is_calibrated
+
+    def connect(self, calibrate: bool = True) -> None:
+        if not self._arm_control_enabled:
+            logger.warning("Exo ports not fully configured; teleop will send joystick only (no arm actions)")
+            return
+
+        self.left_arm.connect(calibrate)
+        self.right_arm.connect(calibrate)
+
+        frozen_joints = [j.strip() for j in self.config.frozen_joints.split(",") if j.strip()]
+        self.ik_helper = ExoskeletonIKHelper(frozen_joints=frozen_joints)
+        logger.info("IK helper initialized")
+
+        time.sleep(0.1)  # Give serial time to populate buffer
+
+        left_raw = self.left_arm.read_raw()
+        right_raw = self.right_arm.read_raw()
+        self.remote_controller.calibrate_center(left_raw, "left")
+        self.remote_controller.calibrate_center(right_raw, "right")
+
+    def calibrate(self) -> None:
+        if not self.left_arm.is_calibrated:
+            logger.info("Starting calibration for left arm...")
+            self.left_arm.calibrate()
+        else:
+            logger.info("Left arm already calibrated. Skipping.")
+
+        if not self.right_arm.is_calibrated:
+            logger.info("Starting calibration for right arm...")
+            self.right_arm.calibrate()
+        else:
+            logger.info("Right arm already calibrated. Skipping.")
+
+        logger.info("Starting visualization to verify calibration...")
+        self.run_visualization_loop()
+
+    def configure(self) -> None:
+        pass
+
+    def get_action(self) -> dict[str, float]:
+        joint_action = {}
+        left_raw = None
+        right_raw = None
+        if self._arm_control_enabled:
+            left_raw = self.left_arm.read_raw()
+            right_raw = self.right_arm.read_raw()
+
+            left_angles = self.left_arm.get_angles()
+            right_angles = self.right_arm.get_angles()
+            joint_action = self.ik_helper.compute_g1_joints_from_exo(left_angles, right_angles)
+
+        # Wireless remote has priority when non-zero; otherwise, use exo joystick.
+        rc = self.remote_controller
+        wireless_active = (
+            abs(rc.lx) > 1e-3 or abs(rc.ly) > 1e-3 or abs(rc.rx) > 1e-3 or abs(rc.ry) > 1e-3
+        ) or any(rc.button)
+        if self._arm_control_enabled and not wireless_active:
+            rc.set_from_exo(left_raw, "left")
+            rc.set_from_exo(right_raw, "right")
+
+        rc._sync_remote_action()
+        return {**joint_action, **rc.remote_action}
+
+    def send_feedback(self, feedback: dict[str, Any]) -> None:
+        wireless_remote = feedback.get("wireless_remote")
+        if wireless_remote is not None:
+            self.remote_controller.set_from_wireless(wireless_remote)
+
+    def disconnect(self) -> None:
+        self.left_arm.disconnect()
+        self.right_arm.disconnect()
+
+    def run_visualization_loop(self):
+        """Run interactive Meshcat visualization loop to verify tracking."""
+        if self.ik_helper is None:
+            frozen_joints = [j.strip() for j in self.config.frozen_joints.split(",") if j.strip()]
+            self.ik_helper = ExoskeletonIKHelper(frozen_joints=frozen_joints)
+
+        self.ik_helper.init_visualization()
+
+        print("\n" + "=" * 60)
+        print("Visualization running! Move the exoskeletons to test tracking.")
+        print("Press Ctrl+C to exit.")
+        print("=" * 60 + "\n")
+
+        try:
+            while True:
+                left_angles = self.left_arm.get_angles()
+                right_angles = self.right_arm.get_angles()
+
+                self.ik_helper.compute_g1_joints_from_exo(left_angles, right_angles)
+                self.ik_helper.update_visualization()
+
+                time.sleep(0.01)
+
+        except KeyboardInterrupt:
+            print("\n\nVisualization stopped.")
+
+    @cached_property
+    def _g1_arm_joint_names(self) -> list[str]:
+        return [joint.name for joint in G1_29_JointArmIndex]
diff --git a/lerobot/src/lerobot/teleoperators/utils.py b/lerobot/src/lerobot/teleoperators/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..db685f396340efdec51659451abe79ed2fa60348
--- /dev/null
+++ b/lerobot/src/lerobot/teleoperators/utils.py
@@ -0,0 +1,106 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from enum import Enum
+from typing import TYPE_CHECKING, cast
+
+from lerobot.utils.import_utils import make_device_from_device_class
+
+from .config import TeleoperatorConfig
+
+if TYPE_CHECKING:
+    from .teleoperator import Teleoperator
+
+
+class TeleopEvents(Enum):
+    """Shared constants for teleoperator events across teleoperators."""
+
+    SUCCESS = "success"
+    FAILURE = "failure"
+    RERECORD_EPISODE = "rerecord_episode"
+    IS_INTERVENTION = "is_intervention"
+    TERMINATE_EPISODE = "terminate_episode"
+
+
+def make_teleoperator_from_config(config: TeleoperatorConfig) -> "Teleoperator":
+    # TODO(Steven): Consider just using the make_device_from_device_class for all types
+    if config.type == "keyboard":
+        from .keyboard import KeyboardTeleop
+
+        return KeyboardTeleop(config)
+    elif config.type == "koch_leader":
+        from .koch_leader import KochLeader
+
+        return KochLeader(config)
+    elif config.type == "omx_leader":
+        from .omx_leader import OmxLeader
+
+        return OmxLeader(config)
+    elif config.type == "so100_leader":
+        from .so_leader import SO100Leader
+
+        return SO100Leader(config)
+    elif config.type == "so101_leader":
+        from .so_leader import SO101Leader
+
+        return SO101Leader(config)
+    elif config.type == "mock_teleop":
+        from tests.mocks.mock_teleop import MockTeleop
+
+        return MockTeleop(config)
+    elif config.type == "gamepad":
+        from .gamepad.teleop_gamepad import GamepadTeleop
+
+        return GamepadTeleop(config)
+    elif config.type == "keyboard_ee":
+        from .keyboard.teleop_keyboard import KeyboardEndEffectorTeleop
+
+        return KeyboardEndEffectorTeleop(config)
+    elif config.type == "homunculus_glove":
+        from .homunculus import HomunculusGlove
+
+        return HomunculusGlove(config)
+    elif config.type == "homunculus_arm":
+        from .homunculus import HomunculusArm
+
+        return HomunculusArm(config)
+    elif config.type == "unitree_g1":
+        from .unitree_g1 import UnitreeG1Teleoperator
+
+        return UnitreeG1Teleoperator(config)
+    elif config.type == "bi_so_leader":
+        from .bi_so_leader import BiSOLeader
+
+        return BiSOLeader(config)
+    elif config.type == "reachy2_teleoperator":
+        from .reachy2_teleoperator import Reachy2Teleoperator
+
+        return Reachy2Teleoperator(config)
+    elif config.type == "openarm_leader":
+        from .openarm_leader import OpenArmLeader
+
+        return OpenArmLeader(config)
+    elif config.type == "bi_openarm_leader":
+        from .bi_openarm_leader import BiOpenArmLeader
+
+        return BiOpenArmLeader(config)
+    elif config.type == "openarm_mini":
+        from .openarm_mini import OpenArmMini
+
+        return OpenArmMini(config)
+    else:
+        try:
+            return cast("Teleoperator", make_device_from_device_class(config))
+        except Exception as e:
+            raise ValueError(f"Error creating robot with config {config}: {e}") from e
diff --git a/lerobot/src/lerobot/templates/lerobot_modelcard_template.md b/lerobot/src/lerobot/templates/lerobot_modelcard_template.md
new file mode 100644
index 0000000000000000000000000000000000000000..c59cf418385d4cc9bc6ee2e2479f08eceedc171b
--- /dev/null
+++ b/lerobot/src/lerobot/templates/lerobot_modelcard_template.md
@@ -0,0 +1,91 @@
+---
+# For reference on model card metadata, see the spec: https://github.com/huggingface/hub-docs/blob/main/modelcard.md?plain=1
+# Doc / guide: https://huggingface.co/docs/hub/model-cards
+# prettier-ignore
+{{card_data}}
+---
+
+# Model Card for {{ model_name | default("Model ID", true) }}
+
+<!-- Provide a quick summary of what the model is/does. -->
+
+{% if model_name == "smolvla" %}
+[SmolVLA](https://huggingface.co/papers/2506.01844) is a compact, efficient vision-language-action model that achieves competitive performance at reduced computational costs and can be deployed on consumer-grade hardware.
+{% elif model_name == "act" %}
+[Action Chunking with Transformers (ACT)](https://huggingface.co/papers/2304.13705) is an imitation-learning method that predicts short action chunks instead of single steps. It learns from teleoperated data and often achieves high success rates.
+{% elif model_name == "tdmpc" %}
+[TD-MPC](https://huggingface.co/papers/2203.04955) combines model-free and model-based approaches to improve sample efficiency and performance in continuous control tasks by using a learned latent dynamics model and terminal value function.
+{% elif model_name == "diffusion" %}
+[Diffusion Policy](https://huggingface.co/papers/2303.04137) treats visuomotor control as a generative diffusion process, producing smooth, multi-step action trajectories that excel at contact-rich manipulation.
+{% elif model_name == "vqbet" %}
+[VQ-BET](https://huggingface.co/papers/2403.03181) combines vector-quantised action tokens with Behaviour Transformers to discretise control and achieve data-efficient imitation across diverse skills.
+{% elif model_name == "pi0" %}
+**π₀ (Pi0)**
+
+π₀ is a Vision-Language-Action model for general robot control, from Physical Intelligence. The LeRobot implementation is adapted from their open source OpenPI repository.
+
+**Model Overview**
+
+π₀ represents a breakthrough in robotics as the first general-purpose robot foundation model developed by Physical Intelligence. Unlike traditional robots that are narrow specialists programmed for repetitive motions, π₀ is designed to be a generalist policy that can understand visual inputs, interpret natural language instructions, and control a variety of different robots across diverse tasks.
+
+For more details, see the [Physical Intelligence π₀ blog post](https://www.physicalintelligence.company/blog/pi0).
+{% elif model_name == "pi05" %}
+**π₀.₅ (Pi05) Policy**
+
+π₀.₅ is a Vision-Language-Action model with open-world generalization, from Physical Intelligence. The LeRobot implementation is adapted from their open source OpenPI repository.
+
+**Model Overview**
+
+π₀.₅ represents a significant evolution from π₀, developed by Physical Intelligence to address a big challenge in robotics: open-world generalization. While robots can perform impressive tasks in controlled environments, π₀.₅ is designed to generalize to entirely new environments and situations that were never seen during training.
+
+For more details, see the [Physical Intelligence π₀.₅ blog post](https://www.physicalintelligence.company/blog/pi05).
+{% elif model_name == "sac" %}
+[Soft Actor-Critic (SAC)](https://huggingface.co/papers/1801.01290) is an entropy-regularised actor-critic algorithm offering stable, sample-efficient learning in continuous-control environments.
+{% elif model_name == "reward_classifier" %}
+A reward classifier is a lightweight neural network that scores observations or trajectories for task success, providing a learned reward signal or offline evaluation when explicit rewards are unavailable.
+{% else %}
+_Model type not recognized — please update this template._
+{% endif %}
+
+This policy has been trained and pushed to the Hub using [LeRobot](https://github.com/huggingface/lerobot).
+See the full documentation at [LeRobot Docs](https://huggingface.co/docs/lerobot/index).
+
+---
+
+## How to Get Started with the Model
+
+For a complete walkthrough, see the [training guide](https://huggingface.co/docs/lerobot/il_robots#train-a-policy).
+Below is the short version on how to train and run inference/eval:
+
+### Train from scratch
+
+```bash
+lerobot-train \
+  --dataset.repo_id=${HF_USER}/<dataset> \
+  --policy.type=act \
+  --output_dir=outputs/train/<desired_policy_repo_id> \
+  --job_name=lerobot_training \
+  --policy.device=cuda \
+  --policy.repo_id=${HF_USER}/<desired_policy_repo_id>
+  --wandb.enable=true
+```
+
+_Writes checkpoints to `outputs/train/<desired_policy_repo_id>/checkpoints/`._
+
+### Evaluate the policy/run inference
+
+```bash
+lerobot-record \
+  --robot.type=so100_follower \
+  --dataset.repo_id=<hf_user>/eval_<dataset> \
+  --policy.path=<hf_user>/<desired_policy_repo_id> \
+  --episodes=10
+```
+
+Prefix the dataset repo with **eval\_** and supply `--policy.path` pointing to a local or hub checkpoint.
+
+---
+
+## Model Details
+
+- **License:** {{ license | default("\[More Information Needed]", true) }}
diff --git a/lerobot/src/lerobot/transport/services.proto b/lerobot/src/lerobot/transport/services.proto
new file mode 100644
index 0000000000000000000000000000000000000000..ea0c12de673564d1479ff1181741d284414be884
--- /dev/null
+++ b/lerobot/src/lerobot/transport/services.proto
@@ -0,0 +1,87 @@
+//  Copyright 2024 The HuggingFace Inc. team.
+//  All rights reserved.
+
+//  Licensed under the Apache License, Version 2.0 (the "License");
+//  you may not use this file except in compliance with the License.
+//  You may obtain a copy of the License at
+
+//      http://www.apache.org/licenses/LICENSE-2.0
+
+//  Unless required by applicable law or agreed to in writing, software
+//  distributed under the License is distributed on an "AS IS" BASIS,
+//  WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+//  See the License for the specific language governing permissions and
+//  limitations under the License.python -m grpc_tools.protoc -I src --python_out=src --grpc_python_out=src src/lerobot/transport/services.proto
+
+// To generate a classes for transport part (services_pb2.py and services_pb2_grpc.py) use the following command:
+//
+// python -m grpc_tools.protoc -I src --python_out=src --grpc_python_out=src src/lerobot/transport/services.proto
+//
+// The command should be launched from the root of the project.
+
+syntax = "proto3";
+
+package transport;
+
+// LearnerService: the Actor calls this to push transitions.
+// The Learner implements this service.
+service LearnerService {
+  // Actor -> Learner to store transitions
+  rpc StreamParameters(Empty) returns (stream Parameters);
+  rpc SendTransitions(stream Transition) returns (Empty);
+  rpc SendInteractions(stream InteractionMessage) returns (Empty);
+  rpc Ready(Empty) returns (Empty);
+}
+
+// AsyncInference: from Robot perspective
+// Robot send observations to & executes action received from a remote Policy server
+service AsyncInference {
+  // Robot -> Policy to share observations with a remote inference server
+  // Policy -> Robot to share actions predicted for given observations
+  rpc SendObservations(stream Observation) returns (Empty);
+  rpc GetActions(Empty) returns (Actions);
+  rpc SendPolicyInstructions(PolicySetup) returns (Empty);
+  rpc Ready(Empty) returns (Empty);
+}
+
+enum TransferState {
+    TRANSFER_UNKNOWN = 0;
+    TRANSFER_BEGIN = 1;
+    TRANSFER_MIDDLE = 2;
+    TRANSFER_END = 3;
+}
+
+// Messages
+message Transition {
+  TransferState transfer_state = 1;
+  bytes data = 2;
+}
+
+message Parameters {
+  TransferState transfer_state = 1;
+  bytes data = 2;
+}
+
+message InteractionMessage {
+  TransferState transfer_state = 1;
+  bytes data = 2;
+}
+
+// Messages
+message Observation {
+  // sent by Robot, to remote Policy
+  TransferState transfer_state = 1;  // Observations can be streamed exceeding 4MB of size
+  bytes data = 2;
+}
+
+message Actions {
+  // sent by remote Policy, to Robot
+  bytes data = 1;
+}
+
+message PolicySetup {
+  // sent by Robot to remote server, to init Policy
+  bytes data = 1;
+}
+
+message Empty {}
diff --git a/lerobot/src/lerobot/transport/services_pb2.py b/lerobot/src/lerobot/transport/services_pb2.py
new file mode 100644
index 0000000000000000000000000000000000000000..05f2d174fd4486f9e3c6a39c9556d4527b65b400
--- /dev/null
+++ b/lerobot/src/lerobot/transport/services_pb2.py
@@ -0,0 +1,53 @@
+# Generated by the protocol buffer compiler.  DO NOT EDIT!
+# NO CHECKED-IN PROTOBUF GENCODE
+# source: lerobot/transport/services.proto
+# Protobuf Python Version: 6.31.0
+"""Generated protocol buffer code."""
+from google.protobuf import descriptor as _descriptor
+from google.protobuf import descriptor_pool as _descriptor_pool
+from google.protobuf import runtime_version as _runtime_version
+from google.protobuf import symbol_database as _symbol_database
+from google.protobuf.internal import builder as _builder
+_runtime_version.ValidateProtobufRuntimeVersion(
+    _runtime_version.Domain.PUBLIC,
+    6,
+    31,
+    0,
+    '',
+    'lerobot/transport/services.proto'
+)
+# @@protoc_insertion_point(imports)
+
+_sym_db = _symbol_database.Default()
+
+
+
+
+DESCRIPTOR = _descriptor_pool.Default().AddSerializedFile(b'\n lerobot/transport/services.proto\x12\ttransport\"L\n\nTransition\x12\x30\n\x0etransfer_state\x18\x01 \x01(\x0e\x32\x18.transport.TransferState\x12\x0c\n\x04\x64\x61ta\x18\x02 \x01(\x0c\"L\n\nParameters\x12\x30\n\x0etransfer_state\x18\x01 \x01(\x0e\x32\x18.transport.TransferState\x12\x0c\n\x04\x64\x61ta\x18\x02 \x01(\x0c\"T\n\x12InteractionMessage\x12\x30\n\x0etransfer_state\x18\x01 \x01(\x0e\x32\x18.transport.TransferState\x12\x0c\n\x04\x64\x61ta\x18\x02 \x01(\x0c\"M\n\x0bObservation\x12\x30\n\x0etransfer_state\x18\x01 \x01(\x0e\x32\x18.transport.TransferState\x12\x0c\n\x04\x64\x61ta\x18\x02 \x01(\x0c\"\x17\n\x07\x41\x63tions\x12\x0c\n\x04\x64\x61ta\x18\x01 \x01(\x0c\"\x1b\n\x0bPolicySetup\x12\x0c\n\x04\x64\x61ta\x18\x01 \x01(\x0c\"\x07\n\x05\x45mpty*`\n\rTransferState\x12\x14\n\x10TRANSFER_UNKNOWN\x10\x00\x12\x12\n\x0eTRANSFER_BEGIN\x10\x01\x12\x13\n\x0fTRANSFER_MIDDLE\x10\x02\x12\x10\n\x0cTRANSFER_END\x10\x03\x32\x81\x02\n\x0eLearnerService\x12=\n\x10StreamParameters\x12\x10.transport.Empty\x1a\x15.transport.Parameters0\x01\x12<\n\x0fSendTransitions\x12\x15.transport.Transition\x1a\x10.transport.Empty(\x01\x12\x45\n\x10SendInteractions\x12\x1d.transport.InteractionMessage\x1a\x10.transport.Empty(\x01\x12+\n\x05Ready\x12\x10.transport.Empty\x1a\x10.transport.Empty2\xf5\x01\n\x0e\x41syncInference\x12>\n\x10SendObservations\x12\x16.transport.Observation\x1a\x10.transport.Empty(\x01\x12\x32\n\nGetActions\x12\x10.transport.Empty\x1a\x12.transport.Actions\x12\x42\n\x16SendPolicyInstructions\x12\x16.transport.PolicySetup\x1a\x10.transport.Empty\x12+\n\x05Ready\x12\x10.transport.Empty\x1a\x10.transport.Emptyb\x06proto3')
+
+_globals = globals()
+_builder.BuildMessageAndEnumDescriptors(DESCRIPTOR, _globals)
+_builder.BuildTopDescriptorsAndMessages(DESCRIPTOR, 'lerobot.transport.services_pb2', _globals)
+if not _descriptor._USE_C_DESCRIPTORS:
+  DESCRIPTOR._loaded_options = None
+  _globals['_TRANSFERSTATE']._serialized_start=431
+  _globals['_TRANSFERSTATE']._serialized_end=527
+  _globals['_TRANSITION']._serialized_start=47
+  _globals['_TRANSITION']._serialized_end=123
+  _globals['_PARAMETERS']._serialized_start=125
+  _globals['_PARAMETERS']._serialized_end=201
+  _globals['_INTERACTIONMESSAGE']._serialized_start=203
+  _globals['_INTERACTIONMESSAGE']._serialized_end=287
+  _globals['_OBSERVATION']._serialized_start=289
+  _globals['_OBSERVATION']._serialized_end=366
+  _globals['_ACTIONS']._serialized_start=368
+  _globals['_ACTIONS']._serialized_end=391
+  _globals['_POLICYSETUP']._serialized_start=393
+  _globals['_POLICYSETUP']._serialized_end=420
+  _globals['_EMPTY']._serialized_start=422
+  _globals['_EMPTY']._serialized_end=429
+  _globals['_LEARNERSERVICE']._serialized_start=530
+  _globals['_LEARNERSERVICE']._serialized_end=787
+  _globals['_ASYNCINFERENCE']._serialized_start=790
+  _globals['_ASYNCINFERENCE']._serialized_end=1035
+# @@protoc_insertion_point(module_scope)
diff --git a/lerobot/src/lerobot/transport/services_pb2_grpc.py b/lerobot/src/lerobot/transport/services_pb2_grpc.py
new file mode 100644
index 0000000000000000000000000000000000000000..35a01b6754ecf6d8dab41a9629da4badc83d5632
--- /dev/null
+++ b/lerobot/src/lerobot/transport/services_pb2_grpc.py
@@ -0,0 +1,442 @@
+# Generated by the gRPC Python protocol compiler plugin. DO NOT EDIT!
+"""Client and server classes corresponding to protobuf-defined services."""
+import grpc
+import warnings
+
+from lerobot.transport import services_pb2 as lerobot_dot_transport_dot_services__pb2
+
+GRPC_GENERATED_VERSION = '1.73.1'
+GRPC_VERSION = grpc.__version__
+_version_not_supported = False
+
+try:
+    from grpc._utilities import first_version_is_lower
+    _version_not_supported = first_version_is_lower(GRPC_VERSION, GRPC_GENERATED_VERSION)
+except ImportError:
+    _version_not_supported = True
+
+if _version_not_supported:
+    raise RuntimeError(
+        f'The grpc package installed is at version {GRPC_VERSION},'
+        + f' but the generated code in lerobot/transport/services_pb2_grpc.py depends on'
+        + f' grpcio>={GRPC_GENERATED_VERSION}.'
+        + f' Please upgrade your grpc module to grpcio>={GRPC_GENERATED_VERSION}'
+        + f' or downgrade your generated code using grpcio-tools<={GRPC_VERSION}.'
+    )
+
+
+class LearnerServiceStub:
+    """LearnerService: the Actor calls this to push transitions.
+    The Learner implements this service.
+    """
+
+    def __init__(self, channel):
+        """Constructor.
+
+        Args:
+            channel: A grpc.Channel.
+        """
+        self.StreamParameters = channel.unary_stream(
+                '/transport.LearnerService/StreamParameters',
+                request_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+                response_deserializer=lerobot_dot_transport_dot_services__pb2.Parameters.FromString,
+                _registered_method=True)
+        self.SendTransitions = channel.stream_unary(
+                '/transport.LearnerService/SendTransitions',
+                request_serializer=lerobot_dot_transport_dot_services__pb2.Transition.SerializeToString,
+                response_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                _registered_method=True)
+        self.SendInteractions = channel.stream_unary(
+                '/transport.LearnerService/SendInteractions',
+                request_serializer=lerobot_dot_transport_dot_services__pb2.InteractionMessage.SerializeToString,
+                response_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                _registered_method=True)
+        self.Ready = channel.unary_unary(
+                '/transport.LearnerService/Ready',
+                request_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+                response_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                _registered_method=True)
+
+
+class LearnerServiceServicer:
+    """LearnerService: the Actor calls this to push transitions.
+    The Learner implements this service.
+    """
+
+    def StreamParameters(self, request, context):
+        """Actor -> Learner to store transitions
+        """
+        context.set_code(grpc.StatusCode.UNIMPLEMENTED)
+        context.set_details('Method not implemented!')
+        raise NotImplementedError('Method not implemented!')
+
+    def SendTransitions(self, request_iterator, context):
+        """Missing associated documentation comment in .proto file."""
+        context.set_code(grpc.StatusCode.UNIMPLEMENTED)
+        context.set_details('Method not implemented!')
+        raise NotImplementedError('Method not implemented!')
+
+    def SendInteractions(self, request_iterator, context):
+        """Missing associated documentation comment in .proto file."""
+        context.set_code(grpc.StatusCode.UNIMPLEMENTED)
+        context.set_details('Method not implemented!')
+        raise NotImplementedError('Method not implemented!')
+
+    def Ready(self, request, context):
+        """Missing associated documentation comment in .proto file."""
+        context.set_code(grpc.StatusCode.UNIMPLEMENTED)
+        context.set_details('Method not implemented!')
+        raise NotImplementedError('Method not implemented!')
+
+
+def add_LearnerServiceServicer_to_server(servicer, server):
+    rpc_method_handlers = {
+            'StreamParameters': grpc.unary_stream_rpc_method_handler(
+                    servicer.StreamParameters,
+                    request_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                    response_serializer=lerobot_dot_transport_dot_services__pb2.Parameters.SerializeToString,
+            ),
+            'SendTransitions': grpc.stream_unary_rpc_method_handler(
+                    servicer.SendTransitions,
+                    request_deserializer=lerobot_dot_transport_dot_services__pb2.Transition.FromString,
+                    response_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            ),
+            'SendInteractions': grpc.stream_unary_rpc_method_handler(
+                    servicer.SendInteractions,
+                    request_deserializer=lerobot_dot_transport_dot_services__pb2.InteractionMessage.FromString,
+                    response_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            ),
+            'Ready': grpc.unary_unary_rpc_method_handler(
+                    servicer.Ready,
+                    request_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                    response_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            ),
+    }
+    generic_handler = grpc.method_handlers_generic_handler(
+            'transport.LearnerService', rpc_method_handlers)
+    server.add_generic_rpc_handlers((generic_handler,))
+    server.add_registered_method_handlers('transport.LearnerService', rpc_method_handlers)
+
+
+ # This class is part of an EXPERIMENTAL API.
+class LearnerService:
+    """LearnerService: the Actor calls this to push transitions.
+    The Learner implements this service.
+    """
+
+    @staticmethod
+    def StreamParameters(request,
+            target,
+            options=(),
+            channel_credentials=None,
+            call_credentials=None,
+            insecure=False,
+            compression=None,
+            wait_for_ready=None,
+            timeout=None,
+            metadata=None):
+        return grpc.experimental.unary_stream(
+            request,
+            target,
+            '/transport.LearnerService/StreamParameters',
+            lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            lerobot_dot_transport_dot_services__pb2.Parameters.FromString,
+            options,
+            channel_credentials,
+            insecure,
+            call_credentials,
+            compression,
+            wait_for_ready,
+            timeout,
+            metadata,
+            _registered_method=True)
+
+    @staticmethod
+    def SendTransitions(request_iterator,
+            target,
+            options=(),
+            channel_credentials=None,
+            call_credentials=None,
+            insecure=False,
+            compression=None,
+            wait_for_ready=None,
+            timeout=None,
+            metadata=None):
+        return grpc.experimental.stream_unary(
+            request_iterator,
+            target,
+            '/transport.LearnerService/SendTransitions',
+            lerobot_dot_transport_dot_services__pb2.Transition.SerializeToString,
+            lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+            options,
+            channel_credentials,
+            insecure,
+            call_credentials,
+            compression,
+            wait_for_ready,
+            timeout,
+            metadata,
+            _registered_method=True)
+
+    @staticmethod
+    def SendInteractions(request_iterator,
+            target,
+            options=(),
+            channel_credentials=None,
+            call_credentials=None,
+            insecure=False,
+            compression=None,
+            wait_for_ready=None,
+            timeout=None,
+            metadata=None):
+        return grpc.experimental.stream_unary(
+            request_iterator,
+            target,
+            '/transport.LearnerService/SendInteractions',
+            lerobot_dot_transport_dot_services__pb2.InteractionMessage.SerializeToString,
+            lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+            options,
+            channel_credentials,
+            insecure,
+            call_credentials,
+            compression,
+            wait_for_ready,
+            timeout,
+            metadata,
+            _registered_method=True)
+
+    @staticmethod
+    def Ready(request,
+            target,
+            options=(),
+            channel_credentials=None,
+            call_credentials=None,
+            insecure=False,
+            compression=None,
+            wait_for_ready=None,
+            timeout=None,
+            metadata=None):
+        return grpc.experimental.unary_unary(
+            request,
+            target,
+            '/transport.LearnerService/Ready',
+            lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+            options,
+            channel_credentials,
+            insecure,
+            call_credentials,
+            compression,
+            wait_for_ready,
+            timeout,
+            metadata,
+            _registered_method=True)
+
+
+class AsyncInferenceStub:
+    """AsyncInference: from Robot perspective
+    Robot send observations to & executes action received from a remote Policy server
+    """
+
+    def __init__(self, channel):
+        """Constructor.
+
+        Args:
+            channel: A grpc.Channel.
+        """
+        self.SendObservations = channel.stream_unary(
+                '/transport.AsyncInference/SendObservations',
+                request_serializer=lerobot_dot_transport_dot_services__pb2.Observation.SerializeToString,
+                response_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                _registered_method=True)
+        self.GetActions = channel.unary_unary(
+                '/transport.AsyncInference/GetActions',
+                request_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+                response_deserializer=lerobot_dot_transport_dot_services__pb2.Actions.FromString,
+                _registered_method=True)
+        self.SendPolicyInstructions = channel.unary_unary(
+                '/transport.AsyncInference/SendPolicyInstructions',
+                request_serializer=lerobot_dot_transport_dot_services__pb2.PolicySetup.SerializeToString,
+                response_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                _registered_method=True)
+        self.Ready = channel.unary_unary(
+                '/transport.AsyncInference/Ready',
+                request_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+                response_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                _registered_method=True)
+
+
+class AsyncInferenceServicer:
+    """AsyncInference: from Robot perspective
+    Robot send observations to & executes action received from a remote Policy server
+    """
+
+    def SendObservations(self, request_iterator, context):
+        """Robot -> Policy to share observations with a remote inference server
+        Policy -> Robot to share actions predicted for given observations
+        """
+        context.set_code(grpc.StatusCode.UNIMPLEMENTED)
+        context.set_details('Method not implemented!')
+        raise NotImplementedError('Method not implemented!')
+
+    def GetActions(self, request, context):
+        """Missing associated documentation comment in .proto file."""
+        context.set_code(grpc.StatusCode.UNIMPLEMENTED)
+        context.set_details('Method not implemented!')
+        raise NotImplementedError('Method not implemented!')
+
+    def SendPolicyInstructions(self, request, context):
+        """Missing associated documentation comment in .proto file."""
+        context.set_code(grpc.StatusCode.UNIMPLEMENTED)
+        context.set_details('Method not implemented!')
+        raise NotImplementedError('Method not implemented!')
+
+    def Ready(self, request, context):
+        """Missing associated documentation comment in .proto file."""
+        context.set_code(grpc.StatusCode.UNIMPLEMENTED)
+        context.set_details('Method not implemented!')
+        raise NotImplementedError('Method not implemented!')
+
+
+def add_AsyncInferenceServicer_to_server(servicer, server):
+    rpc_method_handlers = {
+            'SendObservations': grpc.stream_unary_rpc_method_handler(
+                    servicer.SendObservations,
+                    request_deserializer=lerobot_dot_transport_dot_services__pb2.Observation.FromString,
+                    response_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            ),
+            'GetActions': grpc.unary_unary_rpc_method_handler(
+                    servicer.GetActions,
+                    request_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                    response_serializer=lerobot_dot_transport_dot_services__pb2.Actions.SerializeToString,
+            ),
+            'SendPolicyInstructions': grpc.unary_unary_rpc_method_handler(
+                    servicer.SendPolicyInstructions,
+                    request_deserializer=lerobot_dot_transport_dot_services__pb2.PolicySetup.FromString,
+                    response_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            ),
+            'Ready': grpc.unary_unary_rpc_method_handler(
+                    servicer.Ready,
+                    request_deserializer=lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+                    response_serializer=lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            ),
+    }
+    generic_handler = grpc.method_handlers_generic_handler(
+            'transport.AsyncInference', rpc_method_handlers)
+    server.add_generic_rpc_handlers((generic_handler,))
+    server.add_registered_method_handlers('transport.AsyncInference', rpc_method_handlers)
+
+
+ # This class is part of an EXPERIMENTAL API.
+class AsyncInference:
+    """AsyncInference: from Robot perspective
+    Robot send observations to & executes action received from a remote Policy server
+    """
+
+    @staticmethod
+    def SendObservations(request_iterator,
+            target,
+            options=(),
+            channel_credentials=None,
+            call_credentials=None,
+            insecure=False,
+            compression=None,
+            wait_for_ready=None,
+            timeout=None,
+            metadata=None):
+        return grpc.experimental.stream_unary(
+            request_iterator,
+            target,
+            '/transport.AsyncInference/SendObservations',
+            lerobot_dot_transport_dot_services__pb2.Observation.SerializeToString,
+            lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+            options,
+            channel_credentials,
+            insecure,
+            call_credentials,
+            compression,
+            wait_for_ready,
+            timeout,
+            metadata,
+            _registered_method=True)
+
+    @staticmethod
+    def GetActions(request,
+            target,
+            options=(),
+            channel_credentials=None,
+            call_credentials=None,
+            insecure=False,
+            compression=None,
+            wait_for_ready=None,
+            timeout=None,
+            metadata=None):
+        return grpc.experimental.unary_unary(
+            request,
+            target,
+            '/transport.AsyncInference/GetActions',
+            lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            lerobot_dot_transport_dot_services__pb2.Actions.FromString,
+            options,
+            channel_credentials,
+            insecure,
+            call_credentials,
+            compression,
+            wait_for_ready,
+            timeout,
+            metadata,
+            _registered_method=True)
+
+    @staticmethod
+    def SendPolicyInstructions(request,
+            target,
+            options=(),
+            channel_credentials=None,
+            call_credentials=None,
+            insecure=False,
+            compression=None,
+            wait_for_ready=None,
+            timeout=None,
+            metadata=None):
+        return grpc.experimental.unary_unary(
+            request,
+            target,
+            '/transport.AsyncInference/SendPolicyInstructions',
+            lerobot_dot_transport_dot_services__pb2.PolicySetup.SerializeToString,
+            lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+            options,
+            channel_credentials,
+            insecure,
+            call_credentials,
+            compression,
+            wait_for_ready,
+            timeout,
+            metadata,
+            _registered_method=True)
+
+    @staticmethod
+    def Ready(request,
+            target,
+            options=(),
+            channel_credentials=None,
+            call_credentials=None,
+            insecure=False,
+            compression=None,
+            wait_for_ready=None,
+            timeout=None,
+            metadata=None):
+        return grpc.experimental.unary_unary(
+            request,
+            target,
+            '/transport.AsyncInference/Ready',
+            lerobot_dot_transport_dot_services__pb2.Empty.SerializeToString,
+            lerobot_dot_transport_dot_services__pb2.Empty.FromString,
+            options,
+            channel_credentials,
+            insecure,
+            call_credentials,
+            compression,
+            wait_for_ready,
+            timeout,
+            metadata,
+            _registered_method=True)
diff --git a/lerobot/src/lerobot/transport/utils.py b/lerobot/src/lerobot/transport/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..8da33804409da1add464f31022f926699e74bf8e
--- /dev/null
+++ b/lerobot/src/lerobot/transport/utils.py
@@ -0,0 +1,189 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team.
+# All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import io
+import json
+import logging
+import pickle  # nosec B403: Safe usage for internal serialization only
+from multiprocessing.synchronize import Event as MpEvent
+from queue import Queue
+from typing import Any
+
+import torch
+
+from lerobot.transport import services_pb2
+from lerobot.utils.transition import Transition
+
+# FIX for protobuf: Assign the enum to a variable and ignore the type error once
+TransferState = services_pb2.TransferState  # type: ignore[attr-defined]
+
+CHUNK_SIZE = 2 * 1024 * 1024  # 2 MB
+MAX_MESSAGE_SIZE = 4 * 1024 * 1024  # 4 MB
+
+
+def bytes_buffer_size(buffer: io.BytesIO) -> int:
+    buffer.seek(0, io.SEEK_END)
+    result = buffer.tell()
+    buffer.seek(0)
+    return result
+
+
+def send_bytes_in_chunks(buffer: bytes, message_class: Any, log_prefix: str = "", silent: bool = True):
+    bytes_buffer: io.BytesIO = io.BytesIO(buffer)
+    size_in_bytes = bytes_buffer_size(bytes_buffer)
+
+    sent_bytes = 0
+
+    logging_method = logging.info if not silent else logging.debug
+
+    logging_method(f"{log_prefix} Buffer size {size_in_bytes / 1024 / 1024} MB with")
+
+    while sent_bytes < size_in_bytes:
+        transfer_state = TransferState.TRANSFER_MIDDLE
+
+        if sent_bytes + CHUNK_SIZE >= size_in_bytes:
+            transfer_state = TransferState.TRANSFER_END
+        elif sent_bytes == 0:
+            transfer_state = TransferState.TRANSFER_BEGIN
+
+        size_to_read = min(CHUNK_SIZE, size_in_bytes - sent_bytes)
+        chunk = bytes_buffer.read(size_to_read)
+
+        yield message_class(transfer_state=transfer_state, data=chunk)
+        sent_bytes += size_to_read
+        logging_method(f"{log_prefix} Sent {sent_bytes}/{size_in_bytes} bytes with state {transfer_state}")
+
+    logging_method(f"{log_prefix} Published {sent_bytes / 1024 / 1024} MB")
+
+
+def receive_bytes_in_chunks(iterator, queue: Queue | None, shutdown_event: MpEvent, log_prefix: str = ""):
+    bytes_buffer = io.BytesIO()
+    step = 0
+
+    logging.info(f"{log_prefix} Starting receiver")
+    for item in iterator:
+        logging.debug(f"{log_prefix} Received item")
+        if shutdown_event.is_set():
+            logging.info(f"{log_prefix} Shutting down receiver")
+            return
+
+        if item.transfer_state == TransferState.TRANSFER_BEGIN:
+            bytes_buffer.seek(0)
+            bytes_buffer.truncate(0)
+            bytes_buffer.write(item.data)
+            logging.debug(f"{log_prefix} Received data at step 0")
+            step = 0
+        elif item.transfer_state == TransferState.TRANSFER_MIDDLE:
+            bytes_buffer.write(item.data)
+            step += 1
+            logging.debug(f"{log_prefix} Received data at step {step}")
+        elif item.transfer_state == TransferState.TRANSFER_END:
+            bytes_buffer.write(item.data)
+            logging.debug(f"{log_prefix} Received data at step end size {bytes_buffer_size(bytes_buffer)}")
+
+            if queue is not None:
+                queue.put(bytes_buffer.getvalue())
+            else:
+                return bytes_buffer.getvalue()
+
+            bytes_buffer.seek(0)
+            bytes_buffer.truncate(0)
+            step = 0
+
+            logging.debug(f"{log_prefix} Queue updated")
+        else:
+            logging.warning(f"{log_prefix} Received unknown transfer state {item.transfer_state}")
+            raise ValueError(f"Received unknown transfer state {item.transfer_state}")
+
+
+def state_to_bytes(state_dict: dict[str, torch.Tensor]) -> bytes:
+    """Convert model state dict to flat array for transmission"""
+    bytes_buffer = io.BytesIO()
+
+    torch.save(state_dict, bytes_buffer)
+
+    return bytes_buffer.getvalue()
+
+
+def bytes_to_state_dict(buffer: bytes) -> dict[str, torch.Tensor]:
+    bytes_buffer = io.BytesIO(buffer)
+    bytes_buffer.seek(0)
+    return torch.load(bytes_buffer, weights_only=True)
+
+
+def python_object_to_bytes(python_object: Any) -> bytes:
+    return pickle.dumps(python_object)
+
+
+def bytes_to_python_object(buffer: bytes) -> Any:
+    bytes_buffer = io.BytesIO(buffer)
+    bytes_buffer.seek(0)
+    obj = pickle.load(bytes_buffer)  # nosec B301: Safe usage of pickle.load
+    # Add validation checks here
+    return obj
+
+
+def bytes_to_transitions(buffer: bytes) -> list[Transition]:
+    bytes_buffer = io.BytesIO(buffer)
+    bytes_buffer.seek(0)
+    transitions = torch.load(bytes_buffer, weights_only=True)
+    return transitions
+
+
+def transitions_to_bytes(transitions: list[Transition]) -> bytes:
+    bytes_buffer = io.BytesIO()
+    torch.save(transitions, bytes_buffer)
+    return bytes_buffer.getvalue()
+
+
+def grpc_channel_options(
+    max_receive_message_length: int = MAX_MESSAGE_SIZE,
+    max_send_message_length: int = MAX_MESSAGE_SIZE,
+    enable_retries: bool = True,
+    initial_backoff: str = "0.1s",
+    max_attempts: int = 5,
+    backoff_multiplier: float = 2,
+    max_backoff: str = "2s",
+):
+    service_config = {
+        "methodConfig": [
+            {
+                "name": [{}],  # Applies to ALL methods in ALL services
+                "retryPolicy": {
+                    "maxAttempts": max_attempts,  # Max retries (total attempts = 5)
+                    "initialBackoff": initial_backoff,  # First retry after 0.1s
+                    "maxBackoff": max_backoff,  # Max wait time between retries
+                    "backoffMultiplier": backoff_multiplier,  # Exponential backoff factor
+                    "retryableStatusCodes": [
+                        "UNAVAILABLE",
+                        "DEADLINE_EXCEEDED",
+                    ],  # Retries on network failures
+                },
+            }
+        ]
+    }
+
+    service_config_json = json.dumps(service_config)
+
+    retries_option = 1 if enable_retries else 0
+
+    return [
+        ("grpc.max_receive_message_length", max_receive_message_length),
+        ("grpc.max_send_message_length", max_send_message_length),
+        ("grpc.enable_retries", retries_option),
+        ("grpc.service_config", service_config_json),
+    ]
diff --git a/lerobot/src/lerobot/types.py b/lerobot/src/lerobot/types.py
new file mode 100644
index 0000000000000000000000000000000000000000..d9b8166c54d2f8a1584faabfdfb294ace6c37c99
--- /dev/null
+++ b/lerobot/src/lerobot/types.py
@@ -0,0 +1,56 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from __future__ import annotations
+
+from enum import Enum
+from typing import Any, TypedDict
+
+import numpy as np
+import torch
+
+
+class TransitionKey(str, Enum):
+    """Keys for accessing EnvTransition dictionary components."""
+
+    # TODO(Steven): Use consts
+    OBSERVATION = "observation"
+    ACTION = "action"
+    REWARD = "reward"
+    DONE = "done"
+    TRUNCATED = "truncated"
+    INFO = "info"
+    COMPLEMENTARY_DATA = "complementary_data"
+
+
+PolicyAction = torch.Tensor
+RobotAction = dict[str, Any]
+EnvAction = np.ndarray
+RobotObservation = dict[str, Any]
+
+
+EnvTransition = TypedDict(
+    "EnvTransition",
+    {
+        TransitionKey.OBSERVATION.value: RobotObservation | None,
+        TransitionKey.ACTION.value: PolicyAction | RobotAction | EnvAction | None,
+        TransitionKey.REWARD.value: float | torch.Tensor | None,
+        TransitionKey.DONE.value: bool | torch.Tensor | None,
+        TransitionKey.TRUNCATED.value: bool | torch.Tensor | None,
+        TransitionKey.INFO.value: dict[str, Any] | None,
+        TransitionKey.COMPLEMENTARY_DATA.value: dict[str, Any] | None,
+    },
+)
diff --git a/lerobot/src/lerobot/utils/constants.py b/lerobot/src/lerobot/utils/constants.py
new file mode 100644
index 0000000000000000000000000000000000000000..ecd54844c98be8b2abf2e08f8d503e3ff8a2cdeb
--- /dev/null
+++ b/lerobot/src/lerobot/utils/constants.py
@@ -0,0 +1,91 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# keys
+import os
+from pathlib import Path
+
+from huggingface_hub.constants import HF_HOME
+
+OBS_STR = "observation"
+OBS_PREFIX = OBS_STR + "."
+OBS_ENV_STATE = OBS_STR + ".environment_state"
+OBS_STATE = OBS_STR + ".state"
+OBS_IMAGE = OBS_STR + ".image"
+OBS_IMAGES = OBS_IMAGE + "s"
+OBS_LANGUAGE = OBS_STR + ".language"
+OBS_LANGUAGE_TOKENS = OBS_LANGUAGE + ".tokens"
+OBS_LANGUAGE_ATTENTION_MASK = OBS_LANGUAGE + ".attention_mask"
+OBS_LANGUAGE_SUBTASK = OBS_STR + ".subtask"
+OBS_LANGUAGE_SUBTASK_TOKENS = OBS_LANGUAGE_SUBTASK + ".tokens"
+OBS_LANGUAGE_SUBTASK_ATTENTION_MASK = OBS_LANGUAGE_SUBTASK + ".attention_mask"
+
+ACTION = "action"
+ACTION_PREFIX = ACTION + "."
+ACTION_TOKENS = ACTION + ".tokens"
+ACTION_TOKEN_MASK = ACTION + ".token_mask"
+REWARD = "next.reward"
+TRUNCATED = "next.truncated"
+DONE = "next.done"
+INFO = "info"
+
+ROBOTS = "robots"
+TELEOPERATORS = "teleoperators"
+
+# files & directories
+CHECKPOINTS_DIR = "checkpoints"
+LAST_CHECKPOINT_LINK = "last"
+PRETRAINED_MODEL_DIR = "pretrained_model"
+TRAINING_STATE_DIR = "training_state"
+RNG_STATE = "rng_state.safetensors"
+TRAINING_STEP = "training_step.json"
+OPTIMIZER_STATE = "optimizer_state.safetensors"
+OPTIMIZER_PARAM_GROUPS = "optimizer_param_groups.json"
+SCHEDULER_STATE = "scheduler_state.json"
+
+POLICY_PREPROCESSOR_DEFAULT_NAME = "policy_preprocessor"
+POLICY_POSTPROCESSOR_DEFAULT_NAME = "policy_postprocessor"
+
+if "LEROBOT_HOME" in os.environ:
+    raise ValueError(
+        f"You have a 'LEROBOT_HOME' environment variable set to '{os.getenv('LEROBOT_HOME')}'.\n"
+        "'LEROBOT_HOME' is deprecated, please use 'HF_LEROBOT_HOME' instead."
+    )
+
+# cache dir
+default_cache_path = Path(HF_HOME) / "lerobot"
+HF_LEROBOT_HOME = Path(os.getenv("HF_LEROBOT_HOME", default_cache_path)).expanduser()
+
+# calibration dir
+default_calibration_path = HF_LEROBOT_HOME / "calibration"
+HF_LEROBOT_CALIBRATION = Path(os.getenv("HF_LEROBOT_CALIBRATION", default_calibration_path)).expanduser()
+
+
+# streaming datasets
+LOOKBACK_BACKTRACKTABLE = 100
+LOOKAHEAD_BACKTRACKTABLE = 100
+
+# openpi
+OPENPI_ATTENTION_MASK_VALUE = -2.3819763e38  # TODO(pepijn): Modify this when extending support to fp8 models
+
+# Constants for LIBERO observation keys
+LIBERO_KEY_EEF_POS = "robot_state/eef/pos"
+LIBERO_KEY_EEF_QUAT = "robot_state/eef/quat"
+LIBERO_KEY_EEF_MAT = "robot_state/eef/mat"
+LIBERO_KEY_EEF_AXISANGLE = "robot_state/eef/axisangle"
+LIBERO_KEY_GRIPPER_QPOS = "robot_state/gripper/qpos"
+LIBERO_KEY_GRIPPER_QVEL = "robot_state/gripper/qvel"
+LIBERO_KEY_JOINTS_POS = "robot_state/joints/pos"
+LIBERO_KEY_JOINTS_VEL = "robot_state/joints/vel"
+LIBERO_KEY_PIXELS_AGENTVIEW = "pixels/agentview_image"
+LIBERO_KEY_PIXELS_EYE_IN_HAND = "pixels/robot0_eye_in_hand_image"
diff --git a/lerobot/src/lerobot/utils/control_utils.py b/lerobot/src/lerobot/utils/control_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..94cd82fa11b3e535d97cae780c8d57e6730fa98b
--- /dev/null
+++ b/lerobot/src/lerobot/utils/control_utils.py
@@ -0,0 +1,236 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+########################################################################################
+# Utilities
+########################################################################################
+
+
+import logging
+import traceback
+from contextlib import nullcontext
+from copy import copy
+from functools import cache
+from typing import Any
+
+import numpy as np
+import torch
+from deepdiff import DeepDiff
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.utils import DEFAULT_FEATURES
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.utils import prepare_observation_for_inference
+from lerobot.processor import PolicyProcessorPipeline
+from lerobot.robots import Robot
+from lerobot.types import PolicyAction
+
+
+@cache
+def is_headless():
+    """
+    Detects if the Python script is running in a headless environment (e.g., without a display).
+
+    This function attempts to import `pynput`, a library that requires a graphical environment.
+    If the import fails, it assumes the environment is headless. The result is cached to avoid
+    re-running the check.
+
+    Returns:
+        True if the environment is determined to be headless, False otherwise.
+    """
+    try:
+        import pynput  # noqa
+
+        return False
+    except Exception:
+        print(
+            "Error trying to import pynput. Switching to headless mode. "
+            "As a result, the video stream from the cameras won't be shown, "
+            "and you won't be able to change the control flow with keyboards. "
+            "For more info, see traceback below.\n"
+        )
+        traceback.print_exc()
+        print()
+        return True
+
+
+def predict_action(
+    observation: dict[str, np.ndarray],
+    policy: PreTrainedPolicy,
+    device: torch.device,
+    preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    postprocessor: PolicyProcessorPipeline[PolicyAction, PolicyAction],
+    use_amp: bool,
+    task: str | None = None,
+    robot_type: str | None = None,
+):
+    """
+    Performs a single-step inference to predict a robot action from an observation.
+
+    This function encapsulates the full inference pipeline:
+    1. Prepares the observation by converting it to PyTorch tensors and adding a batch dimension.
+    2. Runs the preprocessor pipeline on the observation.
+    3. Feeds the processed observation to the policy to get a raw action.
+    4. Runs the postprocessor pipeline on the raw action.
+    5. Formats the final action by removing the batch dimension and moving it to the CPU.
+
+    Args:
+        observation: A dictionary of NumPy arrays representing the robot's current observation.
+        policy: The `PreTrainedPolicy` model to use for action prediction.
+        device: The `torch.device` (e.g., 'cuda' or 'cpu') to run inference on.
+        preprocessor: The `PolicyProcessorPipeline` for preprocessing observations.
+        postprocessor: The `PolicyProcessorPipeline` for postprocessing actions.
+        use_amp: A boolean to enable/disable Automatic Mixed Precision for CUDA inference.
+        task: An optional string identifier for the task.
+        robot_type: An optional string identifier for the robot type.
+
+    Returns:
+        A `torch.Tensor` containing the predicted action, ready for the robot.
+    """
+    observation = copy(observation)
+    with (
+        torch.inference_mode(),
+        torch.autocast(device_type=device.type) if device.type == "cuda" and use_amp else nullcontext(),
+    ):
+        # Convert to pytorch format: channel first and float32 in [0,1] with batch dimension
+        observation = prepare_observation_for_inference(observation, device, task, robot_type)
+        observation = preprocessor(observation)
+
+        # Compute the next action with the policy
+        # based on the current observation
+        action = policy.select_action(observation)
+
+        action = postprocessor(action)
+
+    return action
+
+
+def init_keyboard_listener():
+    """
+    Initializes a non-blocking keyboard listener for real-time user interaction.
+
+    This function sets up a listener for specific keys (right arrow, left arrow, escape) to control
+    the program flow during execution, such as stopping recording or exiting loops. It gracefully
+    handles headless environments where keyboard listening is not possible.
+
+    Returns:
+        A tuple containing:
+        - The `pynput.keyboard.Listener` instance, or `None` if in a headless environment.
+        - A dictionary of event flags (e.g., `exit_early`) that are set by key presses.
+    """
+    # Allow to exit early while recording an episode or resetting the environment,
+    # by tapping the right arrow key '->'. This might require a sudo permission
+    # to allow your terminal to monitor keyboard events.
+    events = {}
+    events["exit_early"] = False
+    events["rerecord_episode"] = False
+    events["stop_recording"] = False
+
+    if is_headless():
+        logging.warning(
+            "Headless environment detected. On-screen cameras display and keyboard inputs will not be available."
+        )
+        listener = None
+        return listener, events
+
+    # Only import pynput if not in a headless environment
+    from pynput import keyboard
+
+    def on_press(key):
+        try:
+            if key == keyboard.Key.right:
+                print("Right arrow key pressed. Exiting loop...")
+                events["exit_early"] = True
+            elif key == keyboard.Key.left:
+                print("Left arrow key pressed. Exiting loop and rerecord the last episode...")
+                events["rerecord_episode"] = True
+                events["exit_early"] = True
+            elif key == keyboard.Key.esc:
+                print("Escape key pressed. Stopping data recording...")
+                events["stop_recording"] = True
+                events["exit_early"] = True
+        except Exception as e:
+            print(f"Error handling key press: {e}")
+
+    listener = keyboard.Listener(on_press=on_press)
+    listener.start()
+
+    return listener, events
+
+
+def sanity_check_dataset_name(repo_id, policy_cfg):
+    """
+    Validates the dataset repository name against the presence of a policy configuration.
+
+    This function enforces a naming convention: a dataset repository ID should start with "eval_"
+    if and only if a policy configuration is provided for evaluation purposes.
+
+    Args:
+        repo_id: The Hugging Face Hub repository ID of the dataset.
+        policy_cfg: The configuration object for the policy, or `None`.
+
+    Raises:
+        ValueError: If the naming convention is violated.
+    """
+    _, dataset_name = repo_id.split("/")
+    # either repo_id doesnt start with "eval_" and there is no policy
+    # or repo_id starts with "eval_" and there is a policy
+
+    # Check if dataset_name starts with "eval_" but policy is missing
+    if dataset_name.startswith("eval_") and policy_cfg is None:
+        raise ValueError(
+            f"Your dataset name begins with 'eval_' ({dataset_name}), but no policy is provided."
+        )
+
+    # Check if dataset_name does not start with "eval_" but policy is provided
+    if not dataset_name.startswith("eval_") and policy_cfg is not None:
+        raise ValueError(
+            f"Your dataset name does not begin with 'eval_' ({dataset_name}), but a policy is provided ({policy_cfg.type})."
+        )
+
+
+def sanity_check_dataset_robot_compatibility(
+    dataset: LeRobotDataset, robot: Robot, fps: int, features: dict
+) -> None:
+    """
+    Checks if a dataset's metadata is compatible with the current robot and recording setup.
+
+    This function compares key metadata fields (`robot_type`, `fps`, and `features`) from the
+    dataset against the current configuration to ensure that appended data will be consistent.
+
+    Args:
+        dataset: The `LeRobotDataset` instance to check.
+        robot: The `Robot` instance representing the current hardware setup.
+        fps: The current recording frequency (frames per second).
+        features: The dictionary of features for the current recording session.
+
+    Raises:
+        ValueError: If any of the checked metadata fields do not match.
+    """
+    fields = [
+        ("robot_type", dataset.meta.robot_type, robot.robot_type),
+        ("fps", dataset.fps, fps),
+        ("features", dataset.features, {**features, **DEFAULT_FEATURES}),
+    ]
+
+    mismatches = []
+    for field, dataset_value, present_value in fields:
+        diff = DeepDiff(dataset_value, present_value, exclude_regex_paths=[r".*\['info'\]$"])
+        if diff:
+            mismatches.append(f"{field}: expected {present_value}, got {dataset_value}")
+
+    if mismatches:
+        raise ValueError(
+            "Dataset metadata compatibility check failed with mismatches:\n" + "\n".join(mismatches)
+        )
diff --git a/lerobot/src/lerobot/utils/decorators.py b/lerobot/src/lerobot/utils/decorators.py
new file mode 100644
index 0000000000000000000000000000000000000000..8fc2f9a07ac8645faa53de45c9dd88e9eb89b290
--- /dev/null
+++ b/lerobot/src/lerobot/utils/decorators.py
@@ -0,0 +1,41 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from functools import wraps
+
+from lerobot.utils.errors import DeviceAlreadyConnectedError, DeviceNotConnectedError
+
+
+def check_if_not_connected(func):
+    @wraps(func)
+    def wrapper(self, *args, **kwargs):
+        if not self.is_connected:
+            raise DeviceNotConnectedError(
+                f"{self.__class__.__name__} is not connected. Run `.connect()` first."
+            )
+        return func(self, *args, **kwargs)
+
+    return wrapper
+
+
+def check_if_already_connected(func):
+    @wraps(func)
+    def wrapper(self, *args, **kwargs):
+        if self.is_connected:
+            raise DeviceAlreadyConnectedError(f"{self.__class__.__name__} is already connected.")
+        return func(self, *args, **kwargs)
+
+    return wrapper
diff --git a/lerobot/src/lerobot/utils/device_utils.py b/lerobot/src/lerobot/utils/device_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..37981f07fc09becc40eaa1b9d3a06df7714b7ec0
--- /dev/null
+++ b/lerobot/src/lerobot/utils/device_utils.py
@@ -0,0 +1,109 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+
+import torch
+
+
+def auto_select_torch_device() -> torch.device:
+    """Tries to select automatically a torch device."""
+    if torch.cuda.is_available():
+        logging.info("Cuda backend detected, using cuda.")
+        return torch.device("cuda")
+    elif torch.backends.mps.is_available():
+        logging.info("Metal backend detected, using mps.")
+        return torch.device("mps")
+    elif torch.xpu.is_available():
+        logging.info("Intel XPU backend detected, using xpu.")
+        return torch.device("xpu")
+    else:
+        logging.warning("No accelerated backend detected. Using default cpu, this will be slow.")
+        return torch.device("cpu")
+
+
+# TODO(Steven): Remove log. log shouldn't be an argument, this should be handled by the logger level
+def get_safe_torch_device(try_device: str, log: bool = False) -> torch.device:
+    """Given a string, return a torch.device with checks on whether the device is available."""
+    try_device = str(try_device)
+    if try_device.startswith("cuda"):
+        assert torch.cuda.is_available()
+        device = torch.device(try_device)
+    elif try_device == "mps":
+        assert torch.backends.mps.is_available()
+        device = torch.device("mps")
+    elif try_device == "xpu":
+        assert torch.xpu.is_available()
+        device = torch.device("xpu")
+    elif try_device == "cpu":
+        device = torch.device("cpu")
+        if log:
+            logging.warning("Using CPU, this will be slow.")
+    else:
+        device = torch.device(try_device)
+        if log:
+            logging.warning(f"Using custom {try_device} device.")
+    return device
+
+
+def get_safe_dtype(dtype: torch.dtype, device: str | torch.device):
+    """
+    mps is currently not compatible with float64
+    """
+    if isinstance(device, torch.device):
+        device = device.type
+    if device == "mps" and dtype == torch.float64:
+        return torch.float32
+    if device == "xpu" and dtype == torch.float64:
+        if hasattr(torch.xpu, "get_device_capability"):
+            device_capability = torch.xpu.get_device_capability()
+            # NOTE: Some Intel XPU devices do not support double precision (FP64).
+            # The `has_fp64` flag is returned by `torch.xpu.get_device_capability()`
+            # when available; if False, we fall back to float32 for compatibility.
+            if not device_capability.get("has_fp64", False):
+                logging.warning(f"Device {device} does not support float64, using float32 instead.")
+                return torch.float32
+        else:
+            logging.warning(
+                f"Device {device} capability check failed. Assuming no support for float64, using float32 instead."
+            )
+            return torch.float32
+        return dtype
+    else:
+        return dtype
+
+
+def is_torch_device_available(try_device: str) -> bool:
+    try_device = str(try_device)  # Ensure try_device is a string
+    if try_device.startswith("cuda"):
+        return torch.cuda.is_available()
+    elif try_device == "mps":
+        return torch.backends.mps.is_available()
+    elif try_device == "xpu":
+        return torch.xpu.is_available()
+    elif try_device == "cpu":
+        return True
+    else:
+        raise ValueError(f"Unknown device {try_device}. Supported devices are: cuda, mps, xpu or cpu.")
+
+
+def is_amp_available(device: str):
+    if device in ["cuda", "xpu", "cpu"]:
+        return True
+    elif device == "mps":
+        return False
+    else:
+        raise ValueError(f"Unknown device '{device}.")
diff --git a/lerobot/src/lerobot/utils/errors.py b/lerobot/src/lerobot/utils/errors.py
new file mode 100644
index 0000000000000000000000000000000000000000..31b73eacabc8e1065fd536f0e8822b31869b644c
--- /dev/null
+++ b/lerobot/src/lerobot/utils/errors.py
@@ -0,0 +1,32 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+
+class DeviceNotConnectedError(ConnectionError):
+    """Exception raised when the device is not connected."""
+
+    def __init__(self, message="This device is not connected. Try calling `connect()` first."):
+        self.message = message
+        super().__init__(self.message)
+
+
+class DeviceAlreadyConnectedError(ConnectionError):
+    """Exception raised when the device is already connected."""
+
+    def __init__(
+        self,
+        message="This device is already connected. Try not calling `connect()` twice.",
+    ):
+        self.message = message
+        super().__init__(self.message)
diff --git a/lerobot/src/lerobot/utils/hub.py b/lerobot/src/lerobot/utils/hub.py
new file mode 100644
index 0000000000000000000000000000000000000000..566701b31c4433ad9974cfacaf7316239ecff0b9
--- /dev/null
+++ b/lerobot/src/lerobot/utils/hub.py
@@ -0,0 +1,203 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import builtins
+from pathlib import Path
+from tempfile import TemporaryDirectory
+from typing import Any, TypeVar
+
+from huggingface_hub import HfApi
+from huggingface_hub.utils import validate_hf_hub_args
+
+T = TypeVar("T", bound="HubMixin")
+
+
+class HubMixin:
+    """
+    A Mixin containing the functionality to push an object to the hub.
+
+    This is similar to huggingface_hub.ModelHubMixin but is lighter and makes less assumptions about its
+    subclasses (in particular, the fact that it's not necessarily a model).
+
+    The inheriting classes must implement '_save_pretrained' and 'from_pretrained'.
+    """
+
+    def save_pretrained(
+        self,
+        save_directory: str | Path,
+        *,
+        repo_id: str | None = None,
+        push_to_hub: bool = False,
+        card_kwargs: dict[str, Any] | None = None,
+        **push_to_hub_kwargs,
+    ) -> str | None:
+        """
+        Save object in local directory.
+
+        Args:
+            save_directory (`str` or `Path`):
+                Path to directory in which the object will be saved.
+            push_to_hub (`bool`, *optional*, defaults to `False`):
+                Whether or not to push your object to the Huggingface Hub after saving it.
+            repo_id (`str`, *optional*):
+                ID of your repository on the Hub. Used only if `push_to_hub=True`. Will default to the folder name if
+                not provided.
+            card_kwargs (`Dict[str, Any]`, *optional*):
+                Additional arguments passed to the card template to customize the card.
+            push_to_hub_kwargs:
+                Additional key word arguments passed along to the [`~HubMixin.push_to_hub`] method.
+        Returns:
+            `str` or `None`: url of the commit on the Hub if `push_to_hub=True`, `None` otherwise.
+        """
+        save_directory = Path(save_directory)
+        save_directory.mkdir(parents=True, exist_ok=True)
+
+        # save object (weights, files, etc.)
+        self._save_pretrained(save_directory)
+
+        # push to the Hub if required
+        if push_to_hub:
+            if repo_id is None:
+                repo_id = save_directory.name  # Defaults to `save_directory` name
+            return self.push_to_hub(repo_id=repo_id, card_kwargs=card_kwargs, **push_to_hub_kwargs)
+        return None
+
+    def _save_pretrained(self, save_directory: Path) -> None:
+        """
+        Overwrite this method in subclass to define how to save your object.
+
+        Args:
+            save_directory (`str` or `Path`):
+                Path to directory in which the object files will be saved.
+        """
+        raise NotImplementedError
+
+    @classmethod
+    @validate_hf_hub_args
+    def from_pretrained(
+        cls: builtins.type[T],
+        pretrained_name_or_path: str | Path,
+        *,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        **kwargs,
+    ) -> T:
+        """
+        Download the object from the Huggingface Hub and instantiate it.
+
+        Args:
+            pretrained_name_or_path (`str`, `Path`):
+                - Either the `repo_id` (string) of the object hosted on the Hub, e.g. `lerobot/diffusion_pusht`.
+                - Or a path to a `directory` containing the object files saved using `.save_pretrained`,
+                    e.g., `../path/to/my_model_directory/`.
+            revision (`str`, *optional*):
+                Revision on the Hub. Can be a branch name, a git tag or any commit id.
+                Defaults to the latest commit on `main` branch.
+            force_download (`bool`, *optional*, defaults to `False`):
+                Whether to force (re-)downloading the files from the Hub, overriding the existing cache.
+            proxies (`Dict[str, str]`, *optional*):
+                A dictionary of proxy servers to use by protocol or endpoint, e.g., `{'http': 'foo.bar:3128',
+                'http://hostname': 'foo.bar:4012'}`. The proxies are used on every request.
+            token (`str` or `bool`, *optional*):
+                The token to use as HTTP bearer authorization for remote files. By default, it will use the token
+                cached when running `huggingface-cli login`.
+            cache_dir (`str`, `Path`, *optional*):
+                Path to the folder where cached files are stored.
+            local_files_only (`bool`, *optional*, defaults to `False`):
+                If `True`, avoid downloading the file and return the path to the local cached file if it exists.
+            kwargs (`Dict`, *optional*):
+                Additional kwargs to pass to the object during initialization.
+        """
+        raise NotImplementedError
+
+    @validate_hf_hub_args
+    def push_to_hub(
+        self,
+        repo_id: str,
+        *,
+        commit_message: str | None = None,
+        private: bool | None = None,
+        token: str | None = None,
+        branch: str | None = None,
+        create_pr: bool | None = None,
+        allow_patterns: list[str] | str | None = None,
+        ignore_patterns: list[str] | str | None = None,
+        delete_patterns: list[str] | str | None = None,
+        card_kwargs: dict[str, Any] | None = None,
+    ) -> str:
+        """
+        Upload model checkpoint to the Hub.
+
+        Use `allow_patterns` and `ignore_patterns` to precisely filter which files should be pushed to the hub. Use
+        `delete_patterns` to delete existing remote files in the same commit. See [`upload_folder`] reference for more
+        details.
+
+        Args:
+            repo_id (`str`):
+                ID of the repository to push to (example: `"username/my-model"`).
+            commit_message (`str`, *optional*):
+                Message to commit while pushing.
+            private (`bool`, *optional*):
+                Whether the repository created should be private.
+                If `None` (default), the repo will be public unless the organization's default is private.
+            token (`str`, *optional*):
+                The token to use as HTTP bearer authorization for remote files. By default, it will use the token
+                cached when running `huggingface-cli login`.
+            branch (`str`, *optional*):
+                The git branch on which to push the model. This defaults to `"main"`.
+            create_pr (`boolean`, *optional*):
+                Whether or not to create a Pull Request from `branch` with that commit. Defaults to `False`.
+            allow_patterns (`List[str]` or `str`, *optional*):
+                If provided, only files matching at least one pattern are pushed.
+            ignore_patterns (`List[str]` or `str`, *optional*):
+                If provided, files matching any of the patterns are not pushed.
+            delete_patterns (`List[str]` or `str`, *optional*):
+                If provided, remote files matching any of the patterns will be deleted from the repo.
+            card_kwargs (`Dict[str, Any]`, *optional*):
+                Additional arguments passed to the card template to customize the card.
+
+        Returns:
+            The url of the commit of your object in the given repository.
+        """
+        api = HfApi(token=token)
+        repo_id = api.create_repo(repo_id=repo_id, private=private, exist_ok=True).repo_id
+
+        if commit_message is None:
+            if "Policy" in self.__class__.__name__:
+                commit_message = "Upload policy"
+            elif "Config" in self.__class__.__name__:
+                commit_message = "Upload config"
+            else:
+                commit_message = f"Upload {self.__class__.__name__}"
+
+        # Push the files to the repo in a single commit
+        with TemporaryDirectory(ignore_cleanup_errors=True) as tmp:
+            saved_path = Path(tmp) / repo_id
+            self.save_pretrained(saved_path, card_kwargs=card_kwargs)
+            return api.upload_folder(
+                repo_id=repo_id,
+                repo_type="model",
+                folder_path=saved_path,
+                commit_message=commit_message,
+                revision=branch,
+                create_pr=create_pr,
+                allow_patterns=allow_patterns,
+                ignore_patterns=ignore_patterns,
+                delete_patterns=delete_patterns,
+            )
diff --git a/lerobot/src/lerobot/utils/import_utils.py b/lerobot/src/lerobot/utils/import_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..2b26b2302f31465c714683475d839723c153f25c
--- /dev/null
+++ b/lerobot/src/lerobot/utils/import_utils.py
@@ -0,0 +1,174 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import importlib
+import importlib.metadata
+import logging
+from typing import Any
+
+from draccus.choice_types import ChoiceRegistry
+
+
+def is_package_available(
+    pkg_name: str, import_name: str | None = None, return_version: bool = False
+) -> tuple[bool, str] | bool:
+    """
+    Check if the package spec exists and grab its version to avoid importing a local directory.
+
+    Args:
+        pkg_name: The name of the package as installed via pip (e.g. "python-can").
+        import_name: The actual name used to import the package (e.g. "can").
+                     Defaults to pkg_name if not provided.
+        return_version: Whether to return the version string.
+    """
+    if import_name is None:
+        import_name = pkg_name
+
+    # Check if the module spec exists using the import name
+    package_exists = importlib.util.find_spec(import_name) is not None
+    package_version = "N/A"
+    if package_exists:
+        try:
+            # Primary method to get the package version
+            package_version = importlib.metadata.version(pkg_name)
+
+        except importlib.metadata.PackageNotFoundError:
+            # Fallback method: Only for "torch" and versions containing "dev"
+            if pkg_name == "torch":
+                try:
+                    package = importlib.import_module(import_name)
+                    temp_version = getattr(package, "__version__", "N/A")
+                    # Check if the version contains "dev"
+                    if "dev" in temp_version:
+                        package_version = temp_version
+                        package_exists = True
+                    else:
+                        package_exists = False
+                except ImportError:
+                    # If the package can't be imported, it's not available
+                    package_exists = False
+            else:
+                # For packages other than "torch", don't attempt the fallback and set as not available
+                package_exists = False
+        logging.debug(f"Detected {pkg_name} version: {package_version}")
+    if return_version:
+        return package_exists, package_version
+    else:
+        return package_exists
+
+
+_transformers_available = is_package_available("transformers")
+_peft_available = is_package_available("peft")
+_scipy_available = is_package_available("scipy")
+_reachy2_sdk_available = is_package_available("reachy2_sdk")
+_can_available = is_package_available("python-can", "can")
+_unitree_sdk_available = is_package_available("unitree-sdk2py", "unitree_sdk2py")
+_pygame_available = is_package_available("pygame")
+
+
+def make_device_from_device_class(config: ChoiceRegistry) -> Any:
+    """
+    Dynamically instantiates an object from its `ChoiceRegistry` configuration.
+
+    This factory uses the module path and class name from the `config` object's
+    type to locate and instantiate the corresponding device class (not the config).
+    It derives the device class name by removing a trailing 'Config' from the config
+    class name and tries a few candidate modules where the device implementation is
+    commonly located.
+    """
+    if not isinstance(config, ChoiceRegistry):
+        raise ValueError(f"Config should be an instance of `ChoiceRegistry`, got {type(config)}")
+
+    config_cls = config.__class__
+    module_path = config_cls.__module__  # typical: lerobot_teleop_mydevice.config_mydevice
+    config_name = config_cls.__name__  # typical: MyDeviceConfig
+
+    # Derive device class name (strip "Config")
+    if not config_name.endswith("Config"):
+        raise ValueError(f"Config class name '{config_name}' does not end with 'Config'")
+
+    device_class_name = config_name[:-6]  # typical: MyDeviceConfig -> MyDevice
+
+    # Build candidate modules to search for the device class
+    parts = module_path.split(".")
+    parent_module = ".".join(parts[:-1]) if len(parts) > 1 else module_path
+    candidates = [
+        parent_module,  # typical: lerobot_teleop_mydevice
+        parent_module + "." + device_class_name.lower(),  # typical: lerobot_teleop_mydevice.mydevice
+    ]
+
+    # handle modules named like "config_xxx" -> try replacing that piece with "xxx"
+    last = parts[-1] if parts else ""
+    if last.startswith("config_"):
+        candidates.append(".".join(parts[:-1] + [last.replace("config_", "")]))
+
+    # de-duplicate while preserving order
+    seen: set[str] = set()
+    candidates = [c for c in candidates if not (c in seen or seen.add(c))]
+
+    tried: list[str] = []
+    for candidate in candidates:
+        tried.append(candidate)
+        try:
+            module = importlib.import_module(candidate)
+        except ImportError:
+            continue
+
+        if hasattr(module, device_class_name):
+            cls = getattr(module, device_class_name)
+            if callable(cls):
+                try:
+                    return cls(config)
+                except TypeError as e:
+                    raise TypeError(
+                        f"Failed to instantiate '{device_class_name}' from module '{candidate}': {e}"
+                    ) from e
+
+    raise ImportError(
+        f"Could not locate device class '{device_class_name}' for config '{config_name}'. "
+        f"Tried modules: {tried}. Ensure your device class name is the config class name without "
+        f"'Config' and that it's importable from one of those modules."
+    )
+
+
+def register_third_party_plugins() -> None:
+    """
+    Discover and import third-party LeRobot plugins so they can register themselves.
+
+    This function uses `importlib.metadata` to find packages installed in the environment
+    (including editable installs) starting with 'lerobot_robot_', 'lerobot_camera_',
+    'lerobot_teleoperator_', or 'lerobot_policy_' and imports them.
+    """
+    prefixes = ("lerobot_robot_", "lerobot_camera_", "lerobot_teleoperator_", "lerobot_policy_")
+    imported: list[str] = []
+    failed: list[str] = []
+
+    def attempt_import(module_name: str):
+        try:
+            importlib.import_module(module_name)
+            imported.append(module_name)
+            logging.info("Imported third-party plugin: %s", module_name)
+        except Exception:
+            logging.exception("Could not import third-party plugin: %s", module_name)
+            failed.append(module_name)
+
+    for dist in importlib.metadata.distributions():
+        dist_name = dist.metadata.get("Name")
+        if not dist_name:
+            continue
+        if dist_name.startswith(prefixes):
+            attempt_import(dist_name)
+
+    logging.debug("Third-party plugin import summary: imported=%s failed=%s", imported, failed)
diff --git a/lerobot/src/lerobot/utils/io_utils.py b/lerobot/src/lerobot/utils/io_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..d70ea8b6a7f3fc2e51b34888d1fabcd72a37898f
--- /dev/null
+++ b/lerobot/src/lerobot/utils/io_utils.py
@@ -0,0 +1,109 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import json
+import warnings
+from pathlib import Path
+
+import imageio
+
+JsonLike = str | int | float | bool | None | list["JsonLike"] | dict[str, "JsonLike"] | tuple["JsonLike", ...]
+
+
+def write_video(video_path, stacked_frames, fps):
+    # Filter out DeprecationWarnings raised from pkg_resources
+    with warnings.catch_warnings():
+        warnings.filterwarnings(
+            "ignore", "pkg_resources is deprecated as an API", category=DeprecationWarning
+        )
+        imageio.mimsave(video_path, stacked_frames, fps=fps)
+
+
+def deserialize_json_into_object[T: JsonLike](fpath: Path, obj: T) -> T:
+    """
+    Loads the JSON data from `fpath` and recursively fills `obj` with the
+    corresponding values (strictly matching structure and types).
+    Tuples in `obj` are expected to be lists in the JSON data, which will be
+    converted back into tuples.
+    """
+    with open(fpath, encoding="utf-8") as f:
+        data = json.load(f)
+
+    def _deserialize(target, source):
+        """
+        Recursively overwrite the structure in `target` with data from `source`,
+        performing strict checks on structure and type.
+        Returns the updated version of `target` (especially important for tuples).
+        """
+
+        # If the target is a dictionary, source must be a dictionary as well.
+        if isinstance(target, dict):
+            if not isinstance(source, dict):
+                raise TypeError(f"Type mismatch: expected dict, got {type(source)}")
+
+            # Check that they have exactly the same set of keys.
+            if target.keys() != source.keys():
+                raise ValueError(
+                    f"Dictionary keys do not match.\nExpected: {target.keys()}, got: {source.keys()}"
+                )
+
+            # Recursively update each key.
+            for k in target:
+                target[k] = _deserialize(target[k], source[k])
+
+            return target
+
+        # If the target is a list, source must be a list as well.
+        elif isinstance(target, list):
+            if not isinstance(source, list):
+                raise TypeError(f"Type mismatch: expected list, got {type(source)}")
+
+            # Check length
+            if len(target) != len(source):
+                raise ValueError(f"List length mismatch: expected {len(target)}, got {len(source)}")
+
+            # Recursively update each element.
+            for i in range(len(target)):
+                target[i] = _deserialize(target[i], source[i])
+
+            return target
+
+        # If the target is a tuple, the source must be a list in JSON,
+        # which we'll convert back to a tuple.
+        elif isinstance(target, tuple):
+            if not isinstance(source, list):
+                raise TypeError(f"Type mismatch: expected list (for tuple), got {type(source)}")
+
+            if len(target) != len(source):
+                raise ValueError(f"Tuple length mismatch: expected {len(target)}, got {len(source)}")
+
+            # Convert each element, forming a new tuple.
+            converted_items = []
+            for t_item, s_item in zip(target, source, strict=False):
+                converted_items.append(_deserialize(t_item, s_item))
+
+            # Return a brand new tuple (tuples are immutable in Python).
+            return tuple(converted_items)
+
+        # Otherwise, we're dealing with a "primitive" (int, float, str, bool, None).
+        else:
+            # Check the exact type.  If these must match 1:1, do:
+            if type(target) is not type(source):
+                raise TypeError(f"Type mismatch: expected {type(target)}, got {type(source)}")
+            return source
+
+    # Perform the in-place/recursive deserialization
+    updated_obj = _deserialize(obj, data)
+    return updated_obj
diff --git a/lerobot/src/lerobot/utils/logging_utils.py b/lerobot/src/lerobot/utils/logging_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..1497c0585963cae5ef9483a0488b5217bb849290
--- /dev/null
+++ b/lerobot/src/lerobot/utils/logging_utils.py
@@ -0,0 +1,169 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from collections.abc import Callable
+from typing import Any
+
+from lerobot.utils.utils import format_big_number
+
+
+class AverageMeter:
+    """
+    Computes and stores the average and current value
+    Adapted from https://github.com/pytorch/examples/blob/main/imagenet/main.py
+    """
+
+    def __init__(self, name: str, fmt: str = ":f"):
+        self.name = name
+        self.fmt = fmt
+        self.reset()
+
+    def reset(self) -> None:
+        self.val = 0.0
+        self.avg = 0.0
+        self.sum = 0.0
+        self.count = 0.0
+
+    def update(self, val: float, n: int = 1) -> None:
+        self.val = val
+        self.sum += val * n
+        self.count += n
+        self.avg = self.sum / self.count
+
+    def __str__(self):
+        fmtstr = "{name}:{avg" + self.fmt + "}"
+        return fmtstr.format(**self.__dict__)
+
+
+class MetricsTracker:
+    """
+    A helper class to track and log metrics over time.
+
+    Usage pattern:
+
+    ```python
+    # initialize, potentially with non-zero initial step (e.g. if resuming run)
+    metrics = {"loss": AverageMeter("loss", ":.3f")}
+    train_metrics = MetricsTracker(cfg, dataset, metrics, initial_step=step)
+
+    # update metrics derived from step (samples, episodes, epochs) at each training step
+    train_metrics.step()
+
+    # update various metrics
+    loss = policy.forward(batch)
+    train_metrics.loss = loss
+
+    # display current metrics
+    logging.info(train_metrics)
+
+    # export for wandb
+    wandb.log(train_metrics.to_dict())
+
+    # reset averages after logging
+    train_metrics.reset_averages()
+    ```
+    """
+
+    __keys__ = [
+        "_batch_size",
+        "_num_frames",
+        "_avg_samples_per_ep",
+        "metrics",
+        "steps",
+        "samples",
+        "episodes",
+        "epochs",
+        "accelerator",
+    ]
+
+    def __init__(
+        self,
+        batch_size: int,
+        num_frames: int,
+        num_episodes: int,
+        metrics: dict[str, AverageMeter],
+        initial_step: int = 0,
+        accelerator: Callable | None = None,
+    ):
+        self.__dict__.update(dict.fromkeys(self.__keys__))
+        self._batch_size = batch_size
+        self._num_frames = num_frames
+        self._avg_samples_per_ep = num_frames / num_episodes
+        self.metrics = metrics
+
+        self.steps = initial_step
+        world_size = accelerator.num_processes if accelerator else 1
+        # A sample is an (observation,action) pair, where observation and action
+        # can be on multiple timestamps. In a batch, we have `batch_size` number of samples.
+        self.samples = self.steps * self._batch_size * world_size
+        self.episodes = self.samples / self._avg_samples_per_ep
+        self.epochs = self.samples / self._num_frames
+        self.accelerator = accelerator
+
+    def __getattr__(self, name: str) -> int | dict[str, AverageMeter] | AverageMeter | Any:
+        if name in self.__dict__:
+            return self.__dict__[name]
+        elif name in self.metrics:
+            return self.metrics[name]
+        else:
+            raise AttributeError(f"'{self.__class__.__name__}' object has no attribute '{name}'")
+
+    def __setattr__(self, name: str, value: Any) -> None:
+        if name in self.__dict__:
+            super().__setattr__(name, value)
+        elif name in self.metrics:
+            self.metrics[name].update(value)
+        else:
+            raise AttributeError(f"'{self.__class__.__name__}' object has no attribute '{name}'")
+
+    def step(self) -> None:
+        """
+        Updates metrics that depend on 'step' for one step.
+        """
+        self.steps += 1
+        world_size = self.accelerator.num_processes if self.accelerator else 1
+        self.samples += self._batch_size * world_size
+        self.episodes = self.samples / self._avg_samples_per_ep
+        self.epochs = self.samples / self._num_frames
+
+    def __str__(self) -> str:
+        display_list = [
+            f"step:{format_big_number(self.steps)}",
+            # number of samples seen during training
+            f"smpl:{format_big_number(self.samples)}",
+            # number of episodes seen during training
+            f"ep:{format_big_number(self.episodes)}",
+            # number of time all unique samples are seen
+            f"epch:{self.epochs:.2f}",
+            *[str(m) for m in self.metrics.values()],
+        ]
+        return " ".join(display_list)
+
+    def to_dict(self, use_avg: bool = True) -> dict[str, int | float]:
+        """
+        Returns the current metric values (or averages if `use_avg=True`) as a dict.
+        """
+        return {
+            "steps": self.steps,
+            "samples": self.samples,
+            "episodes": self.episodes,
+            "epochs": self.epochs,
+            **{k: m.avg if use_avg else m.val for k, m in self.metrics.items()},
+        }
+
+    def reset_averages(self) -> None:
+        """Resets average meters."""
+        for m in self.metrics.values():
+            m.reset()
diff --git a/lerobot/src/lerobot/utils/rabc.py b/lerobot/src/lerobot/utils/rabc.py
new file mode 100644
index 0000000000000000000000000000000000000000..dc0c61c6964580f502fd8d5a6f4edf46d7d7b95f
--- /dev/null
+++ b/lerobot/src/lerobot/utils/rabc.py
@@ -0,0 +1,288 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+from pathlib import Path
+
+import numpy as np
+import pandas as pd
+import torch
+from huggingface_hub import hf_hub_download
+
+
+def resolve_hf_path(path: str | Path) -> Path:
+    """Resolve a path that may be a HuggingFace URL (hf://datasets/...) to a local path."""
+    path_str = str(path)
+    if path_str.startswith("hf://datasets/"):
+        parts = path_str.replace("hf://datasets/", "").split("/")
+        repo_id = "/".join(parts[:2])
+        filename = "/".join(parts[2:])
+        return Path(hf_hub_download(repo_id=repo_id, filename=filename, repo_type="dataset"))
+    return Path(path)
+
+
+class RABCWeights:
+    """
+    Load precomputed SARM progress values and compute RA-BC weights during training.
+
+    Progress values are loaded from a parquet file (generated by compute_rabc_weights.py).
+    During training, computes:
+        - progress_delta = progress[t + chunk_size] - progress[t]
+        - rabc_weight based on the delta (paper Eq. 8-9)
+
+    Args:
+        progress_path: Path to parquet file with precomputed progress values
+        chunk_size: Number of frames ahead for computing progress delta
+        head_mode: Which SARM head to use ("sparse" or "dense")
+        kappa: Hard threshold for high-quality samples (default: 0.01)
+        epsilon: Small constant for numerical stability (default: 1e-6)
+        fallback_weight: Weight to use for frames without valid delta (default: 1.0)
+        device: Device to return tensors on
+    """
+
+    def __init__(
+        self,
+        progress_path: str | Path,
+        chunk_size: int = 50,
+        head_mode: str = "sparse",
+        kappa: float = 0.01,
+        epsilon: float = 1e-6,
+        fallback_weight: float = 1.0,
+        device: torch.device = None,
+    ):
+        self.progress_path = resolve_hf_path(progress_path)
+        self.chunk_size = chunk_size
+        self.head_mode = head_mode
+        self.kappa = kappa
+        self.epsilon = epsilon
+        self.fallback_weight = fallback_weight
+        self.device = device or torch.device("cuda" if torch.cuda.is_available() else "cpu")
+
+        # Determine progress column name
+        self.progress_column = f"progress_{head_mode}"
+
+        # Load progress values
+        logging.info(f"Loading SARM progress values from {self.progress_path}")
+        self.df = pd.read_parquet(self.progress_path)
+
+        # Check if the requested head mode column exists
+        if self.progress_column not in self.df.columns:
+            available = [c for c in self.df.columns if c.startswith("progress")]
+            raise ValueError(
+                f"Column '{self.progress_column}' not found. Available progress columns: {available}"
+            )
+
+        logging.info(f"Using progress column: {self.progress_column}")
+
+        self.progress_lookup = {}
+        self.episode_lookup = {}
+
+        for _, row in self.df.iterrows():
+            global_idx = int(row["index"])
+            progress = row[self.progress_column]
+            episode_idx = int(row["episode_index"])
+
+            if not np.isnan(progress):
+                self.progress_lookup[global_idx] = float(progress)
+            self.episode_lookup[global_idx] = episode_idx
+
+        # Build episode boundaries for delta computation
+        self.episode_boundaries = {}
+        for episode_idx in self.df["episode_index"].unique():
+            ep_df = self.df[self.df["episode_index"] == episode_idx]
+            self.episode_boundaries[int(episode_idx)] = {
+                "start": int(ep_df["index"].min()),
+                "end": int(ep_df["index"].max()) + 1,
+            }
+
+        logging.info(f"Loaded {len(self.progress_lookup)} frame progress values")
+        logging.info(f"Chunk size for delta computation: {chunk_size}")
+
+        # Compute global statistics for weight computation
+        self._compute_global_stats()
+
+    def _compute_global_stats(self):
+        """Compute global mean and std of progress deltas for weight calculation."""
+        all_deltas = []
+
+        for global_idx, progress in self.progress_lookup.items():
+            episode_idx = self.episode_lookup.get(global_idx)
+            if episode_idx is None:
+                continue
+
+            bounds = self.episode_boundaries.get(episode_idx)
+            if bounds is None:
+                continue
+
+            future_idx = global_idx + self.chunk_size
+            if future_idx >= bounds["end"]:
+                # Near end of episode: use last frame's progress
+                future_idx = bounds["end"] - 1
+
+            future_progress = self.progress_lookup.get(future_idx)
+            if future_progress is not None:
+                delta = future_progress - progress
+                all_deltas.append(delta)
+
+        if all_deltas:
+            self.delta_mean = max(np.mean(all_deltas), 0.0)
+            self.delta_std = max(np.std(all_deltas), self.epsilon)
+            logging.info(f"Progress delta stats: mean={self.delta_mean:.4f}, std={self.delta_std:.4f}")
+        else:
+            self.delta_mean = 0.0
+            self.delta_std = self.epsilon
+            logging.warning("No valid progress deltas found, using default stats")
+
+    def compute_batch_weights(self, batch: dict) -> tuple[torch.Tensor, dict]:
+        """
+        Compute RA-BC weights for a batch.
+
+        For each sample:
+        1. Get progress at current frame
+        2. Get progress at frame + chunk_size (within same episode)
+        3. Compute delta = future_progress - current_progress
+        4. Compute weight using paper Eq. 8-9
+
+        Args:
+            batch: Training batch containing "index" key with global frame indices
+
+        Returns:
+            Tuple of:
+            - Weights tensor (batch_size,) normalized to sum to batch_size
+            - Stats dict with raw_mean_weight, num_zero_weight, num_full_weight
+        """
+        indices = batch.get("index")
+        if indices is None:
+            logging.warning("RA-BC: Batch missing 'index' key, using uniform weights")
+            batch_size = self._get_batch_size(batch)
+            return torch.ones(batch_size, device=self.device), {"raw_mean_weight": 1.0}
+
+        # Convert to list of ints
+        if isinstance(indices, torch.Tensor):
+            indices = indices.cpu().numpy().tolist()
+        elif isinstance(indices, np.ndarray):
+            indices = indices.tolist()
+
+        # Compute deltas and weights for each sample
+        deltas = []
+        for idx in indices:
+            idx = int(idx)
+            delta = self._compute_delta(idx)
+            deltas.append(delta)
+
+        deltas = np.array(deltas, dtype=np.float32)
+
+        # Compute weights from deltas
+        weights = self._compute_weights(deltas)
+
+        # Compute stats before normalization for logging
+        raw_mean_weight = float(np.nanmean(weights))
+        num_zero_weight = int(np.sum(weights == 0))
+        num_full_weight = int(np.sum(weights == 1.0))
+        batch_stats = {
+            "raw_mean_weight": raw_mean_weight,
+            "num_zero_weight": num_zero_weight,
+            "num_full_weight": num_full_weight,
+        }
+
+        weights = torch.tensor(weights, device=self.device, dtype=torch.float32)
+
+        # Normalize to sum to batch_size
+        batch_size = len(weights)
+        weight_sum = weights.sum() + self.epsilon
+        weights = weights * batch_size / weight_sum
+
+        return weights, batch_stats
+
+    def _compute_delta(self, global_idx: int) -> float:
+        """Compute progress delta for a single frame."""
+        current_progress = self.progress_lookup.get(global_idx)
+        if current_progress is None:
+            return np.nan
+
+        episode_idx = self.episode_lookup.get(global_idx)
+        if episode_idx is None:
+            return np.nan
+
+        bounds = self.episode_boundaries.get(episode_idx)
+        if bounds is None:
+            return np.nan
+
+        future_idx = global_idx + self.chunk_size  # Δ = chunk_size
+        if future_idx >= bounds["end"]:
+            # Near end of episode: use last frame's progress instead
+            future_idx = bounds["end"] - 1
+
+        future_progress = self.progress_lookup.get(future_idx)
+        if future_progress is None:
+            return np.nan
+
+        return future_progress - current_progress
+
+    def _compute_weights(self, deltas: np.ndarray) -> np.ndarray:
+        """
+        Compute RA-BC weights from progress deltas.
+
+        Following paper Eq. 8-9:
+        - Soft weight: ˜wi = clip((ri − (µ − 2σ)) / (4σ + ε), 0, 1)
+        - Final weight: wi = 1{ri > κ} + 1{0 ≤ ri ≤ κ}˜wi
+
+        Returns:
+            Array of weights
+        """
+        valid_mask = ~np.isnan(deltas)
+
+        # Compute soft weights using global statistics
+        lower_bound = self.delta_mean - 2 * self.delta_std
+        soft_weights = (deltas - lower_bound) / (4 * self.delta_std + self.epsilon)
+        soft_weights = np.clip(soft_weights, 0.0, 1.0)
+
+        # Apply paper's Eq. 9
+        weights = np.zeros_like(deltas, dtype=np.float32)
+
+        # High quality: ri > kappa → weight = 1
+        high_quality_mask = deltas > self.kappa
+        weights[high_quality_mask] = 1.0
+
+        # Moderate quality: 0 <= ri <= kappa → weight = soft_weight
+        moderate_mask = (deltas >= 0) & (deltas <= self.kappa)
+        weights[moderate_mask] = soft_weights[moderate_mask]
+
+        # Negative progress: ri < 0 → weight = 0 (already 0)
+        # Invalid (NaN): use fallback weight
+        weights[~valid_mask] = self.fallback_weight
+
+        return weights
+
+    def _get_batch_size(self, batch: dict) -> int:
+        """Determine batch size from batch."""
+        for key in ["action", "index"]:
+            if key in batch:
+                val = batch[key]
+                if isinstance(val, (torch.Tensor, np.ndarray)):
+                    return val.shape[0]
+        return 1
+
+    def get_stats(self) -> dict:
+        """Get statistics."""
+        return {
+            "num_frames": len(self.progress_lookup),
+            "chunk_size": self.chunk_size,
+            "head_mode": self.head_mode,
+            "delta_mean": self.delta_mean,
+            "delta_std": self.delta_std,
+            "kappa": self.kappa,
+        }
diff --git a/lerobot/src/lerobot/utils/random_utils.py b/lerobot/src/lerobot/utils/random_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..b34d357aa5da6448b04d7274c8355d5d196db693
--- /dev/null
+++ b/lerobot/src/lerobot/utils/random_utils.py
@@ -0,0 +1,198 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import random
+from collections.abc import Callable, Generator
+from contextlib import contextmanager
+from pathlib import Path
+from typing import Any
+
+import numpy as np
+import torch
+from safetensors.torch import load_file, save_file
+
+from lerobot.datasets.utils import flatten_dict, unflatten_dict
+from lerobot.utils.constants import RNG_STATE
+
+
+def serialize_python_rng_state() -> dict[str, torch.Tensor]:
+    """
+    Returns the rng state for `random` in the form of a flat dict[str, torch.Tensor] to be saved using
+    `safetensors.save_file()` or `torch.save()`.
+    """
+    py_state = random.getstate()
+    return {
+        "py_rng_version": torch.tensor([py_state[0]], dtype=torch.int64),
+        "py_rng_state": torch.tensor(py_state[1], dtype=torch.int64),
+    }
+
+
+def deserialize_python_rng_state(rng_state_dict: dict[str, torch.Tensor]) -> None:
+    """
+    Restores the rng state for `random` from a dictionary produced by `serialize_python_rng_state()`.
+    """
+    py_state = (rng_state_dict["py_rng_version"].item(), tuple(rng_state_dict["py_rng_state"].tolist()), None)
+    random.setstate(py_state)
+
+
+def serialize_numpy_rng_state() -> dict[str, torch.Tensor]:
+    """
+    Returns the rng state for `numpy` in the form of a flat dict[str, torch.Tensor] to be saved using
+    `safetensors.save_file()` or `torch.save()`.
+    """
+    np_state = np.random.get_state()
+    # Ensure no breaking changes from numpy
+    assert np_state[0] == "MT19937"
+    return {
+        "np_rng_state_values": torch.tensor(np_state[1], dtype=torch.int64),
+        "np_rng_state_index": torch.tensor([np_state[2]], dtype=torch.int64),
+        "np_rng_has_gauss": torch.tensor([np_state[3]], dtype=torch.int64),
+        "np_rng_cached_gaussian": torch.tensor([np_state[4]], dtype=torch.float32),
+    }
+
+
+def deserialize_numpy_rng_state(rng_state_dict: dict[str, torch.Tensor]) -> None:
+    """
+    Restores the rng state for `numpy` from a dictionary produced by `serialize_numpy_rng_state()`.
+    """
+    np_state = (
+        "MT19937",
+        rng_state_dict["np_rng_state_values"].numpy(),
+        rng_state_dict["np_rng_state_index"].item(),
+        rng_state_dict["np_rng_has_gauss"].item(),
+        rng_state_dict["np_rng_cached_gaussian"].item(),
+    )
+    np.random.set_state(np_state)
+
+
+def serialize_torch_rng_state() -> dict[str, torch.Tensor]:
+    """
+    Returns the rng state for `torch` in the form of a flat dict[str, torch.Tensor] to be saved using
+    `safetensors.save_file()` or `torch.save()`.
+    """
+    torch_rng_state_dict = {"torch_rng_state": torch.get_rng_state()}
+    if torch.cuda.is_available():
+        torch_rng_state_dict["torch_cuda_rng_state"] = torch.cuda.get_rng_state()
+    return torch_rng_state_dict
+
+
+def deserialize_torch_rng_state(rng_state_dict: dict[str, torch.Tensor]) -> None:
+    """
+    Restores the rng state for `torch` from a dictionary produced by `serialize_torch_rng_state()`.
+    """
+    torch.set_rng_state(rng_state_dict["torch_rng_state"])
+    if torch.cuda.is_available() and "torch_cuda_rng_state" in rng_state_dict:
+        torch.cuda.set_rng_state(rng_state_dict["torch_cuda_rng_state"])
+
+
+def serialize_rng_state() -> dict[str, torch.Tensor]:
+    """
+    Returns the rng state for `random`, `numpy`, and `torch`, in the form of a flat
+    dict[str, torch.Tensor] to be saved using `safetensors.save_file()` `torch.save()`.
+    """
+    py_rng_state_dict = serialize_python_rng_state()
+    np_rng_state_dict = serialize_numpy_rng_state()
+    torch_rng_state_dict = serialize_torch_rng_state()
+
+    return {
+        **py_rng_state_dict,
+        **np_rng_state_dict,
+        **torch_rng_state_dict,
+    }
+
+
+def deserialize_rng_state(rng_state_dict: dict[str, torch.Tensor]) -> None:
+    """
+    Restores the rng state for `random`, `numpy`, and `torch` from a dictionary produced by
+    `serialize_rng_state()`.
+    """
+    py_rng_state_dict = {k: v for k, v in rng_state_dict.items() if k.startswith("py")}
+    np_rng_state_dict = {k: v for k, v in rng_state_dict.items() if k.startswith("np")}
+    torch_rng_state_dict = {k: v for k, v in rng_state_dict.items() if k.startswith("torch")}
+
+    deserialize_python_rng_state(py_rng_state_dict)
+    deserialize_numpy_rng_state(np_rng_state_dict)
+    deserialize_torch_rng_state(torch_rng_state_dict)
+
+
+def save_rng_state(save_dir: Path) -> None:
+    rng_state_dict = serialize_rng_state()
+    flat_rng_state_dict = flatten_dict(rng_state_dict)
+    save_file(flat_rng_state_dict, save_dir / RNG_STATE)
+
+
+def load_rng_state(save_dir: Path) -> None:
+    flat_rng_state_dict = load_file(save_dir / RNG_STATE)
+    rng_state_dict = unflatten_dict(flat_rng_state_dict)
+    deserialize_rng_state(rng_state_dict)
+
+
+def get_rng_state() -> dict[str, Any]:
+    """Get the random state for `random`, `numpy`, and `torch`."""
+    random_state_dict = {
+        "random_state": random.getstate(),
+        "numpy_random_state": np.random.get_state(),
+        "torch_random_state": torch.random.get_rng_state(),
+    }
+    if torch.cuda.is_available():
+        random_state_dict["torch_cuda_random_state"] = torch.cuda.random.get_rng_state()
+    return random_state_dict
+
+
+def set_rng_state(random_state_dict: dict[str, Any]):
+    """Set the random state for `random`, `numpy`, and `torch`.
+
+    Args:
+        random_state_dict: A dictionary of the form returned by `get_rng_state`.
+    """
+    random.setstate(random_state_dict["random_state"])
+    np.random.set_state(random_state_dict["numpy_random_state"])
+    torch.random.set_rng_state(random_state_dict["torch_random_state"])
+    if torch.cuda.is_available():
+        torch.cuda.random.set_rng_state(random_state_dict["torch_cuda_random_state"])
+
+
+def set_seed(seed, accelerator: Callable | None = None) -> None:
+    """Set seed for reproducibility."""
+    random.seed(seed)
+    np.random.seed(seed)
+    torch.manual_seed(seed)
+
+    if torch.cuda.is_available():
+        torch.cuda.manual_seed_all(seed)
+
+    if accelerator:
+        from accelerate.utils import set_seed as _accelerate_set_seed
+
+        _accelerate_set_seed(seed)
+
+
+@contextmanager
+def seeded_context(seed: int) -> Generator[None, None, None]:
+    """Set the seed when entering a context, and restore the prior random state at exit.
+
+    Example usage:
+
+    ```
+    a = random.random()  # produces some random number
+    with seeded_context(1337):
+        b = random.random()  # produces some other random number
+    c = random.random()  # produces yet another random number, but the same it would have if we never made `b`
+    ```
+    """
+    random_state_dict = get_rng_state()
+    set_seed(seed)
+    yield None
+    set_rng_state(random_state_dict)
diff --git a/lerobot/src/lerobot/utils/robot_utils.py b/lerobot/src/lerobot/utils/robot_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..656dc2649173af730ebb28cc179718fd0e3d49e6
--- /dev/null
+++ b/lerobot/src/lerobot/utils/robot_utils.py
@@ -0,0 +1,55 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import platform
+import time
+
+
+def precise_sleep(seconds: float, spin_threshold: float = 0.010, sleep_margin: float = 0.005):
+    """
+    Wait for `seconds` with better precision than time.sleep alone at the expense of more CPU usage.
+
+    Parameters:
+      - seconds: duration to wait
+      - spin_threshold: if remaining <= spin_threshold -> spin; otherwise sleep (seconds). Default 10ms
+      - sleep_margin: when sleeping leave this much time before deadline to avoid oversleep. Default 5ms
+
+    Note:
+        The default parameters are chosen to prioritize timing accuracy over CPU usage for the common 30 FPS use case.
+    """
+    if seconds <= 0:
+        return
+
+    system = platform.system()
+    # On macOS and Windows the scheduler / sleep granularity can make
+    # short sleeps inaccurate. Instead of burning CPU for the whole
+    # duration, sleep for most of the time and spin for the final few
+    # milliseconds to achieve good accuracy with much lower CPU usage.
+    if system in ("Darwin", "Windows"):
+        end_time = time.perf_counter() + seconds
+        while True:
+            remaining = end_time - time.perf_counter()
+            if remaining <= 0:
+                break
+            # If there's more than a couple milliseconds left, sleep most
+            # of the remaining time and leave a small margin for the final spin.
+            if remaining > spin_threshold:
+                # Sleep but avoid sleeping past the end by leaving a small margin.
+                time.sleep(max(remaining - sleep_margin, 0))
+            else:
+                # Final short spin to hit precise timing without long sleeps.
+                pass
+    else:
+        # On Linux time.sleep is accurate enough for most uses
+        time.sleep(seconds)
diff --git a/lerobot/src/lerobot/utils/rotation.py b/lerobot/src/lerobot/utils/rotation.py
new file mode 100644
index 0000000000000000000000000000000000000000..41b6529478aefae6f2ab77cba50b1b7245c186fc
--- /dev/null
+++ b/lerobot/src/lerobot/utils/rotation.py
@@ -0,0 +1,270 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Custom rotation utilities to replace scipy.spatial.transform.Rotation."""
+
+import numpy as np
+
+
+class Rotation:
+    """
+    Custom rotation class that provides a subset of scipy.spatial.transform.Rotation functionality.
+
+    Supports conversions between rotation vectors, rotation matrices, and quaternions.
+    """
+
+    def __init__(self, quat: np.ndarray) -> None:
+        """Initialize rotation from quaternion [x, y, z, w]."""
+        self._quat = np.asarray(quat, dtype=float)
+        # Normalize quaternion
+        norm = np.linalg.norm(self._quat)
+        if norm > 0:
+            self._quat = self._quat / norm
+
+    @classmethod
+    def from_rotvec(cls, rotvec: np.ndarray) -> "Rotation":
+        """
+        Create rotation from rotation vector using Rodrigues' formula.
+
+        Args:
+            rotvec: Rotation vector [x, y, z] where magnitude is angle in radians
+
+        Returns:
+            Rotation instance
+        """
+        rotvec = np.asarray(rotvec, dtype=float)
+        angle = np.linalg.norm(rotvec)
+
+        if angle < 1e-8:
+            # For very small angles, use identity quaternion
+            quat = np.array([0.0, 0.0, 0.0, 1.0])
+        else:
+            axis = rotvec / angle
+            half_angle = angle / 2.0
+            sin_half = np.sin(half_angle)
+            cos_half = np.cos(half_angle)
+
+            # Quaternion [x, y, z, w]
+            quat = np.array([axis[0] * sin_half, axis[1] * sin_half, axis[2] * sin_half, cos_half])
+
+        return cls(quat)
+
+    @classmethod
+    def from_matrix(cls, matrix: np.ndarray) -> "Rotation":
+        """
+        Create rotation from 3x3 rotation matrix.
+
+        Args:
+            matrix: 3x3 rotation matrix
+
+        Returns:
+            Rotation instance
+        """
+        matrix = np.asarray(matrix, dtype=float)
+
+        # Shepherd's method for converting rotation matrix to quaternion
+        trace = np.trace(matrix)
+
+        if trace > 0:
+            s = np.sqrt(trace + 1.0) * 2  # s = 4 * qw
+            qw = 0.25 * s
+            qx = (matrix[2, 1] - matrix[1, 2]) / s
+            qy = (matrix[0, 2] - matrix[2, 0]) / s
+            qz = (matrix[1, 0] - matrix[0, 1]) / s
+        elif matrix[0, 0] > matrix[1, 1] and matrix[0, 0] > matrix[2, 2]:
+            s = np.sqrt(1.0 + matrix[0, 0] - matrix[1, 1] - matrix[2, 2]) * 2  # s = 4 * qx
+            qw = (matrix[2, 1] - matrix[1, 2]) / s
+            qx = 0.25 * s
+            qy = (matrix[0, 1] + matrix[1, 0]) / s
+            qz = (matrix[0, 2] + matrix[2, 0]) / s
+        elif matrix[1, 1] > matrix[2, 2]:
+            s = np.sqrt(1.0 + matrix[1, 1] - matrix[0, 0] - matrix[2, 2]) * 2  # s = 4 * qy
+            qw = (matrix[0, 2] - matrix[2, 0]) / s
+            qx = (matrix[0, 1] + matrix[1, 0]) / s
+            qy = 0.25 * s
+            qz = (matrix[1, 2] + matrix[2, 1]) / s
+        else:
+            s = np.sqrt(1.0 + matrix[2, 2] - matrix[0, 0] - matrix[1, 1]) * 2  # s = 4 * qz
+            qw = (matrix[1, 0] - matrix[0, 1]) / s
+            qx = (matrix[0, 2] + matrix[2, 0]) / s
+            qy = (matrix[1, 2] + matrix[2, 1]) / s
+            qz = 0.25 * s
+
+        quat = np.array([qx, qy, qz, qw])
+        return cls(quat)
+
+    @classmethod
+    def from_quat(cls, quat: np.ndarray) -> "Rotation":
+        """
+        Create rotation from quaternion.
+
+        Args:
+            quat: Quaternion [x, y, z, w] or [w, x, y, z] (specify convention in docstring)
+                  This implementation expects [x, y, z, w] format
+
+        Returns:
+            Rotation instance
+        """
+        return cls(quat)
+
+    def as_matrix(self) -> np.ndarray:
+        """
+        Convert rotation to 3x3 rotation matrix.
+
+        Returns:
+            3x3 rotation matrix
+        """
+        qx, qy, qz, qw = self._quat
+
+        # Compute rotation matrix from quaternion
+        return np.array(
+            [
+                [1 - 2 * (qy * qy + qz * qz), 2 * (qx * qy - qz * qw), 2 * (qx * qz + qy * qw)],
+                [2 * (qx * qy + qz * qw), 1 - 2 * (qx * qx + qz * qz), 2 * (qy * qz - qx * qw)],
+                [2 * (qx * qz - qy * qw), 2 * (qy * qz + qx * qw), 1 - 2 * (qx * qx + qy * qy)],
+            ],
+            dtype=float,
+        )
+
+    def as_rotvec(self) -> np.ndarray:
+        """
+        Convert rotation to rotation vector.
+
+        Returns:
+            Rotation vector [x, y, z] where magnitude is angle in radians
+        """
+        qx, qy, qz, qw = self._quat
+
+        # Ensure qw is positive for unique representation
+        if qw < 0:
+            qx, qy, qz, qw = -qx, -qy, -qz, -qw
+
+        # Compute angle and axis
+        angle = 2.0 * np.arccos(np.clip(abs(qw), 0.0, 1.0))
+        sin_half_angle = np.sqrt(1.0 - qw * qw)
+
+        if sin_half_angle < 1e-8:
+            # For very small angles, use linearization: rotvec ≈ 2 * [qx, qy, qz]
+            return 2.0 * np.array([qx, qy, qz])
+
+        # Extract axis and scale by angle
+        axis = np.array([qx, qy, qz]) / sin_half_angle
+        return angle * axis
+
+    def as_quat(self) -> np.ndarray:
+        """
+        Get quaternion representation.
+
+        Returns:
+            Quaternion [x, y, z, w]
+        """
+        return self._quat.copy()
+
+    def apply(self, vectors: np.ndarray, inverse: bool = False) -> np.ndarray:
+        """
+        Apply this rotation to a set of vectors.
+
+        This is equivalent to applying the rotation matrix to the vectors:
+        self.as_matrix() @ vectors (or self.as_matrix().T @ vectors if inverse=True).
+
+        Args:
+            vectors: Array of shape (3,) or (N, 3) representing vectors in 3D space
+            inverse: If True, apply the inverse of the rotation. Default is False.
+
+        Returns:
+            Rotated vectors with shape:
+            - (3,) if input was single vector with shape (3,)
+            - (N, 3) in all other cases
+        """
+        vectors = np.asarray(vectors, dtype=float)
+        original_shape = vectors.shape
+
+        # Handle single vector case - ensure it's 2D for matrix multiplication
+        if vectors.ndim == 1:
+            if len(vectors) != 3:
+                raise ValueError("Single vector must have length 3")
+            vectors = vectors.reshape(1, 3)
+            single_vector = True
+        elif vectors.ndim == 2:
+            if vectors.shape[1] != 3:
+                raise ValueError("Vectors must have shape (N, 3)")
+            single_vector = False
+        else:
+            raise ValueError("Vectors must be 1D or 2D array")
+
+        # Get rotation matrix
+        rotation_matrix = self.as_matrix()
+
+        # Apply inverse if requested (transpose for orthogonal rotation matrices)
+        if inverse:
+            rotation_matrix = rotation_matrix.T
+
+        # Apply rotation: (N, 3) @ (3, 3).T -> (N, 3)
+        rotated_vectors = vectors @ rotation_matrix.T
+
+        # Return original shape for single vector case
+        if single_vector and original_shape == (3,):
+            return rotated_vectors.flatten()
+
+        return rotated_vectors
+
+    def inv(self) -> "Rotation":
+        """
+        Invert this rotation.
+
+        Composition of a rotation with its inverse results in an identity transformation.
+
+        Returns:
+            Rotation instance containing the inverse of this rotation
+        """
+        qx, qy, qz, qw = self._quat
+
+        # For a unit quaternion, the inverse is the conjugate: [-x, -y, -z, w]
+        inverse_quat = np.array([-qx, -qy, -qz, qw])
+
+        return Rotation(inverse_quat)
+
+    def __mul__(self, other: "Rotation") -> "Rotation":
+        """
+        Compose this rotation with another rotation using the * operator.
+
+        The composition `r2 * r1` means "apply r1 first, then r2".
+        This is equivalent to applying rotation matrices: r2.as_matrix() @ r1.as_matrix()
+
+        Args:
+            other: Another Rotation instance to compose with
+
+        Returns:
+            Rotation instance representing the composition of rotations
+        """
+        if not isinstance(other, Rotation):
+            return NotImplemented
+
+        # Get quaternions [x, y, z, w]
+        x1, y1, z1, w1 = other._quat  # Apply first
+        x2, y2, z2, w2 = self._quat  # Apply second
+
+        # Quaternion multiplication: q2 * q1 (apply q1 first, then q2)
+        composed_quat = np.array(
+            [
+                w2 * x1 + x2 * w1 + y2 * z1 - z2 * y1,  # x component
+                w2 * y1 - x2 * z1 + y2 * w1 + z2 * x1,  # y component
+                w2 * z1 + x2 * y1 - y2 * x1 + z2 * w1,  # z component
+                w2 * w1 - x2 * x1 - y2 * y1 - z2 * z1,  # w component
+            ]
+        )
+
+        return Rotation(composed_quat)
diff --git a/lerobot/src/lerobot/utils/train_utils.py b/lerobot/src/lerobot/utils/train_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..02f6aebb39831bc0f16fd95e5fa7f67d8b8a2946
--- /dev/null
+++ b/lerobot/src/lerobot/utils/train_utils.py
@@ -0,0 +1,169 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from pathlib import Path
+
+from torch.optim import Optimizer
+from torch.optim.lr_scheduler import LRScheduler
+
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.datasets.io_utils import load_json, write_json
+from lerobot.optim.optimizers import load_optimizer_state, save_optimizer_state
+from lerobot.optim.schedulers import load_scheduler_state, save_scheduler_state
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.processor import PolicyProcessorPipeline
+from lerobot.utils.constants import (
+    CHECKPOINTS_DIR,
+    LAST_CHECKPOINT_LINK,
+    PRETRAINED_MODEL_DIR,
+    TRAINING_STATE_DIR,
+    TRAINING_STEP,
+)
+from lerobot.utils.random_utils import load_rng_state, save_rng_state
+
+
+def get_step_identifier(step: int, total_steps: int) -> str:
+    num_digits = max(6, len(str(total_steps)))
+    return f"{step:0{num_digits}d}"
+
+
+def get_step_checkpoint_dir(output_dir: Path, total_steps: int, step: int) -> Path:
+    """Returns the checkpoint sub-directory corresponding to the step number."""
+    step_identifier = get_step_identifier(step, total_steps)
+    return output_dir / CHECKPOINTS_DIR / step_identifier
+
+
+def save_training_step(step: int, save_dir: Path) -> None:
+    write_json({"step": step}, save_dir / TRAINING_STEP)
+
+
+def load_training_step(save_dir: Path) -> int:
+    training_step = load_json(save_dir / TRAINING_STEP)
+    return training_step["step"]
+
+
+def update_last_checkpoint(checkpoint_dir: Path) -> Path:
+    last_checkpoint_dir = checkpoint_dir.parent / LAST_CHECKPOINT_LINK
+    if last_checkpoint_dir.is_symlink():
+        last_checkpoint_dir.unlink()
+    relative_target = checkpoint_dir.relative_to(checkpoint_dir.parent)
+    last_checkpoint_dir.symlink_to(relative_target)
+
+
+def save_checkpoint(
+    checkpoint_dir: Path,
+    step: int,
+    cfg: TrainPipelineConfig,
+    policy: PreTrainedPolicy,
+    optimizer: Optimizer,
+    scheduler: LRScheduler | None = None,
+    preprocessor: PolicyProcessorPipeline | None = None,
+    postprocessor: PolicyProcessorPipeline | None = None,
+) -> None:
+    """This function creates the following directory structure:
+
+    005000/  #  training step at checkpoint
+    ├── pretrained_model/
+    │   ├── config.json  # policy config
+    │   ├── model.safetensors  # policy weights
+    │   ├── train_config.json  # train config
+    │   ├── processor.json  # processor config (if preprocessor provided)
+    │   └── step_*.safetensors  # processor state files (if any)
+    └── training_state/
+        ├── optimizer_param_groups.json  #  optimizer param groups
+        ├── optimizer_state.safetensors  # optimizer state
+        ├── rng_state.safetensors  # rng states
+        ├── scheduler_state.json  # scheduler state
+        └── training_step.json  # training step
+
+    Args:
+        cfg (TrainPipelineConfig): The training config used for this run.
+        step (int): The training step at that checkpoint.
+        policy (PreTrainedPolicy): The policy to save.
+        optimizer (Optimizer | None, optional): The optimizer to save the state from. Defaults to None.
+        scheduler (LRScheduler | None, optional): The scheduler to save the state from. Defaults to None.
+        preprocessor: The preprocessor/pipeline to save. Defaults to None.
+    """
+    pretrained_dir = checkpoint_dir / PRETRAINED_MODEL_DIR
+    policy.save_pretrained(pretrained_dir)
+    cfg.save_pretrained(pretrained_dir)
+    if cfg.peft is not None:
+        # When using PEFT, policy.save_pretrained will only write the adapter weights + config, not the
+        # policy config which we need for loading the model. In this case we'll write it ourselves.
+        policy.config.save_pretrained(pretrained_dir)
+    if preprocessor is not None:
+        preprocessor.save_pretrained(pretrained_dir)
+    if postprocessor is not None:
+        postprocessor.save_pretrained(pretrained_dir)
+    save_training_state(checkpoint_dir, step, optimizer, scheduler)
+
+
+def save_training_state(
+    checkpoint_dir: Path,
+    train_step: int,
+    optimizer: Optimizer | None = None,
+    scheduler: LRScheduler | None = None,
+) -> None:
+    """
+    Saves the training step, optimizer state, scheduler state, and rng state.
+
+    Args:
+        save_dir (Path): The directory to save artifacts to.
+        train_step (int): Current training step.
+        optimizer (Optimizer | None, optional): The optimizer from which to save the state_dict.
+            Defaults to None.
+        scheduler (LRScheduler | None, optional): The scheduler from which to save the state_dict.
+            Defaults to None.
+    """
+    save_dir = checkpoint_dir / TRAINING_STATE_DIR
+    save_dir.mkdir(parents=True, exist_ok=True)
+    save_training_step(train_step, save_dir)
+    save_rng_state(save_dir)
+    if optimizer is not None:
+        save_optimizer_state(optimizer, save_dir)
+    if scheduler is not None:
+        save_scheduler_state(scheduler, save_dir)
+
+
+def load_training_state(
+    checkpoint_dir: Path, optimizer: Optimizer, scheduler: LRScheduler | None
+) -> tuple[int, Optimizer, LRScheduler | None]:
+    """
+    Loads the training step, optimizer state, scheduler state, and rng state.
+    This is used to resume a training run.
+
+    Args:
+        checkpoint_dir (Path): The checkpoint directory. Should contain a 'training_state' dir.
+        optimizer (Optimizer): The optimizer to load the state_dict to.
+        scheduler (LRScheduler | None): The scheduler to load the state_dict to (can be None).
+
+    Raises:
+        NotADirectoryError: If 'checkpoint_dir' doesn't contain a 'training_state' dir
+
+    Returns:
+        tuple[int, Optimizer, LRScheduler | None]: training step, optimizer and scheduler with their
+            state_dict loaded.
+    """
+    training_state_dir = checkpoint_dir / TRAINING_STATE_DIR
+    if not training_state_dir.is_dir():
+        raise NotADirectoryError(training_state_dir)
+
+    load_rng_state(training_state_dir)
+    step = load_training_step(training_state_dir)
+    optimizer = load_optimizer_state(optimizer, training_state_dir)
+    if scheduler is not None:
+        scheduler = load_scheduler_state(scheduler, training_state_dir)
+
+    return step, optimizer, scheduler
diff --git a/lerobot/src/lerobot/utils/transition.py b/lerobot/src/lerobot/utils/transition.py
new file mode 100644
index 0000000000000000000000000000000000000000..fe36208615b95e865a955cbe9fe5b20b48c256d7
--- /dev/null
+++ b/lerobot/src/lerobot/utils/transition.py
@@ -0,0 +1,87 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from typing import TypedDict
+
+import torch
+
+from lerobot.utils.constants import ACTION
+
+
+class Transition(TypedDict):
+    state: dict[str, torch.Tensor]
+    action: torch.Tensor
+    reward: float
+    next_state: dict[str, torch.Tensor]
+    done: bool
+    truncated: bool
+    complementary_info: dict[str, torch.Tensor | float | int] | None = None
+
+
+def move_transition_to_device(transition: Transition, device: str = "cpu") -> Transition:
+    device = torch.device(device)
+    non_blocking = device.type == "cuda"
+
+    # Move state tensors to device
+    transition["state"] = {
+        key: val.to(device, non_blocking=non_blocking) for key, val in transition["state"].items()
+    }
+
+    # Move action to device
+    transition[ACTION] = transition[ACTION].to(device, non_blocking=non_blocking)
+
+    # Move reward and done if they are tensors
+    if isinstance(transition["reward"], torch.Tensor):
+        transition["reward"] = transition["reward"].to(device, non_blocking=non_blocking)
+
+    if isinstance(transition["done"], torch.Tensor):
+        transition["done"] = transition["done"].to(device, non_blocking=non_blocking)
+
+    if isinstance(transition["truncated"], torch.Tensor):
+        transition["truncated"] = transition["truncated"].to(device, non_blocking=non_blocking)
+
+    # Move next_state tensors to device
+    transition["next_state"] = {
+        key: val.to(device, non_blocking=non_blocking) for key, val in transition["next_state"].items()
+    }
+
+    # Move complementary_info tensors if present
+    if transition.get("complementary_info") is not None:
+        for key, val in transition["complementary_info"].items():
+            if isinstance(val, torch.Tensor):
+                transition["complementary_info"][key] = val.to(device, non_blocking=non_blocking)
+            elif isinstance(val, (int | float | bool)):
+                transition["complementary_info"][key] = torch.tensor(val, device=device)
+            else:
+                raise ValueError(f"Unsupported type {type(val)} for complementary_info[{key}]")
+    return transition
+
+
+def move_state_dict_to_device(state_dict, device="cpu"):
+    """
+    Recursively move all tensors in a (potentially) nested
+    dict/list/tuple structure to the CPU.
+    """
+    if isinstance(state_dict, torch.Tensor):
+        return state_dict.to(device)
+    elif isinstance(state_dict, dict):
+        return {k: move_state_dict_to_device(v, device=device) for k, v in state_dict.items()}
+    elif isinstance(state_dict, list):
+        return [move_state_dict_to_device(v, device=device) for v in state_dict]
+    elif isinstance(state_dict, tuple):
+        return tuple(move_state_dict_to_device(v, device=device) for v in state_dict)
+    else:
+        return state_dict
diff --git a/lerobot/src/lerobot/utils/utils.py b/lerobot/src/lerobot/utils/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..b9f8441d6748b4bdc69d533e2816e312020042b1
--- /dev/null
+++ b/lerobot/src/lerobot/utils/utils.py
@@ -0,0 +1,327 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from __future__ import annotations
+
+import logging
+import os
+import platform
+import select
+import subprocess
+import sys
+import time
+from copy import copy, deepcopy
+from datetime import datetime
+from pathlib import Path
+from statistics import mean
+from typing import TYPE_CHECKING
+
+import numpy as np
+
+if TYPE_CHECKING:
+    from accelerate import Accelerator
+
+
+def inside_slurm():
+    """Check whether the python process was launched through slurm"""
+    # TODO(rcadene): return False for interactive mode `--pty bash`
+    return "SLURM_JOB_ID" in os.environ
+
+
+def init_logging(
+    log_file: Path | None = None,
+    display_pid: bool = False,
+    console_level: str = "INFO",
+    file_level: str = "DEBUG",
+    accelerator: Accelerator | None = None,
+):
+    """Initialize logging configuration for LeRobot.
+
+    In multi-GPU training, only the main process logs to console to avoid duplicate output.
+    Non-main processes have console logging suppressed but can still log to file.
+
+    Args:
+        log_file: Optional file path to write logs to
+        display_pid: Include process ID in log messages (useful for debugging multi-process)
+        console_level: Logging level for console output
+        file_level: Logging level for file output
+        accelerator: Optional Accelerator instance (for multi-GPU detection)
+    """
+
+    def custom_format(record: logging.LogRecord) -> str:
+        dt = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
+        fnameline = f"{record.pathname}:{record.lineno}"
+        pid_str = f"[PID: {os.getpid()}] " if display_pid else ""
+        return f"{record.levelname} {pid_str}{dt} {fnameline[-15:]:>15} {record.getMessage()}"
+
+    formatter = logging.Formatter()
+    formatter.format = custom_format
+
+    logger = logging.getLogger()
+    logger.setLevel(logging.NOTSET)
+
+    # Clear any existing handlers
+    logger.handlers.clear()
+
+    # Determine if this is a non-main process in distributed training
+    is_main_process = accelerator.is_main_process if accelerator is not None else True
+
+    # Console logging (main process only)
+    if is_main_process:
+        console_handler = logging.StreamHandler()
+        console_handler.setFormatter(formatter)
+        console_handler.setLevel(console_level.upper())
+        logger.addHandler(console_handler)
+    else:
+        # Suppress console output for non-main processes
+        logger.addHandler(logging.NullHandler())
+        logger.setLevel(logging.ERROR)
+
+    if log_file is not None:
+        file_handler = logging.FileHandler(log_file)
+        file_handler.setFormatter(formatter)
+        file_handler.setLevel(file_level.upper())
+        logger.addHandler(file_handler)
+
+
+def format_big_number(num, precision=0):
+    suffixes = ["", "K", "M", "B", "T", "Q"]
+    divisor = 1000.0
+
+    for suffix in suffixes:
+        if abs(num) < divisor:
+            return f"{num:.{precision}f}{suffix}"
+        num /= divisor
+
+    return num
+
+
+def say(text: str, blocking: bool = False):
+    system = platform.system()
+
+    if system == "Darwin":
+        cmd = ["say", text]
+
+    elif system == "Linux":
+        cmd = ["spd-say", text]
+        if blocking:
+            cmd.append("--wait")
+
+    elif system == "Windows":
+        cmd = [
+            "PowerShell",
+            "-Command",
+            "Add-Type -AssemblyName System.Speech; "
+            f"(New-Object System.Speech.Synthesis.SpeechSynthesizer).Speak('{text}')",
+        ]
+
+    else:
+        raise RuntimeError("Unsupported operating system for text-to-speech.")
+
+    if blocking:
+        subprocess.run(cmd, check=True)
+    else:
+        subprocess.Popen(cmd, creationflags=subprocess.CREATE_NO_WINDOW if system == "Windows" else 0)
+
+
+def log_say(text: str, play_sounds: bool = True, blocking: bool = False):
+    logging.info(text)
+
+    if play_sounds:
+        say(text, blocking)
+
+
+def get_channel_first_image_shape(image_shape: tuple) -> tuple:
+    shape = copy(image_shape)
+    if shape[2] < shape[0] and shape[2] < shape[1]:  # (h, w, c) -> (c, h, w)
+        shape = (shape[2], shape[0], shape[1])
+    elif not (shape[0] < shape[1] and shape[0] < shape[2]):
+        raise ValueError(image_shape)
+
+    return shape
+
+
+def has_method(cls: object, method_name: str) -> bool:
+    return hasattr(cls, method_name) and callable(getattr(cls, method_name))
+
+
+def is_valid_numpy_dtype_string(dtype_str: str) -> bool:
+    """
+    Return True if a given string can be converted to a numpy dtype.
+    """
+    try:
+        # Attempt to convert the string to a numpy dtype
+        np.dtype(dtype_str)
+        return True
+    except TypeError:
+        # If a TypeError is raised, the string is not a valid dtype
+        return False
+
+
+def enter_pressed() -> bool:
+    if platform.system() == "Windows":
+        import msvcrt
+
+        if msvcrt.kbhit():
+            key = msvcrt.getch()
+            return key in (b"\r", b"\n")  # enter key
+        return False
+    else:
+        return select.select([sys.stdin], [], [], 0)[0] and sys.stdin.readline().strip() == ""
+
+
+def move_cursor_up(lines):
+    """Move the cursor up by a specified number of lines."""
+    print(f"\033[{lines}A", end="")
+
+
+def get_elapsed_time_in_days_hours_minutes_seconds(elapsed_time_s: float):
+    days = int(elapsed_time_s // (24 * 3600))
+    elapsed_time_s %= 24 * 3600
+    hours = int(elapsed_time_s // 3600)
+    elapsed_time_s %= 3600
+    minutes = int(elapsed_time_s // 60)
+    seconds = elapsed_time_s % 60
+    return days, hours, minutes, seconds
+
+
+class SuppressProgressBars:
+    """
+    Context manager to suppress progress bars.
+
+    Example
+    --------
+    ```python
+    with SuppressProgressBars():
+        # Code that would normally show progress bars
+    ```
+    """
+
+    def __enter__(self):
+        from datasets.utils.logging import disable_progress_bar
+
+        disable_progress_bar()
+
+    def __exit__(self, exc_type, exc_val, exc_tb):
+        from datasets.utils.logging import enable_progress_bar
+
+        enable_progress_bar()
+
+
+class TimerManager:
+    """
+    Lightweight utility to measure elapsed time.
+
+    Examples
+    --------
+    ```python
+    # Example 1: Using context manager
+    timer = TimerManager("Policy", log=False)
+    for _ in range(3):
+        with timer:
+            time.sleep(0.01)
+    print(timer.last, timer.fps_avg, timer.percentile(90))  # Prints: 0.01 100.0 0.01
+    ```
+
+    ```python
+    # Example 2: Using start/stop methods
+    timer = TimerManager("Policy", log=False)
+    timer.start()
+    time.sleep(0.01)
+    timer.stop()
+    print(timer.last, timer.fps_avg, timer.percentile(90))  # Prints: 0.01 100.0 0.01
+    ```
+    """
+
+    def __init__(
+        self,
+        label: str = "Elapsed-time",
+        log: bool = True,
+        logger: logging.Logger | None = None,
+    ):
+        self.label = label
+        self.log = log
+        self.logger = logger
+        self._start: float | None = None
+        self._history: list[float] = []
+
+    def __enter__(self):
+        return self.start()
+
+    def __exit__(self, exc_type, exc_val, exc_tb):
+        self.stop()
+
+    def start(self):
+        self._start = time.perf_counter()
+        return self
+
+    def stop(self) -> float:
+        if self._start is None:
+            raise RuntimeError("Timer was never started.")
+        elapsed = time.perf_counter() - self._start
+        self._history.append(elapsed)
+        self._start = None
+        if self.log:
+            if self.logger is not None:
+                self.logger.info(f"{self.label}: {elapsed:.6f} s")
+            else:
+                logging.info(f"{self.label}: {elapsed:.6f} s")
+        return elapsed
+
+    def reset(self):
+        self._history.clear()
+
+    @property
+    def last(self) -> float:
+        return self._history[-1] if self._history else 0.0
+
+    @property
+    def avg(self) -> float:
+        return mean(self._history) if self._history else 0.0
+
+    @property
+    def total(self) -> float:
+        return sum(self._history)
+
+    @property
+    def count(self) -> int:
+        return len(self._history)
+
+    @property
+    def history(self) -> list[float]:
+        return deepcopy(self._history)
+
+    @property
+    def fps_last(self) -> float:
+        return 0.0 if self.last == 0 else 1.0 / self.last
+
+    @property
+    def fps_avg(self) -> float:
+        return 0.0 if self.avg == 0 else 1.0 / self.avg
+
+    def percentile(self, p: float) -> float:
+        """
+        Return the p-th percentile of recorded times.
+        """
+        if not self._history:
+            return 0.0
+        return float(np.percentile(self._history, p))
+
+    def fps_percentile(self, p: float) -> float:
+        """
+        FPS corresponding to the p-th percentile time.
+        """
+        val = self.percentile(p)
+        return 0.0 if val == 0 else 1.0 / val
diff --git a/lerobot/src/lerobot/utils/visualization_utils.py b/lerobot/src/lerobot/utils/visualization_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..782358c9ef71b41cc62566235646b0a1ca54cddf
--- /dev/null
+++ b/lerobot/src/lerobot/utils/visualization_utils.py
@@ -0,0 +1,112 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import numbers
+import os
+
+import numpy as np
+import rerun as rr
+
+from lerobot.types import RobotAction, RobotObservation
+
+from .constants import ACTION, ACTION_PREFIX, OBS_PREFIX, OBS_STR
+
+
+def init_rerun(
+    session_name: str = "lerobot_control_loop", ip: str | None = None, port: int | None = None
+) -> None:
+    """
+    Initializes the Rerun SDK for visualizing the control loop.
+
+    Args:
+        session_name: Name of the Rerun session.
+        ip: Optional IP for connecting to a Rerun server.
+        port: Optional port for connecting to a Rerun server.
+    """
+    batch_size = os.getenv("RERUN_FLUSH_NUM_BYTES", "8000")
+    os.environ["RERUN_FLUSH_NUM_BYTES"] = batch_size
+    rr.init(session_name)
+    memory_limit = os.getenv("LEROBOT_RERUN_MEMORY_LIMIT", "10%")
+    if ip and port:
+        rr.connect_grpc(url=f"rerun+http://{ip}:{port}/proxy")
+    else:
+        rr.spawn(memory_limit=memory_limit)
+
+
+def _is_scalar(x):
+    return isinstance(x, (float | numbers.Real | np.integer | np.floating)) or (
+        isinstance(x, np.ndarray) and x.ndim == 0
+    )
+
+
+def log_rerun_data(
+    observation: RobotObservation | None = None,
+    action: RobotAction | None = None,
+    compress_images: bool = False,
+) -> None:
+    """
+    Logs observation and action data to Rerun for real-time visualization.
+
+    This function iterates through the provided observation and action dictionaries and sends their contents
+    to the Rerun viewer. It handles different data types appropriately:
+    - Scalars values (floats, ints) are logged as `rr.Scalars`.
+    - 3D NumPy arrays that resemble images (e.g., with 1, 3, or 4 channels first) are transposed
+      from CHW to HWC format, (optionally) compressed to JPEG and logged as `rr.Image` or `rr.EncodedImage`.
+    - 1D NumPy arrays are logged as a series of individual scalars, with each element indexed.
+    - Other multi-dimensional arrays are flattened and logged as individual scalars.
+
+    Keys are automatically namespaced with "observation." or "action." if not already present.
+
+    Args:
+        observation: An optional dictionary containing observation data to log.
+        action: An optional dictionary containing action data to log.
+        compress_images: Whether to compress images before logging to save bandwidth & memory in exchange for cpu and quality.
+    """
+    if observation:
+        for k, v in observation.items():
+            if v is None:
+                continue
+            key = k if str(k).startswith(OBS_PREFIX) else f"{OBS_STR}.{k}"
+
+            if _is_scalar(v):
+                rr.log(key, rr.Scalars(float(v)))
+            elif isinstance(v, np.ndarray):
+                arr = v
+                # Convert CHW -> HWC when needed
+                if arr.ndim == 3 and arr.shape[0] in (1, 3, 4) and arr.shape[-1] not in (1, 3, 4):
+                    arr = np.transpose(arr, (1, 2, 0))
+                if arr.ndim == 1:
+                    for i, vi in enumerate(arr):
+                        rr.log(f"{key}_{i}", rr.Scalars(float(vi)))
+                else:
+                    img_entity = rr.Image(arr).compress() if compress_images else rr.Image(arr)
+                    rr.log(key, entity=img_entity, static=True)
+
+    if action:
+        for k, v in action.items():
+            if v is None:
+                continue
+            key = k if str(k).startswith(ACTION_PREFIX) else f"{ACTION}.{k}"
+
+            if _is_scalar(v):
+                rr.log(key, rr.Scalars(float(v)))
+            elif isinstance(v, np.ndarray):
+                if v.ndim == 1:
+                    for i, vi in enumerate(v):
+                        rr.log(f"{key}_{i}", rr.Scalars(float(vi)))
+                else:
+                    # Fall back to flattening higher-dimensional arrays
+                    flat = v.flatten()
+                    for i, vi in enumerate(flat):
+                        rr.log(f"{key}_{i}", rr.Scalars(float(vi)))
diff --git a/lerobot/tests/__init__.py b/lerobot/tests/__init__.py
new file mode 100644
index 0000000000000000000000000000000000000000..f52df1bd7aa0987f546da171e1b7e648596540ba
--- /dev/null
+++ b/lerobot/tests/__init__.py
@@ -0,0 +1,13 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
diff --git a/lerobot/tests/artifacts/cameras/image_128x128.png b/lerobot/tests/artifacts/cameras/image_128x128.png
new file mode 100644
index 0000000000000000000000000000000000000000..b117f49f27bda4c40dcb02e456389bdbbd19aeaa
--- /dev/null
+++ b/lerobot/tests/artifacts/cameras/image_128x128.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9dc9df05797dc0e7b92edc845caab2e4c37c3cfcabb4ee6339c67212b5baba3b
+size 38023
diff --git a/lerobot/tests/artifacts/cameras/image_160x120.png b/lerobot/tests/artifacts/cameras/image_160x120.png
new file mode 100644
index 0000000000000000000000000000000000000000..cdc681d183e8273746b56ae58276bcb03abb48aa
--- /dev/null
+++ b/lerobot/tests/artifacts/cameras/image_160x120.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7e11af87616b83c1cdb30330e951b91e86b51c64a1326e1ba5b4a3fbcdec1a11
+size 55698
diff --git a/lerobot/tests/artifacts/cameras/image_320x180.png b/lerobot/tests/artifacts/cameras/image_320x180.png
new file mode 100644
index 0000000000000000000000000000000000000000..4cfd511a71e49ded7af38b77d61ea75123134abc
--- /dev/null
+++ b/lerobot/tests/artifacts/cameras/image_320x180.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b8840fb643afe903191248703b1f95a57faf5812ecd9978ac502ee939646fdb2
+size 121115
diff --git a/lerobot/tests/artifacts/cameras/image_480x270.png b/lerobot/tests/artifacts/cameras/image_480x270.png
new file mode 100644
index 0000000000000000000000000000000000000000..b564d5424cf504f5adb0c5d2edf8780e36a4e7e2
--- /dev/null
+++ b/lerobot/tests/artifacts/cameras/image_480x270.png
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f79d14daafb1c0cf2fec5d46ee8029a73fe357402fdd31a7cd4a4794d7319a7c
+size 260367
diff --git a/lerobot/tests/artifacts/cameras/test_rs.bag b/lerobot/tests/artifacts/cameras/test_rs.bag
new file mode 100644
index 0000000000000000000000000000000000000000..1b9662c35603734edc202d1714b71ca50e89fa59
--- /dev/null
+++ b/lerobot/tests/artifacts/cameras/test_rs.bag
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a8d6e64d6cb0e02c94ae125630ee758055bd2e695772c0463a30d63ddc6c5e17
+size 3520862
diff --git a/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_0.safetensors b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_0.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..1b1994ccc0358b6b15efd623b65ddd2392c74af2
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_0.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6bdf22208d49cd36d24bc844d4d8bda5e321eafe39d2b470e4fc95c7812fdb24
+size 3687117
diff --git a/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_1.safetensors b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_1.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..a36663bf4e26c371fa1fc97f983a06602d7fb90d
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_1.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8920d5ebab36ffcba9aa74dcd91677c121f504b4d945b472352d379f9272fabf
+size 3687117
diff --git a/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_250.safetensors b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_250.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..b6e6e0e83a42f9c46b5b014477951d49dd580c8e
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_250.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:35723f2db499da3d9d121aa79d2ff4c748effd7c2ea92f277ec543a82fb843ca
+size 3687117
diff --git a/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_251.safetensors b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_251.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..ca750b909e7561aaa65c76a62ba2b766aa1322b9
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_251.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:53172b773d4a78bb3140f10280105c2c4ebcb467f3097579988d42cb87790ab9
+size 3687117
diff --git a/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_498.safetensors b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_498.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..9eb2e149d987c4333d4a8504bc88f42cae9b0d04
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_498.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:58a5d91573e7dd2352a1454a5c9118c9ad3798428a0104e5e0b57fc01f780ae7
+size 3687117
diff --git a/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_499.safetensors b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_499.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..849c44bc66b778bb388cfc576e6da233d01b5367
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/aloha_sim_insertion_human/frame_499.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:bb65a25e989a32a8b6258d368bd077e4548379c74ab5ada01cc532d658670df0
+size 3687117
diff --git a/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_0.safetensors b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_0.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..0a7ced50507c373f1150e63504217beeab2d9501
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_0.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c3dcff0a705ebfdaf11b7f49ad85b464eff03477ace3d63ce45d6a3a10b429d5
+size 111338
diff --git a/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_1.safetensors b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_1.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..f999e25e8a63d7dc3eecf7c206298b4a5b13740b
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_1.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d8ab0274761cdd758bafdf274ce3e6398cd6f0df23393971f3e1b6b465d66ef3
+size 111338
diff --git a/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_159.safetensors b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_159.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..f49a88471579376212210b4237548be40d3c6fd4
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_159.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:aee60956925da9687546aafa770d5e6a04f99576f903b08d0bd5f8003a7f4f3e
+size 111338
diff --git a/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_160.safetensors b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_160.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..dee72c6e6804be61af69b53060825d6432f313bd
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_160.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c8d9f9cc9e232820760fe4a46b47000c921fa5d868420e55d8dbc05dae56e8bd
+size 111338
diff --git a/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_80.safetensors b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_80.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..9189c4d47d152d628f99aba86fc3e37a8c56a290
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_80.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:01cfe50c537e3aef0cd5947ec0b15b321b54ecb461baf7b4f2506897158eebc8
+size 111338
diff --git a/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_81.safetensors b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_81.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..2537af3130216aaab8b6e60a0e7a9c43cf55ed6a
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/pusht/frame_81.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:96431ca3479eef2379406ef901cad7ba5eac4f7edcc48ecc9e8d1fa0e99d8017
+size 111338
diff --git a/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_0.safetensors b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_0.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..00db26a67eebaaa0d6ecf64cb5a1425d99c19004
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_0.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3763d7bff7873cb40ea9d6f2f98d45fcf163addcd2809b6c59f273b6c3627ad5
+size 85353
diff --git a/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_1.safetensors b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_1.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..6f4b0c0d3fc11fd0119a20a3318c3fd28f8ac176
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_1.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:24150994c6959631dc081b43e4001a8664e13b194ac194a32100f7d3fd2c0d0f
+size 85353
diff --git a/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_12.safetensors b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_12.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..fa42365b86db01f8868425f6801579b546da35c0
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_12.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c9c3fdf34debe47d4b80570a19e676185449df749f37daa2111184c1f439ae5f
+size 85353
diff --git a/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_13.safetensors b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_13.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..c010a484e10d201597e84e9ec4a0c8c4c5866cf0
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_13.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f8cfbe444c14d643da2faea9f6a402ddb37114ab15395c381f1a7982e541f868
+size 85353
diff --git a/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_23.safetensors b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_23.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..056f9f15b95a20c9f6374e6a21f701479b182a1b
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_23.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:07c5c1a63998884ee747a6d0aa8f49217da3c32af2760dad2a9da794d3517003
+size 85353
diff --git a/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_24.safetensors b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_24.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..41a384d811d67d4e3f25b1458c5e891a4823a586
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/lerobot/xarm_lift_medium/frame_24.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9927ec508e3335f8b10cf3682e41dedb7e647f92a2063a4196f1e48749c47bc5
+size 85353
diff --git a/lerobot/tests/artifacts/datasets/save_dataset_to_safetensors.py b/lerobot/tests/artifacts/datasets/save_dataset_to_safetensors.py
new file mode 100644
index 0000000000000000000000000000000000000000..3df42f35cd27a46d7272b159ff8a9d1a1fc0c279
--- /dev/null
+++ b/lerobot/tests/artifacts/datasets/save_dataset_to_safetensors.py
@@ -0,0 +1,75 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""
+This script provides a utility for saving a dataset as safetensors files for the purpose of testing backward compatibility
+when updating the data format. It uses the `PushtDataset` to create a DataLoader and saves selected frame from the
+dataset into a corresponding safetensors file in a specified output directory.
+
+If you know that your change will break backward compatibility, you should write a shortlived test by modifying
+`tests/test_datasets.py::test_backward_compatibility` accordingly, and make sure this custom test pass. Your custom test
+doesnt need to be merged into the `main` branch. Then you need to run this script and update the tests artifacts.
+
+Example usage:
+    `python tests/artifacts/datasets/save_dataset_to_safetensors.py`
+"""
+
+import shutil
+from pathlib import Path
+
+from safetensors.torch import save_file
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+
+def save_dataset_to_safetensors(output_dir, repo_id="lerobot/pusht"):
+    repo_dir = Path(output_dir) / repo_id
+
+    if repo_dir.exists():
+        shutil.rmtree(repo_dir)
+
+    repo_dir.mkdir(parents=True, exist_ok=True)
+    dataset = LeRobotDataset(
+        repo_id=repo_id,
+        episodes=[0],
+    )
+
+    # save 2 first frames of first episode
+    i = dataset.meta.episodes["dataset_from_index"][0]
+    save_file(dataset[i], repo_dir / f"frame_{i}.safetensors")
+    save_file(dataset[i + 1], repo_dir / f"frame_{i + 1}.safetensors")
+
+    # save 2 frames at the middle of first episode
+    i = int(
+        (dataset.meta.episodes["dataset_to_index"][0] - dataset.meta.episodes["dataset_from_index"][0]) / 2
+    )
+    save_file(dataset[i], repo_dir / f"frame_{i}.safetensors")
+    save_file(dataset[i + 1], repo_dir / f"frame_{i + 1}.safetensors")
+
+    # save 2 last frames of first episode
+    i = dataset.meta.episodes["dataset_to_index"][0]
+    save_file(dataset[i - 2], repo_dir / f"frame_{i - 2}.safetensors")
+    save_file(dataset[i - 1], repo_dir / f"frame_{i - 1}.safetensors")
+
+
+if __name__ == "__main__":
+    for dataset in [
+        "lerobot/pusht",
+        "lerobot/aloha_sim_insertion_human",
+        "lerobot/xarm_lift_medium",
+        "lerobot/nyu_franka_play_dataset",
+        "lerobot/cmu_stretch",
+    ]:
+        save_dataset_to_safetensors("tests/artifacts/datasets", repo_id=dataset)
diff --git a/lerobot/tests/artifacts/image_transforms/default_transforms.safetensors b/lerobot/tests/artifacts/image_transforms/default_transforms.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..2c08499f60d37b891096afd5495e5f0d201ac984
--- /dev/null
+++ b/lerobot/tests/artifacts/image_transforms/default_transforms.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6b1e600768a8771c5fe650e038a1193597e3810f032041b2a0d021e4496381c1
+size 3686488
diff --git a/lerobot/tests/artifacts/image_transforms/save_image_transforms_to_safetensors.py b/lerobot/tests/artifacts/image_transforms/save_image_transforms_to_safetensors.py
new file mode 100644
index 0000000000000000000000000000000000000000..ce15d16fdc0d526a48ce304148a8215b41162cf2
--- /dev/null
+++ b/lerobot/tests/artifacts/image_transforms/save_image_transforms_to_safetensors.py
@@ -0,0 +1,75 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from pathlib import Path
+
+import torch
+from safetensors.torch import save_file
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.transforms import (
+    ImageTransformConfig,
+    ImageTransforms,
+    ImageTransformsConfig,
+    make_transform_from_config,
+)
+from lerobot.utils.random_utils import seeded_context
+
+ARTIFACT_DIR = Path("tests/artifacts/image_transforms")
+DATASET_REPO_ID = "lerobot/aloha_static_cups_open"
+
+
+def save_default_config_transform(original_frame: torch.Tensor, output_dir: Path):
+    cfg = ImageTransformsConfig(enable=True)
+    default_tf = ImageTransforms(cfg)
+
+    with seeded_context(1337):
+        img_tf = default_tf(original_frame)
+
+    save_file({"default": img_tf}, output_dir / "default_transforms.safetensors")
+
+
+def save_single_transforms(original_frame: torch.Tensor, output_dir: Path):
+    transforms = {
+        ("ColorJitter", "brightness", [(0.5, 0.5), (2.0, 2.0)]),
+        ("ColorJitter", "contrast", [(0.5, 0.5), (2.0, 2.0)]),
+        ("ColorJitter", "saturation", [(0.5, 0.5), (2.0, 2.0)]),
+        ("ColorJitter", "hue", [(-0.25, -0.25), (0.25, 0.25)]),
+        ("SharpnessJitter", "sharpness", [(0.5, 0.5), (2.0, 2.0)]),
+    }
+
+    frames = {"original_frame": original_frame}
+    for tf_type, tf_name, min_max_values in transforms.items():
+        for min_max in min_max_values:
+            tf_cfg = ImageTransformConfig(type=tf_type, kwargs={tf_name: min_max})
+            tf = make_transform_from_config(tf_cfg)
+            key = f"{tf_name}_{min_max[0]}_{min_max[1]}"
+            frames[key] = tf(original_frame)
+
+    save_file(frames, output_dir / "single_transforms.safetensors")
+
+
+def main():
+    dataset = LeRobotDataset(DATASET_REPO_ID, episodes=[0], image_transforms=None)
+    output_dir = Path(ARTIFACT_DIR)
+    output_dir.mkdir(parents=True, exist_ok=True)
+    original_frame = dataset[0][dataset.meta.camera_keys[0]]
+
+    save_single_transforms(original_frame, output_dir)
+    save_default_config_transform(original_frame, output_dir)
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot/tests/artifacts/image_transforms/single_transforms.safetensors b/lerobot/tests/artifacts/image_transforms/single_transforms.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..7a0599d9ea5b18990d4679554e15d22b63473873
--- /dev/null
+++ b/lerobot/tests/artifacts/image_transforms/single_transforms.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9d4ebab73eabddc58879a4e770289d19e00a1a4cf2fa5fa33cd3a3246992bc90
+size 40551392
diff --git a/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/actions.safetensors b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/actions.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..771af244500f04ff795ec7f00b031e850de9e7cd
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/actions.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ee0c29d3782aa1cadcf4dc6ed767d9460ff00fff9fc70b460502340b832eefcc
+size 5104
diff --git a/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/grad_stats.safetensors b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/grad_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..5209ae6abc2bfdfc2dfc00993c6b33bae5855fdc
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/grad_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1a7a8b1a457149109f843c32bcbb047d09de2201847b9b79f7501b447f77ecf4
+size 31672
diff --git a/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/output_dict.safetensors b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/output_dict.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..736aff94f50bf813190e540d061333fabf918117
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/output_dict.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5e6ce85296b2009e7c2060d336c0429b1c7197d9adb159e7df0ba18003067b36
+size 68
diff --git a/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/param_stats.safetensors b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/param_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..3e8df708e7742907ae9691f39d406290c9a0999f
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_/param_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ea76e6711959fd3f905ec2bdc306f488920f00ec99421e4870d05f6205eb323e
+size 31672
diff --git a/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/actions.safetensors b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/actions.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..dd7d4d0e773446b5cd9ad4cb4eeef2f08becd257
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/actions.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c2b8f8532c7a0b776de5e536b8b54e30b1a0c2e3d5cc25a2d86fe43e40ae5e8c
+size 515400
diff --git a/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/grad_stats.safetensors b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/grad_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..c58bb44bc2902edd2be05516cedf8d080a0609a9
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/grad_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:224b5fa4828aa88171b68c036e8919c1eae563e2113f03b6461eadf5bf8525a6
+size 31672
diff --git a/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/output_dict.safetensors b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/output_dict.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..9b6ef7f5dafec64df7a9a7428f9a775a56a8b90e
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/output_dict.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:016d2fa8fe5f58017dfd46f4632fdc19dfd751e32a2c7cde2077c6f95546d6bd
+size 68
diff --git a/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/param_stats.safetensors b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/param_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..5da67a1af090aa8d91bea03e0b4ff33ab7f06e5e
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/aloha_sim_insertion_human_act_1000_steps/param_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:eca0d87a699620e4fec7e68539b0be91e4cc933f6bf12032da52c182ab6f38cf
+size 31672
diff --git a/lerobot/tests/artifacts/policies/pusht_diffusion_/actions.safetensors b/lerobot/tests/artifacts/policies/pusht_diffusion_/actions.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..70b1411ab1ad033a69f0fd4680a7c8a3136f931f
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/pusht_diffusion_/actions.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:54aecbc1af72a4cd5e9261492f5e7601890517516257aacdf2a0ffb3ce281f1b
+size 992
diff --git a/lerobot/tests/artifacts/policies/pusht_diffusion_/grad_stats.safetensors b/lerobot/tests/artifacts/policies/pusht_diffusion_/grad_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..bea7d4f19fb2304824b3da46875b5a279b7de7aa
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/pusht_diffusion_/grad_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:88a9c3775a2aa1e90a08850521970070a4fcf0f6b82aab43cd8ccc5cf77e0013
+size 47424
diff --git a/lerobot/tests/artifacts/policies/pusht_diffusion_/output_dict.safetensors b/lerobot/tests/artifacts/policies/pusht_diffusion_/output_dict.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..20cc4f547b4386004a0012d142981513f6574dab
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/pusht_diffusion_/output_dict.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:91a2635e05a75fe187a5081504c5f35ce3417378813fa2deaf9ca4e8200e1819
+size 68
diff --git a/lerobot/tests/artifacts/policies/pusht_diffusion_/param_stats.safetensors b/lerobot/tests/artifacts/policies/pusht_diffusion_/param_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..365a453ddbc67eba5487096de6f986a3f3748414
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/pusht_diffusion_/param_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:645bff922ac7bea63ad018ebf77c303c0e4cd2c1c0dc5ef3192865281bef3dc6
+size 47424
diff --git a/lerobot/tests/artifacts/policies/save_policy_to_safetensors.py b/lerobot/tests/artifacts/policies/save_policy_to_safetensors.py
new file mode 100644
index 0000000000000000000000000000000000000000..64b125cc92f51d4110dbc5df900d13e77f1f9b3a
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/save_policy_to_safetensors.py
@@ -0,0 +1,153 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import shutil
+from pathlib import Path
+
+import torch
+from safetensors.torch import save_file
+
+from lerobot.configs.default import DatasetConfig
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.datasets.factory import make_dataset
+from lerobot.optim.factory import make_optimizer_and_scheduler
+from lerobot.policies.factory import make_policy, make_policy_config, make_pre_post_processors
+from lerobot.utils.constants import OBS_STR
+from lerobot.utils.random_utils import set_seed
+
+
+def get_policy_stats(ds_repo_id: str, policy_name: str, policy_kwargs: dict):
+    set_seed(1337)
+    train_cfg = TrainPipelineConfig(
+        # TODO(rcadene, aliberts): remove dataset download
+        dataset=DatasetConfig(repo_id=ds_repo_id, episodes=[0]),
+        policy=make_policy_config(policy_name, push_to_hub=False, **policy_kwargs),
+    )
+    train_cfg.validate()  # Needed for auto-setting some parameters
+
+    dataset = make_dataset(train_cfg)
+    dataset_stats = dataset.meta.stats
+    policy = make_policy(train_cfg.policy, ds_meta=dataset.meta)
+    preprocessor, postprocessor = make_pre_post_processors(train_cfg.policy, dataset_stats=dataset_stats)
+    policy.train()
+
+    optimizer, _ = make_optimizer_and_scheduler(train_cfg, policy)
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        num_workers=0,
+        batch_size=train_cfg.batch_size,
+        shuffle=False,
+    )
+
+    batch = next(iter(dataloader))
+    batch = preprocessor(batch)
+    loss, output_dict = policy.forward(batch)
+
+    if output_dict is not None:
+        output_dict = {k: v for k, v in output_dict.items() if isinstance(v, torch.Tensor)}
+        output_dict["loss"] = loss
+    else:
+        output_dict = {"loss": loss}
+
+    loss.backward()
+    grad_stats = {}
+    for key, param in policy.named_parameters():
+        if param.requires_grad:
+            grad_stats[f"{key}_mean"] = param.grad.mean()
+            grad_stats[f"{key}_std"] = param.grad.std() if param.grad.numel() > 1 else torch.tensor(0.0)
+
+    optimizer.step()
+    param_stats = {}
+    for key, param in policy.named_parameters():
+        param_stats[f"{key}_mean"] = param.mean()
+        param_stats[f"{key}_std"] = param.std() if param.numel() > 1 else torch.tensor(0.0)
+
+    optimizer.zero_grad()
+    policy.reset()
+
+    # HACK: We reload a batch with no delta_indices as `select_action` won't expect a timestamps dimension
+    # We simulate having an environment using a dataset by setting delta_indices to None and dropping tensors
+    # indicating padding (those ending with "_is_pad")
+    dataset.delta_indices = None
+    batch = next(iter(dataloader))
+    obs = {}
+    for k in batch:
+        # TODO: regenerate the safetensors
+        # for backward compatibility
+        if k.endswith("_is_pad"):
+            continue
+        # for backward compatibility
+        if k == "task":
+            continue
+        if k.startswith(OBS_STR):
+            obs[k] = batch[k]
+
+    if hasattr(train_cfg.policy, "n_action_steps"):
+        actions_queue = train_cfg.policy.n_action_steps
+    else:
+        actions_queue = train_cfg.policy.n_action_repeats
+
+    actions = {}
+    for i in range(actions_queue):
+        unnormalized_action = policy.select_action(obs).contiguous()
+        action_robot = postprocessor(unnormalized_action)
+        actions[str(i)] = action_robot
+
+    return output_dict, grad_stats, param_stats, actions
+
+
+def save_policy_to_safetensors(output_dir: Path, ds_repo_id: str, policy_name: str, policy_kwargs: dict):
+    if output_dir.exists():
+        print(f"Overwrite existing safetensors in '{output_dir}':")
+        print(f" - Validate with: `git add {output_dir}`")
+        print(f" - Revert with: `git checkout -- {output_dir}`")
+        shutil.rmtree(output_dir)
+
+    output_dir.mkdir(parents=True, exist_ok=True)
+    output_dict, grad_stats, param_stats, actions = get_policy_stats(ds_repo_id, policy_name, policy_kwargs)
+    save_file(output_dict, output_dir / "output_dict.safetensors")
+    save_file(grad_stats, output_dir / "grad_stats.safetensors")
+    save_file(param_stats, output_dir / "param_stats.safetensors")
+    save_file(actions, output_dir / "actions.safetensors")
+
+
+if __name__ == "__main__":
+    artifacts_cfg = [
+        ("lerobot/xarm_lift_medium", "tdmpc", {"use_mpc": False}, "use_policy"),
+        ("lerobot/xarm_lift_medium", "tdmpc", {"use_mpc": True}, "use_mpc"),
+        (
+            "lerobot/pusht",
+            "diffusion",
+            {
+                "n_action_steps": 8,
+                "num_inference_steps": 10,
+                "down_dims": [128, 256, 512],
+            },
+            "",
+        ),
+        ("lerobot/aloha_sim_insertion_human", "act", {"n_action_steps": 10}, ""),
+        (
+            "lerobot/aloha_sim_insertion_human",
+            "act",
+            {"n_action_steps": 1000, "chunk_size": 1000},
+            "1000_steps",
+        ),
+    ]
+    if len(artifacts_cfg) == 0:
+        raise RuntimeError("No policies were provided!")
+    for ds_repo_id, policy, policy_kwargs, file_name_extra in artifacts_cfg:
+        ds_name = ds_repo_id.split("/")[-1]
+        output_dir = Path("tests/artifacts/policies") / f"{ds_name}_{policy}_{file_name_extra}"
+        save_policy_to_safetensors(output_dir, ds_repo_id, policy, policy_kwargs)
diff --git a/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/actions.safetensors b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/actions.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..e23eacffdfb2839f062322243b1871b18233a00b
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/actions.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d640988f2269cf6aa03c8ee17f9d096edace83d837f90025011fafec5bf53c61
+size 200
diff --git a/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/grad_stats.safetensors b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/grad_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..e665f73c67d35874b02ad66fa49871f814af8056
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/grad_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:32ddf36af25791935b395c7641531cda14d5c4a2cf654a2e76ac45271665d07a
+size 16904
diff --git a/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/output_dict.safetensors b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/output_dict.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..97d783580a2a37c579fe044dcaacd26dea152d20
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/output_dict.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:22a1031a2acfc36a455bff73ffbe097cfeb7742b6485e7422507e78d7a682703
+size 164
diff --git a/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/param_stats.safetensors b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/param_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..3090b70511a3ab6d938dff86c70ec08f50938f0d
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_mpc/param_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b5dca7940998421ae58e9e26b2b2641b058d23b0270b7a147ebf85fbbdce7184
+size 35496
diff --git a/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/actions.safetensors b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/actions.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..5ce44048f59f0f7367184a830092a8d33b0ee3fd
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/actions.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2212ae7b910d14d723214f5af50985e419f7bd0f4261565ef48b1ef495443d6d
+size 200
diff --git a/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/grad_stats.safetensors b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/grad_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..e665f73c67d35874b02ad66fa49871f814af8056
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/grad_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:32ddf36af25791935b395c7641531cda14d5c4a2cf654a2e76ac45271665d07a
+size 16904
diff --git a/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/output_dict.safetensors b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/output_dict.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..97d783580a2a37c579fe044dcaacd26dea152d20
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/output_dict.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:22a1031a2acfc36a455bff73ffbe097cfeb7742b6485e7422507e78d7a682703
+size 164
diff --git a/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/param_stats.safetensors b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/param_stats.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..3090b70511a3ab6d938dff86c70ec08f50938f0d
--- /dev/null
+++ b/lerobot/tests/artifacts/policies/xarm_lift_medium_tdmpc_use_policy/param_stats.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b5dca7940998421ae58e9e26b2b2641b058d23b0270b7a147ebf85fbbdce7184
+size 35496
diff --git a/lerobot/tests/async_inference/test_e2e.py b/lerobot/tests/async_inference/test_e2e.py
new file mode 100644
index 0000000000000000000000000000000000000000..54ca29b484861d82ac5d6f74891e9c4e9fca6460
--- /dev/null
+++ b/lerobot/tests/async_inference/test_e2e.py
@@ -0,0 +1,185 @@
+# Copyright 2025 The HuggingFace Inc. team.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""End-to-end test of the asynchronous inference stack (client  ↔  server).
+
+This test spins up a lightweight gRPC `PolicyServer` instance with a stubbed
+policy network and launches a `RobotClient` that uses a `MockRobot`.  The goal
+is to exercise the full communication loop:
+
+1. Client sends policy specification → Server
+2. Client streams observations → Server
+3. Server streams action chunks → Client
+4. Client executes received actions
+
+The test succeeds if at least one action is executed and the server records at
+least one predicted timestep - demonstrating that the gRPC round-trip works
+end-to-end using real (but lightweight) protocol messages.
+"""
+
+from __future__ import annotations
+
+import threading
+from concurrent import futures
+
+import pytest
+import torch
+
+# Skip entire module if grpc is not available
+pytest.importorskip("grpc")
+
+# -----------------------------------------------------------------------------
+# End-to-end test
+# -----------------------------------------------------------------------------
+
+
+def test_async_inference_e2e(monkeypatch):
+    """Tests the full asynchronous inference pipeline."""
+    # Import grpc-dependent modules inside the test function
+    import grpc
+
+    from lerobot.async_inference.configs import PolicyServerConfig, RobotClientConfig
+    from lerobot.async_inference.helpers import map_robot_keys_to_lerobot_features
+    from lerobot.async_inference.policy_server import PolicyServer
+    from lerobot.async_inference.robot_client import RobotClient
+    from lerobot.robots.utils import make_robot_from_config
+    from lerobot.transport import (
+        services_pb2,  # type: ignore
+        services_pb2_grpc,  # type: ignore
+    )
+    from tests.mocks.mock_robot import MockRobotConfig
+
+    # Create a stub policy similar to test_policy_server.py
+    class MockPolicy:
+        """A minimal mock for an actual policy, returning zeros."""
+
+        class _Config:
+            robot_type = "dummy_robot"
+
+            @property
+            def image_features(self):
+                """Empty image features since this test doesn't use images."""
+                return {}
+
+        def __init__(self):
+            self.config = self._Config()
+
+        def to(self, *args, **kwargs):
+            return self
+
+        def model(self, batch):
+            # Return a chunk of 20 dummy actions.
+            batch_size = len(batch["robot_type"])
+            return torch.zeros(batch_size, 20, 6)
+
+    # ------------------------------------------------------------------
+    # 1. Create PolicyServer instance with mock policy
+    # ------------------------------------------------------------------
+    policy_server_config = PolicyServerConfig(host="localhost", port=9999)
+    policy_server = PolicyServer(policy_server_config)
+    # Replace the real policy with our fast, deterministic stub.
+    policy_server.policy = MockPolicy()
+    policy_server.actions_per_chunk = 20
+    policy_server.device = "cpu"
+    # NOTE(Steven): Smelly tests as the Server is a state machine being partially mocked. Adding these processors as a quick fix.
+    policy_server.preprocessor = lambda obs: obs
+    policy_server.postprocessor = lambda tensor: tensor
+
+    # Set up robot config and features
+    robot_config = MockRobotConfig()
+    mock_robot = make_robot_from_config(robot_config)
+
+    lerobot_features = map_robot_keys_to_lerobot_features(mock_robot)
+    policy_server.lerobot_features = lerobot_features
+
+    # Force server to produce deterministic action chunks in test mode
+    policy_server.policy_type = "act"
+
+    def _fake_get_action_chunk(_self, _obs, _type="test"):
+        action_dim = 6
+        batch_size = 1
+        actions_per_chunk = policy_server.actions_per_chunk
+
+        return torch.zeros(batch_size, actions_per_chunk, action_dim)
+
+    monkeypatch.setattr(PolicyServer, "_get_action_chunk", _fake_get_action_chunk, raising=True)
+
+    # Bypass potentially heavy model loading inside SendPolicyInstructions
+    def _fake_send_policy_instructions(self, request, context):  # noqa: N802
+        return services_pb2.Empty()
+
+    monkeypatch.setattr(PolicyServer, "SendPolicyInstructions", _fake_send_policy_instructions, raising=True)
+
+    # Build gRPC server running a PolicyServer
+    server = grpc.server(futures.ThreadPoolExecutor(max_workers=1, thread_name_prefix="policy_server"))
+    services_pb2_grpc.add_AsyncInferenceServicer_to_server(policy_server, server)
+
+    # Use the host/port specified in the fixture's config
+    server_address = f"{policy_server.config.host}:{policy_server.config.port}"
+    server.add_insecure_port(server_address)
+    server.start()
+
+    # ------------------------------------------------------------------
+    # 2. Create a RobotClient around the MockRobot
+    # ------------------------------------------------------------------
+    client_config = RobotClientConfig(
+        server_address=server_address,
+        robot=robot_config,
+        chunk_size_threshold=0.0,
+        policy_type="test",
+        pretrained_name_or_path="test",
+        actions_per_chunk=20,
+    )
+
+    client = RobotClient(client_config)
+    assert client.start(), "Client failed initial handshake with the server"
+
+    # Track action chunks received and verify device type
+    action_chunks_received = {"count": 0, "actions_on_cpu": True}
+    original_aggregate = client._aggregate_action_queues
+
+    def counting_aggregate(*args, **kwargs):
+        action_chunks_received["count"] += 1
+        # Check that all received actions are on CPU
+        if args:
+            for timed_action in args[0]:  # args[0] is the list of TimedAction
+                action_tensor = timed_action.get_action()
+                if action_tensor.device.type != "cpu":
+                    action_chunks_received["actions_on_cpu"] = False
+        return original_aggregate(*args, **kwargs)
+
+    monkeypatch.setattr(client, "_aggregate_action_queues", counting_aggregate)
+
+    # Start client threads
+    action_thread = threading.Thread(target=client.receive_actions, daemon=True)
+    control_thread = threading.Thread(target=client.control_loop, args=({"task": ""}), daemon=True)
+    action_thread.start()
+    control_thread.start()
+
+    # ------------------------------------------------------------------
+    # 3. System exchanges a few messages
+    # ------------------------------------------------------------------
+    # Wait for 5 seconds
+    server.wait_for_termination(timeout=5)
+
+    assert action_chunks_received["count"] > 0, "Client did not receive any action chunks"
+    assert len(policy_server._predicted_timesteps) > 0, "Server did not record any predicted timesteps"
+
+    # ------------------------------------------------------------------
+    # 4. Stop the system
+    # ------------------------------------------------------------------
+    client.stop()
+    action_thread.join()
+    control_thread.join()
+    policy_server.stop()
+    server.stop(grace=None)
diff --git a/lerobot/tests/async_inference/test_helpers.py b/lerobot/tests/async_inference/test_helpers.py
new file mode 100644
index 0000000000000000000000000000000000000000..a9e53200d0694f676c9dc0122e26a7ebdbfe4c79
--- /dev/null
+++ b/lerobot/tests/async_inference/test_helpers.py
@@ -0,0 +1,450 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import math
+import pickle
+import time
+
+import numpy as np
+import torch
+
+from lerobot.async_inference.helpers import (
+    FPSTracker,
+    TimedAction,
+    TimedObservation,
+    observations_similar,
+    prepare_image,
+    prepare_raw_observation,
+    raw_observation_to_observation,
+    resize_robot_observation_image,
+)
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.utils.constants import OBS_IMAGES, OBS_STATE
+
+# ---------------------------------------------------------------------
+# FPSTracker
+# ---------------------------------------------------------------------
+
+
+def test_fps_tracker_first_observation():
+    """First observation should initialize timestamp and return 0 FPS."""
+    tracker = FPSTracker(target_fps=30.0)
+    timestamp = 1000.0
+
+    metrics = tracker.calculate_fps_metrics(timestamp)
+
+    assert tracker.first_timestamp == timestamp
+    assert tracker.total_obs_count == 1
+    assert metrics["avg_fps"] == 0.0
+    assert metrics["target_fps"] == 30.0
+
+
+def test_fps_tracker_single_interval():
+    """Two observations 1 second apart should give 1 FPS."""
+    tracker = FPSTracker(target_fps=30.0)
+
+    # First observation at t=0
+    metrics1 = tracker.calculate_fps_metrics(0.0)
+    assert metrics1["avg_fps"] == 0.0
+
+    # Second observation at t=1 (1 second later)
+    metrics2 = tracker.calculate_fps_metrics(1.0)
+    expected_fps = 1.0  # (2-1) observations / 1.0 seconds = 1 FPS
+    assert math.isclose(metrics2["avg_fps"], expected_fps, rel_tol=1e-6)
+
+
+def test_fps_tracker_multiple_intervals():
+    """Multiple observations should calculate correct average FPS."""
+    tracker = FPSTracker(target_fps=30.0)
+
+    # Simulate 5 observations over 2 seconds (should be 2 FPS average)
+    timestamps = [0.0, 0.5, 1.0, 1.5, 2.0]
+
+    for i, ts in enumerate(timestamps):
+        metrics = tracker.calculate_fps_metrics(ts)
+
+        if i == 0:
+            assert metrics["avg_fps"] == 0.0
+        elif i == len(timestamps) - 1:
+            # After 5 observations over 2 seconds: (5-1)/2 = 2 FPS
+            expected_fps = 2.0
+            assert math.isclose(metrics["avg_fps"], expected_fps, rel_tol=1e-6)
+
+
+def test_fps_tracker_irregular_intervals():
+    """FPS calculation should work with irregular time intervals."""
+    tracker = FPSTracker(target_fps=30.0)
+
+    # Irregular timestamps: 0, 0.1, 0.5, 2.0, 3.0 seconds
+    timestamps = [0.0, 0.1, 0.5, 2.0, 3.0]
+
+    for ts in timestamps:
+        metrics = tracker.calculate_fps_metrics(ts)
+
+    # 5 observations over 3 seconds: (5-1)/3 = 1.333... FPS
+    expected_fps = 4.0 / 3.0
+    assert math.isclose(metrics["avg_fps"], expected_fps, rel_tol=1e-6)
+
+
+# ---------------------------------------------------------------------
+# TimedData helpers
+# ---------------------------------------------------------------------
+
+
+def test_timed_action_getters():
+    """TimedAction stores & returns timestamp, action tensor and timestep."""
+    ts = time.time()
+    action = torch.arange(10)
+    ta = TimedAction(timestamp=ts, action=action, timestep=0)
+
+    assert math.isclose(ta.get_timestamp(), ts, rel_tol=0, abs_tol=1e-6)
+    torch.testing.assert_close(ta.get_action(), action)
+    assert ta.get_timestep() == 0
+
+
+def test_timed_observation_getters():
+    """TimedObservation stores & returns timestamp, dict and timestep."""
+    ts = time.time()
+    obs_dict = {OBS_STATE: torch.ones(6)}
+    to = TimedObservation(timestamp=ts, observation=obs_dict, timestep=0)
+
+    assert math.isclose(to.get_timestamp(), ts, rel_tol=0, abs_tol=1e-6)
+    assert to.get_observation() is obs_dict
+    assert to.get_timestep() == 0
+
+
+def test_timed_data_deserialization_data_getters():
+    """TimedAction / TimedObservation survive a round-trip through ``pickle``.
+
+    The async-inference stack uses ``pickle.dumps`` to move these objects across
+    the gRPC boundary (see RobotClient.send_observation and PolicyServer.StreamActions).
+    This test ensures that the payload keeps its content intact after
+    the (de)serialization round-trip.
+    """
+    ts = time.time()
+
+    # ------------------------------------------------------------------
+    # TimedAction
+    # ------------------------------------------------------------------
+    original_action = torch.randn(6)
+    ta_in = TimedAction(timestamp=ts, action=original_action, timestep=13)
+
+    # Serialize → bytes → deserialize
+    ta_bytes = pickle.dumps(ta_in)  # nosec
+    ta_out: TimedAction = pickle.loads(ta_bytes)  # nosec B301
+
+    # Identity & content checks
+    assert math.isclose(ta_out.get_timestamp(), ts, rel_tol=0, abs_tol=1e-6)
+    assert ta_out.get_timestep() == 13
+    torch.testing.assert_close(ta_out.get_action(), original_action)
+
+    # ------------------------------------------------------------------
+    # TimedObservation
+    # ------------------------------------------------------------------
+    obs_dict = {OBS_STATE: torch.arange(4).float()}
+    to_in = TimedObservation(timestamp=ts, observation=obs_dict, timestep=7, must_go=True)
+
+    to_bytes = pickle.dumps(to_in)  # nosec
+    to_out: TimedObservation = pickle.loads(to_bytes)  # nosec B301
+
+    assert math.isclose(to_out.get_timestamp(), ts, rel_tol=0, abs_tol=1e-6)
+    assert to_out.get_timestep() == 7
+    assert to_out.must_go is True
+    assert to_out.get_observation().keys() == obs_dict.keys()
+    torch.testing.assert_close(to_out.get_observation()[OBS_STATE], obs_dict[OBS_STATE])
+
+
+# ---------------------------------------------------------------------
+# observations_similar()
+# ---------------------------------------------------------------------
+
+
+def _make_obs(state: torch.Tensor) -> TimedObservation:
+    """Create a TimedObservation with raw robot observation format."""
+    return TimedObservation(
+        timestamp=time.time(),
+        observation={
+            "shoulder": state[0].item() if len(state) > 0 else 0.0,
+            "elbow": state[1].item() if len(state) > 1 else 0.0,
+            "wrist": state[2].item() if len(state) > 2 else 0.0,
+            "gripper": state[3].item() if len(state) > 3 else 0.0,
+        },
+        timestep=0,
+    )
+
+
+def test_observations_similar_true():
+    """Distance below atol → observations considered similar."""
+    # Create mock lerobot features for the similarity check
+    lerobot_features = {
+        OBS_STATE: {
+            "dtype": "float32",
+            "shape": [4],
+            "names": ["shoulder", "elbow", "wrist", "gripper"],
+        }
+    }
+
+    obs1 = _make_obs(torch.zeros(4))
+    obs2 = _make_obs(0.5 * torch.ones(4))
+    assert observations_similar(obs1, obs2, lerobot_features, atol=2.0)
+
+    obs3 = _make_obs(2.0 * torch.ones(4))
+    assert not observations_similar(obs1, obs3, lerobot_features, atol=2.0)
+
+
+# ---------------------------------------------------------------------
+# raw_observation_to_observation and helpers
+# ---------------------------------------------------------------------
+
+
+def _create_mock_robot_observation():
+    """Create a mock robot observation with motor positions and camera images."""
+    return {
+        "shoulder": 1.0,
+        "elbow": 2.0,
+        "wrist": 3.0,
+        "gripper": 0.5,
+        "laptop": np.random.randint(0, 256, size=(480, 640, 3), dtype=np.uint8),
+        "phone": np.random.randint(0, 256, size=(480, 640, 3), dtype=np.uint8),
+    }
+
+
+def _create_mock_lerobot_features():
+    """Create mock lerobot features mapping similar to what hw_to_dataset_features returns."""
+    return {
+        OBS_STATE: {
+            "dtype": "float32",
+            "shape": [4],
+            "names": ["shoulder", "elbow", "wrist", "gripper"],
+        },
+        f"{OBS_IMAGES}.laptop": {
+            "dtype": "image",
+            "shape": [480, 640, 3],
+            "names": ["height", "width", "channels"],
+        },
+        f"{OBS_IMAGES}.phone": {
+            "dtype": "image",
+            "shape": [480, 640, 3],
+            "names": ["height", "width", "channels"],
+        },
+    }
+
+
+def _create_mock_policy_image_features():
+    """Create mock policy image features with different resolutions."""
+    return {
+        f"{OBS_IMAGES}.laptop": PolicyFeature(
+            type=FeatureType.VISUAL,
+            shape=(3, 224, 224),  # Policy expects smaller resolution
+        ),
+        f"{OBS_IMAGES}.phone": PolicyFeature(
+            type=FeatureType.VISUAL,
+            shape=(3, 160, 160),  # Different resolution for second camera
+        ),
+    }
+
+
+def test_prepare_image():
+    """Test image preprocessing: int8 → float32, normalization to [0,1]."""
+    # Create mock int8 image data
+    image_int8 = torch.randint(0, 256, size=(3, 224, 224), dtype=torch.uint8)
+
+    processed = prepare_image(image_int8)
+
+    # Check dtype conversion
+    assert processed.dtype == torch.float32
+
+    # Check normalization range
+    assert processed.min() >= 0.0
+    assert processed.max() <= 1.0
+
+    # Check that values are scaled correctly (255 → 1.0, 0 → 0.0)
+    if image_int8.max() == 255:
+        assert torch.isclose(processed.max(), torch.tensor(1.0), atol=1e-6)
+    if image_int8.min() == 0:
+        assert torch.isclose(processed.min(), torch.tensor(0.0), atol=1e-6)
+
+    # Check memory contiguity
+    assert processed.is_contiguous()
+
+
+def test_resize_robot_observation_image():
+    """Test image resizing from robot resolution to policy resolution."""
+    # Create mock image: (H=480, W=640, C=3)
+    original_image = torch.randint(0, 256, size=(480, 640, 3), dtype=torch.uint8)
+    target_shape = (3, 224, 224)  # (C, H, W)
+
+    resized = resize_robot_observation_image(original_image, target_shape)
+
+    # Check output shape matches target
+    assert resized.shape == target_shape
+
+    # Check that original image had different dimensions
+    assert original_image.shape != resized.shape
+
+    # Check that resizing preserves value range
+    assert resized.min() >= 0
+    assert resized.max() <= 255
+
+
+def test_prepare_raw_observation():
+    """Test the preparation of raw robot observation to lerobot format."""
+    robot_obs = _create_mock_robot_observation()
+    lerobot_features = _create_mock_lerobot_features()
+    policy_image_features = _create_mock_policy_image_features()
+
+    prepared = prepare_raw_observation(robot_obs, lerobot_features, policy_image_features)
+
+    # Check that state is properly extracted and batched
+    assert OBS_STATE in prepared
+    state = prepared[OBS_STATE]
+    assert isinstance(state, torch.Tensor)
+    assert state.shape == (1, 4)  # Batched state
+
+    # Check that images are processed and resized
+    assert f"{OBS_IMAGES}.laptop" in prepared
+    assert f"{OBS_IMAGES}.phone" in prepared
+
+    laptop_img = prepared[f"{OBS_IMAGES}.laptop"]
+    phone_img = prepared[f"{OBS_IMAGES}.phone"]
+
+    # Check image shapes match policy requirements
+    assert laptop_img.shape == policy_image_features[f"{OBS_IMAGES}.laptop"].shape
+    assert phone_img.shape == policy_image_features[f"{OBS_IMAGES}.phone"].shape
+
+    # Check that images are tensors
+    assert isinstance(laptop_img, torch.Tensor)
+    assert isinstance(phone_img, torch.Tensor)
+
+
+def test_raw_observation_to_observation_basic():
+    """Test the main raw_observation_to_observation function."""
+    robot_obs = _create_mock_robot_observation()
+    lerobot_features = _create_mock_lerobot_features()
+    policy_image_features = _create_mock_policy_image_features()
+
+    observation = raw_observation_to_observation(robot_obs, lerobot_features, policy_image_features)
+
+    # Check that all expected keys are present
+    assert OBS_STATE in observation
+    assert f"{OBS_IMAGES}.laptop" in observation
+    assert f"{OBS_IMAGES}.phone" in observation
+
+    # Check state processing
+    state = observation[OBS_STATE]
+    assert isinstance(state, torch.Tensor)
+    assert state.shape == (1, 4)  # Batched
+
+    # Check image processing
+    laptop_img = observation[f"{OBS_IMAGES}.laptop"]
+    phone_img = observation[f"{OBS_IMAGES}.phone"]
+
+    # Images should have batch dimension: (B, C, H, W)
+    assert laptop_img.shape == (1, 3, 224, 224)
+    assert phone_img.shape == (1, 3, 160, 160)
+
+    # Check image dtype and range (should be float32 in [0, 1])
+    assert laptop_img.dtype == torch.float32
+    assert phone_img.dtype == torch.float32
+    assert laptop_img.min() >= 0.0 and laptop_img.max() <= 1.0
+    assert phone_img.min() >= 0.0 and phone_img.max() <= 1.0
+
+
+def test_raw_observation_to_observation_with_non_tensor_data():
+    """Test that non-tensor data (like task strings) is preserved."""
+    robot_obs = _create_mock_robot_observation()
+    robot_obs["task"] = "pick up the red cube"  # Add string instruction
+
+    lerobot_features = _create_mock_lerobot_features()
+    policy_image_features = _create_mock_policy_image_features()
+
+    observation = raw_observation_to_observation(robot_obs, lerobot_features, policy_image_features)
+
+    # Check that task string is preserved
+    assert "task" in observation
+    assert observation["task"] == "pick up the red cube"
+    assert isinstance(observation["task"], str)
+
+
+@torch.no_grad()
+def test_raw_observation_to_observation_device_handling():
+    """Test that tensors are created (device placement is handled by preprocessor)."""
+    robot_obs = _create_mock_robot_observation()
+    lerobot_features = _create_mock_lerobot_features()
+    policy_image_features = _create_mock_policy_image_features()
+
+    observation = raw_observation_to_observation(robot_obs, lerobot_features, policy_image_features)
+
+    # Check that all expected keys produce tensors (device placement handled by preprocessor later)
+    for key, value in observation.items():
+        if isinstance(value, torch.Tensor):
+            assert value.device.type in ["cpu", "cuda", "mps", "xpu"], f"Tensor {key} on unexpected device"
+
+
+def test_raw_observation_to_observation_deterministic():
+    """Test that the function produces consistent results for the same input."""
+    robot_obs = _create_mock_robot_observation()
+    lerobot_features = _create_mock_lerobot_features()
+    policy_image_features = _create_mock_policy_image_features()
+
+    # Run twice with same input
+    obs1 = raw_observation_to_observation(robot_obs, lerobot_features, policy_image_features)
+    obs2 = raw_observation_to_observation(robot_obs, lerobot_features, policy_image_features)
+
+    # Results should be identical
+    assert set(obs1.keys()) == set(obs2.keys())
+
+    for key in obs1:
+        if isinstance(obs1[key], torch.Tensor):
+            torch.testing.assert_close(obs1[key], obs2[key])
+        else:
+            assert obs1[key] == obs2[key]
+
+
+def test_image_processing_pipeline_preserves_content():
+    """Test that the image processing pipeline preserves recognizable patterns."""
+    # Create an image with a specific pattern
+    original_img = np.zeros((100, 100, 3), dtype=np.uint8)
+    original_img[25:75, 25:75, :] = 255  # White square in center
+
+    robot_obs = {"shoulder": 1.0, "elbow": 1.0, "wrist": 1.0, "gripper": 1.0, "laptop": original_img}
+    lerobot_features = {
+        OBS_STATE: {
+            "dtype": "float32",
+            "shape": [4],
+            "names": ["shoulder", "elbow", "wrist", "gripper"],
+        },
+        f"{OBS_IMAGES}.laptop": {
+            "dtype": "image",
+            "shape": [100, 100, 3],
+            "names": ["height", "width", "channels"],
+        },
+    }
+    policy_image_features = {
+        f"{OBS_IMAGES}.laptop": PolicyFeature(
+            type=FeatureType.VISUAL,
+            shape=(3, 50, 50),  # Downsamples from 100x100
+        )
+    }
+
+    observation = raw_observation_to_observation(robot_obs, lerobot_features, policy_image_features)
+
+    processed_img = observation[f"{OBS_IMAGES}.laptop"].squeeze(0)  # Remove batch dim
+
+    # Check that the center region has higher values than corners
+    # Due to bilinear interpolation, exact values will change but pattern should remain
+    center_val = processed_img[:, 25, 25].mean()  # Center of 50x50 image
+    corner_val = processed_img[:, 5, 5].mean()  # Corner
+
+    assert center_val > corner_val, "Image processing should preserve recognizable patterns"
diff --git a/lerobot/tests/async_inference/test_policy_server.py b/lerobot/tests/async_inference/test_policy_server.py
new file mode 100644
index 0000000000000000000000000000000000000000..c3ee37c8fe0dbf386d1de2b40b20eb5126b466f7
--- /dev/null
+++ b/lerobot/tests/async_inference/test_policy_server.py
@@ -0,0 +1,219 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Unit-tests for the `PolicyServer` core logic.
+Monkey-patch the `policy` attribute with a stub so that no real model inference is performed.
+"""
+
+from __future__ import annotations
+
+import time
+
+import pytest
+import torch
+
+from lerobot.configs.types import PolicyFeature
+from lerobot.utils.constants import OBS_STATE
+from tests.utils import require_package
+
+# -----------------------------------------------------------------------------
+# Test fixtures
+# -----------------------------------------------------------------------------
+
+
+class MockPolicy:
+    """A minimal mock for an actual policy, returning zeros.
+    Refer to tests/policies for tests of the individual policies supported."""
+
+    class _Config:
+        robot_type = "dummy_robot"
+
+        @property
+        def image_features(self) -> dict[str, PolicyFeature]:
+            """Empty image features since this test doesn't use images."""
+            return {}
+
+    def predict_action_chunk(self, observation: dict[str, torch.Tensor]) -> torch.Tensor:
+        """Return a chunk of 20 dummy actions."""
+        batch_size = len(observation[OBS_STATE])
+        return torch.zeros(batch_size, 20, 6)
+
+    def __init__(self):
+        self.config = self._Config()
+
+    def to(self, *args, **kwargs):
+        # The server calls `policy.to(device)`. This stub ignores it.
+        return self
+
+    def model(self, batch: dict) -> torch.Tensor:
+        # Return a chunk of 20 dummy actions.
+        batch_size = len(batch["robot_type"])
+        return torch.zeros(batch_size, 20, 6)
+
+
+@pytest.fixture
+@require_package("grpcio", "grpc")
+def policy_server():
+    """Fresh `PolicyServer` instance with a stubbed-out policy model."""
+    # Import only when the test actually runs (after decorator check)
+    from lerobot.async_inference.configs import PolicyServerConfig
+    from lerobot.async_inference.policy_server import PolicyServer
+
+    test_config = PolicyServerConfig(host="localhost", port=9999)
+    server = PolicyServer(test_config)
+    # Replace the real policy with our fast, deterministic stub.
+    server.policy = MockPolicy()
+    server.actions_per_chunk = 20
+    server.device = "cpu"
+
+    # Add mock lerobot_features that the observation similarity functions need
+    server.lerobot_features = {
+        OBS_STATE: {
+            "dtype": "float32",
+            "shape": [6],
+            "names": ["joint1", "joint2", "joint3", "joint4", "joint5", "joint6"],
+        }
+    }
+
+    return server
+
+
+# -----------------------------------------------------------------------------
+# Helper utilities for tests
+# -----------------------------------------------------------------------------
+
+
+def _make_obs(state: torch.Tensor, timestep: int = 0, must_go: bool = False):
+    """Create a TimedObservation with a given state vector."""
+    # Import only when needed
+    from lerobot.async_inference.helpers import TimedObservation
+
+    return TimedObservation(
+        observation={
+            "joint1": state[0].item() if len(state) > 0 else 0.0,
+            "joint2": state[1].item() if len(state) > 1 else 0.0,
+            "joint3": state[2].item() if len(state) > 2 else 0.0,
+            "joint4": state[3].item() if len(state) > 3 else 0.0,
+            "joint5": state[4].item() if len(state) > 4 else 0.0,
+            "joint6": state[5].item() if len(state) > 5 else 0.0,
+        },
+        timestamp=time.time(),
+        timestep=timestep,
+        must_go=must_go,
+    )
+
+
+# -----------------------------------------------------------------------------
+# Tests
+# -----------------------------------------------------------------------------
+
+
+def test_time_action_chunk(policy_server):
+    """Verify that `_time_action_chunk` assigns correct timestamps and timesteps."""
+    start_ts = time.time()
+    start_t = 10
+    # A chunk of 3 action tensors.
+    action_tensors = [torch.randn(6) for _ in range(3)]
+
+    timed_actions = policy_server._time_action_chunk(start_ts, action_tensors, start_t)
+
+    assert len(timed_actions) == 3
+    # Check timesteps
+    assert [ta.get_timestep() for ta in timed_actions] == [10, 11, 12]
+    # Check timestamps
+    expected_timestamps = [
+        start_ts,
+        start_ts + policy_server.config.environment_dt,
+        start_ts + 2 * policy_server.config.environment_dt,
+    ]
+    for ta, expected_ts in zip(timed_actions, expected_timestamps, strict=True):
+        assert abs(ta.get_timestamp() - expected_ts) < 1e-6
+
+
+def test_maybe_enqueue_observation_must_go(policy_server):
+    """An observation with `must_go=True` is always enqueued."""
+    obs = _make_obs(torch.zeros(6), must_go=True)
+    assert policy_server._enqueue_observation(obs) is True
+    assert policy_server.observation_queue.qsize() == 1
+    assert policy_server.observation_queue.get_nowait() is obs
+
+
+def test_maybe_enqueue_observation_dissimilar(policy_server):
+    """A dissimilar observation (not `must_go`) is enqueued."""
+    # Set a last predicted observation.
+    policy_server.last_processed_obs = _make_obs(torch.zeros(6))
+    # Create a new, dissimilar observation.
+    new_obs = _make_obs(torch.ones(6) * 5)  # High norm difference
+
+    assert policy_server._enqueue_observation(new_obs) is True
+    assert policy_server.observation_queue.qsize() == 1
+
+
+def test_maybe_enqueue_observation_is_skipped(policy_server):
+    """A similar observation (not `must_go`) is skipped."""
+    # Set a last predicted observation.
+    policy_server.last_processed_obs = _make_obs(torch.zeros(6))
+    # Create a new, very similar observation.
+    new_obs = _make_obs(torch.zeros(6) + 1e-4)
+
+    assert policy_server._enqueue_observation(new_obs) is False
+    assert policy_server.observation_queue.empty() is True
+
+
+def test_obs_sanity_checks(policy_server):
+    """Unit-test the private `_obs_sanity_checks` helper."""
+    prev = _make_obs(torch.zeros(6), timestep=0)
+
+    # Case 1 – timestep already predicted
+    policy_server._predicted_timesteps.add(1)
+    obs_same_ts = _make_obs(torch.ones(6), timestep=1)
+    assert policy_server._obs_sanity_checks(obs_same_ts, prev) is False
+
+    # Case 2 – observation too similar
+    policy_server._predicted_timesteps.clear()
+    obs_similar = _make_obs(torch.zeros(6) + 1e-4, timestep=2)
+    assert policy_server._obs_sanity_checks(obs_similar, prev) is False
+
+    # Case 3 – genuinely new & dissimilar observation passes
+    obs_ok = _make_obs(torch.ones(6) * 5, timestep=3)
+    assert policy_server._obs_sanity_checks(obs_ok, prev) is True
+
+
+def test_predict_action_chunk(monkeypatch, policy_server):
+    """End-to-end test of `_predict_action_chunk` with a stubbed _get_action_chunk."""
+    # Import only when needed
+    from lerobot.async_inference.policy_server import PolicyServer
+
+    # Force server to act-style policy; patch method to return deterministic tensor
+    policy_server.policy_type = "act"
+    # NOTE(Steven): Smelly tests as the Server is a state machine being partially mocked. Adding these processors as a quick fix.
+    policy_server.preprocessor = lambda obs: obs
+    policy_server.postprocessor = lambda tensor: tensor
+    action_dim = 6
+    batch_size = 1
+    actions_per_chunk = policy_server.actions_per_chunk
+
+    def _fake_get_action_chunk(_self, _obs, _type="act"):
+        return torch.zeros(batch_size, actions_per_chunk, action_dim)
+
+    monkeypatch.setattr(PolicyServer, "_get_action_chunk", _fake_get_action_chunk, raising=True)
+
+    obs = _make_obs(torch.zeros(6), timestep=5)
+    timed_actions = policy_server._predict_action_chunk(obs)
+
+    assert len(timed_actions) == actions_per_chunk
+    assert [ta.get_timestep() for ta in timed_actions] == list(range(5, 5 + actions_per_chunk))
+
+    for i, ta in enumerate(timed_actions):
+        expected_ts = obs.get_timestamp() + i * policy_server.config.environment_dt
+        assert abs(ta.get_timestamp() - expected_ts) < 1e-6
diff --git a/lerobot/tests/async_inference/test_robot_client.py b/lerobot/tests/async_inference/test_robot_client.py
new file mode 100644
index 0000000000000000000000000000000000000000..d7ef5b35042991ad819ad6a28e3b4c23e1542880
--- /dev/null
+++ b/lerobot/tests/async_inference/test_robot_client.py
@@ -0,0 +1,269 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Unit-tests for the `RobotClient` action-queue logic (pure Python, no gRPC).
+
+We monkey-patch `lerobot.robots.utils.make_robot_from_config` so that
+no real hardware is accessed. Only the queue-update mechanism is verified.
+"""
+
+from __future__ import annotations
+
+import time
+from queue import Queue
+
+import pytest
+import torch
+
+# Skip entire module if grpc is not available
+pytest.importorskip("grpc")
+
+# -----------------------------------------------------------------------------
+# Test fixtures
+# -----------------------------------------------------------------------------
+
+
+@pytest.fixture()
+def robot_client():
+    """Fresh `RobotClient` instance for each test case (no threads started).
+    Uses DummyRobot."""
+    # Import only when the test actually runs (after decorator check)
+    from lerobot.async_inference.configs import RobotClientConfig
+    from lerobot.async_inference.robot_client import RobotClient
+    from tests.mocks.mock_robot import MockRobotConfig
+
+    test_config = MockRobotConfig()
+
+    # gRPC channel is not actually used in tests, so using a dummy address
+    test_config = RobotClientConfig(
+        robot=test_config,
+        server_address="localhost:9999",
+        policy_type="test",
+        pretrained_name_or_path="test",
+        actions_per_chunk=20,
+    )
+
+    client = RobotClient(test_config)
+
+    # Initialize attributes that are normally set in start() method
+    client.chunks_received = 0
+    client.available_actions_size = []
+
+    yield client
+
+    if client.robot.is_connected:
+        client.stop()
+
+
+# -----------------------------------------------------------------------------
+# Helper utilities for tests
+# -----------------------------------------------------------------------------
+
+
+def _make_actions(start_ts: float, start_t: int, count: int):
+    """Generate `count` consecutive TimedAction objects starting at timestep `start_t`."""
+    from lerobot.async_inference.helpers import TimedAction
+
+    fps = 30  # emulates most common frame-rate
+    actions = []
+    for i in range(count):
+        timestep = start_t + i
+        timestamp = start_ts + i * (1 / fps)
+        action_tensor = torch.full((6,), timestep, dtype=torch.float32)
+        actions.append(TimedAction(action=action_tensor, timestep=timestep, timestamp=timestamp))
+    return actions
+
+
+# -----------------------------------------------------------------------------
+# Tests
+# -----------------------------------------------------------------------------
+
+
+def test_update_action_queue_discards_stale(robot_client):
+    """`_update_action_queue` must drop actions with `timestep` <= `latest_action`."""
+
+    # Pretend we already executed up to action #4
+    robot_client.latest_action = 4
+
+    # Incoming chunk contains timesteps 3..7 -> expect 5,6,7 kept.
+    incoming = _make_actions(start_ts=time.time(), start_t=3, count=5)  # 3,4,5,6,7
+
+    robot_client._aggregate_action_queues(incoming)
+
+    # Extract timesteps from queue
+    resulting_timesteps = [a.get_timestep() for a in robot_client.action_queue.queue]
+
+    assert resulting_timesteps == [5, 6, 7]
+
+
+@pytest.mark.parametrize(
+    "weight_old, weight_new",
+    [
+        (1.0, 0.0),
+        (0.0, 1.0),
+        (0.5, 0.5),
+        (0.2, 0.8),
+        (0.8, 0.2),
+        (0.1, 0.9),
+        (0.9, 0.1),
+    ],
+)
+def test_aggregate_action_queues_combines_actions_in_overlap(
+    robot_client, weight_old: float, weight_new: float
+):
+    """`_aggregate_action_queues` must combine actions on overlapping timesteps according
+    to the provided aggregate_fn, here tested with multiple coefficients."""
+    from lerobot.async_inference.helpers import TimedAction
+
+    robot_client.chunks_received = 0
+
+    # Pretend we already executed up to action #4, and queue contains actions for timesteps 5..6
+    robot_client.latest_action = 4
+    current_actions = _make_actions(
+        start_ts=time.time(), start_t=5, count=2
+    )  # actions are [torch.ones(6), torch.ones(6), ...]
+    current_actions = [
+        TimedAction(action=10 * a.get_action(), timestep=a.get_timestep(), timestamp=a.get_timestamp())
+        for a in current_actions
+    ]
+
+    for a in current_actions:
+        robot_client.action_queue.put(a)
+
+    # Incoming chunk contains timesteps 3..7 -> expect 5,6,7 kept.
+    incoming = _make_actions(start_ts=time.time(), start_t=3, count=5)  # 3,4,5,6,7
+
+    overlap_timesteps = [5, 6]  # properly tested in test_aggregate_action_queues_discards_stale
+    nonoverlap_timesteps = [7]
+
+    robot_client._aggregate_action_queues(
+        incoming, aggregate_fn=lambda x1, x2: weight_old * x1 + weight_new * x2
+    )
+
+    queue_overlap_actions = []
+    queue_non_overlap_actions = []
+    for a in robot_client.action_queue.queue:
+        if a.get_timestep() in overlap_timesteps:
+            queue_overlap_actions.append(a)
+        elif a.get_timestep() in nonoverlap_timesteps:
+            queue_non_overlap_actions.append(a)
+
+    queue_overlap_actions = sorted(queue_overlap_actions, key=lambda x: x.get_timestep())
+    queue_non_overlap_actions = sorted(queue_non_overlap_actions, key=lambda x: x.get_timestep())
+
+    assert torch.allclose(
+        queue_overlap_actions[0].get_action(),
+        weight_old * current_actions[0].get_action() + weight_new * incoming[-3].get_action(),
+    )
+    assert torch.allclose(
+        queue_overlap_actions[1].get_action(),
+        weight_old * current_actions[1].get_action() + weight_new * incoming[-2].get_action(),
+    )
+    assert torch.allclose(queue_non_overlap_actions[0].get_action(), incoming[-1].get_action())
+
+
+@pytest.mark.parametrize(
+    "chunk_size, queue_len, expected",
+    [
+        (20, 12, False),  # 12 / 20 = 0.6  > g=0.5 threshold, not ready to send
+        (20, 8, True),  # 8  / 20 = 0.4 <= g=0.5, ready to send
+        (10, 5, True),
+        (10, 6, False),
+    ],
+)
+def test_ready_to_send_observation(robot_client, chunk_size: int, queue_len: int, expected: bool):
+    """Validate `_ready_to_send_observation` ratio logic for various sizes."""
+
+    robot_client.action_chunk_size = chunk_size
+
+    # Clear any existing actions then fill with `queue_len` dummy entries ----
+    robot_client.action_queue = Queue()
+
+    dummy_actions = _make_actions(start_ts=time.time(), start_t=0, count=queue_len)
+    for act in dummy_actions:
+        robot_client.action_queue.put(act)
+
+    assert robot_client._ready_to_send_observation() is expected
+
+
+@pytest.mark.parametrize(
+    "g_threshold, expected",
+    [
+        # The condition is `queue_size / chunk_size <= g`.
+        # Here, ratio = 6 / 10 = 0.6.
+        (0.0, False),  # 0.6 <= 0.0 is False
+        (0.1, False),
+        (0.2, False),
+        (0.3, False),
+        (0.4, False),
+        (0.5, False),
+        (0.6, True),  # 0.6 <= 0.6 is True
+        (0.7, True),
+        (0.8, True),
+        (0.9, True),
+        (1.0, True),
+    ],
+)
+def test_ready_to_send_observation_with_varying_threshold(robot_client, g_threshold: float, expected: bool):
+    """Validate `_ready_to_send_observation` with fixed sizes and varying `g`."""
+    # Fixed sizes for this test: ratio = 6 / 10 = 0.6
+    chunk_size = 10
+    queue_len = 6
+
+    robot_client.action_chunk_size = chunk_size
+    # This is the parameter we are testing
+    robot_client._chunk_size_threshold = g_threshold
+
+    # Fill queue with dummy actions
+    robot_client.action_queue = Queue()
+    dummy_actions = _make_actions(start_ts=time.time(), start_t=0, count=queue_len)
+    for act in dummy_actions:
+        robot_client.action_queue.put(act)
+
+    assert robot_client._ready_to_send_observation() is expected
+
+
+# -----------------------------------------------------------------------------
+# Regression test: robot type registry populated by robot_client imports
+# -----------------------------------------------------------------------------
+
+
+def test_robot_client_registers_builtin_robot_types():
+    """Importing robot_client must populate RobotConfig's ChoiceRegistry.
+
+    This is a regression test for a bug introduced in #2425, where removing
+    robot module imports from robot_client.py caused RobotConfig's registry to
+    be empty, breaking CLI argument parsing with:
+      error: argument --robot.type: invalid choice: 'so101_follower' (choose from )
+
+    Robot types are registered via @RobotConfig.register_subclass() decorators
+    at import time, so all supported modules must be explicitly imported.
+    """
+    import lerobot.async_inference.robot_client  # noqa: F401
+    from lerobot.robots.config import RobotConfig
+
+    known_choices = RobotConfig.get_known_choices()
+
+    expected_robot_types = [
+        "so100_follower",
+        "so101_follower",
+        "koch_follower",
+        "omx_follower",
+        "bi_so_follower",
+    ]
+    for robot_type in expected_robot_types:
+        assert robot_type in known_choices, (
+            f"Robot type '{robot_type}' is not registered in RobotConfig's ChoiceRegistry. "
+            f"Ensure the corresponding module is imported in robot_client.py. "
+            f"Known choices: {sorted(known_choices)}"
+        )
diff --git a/lerobot/tests/cameras/test_opencv.py b/lerobot/tests/cameras/test_opencv.py
new file mode 100644
index 0000000000000000000000000000000000000000..720d0c9b33a2fae4f79932ad82f2a37d5adcbcbb
--- /dev/null
+++ b/lerobot/tests/cameras/test_opencv.py
@@ -0,0 +1,299 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# Example of running a specific test:
+# ```bash
+# pytest tests/cameras/test_opencv.py::test_connect
+# ```
+
+from pathlib import Path
+from unittest.mock import patch
+
+import cv2
+import numpy as np
+import pytest
+
+from lerobot.cameras.configs import Cv2Rotation
+from lerobot.cameras.opencv import OpenCVCamera, OpenCVCameraConfig
+from lerobot.utils.errors import DeviceAlreadyConnectedError, DeviceNotConnectedError
+
+RealVideoCapture = cv2.VideoCapture
+
+
+class MockLoopingVideoCapture:
+    """
+    Wraps the real OpenCV VideoCapture.
+    Motivation: cv2.VideoCapture(file.png) is only valid for one read.
+    Strategy: Read the file once & return the cached frame for subsequent reads.
+    Consequence: No recurrent I/O operations, but we keep the test artifacts simple.
+    """
+
+    def __init__(self, *args, **kwargs):
+        args_clean = [str(a) if isinstance(a, Path) else a for a in args]
+        self._real_vc = RealVideoCapture(*args_clean, **kwargs)
+        self._cached_frame = None
+
+    def read(self):
+        ret, frame = self._real_vc.read()
+
+        if ret:
+            self._cached_frame = frame
+            return ret, frame
+
+        if not ret and self._cached_frame is not None:
+            return True, self._cached_frame.copy()
+
+        return ret, frame
+
+    def __getattr__(self, name):
+        return getattr(self._real_vc, name)
+
+
+@pytest.fixture(autouse=True)
+def patch_opencv_videocapture():
+    """
+    Automatically patches cv2.VideoCapture for all tests.
+    """
+    module_path = OpenCVCamera.__module__
+    target = f"{module_path}.cv2.VideoCapture"
+
+    with patch(target, new=MockLoopingVideoCapture):
+        yield
+
+
+# NOTE(Steven): more tests + assertions?
+TEST_ARTIFACTS_DIR = Path(__file__).parent.parent / "artifacts" / "cameras"
+DEFAULT_PNG_FILE_PATH = TEST_ARTIFACTS_DIR / "image_160x120.png"
+TEST_IMAGE_SIZES = ["128x128", "160x120", "320x180", "480x270"]
+TEST_IMAGE_PATHS = [TEST_ARTIFACTS_DIR / f"image_{size}.png" for size in TEST_IMAGE_SIZES]
+
+
+def test_abc_implementation():
+    """Instantiation should raise an error if the class doesn't implement abstract methods/properties."""
+    config = OpenCVCameraConfig(index_or_path=0)
+
+    _ = OpenCVCamera(config)
+
+
+def test_connect():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, warmup_s=0)
+
+    with OpenCVCamera(config) as camera:
+        assert camera.is_connected
+
+
+def test_connect_already_connected():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, warmup_s=0)
+
+    with OpenCVCamera(config) as camera, pytest.raises(DeviceAlreadyConnectedError):
+        camera.connect()
+
+
+def test_connect_invalid_camera_path():
+    config = OpenCVCameraConfig(index_or_path="nonexistent/camera.png")
+
+    camera = OpenCVCamera(config)
+
+    with pytest.raises(ConnectionError):
+        camera.connect(warmup=False)
+
+
+def test_invalid_width_connect():
+    config = OpenCVCameraConfig(
+        index_or_path=DEFAULT_PNG_FILE_PATH,
+        width=99999,  # Invalid width to trigger error
+        height=480,
+    )
+
+    camera = OpenCVCamera(config)
+    with pytest.raises(RuntimeError):
+        camera.connect(warmup=False)
+
+
+@pytest.mark.parametrize("index_or_path", TEST_IMAGE_PATHS, ids=TEST_IMAGE_SIZES)
+def test_read(index_or_path):
+    config = OpenCVCameraConfig(index_or_path=index_or_path, warmup_s=0)
+
+    with OpenCVCamera(config) as camera:
+        img = camera.read()
+        assert isinstance(img, np.ndarray)
+
+
+def test_read_before_connect():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH)
+
+    camera = OpenCVCamera(config)
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.read()
+
+
+def test_disconnect():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH)
+    camera = OpenCVCamera(config)
+    camera.connect(warmup=False)
+
+    camera.disconnect()
+
+    assert not camera.is_connected
+
+
+def test_disconnect_before_connect():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH)
+    camera = OpenCVCamera(config)
+
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.disconnect()
+
+
+@pytest.mark.parametrize("index_or_path", TEST_IMAGE_PATHS, ids=TEST_IMAGE_SIZES)
+def test_async_read(index_or_path):
+    config = OpenCVCameraConfig(index_or_path=index_or_path, warmup_s=0)
+
+    with OpenCVCamera(config) as camera:
+        img = camera.async_read()
+
+        assert camera.thread is not None
+        assert camera.thread.is_alive()
+        assert isinstance(img, np.ndarray)
+
+
+@pytest.mark.skip("Skipping test: async_read  0 timeout behavior may be flaky/non-deterministic.")
+def test_async_read_timeout():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, warmup_s=0)
+
+    with OpenCVCamera(config) as camera, pytest.raises(TimeoutError):
+        camera.async_read(timeout_ms=0)  # consumes any available frame by then
+        camera.async_read(timeout_ms=0)  # request immediately another one
+
+
+def test_async_read_before_connect():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH)
+    camera = OpenCVCamera(config)
+
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.async_read()
+
+
+def test_read_latest():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, warmup_s=0)
+
+    with OpenCVCamera(config) as camera:
+        # ensure at least one fresh frame is captured
+        frame = camera.read()
+        latest = camera.read_latest()
+
+        assert isinstance(latest, np.ndarray)
+        assert latest.shape == frame.shape
+
+
+def test_read_latest_before_connect():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH)
+
+    camera = OpenCVCamera(config)
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.read_latest()
+
+
+def test_read_latest_high_frequency():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, warmup_s=0)
+
+    with OpenCVCamera(config) as camera:
+        # prime to ensure frames are available
+        ref = camera.read()
+
+        for _ in range(20):
+            latest = camera.read_latest()
+            assert isinstance(latest, np.ndarray)
+            assert latest.shape == ref.shape
+
+
+def test_read_latest_too_old():
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, warmup_s=0)
+
+    with OpenCVCamera(config) as camera:
+        # prime to ensure frames are available
+        _ = camera.read()
+
+        with pytest.raises(TimeoutError):
+            _ = camera.read_latest(max_age_ms=0)  # immediately too old
+
+
+def test_fourcc_configuration():
+    """Test FourCC configuration validation and application."""
+
+    # Test MJPG specifically (main use case)
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, fourcc="MJPG")
+    camera = OpenCVCamera(config)
+    assert camera.config.fourcc == "MJPG"
+
+    # Test a few other common formats
+    valid_fourcc_codes = ["YUYV", "YUY2", "RGB3"]
+
+    for fourcc in valid_fourcc_codes:
+        config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, fourcc=fourcc)
+        camera = OpenCVCamera(config)
+        assert camera.config.fourcc == fourcc
+
+    # Test invalid FOURCC codes
+    invalid_fourcc_codes = ["ABC", "ABCDE", ""]
+
+    for fourcc in invalid_fourcc_codes:
+        with pytest.raises(ValueError):
+            OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, fourcc=fourcc)
+
+
+def test_fourcc_with_camera():
+    """Test FourCC functionality with actual camera connection."""
+    config = OpenCVCameraConfig(index_or_path=DEFAULT_PNG_FILE_PATH, fourcc="MJPG", warmup_s=0)
+
+    # Connect should work with MJPG specified
+    with OpenCVCamera(config) as camera:
+        assert camera.is_connected
+
+        # Read should work normally
+        img = camera.read()
+        assert isinstance(img, np.ndarray)
+
+
+@pytest.mark.parametrize("index_or_path", TEST_IMAGE_PATHS, ids=TEST_IMAGE_SIZES)
+@pytest.mark.parametrize(
+    "rotation",
+    [
+        Cv2Rotation.NO_ROTATION,
+        Cv2Rotation.ROTATE_90,
+        Cv2Rotation.ROTATE_180,
+        Cv2Rotation.ROTATE_270,
+    ],
+    ids=["no_rot", "rot90", "rot180", "rot270"],
+)
+def test_rotation(rotation, index_or_path):
+    filename = Path(index_or_path).name
+    dimensions = filename.split("_")[-1].split(".")[0]  # Assumes filenames format (_wxh.png)
+    original_width, original_height = map(int, dimensions.split("x"))
+
+    config = OpenCVCameraConfig(index_or_path=index_or_path, rotation=rotation, warmup_s=0)
+    with OpenCVCamera(config) as camera:
+        img = camera.read()
+        assert isinstance(img, np.ndarray)
+
+        if rotation in (Cv2Rotation.ROTATE_90, Cv2Rotation.ROTATE_270):
+            assert camera.width == original_height
+            assert camera.height == original_width
+            assert img.shape[:2] == (original_width, original_height)
+        else:
+            assert camera.width == original_width
+            assert camera.height == original_height
+            assert img.shape[:2] == (original_height, original_width)
diff --git a/lerobot/tests/cameras/test_reachy2_camera.py b/lerobot/tests/cameras/test_reachy2_camera.py
new file mode 100644
index 0000000000000000000000000000000000000000..2aebfdf0a49c0d44f7adc54619a2496f7aaec365
--- /dev/null
+++ b/lerobot/tests/cameras/test_reachy2_camera.py
@@ -0,0 +1,205 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import time
+from unittest.mock import MagicMock, patch
+
+import numpy as np
+import pytest
+
+pytest.importorskip("reachy2_sdk")
+
+from lerobot.cameras.reachy2_camera import Reachy2Camera, Reachy2CameraConfig
+from lerobot.utils.errors import DeviceNotConnectedError
+
+PARAMS = [
+    ("teleop", "left"),
+    ("teleop", "right"),
+    ("depth", "rgb"),
+    # ("depth", "depth"),  # Depth camera is not available yet
+]
+
+
+def _make_cam_manager_mock():
+    c = MagicMock(name="CameraManagerMock")
+
+    teleop = MagicMock(name="TeleopCam")
+    teleop.width = 640
+    teleop.height = 480
+    teleop.get_frame = MagicMock(
+        side_effect=lambda *_, **__: (
+            np.zeros((480, 640, 3), dtype=np.uint8),
+            time.time(),
+        )
+    )
+
+    depth = MagicMock(name="DepthCam")
+    depth.width = 640
+    depth.height = 480
+    depth.get_frame = MagicMock(
+        side_effect=lambda *_, **__: (
+            np.zeros((480, 640, 3), dtype=np.uint8),
+            time.time(),
+        )
+    )
+
+    c.is_connected.return_value = True
+    c.teleop = teleop
+    c.depth = depth
+
+    def _connect():
+        c.teleop = teleop
+        c.depth = depth
+        c.is_connected.return_value = True
+
+    def _disconnect():
+        c.teleop = None
+        c.depth = None
+        c.is_connected.return_value = False
+
+    c.connect = MagicMock(side_effect=_connect)
+    c.disconnect = MagicMock(side_effect=_disconnect)
+
+    # Mock methods
+    c.initialize_cameras = MagicMock()
+
+    return c
+
+
+@pytest.fixture(
+    params=PARAMS,
+    # ids=["teleop-left", "teleop-right", "torso-rgb", "torso-depth"],
+    ids=["teleop-left", "teleop-right", "torso-rgb"],
+)
+def camera(request):
+    name, image_type = request.param
+    with (
+        patch(
+            "lerobot.cameras.reachy2_camera.reachy2_camera.CameraManager",
+            side_effect=lambda *a, **k: _make_cam_manager_mock(),
+        ),
+    ):
+        config = Reachy2CameraConfig(name=name, image_type=image_type)
+        cam = Reachy2Camera(config)
+        yield cam
+        if cam.is_connected:
+            cam.disconnect()
+
+
+def test_connect(camera):
+    camera.connect()
+    assert camera.is_connected
+    camera.cam_manager.initialize_cameras.assert_called_once()
+
+
+def test_read(camera):
+    camera.connect()
+
+    img = camera.read()
+    if camera.config.name == "teleop":
+        camera.cam_manager.teleop.get_frame.assert_called_once()
+    elif camera.config.name == "depth":
+        camera.cam_manager.depth.get_frame.assert_called_once()
+    assert isinstance(img, np.ndarray)
+    assert img.shape == (480, 640, 3)
+
+
+def test_disconnect(camera):
+    camera.connect()
+
+    camera.disconnect()
+    assert not camera.is_connected
+
+
+def test_async_read(camera):
+    camera.connect()
+    try:
+        img = camera.async_read()
+
+        assert isinstance(img, np.ndarray)
+    finally:
+        if camera.is_connected:
+            camera.disconnect()
+
+
+def test_read_before_connect(camera):
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.read()
+
+
+def test_disconnect_before_connect(camera):
+    with pytest.raises(DeviceNotConnectedError):
+        camera.disconnect()
+
+
+def test_async_read_before_connect(camera):
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.async_read()
+
+
+def test_read_latest(camera):
+    camera.connect()
+
+    frame = camera.read()
+    latest = camera.read_latest()
+
+    assert isinstance(latest, np.ndarray)
+    assert latest.shape == frame.shape
+
+
+def test_read_latest_before_connect(camera):
+    # camera fixture yields an unconnected camera instance
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.read_latest()
+
+
+def test_read_latest_high_frequency(camera):
+    camera.connect()
+
+    # prime to ensure frames are available
+    ref = camera.read()
+
+    for _ in range(20):
+        latest = camera.read_latest()
+        assert isinstance(latest, np.ndarray)
+        assert latest.shape == ref.shape
+
+
+def test_read_latest_too_old(camera):
+    camera.connect()
+
+    # prime to ensure frames are available
+    _ = camera.read()
+
+    with pytest.raises(TimeoutError):
+        _ = camera.read_latest(max_age_ms=0)  # immediately too old
+
+
+def test_wrong_camera_name():
+    with pytest.raises(ValueError):
+        _ = Reachy2CameraConfig(name="wrong-name", image_type="left")
+
+
+def test_wrong_image_type():
+    with pytest.raises(ValueError):
+        _ = Reachy2CameraConfig(name="teleop", image_type="rgb")
+    with pytest.raises(ValueError):
+        _ = Reachy2CameraConfig(name="depth", image_type="left")
+
+
+def test_wrong_color_mode():
+    with pytest.raises(ValueError):
+        _ = Reachy2CameraConfig(name="teleop", image_type="left", color_mode="wrong-color")
diff --git a/lerobot/tests/cameras/test_realsense.py b/lerobot/tests/cameras/test_realsense.py
new file mode 100644
index 0000000000000000000000000000000000000000..1deb73f0514b99f31af2cbdcab496e68ecdec880
--- /dev/null
+++ b/lerobot/tests/cameras/test_realsense.py
@@ -0,0 +1,228 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# Example of running a specific test:
+# ```bash
+# pytest tests/cameras/test_opencv.py::test_connect
+# ```
+
+from pathlib import Path
+from unittest.mock import patch
+
+import numpy as np
+import pytest
+
+from lerobot.cameras.configs import Cv2Rotation
+from lerobot.utils.errors import DeviceAlreadyConnectedError, DeviceNotConnectedError
+
+pytest.importorskip("pyrealsense2")
+
+from lerobot.cameras.realsense import RealSenseCamera, RealSenseCameraConfig
+
+TEST_ARTIFACTS_DIR = Path(__file__).parent.parent / "artifacts" / "cameras"
+BAG_FILE_PATH = TEST_ARTIFACTS_DIR / "test_rs.bag"
+
+# NOTE(Steven): For some reason these tests take ~20sec in macOS but only ~2sec in Linux.
+
+
+def mock_rs_config_enable_device_from_file(rs_config_instance, _sn):
+    return rs_config_instance.enable_device_from_file(str(BAG_FILE_PATH), repeat_playback=True)
+
+
+def mock_rs_config_enable_device_bad_file(rs_config_instance, _sn):
+    return rs_config_instance.enable_device_from_file("non_existent_file.bag", repeat_playback=True)
+
+
+@pytest.fixture(name="patch_realsense", autouse=True)
+def fixture_patch_realsense():
+    """Automatically mock pyrealsense2.config.enable_device for all tests."""
+    with patch(
+        "pyrealsense2.config.enable_device", side_effect=mock_rs_config_enable_device_from_file
+    ) as mock:
+        yield mock
+
+
+def test_abc_implementation():
+    """Instantiation should raise an error if the class doesn't implement abstract methods/properties."""
+    config = RealSenseCameraConfig(serial_number_or_name="042")
+    _ = RealSenseCamera(config)
+
+
+def test_connect():
+    config = RealSenseCameraConfig(serial_number_or_name="042", warmup_s=0)
+
+    with RealSenseCamera(config) as camera:
+        assert camera.is_connected
+
+
+def test_connect_already_connected():
+    config = RealSenseCameraConfig(serial_number_or_name="042", warmup_s=0)
+    with RealSenseCamera(config) as camera, pytest.raises(DeviceAlreadyConnectedError):
+        camera.connect(warmup=False)
+
+
+def test_connect_invalid_camera_path(patch_realsense):
+    patch_realsense.side_effect = mock_rs_config_enable_device_bad_file
+    config = RealSenseCameraConfig(serial_number_or_name="042")
+    camera = RealSenseCamera(config)
+
+    with pytest.raises(ConnectionError):
+        camera.connect(warmup=False)
+
+
+def test_invalid_width_connect():
+    config = RealSenseCameraConfig(serial_number_or_name="042", width=99999, height=480, fps=30)
+    camera = RealSenseCamera(config)
+
+    with pytest.raises(ConnectionError):
+        camera.connect(warmup=False)
+
+
+def test_read():
+    config = RealSenseCameraConfig(serial_number_or_name="042", width=640, height=480, fps=30, warmup_s=0)
+    with RealSenseCamera(config) as camera:
+        img = camera.read()
+        assert isinstance(img, np.ndarray)
+
+
+# TODO(Steven): Fix this test for the latest version of pyrealsense2.
+@pytest.mark.skip("Skipping test: pyrealsense2 version > 2.55.1.6486")
+def test_read_depth():
+    config = RealSenseCameraConfig(serial_number_or_name="042", width=640, height=480, fps=30, use_depth=True)
+    camera = RealSenseCamera(config)
+    camera.connect(warmup=False)
+
+    img = camera.read_depth(timeout_ms=2000)  # NOTE(Steven): Reading depth takes longer in CI environments.
+    assert isinstance(img, np.ndarray)
+
+
+def test_read_before_connect():
+    config = RealSenseCameraConfig(serial_number_or_name="042")
+    camera = RealSenseCamera(config)
+
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.read()
+
+
+def test_disconnect():
+    config = RealSenseCameraConfig(serial_number_or_name="042")
+    camera = RealSenseCamera(config)
+    camera.connect(warmup=False)
+
+    camera.disconnect()
+
+    assert not camera.is_connected
+
+
+def test_disconnect_before_connect():
+    config = RealSenseCameraConfig(serial_number_or_name="042")
+    camera = RealSenseCamera(config)
+
+    with pytest.raises(DeviceNotConnectedError):
+        camera.disconnect()
+
+
+def test_async_read():
+    config = RealSenseCameraConfig(serial_number_or_name="042", width=640, height=480, fps=30, warmup_s=0)
+
+    with RealSenseCamera(config) as camera:
+        img = camera.async_read()
+
+        assert camera.thread is not None
+        assert camera.thread.is_alive()
+        assert isinstance(img, np.ndarray)
+
+
+def test_async_read_timeout():
+    config = RealSenseCameraConfig(serial_number_or_name="042", width=640, height=480, fps=30, warmup_s=0)
+    with RealSenseCamera(config) as camera, pytest.raises(TimeoutError):
+        camera.async_read(timeout_ms=0)  # consumes any available frame by then
+        camera.async_read(timeout_ms=0)  # request immediately another one
+
+
+def test_async_read_before_connect():
+    config = RealSenseCameraConfig(serial_number_or_name="042")
+    camera = RealSenseCamera(config)
+
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.async_read()
+
+
+def test_read_latest():
+    config = RealSenseCameraConfig(serial_number_or_name="042", width=640, height=480, fps=30, warmup_s=0)
+    with RealSenseCamera(config) as camera:
+        img = camera.read()
+        latest = camera.read_latest()
+
+        assert isinstance(latest, np.ndarray)
+        assert latest.shape == img.shape
+
+
+def test_read_latest_high_frequency():
+    config = RealSenseCameraConfig(serial_number_or_name="042", width=640, height=480, fps=30, warmup_s=0)
+    with RealSenseCamera(config) as camera:
+        # prime with one read to ensure frames are available
+        ref = camera.read()
+
+        for _ in range(20):
+            latest = camera.read_latest()
+            assert isinstance(latest, np.ndarray)
+            assert latest.shape == ref.shape
+
+
+def test_read_latest_before_connect():
+    config = RealSenseCameraConfig(serial_number_or_name="042")
+    camera = RealSenseCamera(config)
+
+    with pytest.raises(DeviceNotConnectedError):
+        _ = camera.read_latest()
+
+
+def test_read_latest_too_old():
+    config = RealSenseCameraConfig(serial_number_or_name="042")
+
+    with RealSenseCamera(config) as camera:
+        # prime to ensure frames are available
+        _ = camera.read()
+
+        with pytest.raises(TimeoutError):
+            _ = camera.read_latest(max_age_ms=0)  # immediately too old
+
+
+@pytest.mark.parametrize(
+    "rotation",
+    [
+        Cv2Rotation.NO_ROTATION,
+        Cv2Rotation.ROTATE_90,
+        Cv2Rotation.ROTATE_180,
+        Cv2Rotation.ROTATE_270,
+    ],
+    ids=["no_rot", "rot90", "rot180", "rot270"],
+)
+def test_rotation(rotation):
+    config = RealSenseCameraConfig(serial_number_or_name="042", rotation=rotation, warmup_s=0)
+    with RealSenseCamera(config) as camera:
+        img = camera.read()
+        assert isinstance(img, np.ndarray)
+
+        if rotation in (Cv2Rotation.ROTATE_90, Cv2Rotation.ROTATE_270):
+            assert camera.width == 480
+            assert camera.height == 640
+            assert img.shape[:2] == (640, 480)
+        else:
+            assert camera.width == 640
+            assert camera.height == 480
+            assert img.shape[:2] == (480, 640)
diff --git a/lerobot/tests/configs/test_default.py b/lerobot/tests/configs/test_default.py
new file mode 100644
index 0000000000000000000000000000000000000000..238b8bacd3b24561f9dc5285d11473c2b569ba9d
--- /dev/null
+++ b/lerobot/tests/configs/test_default.py
@@ -0,0 +1,38 @@
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import pytest
+
+from lerobot.configs.default import DatasetConfig
+
+
+def test_dataset_config_valid():
+    DatasetConfig(repo_id="user/repo", episodes=[0, 1, 2])
+
+
+def test_dataset_config_negative_episodes():
+    with pytest.raises(ValueError, match="non-negative"):
+        DatasetConfig(repo_id="user/repo", episodes=[0, -1, 2])
+
+
+def test_dataset_config_duplicate_episodes():
+    with pytest.raises(ValueError, match="duplicates"):
+        DatasetConfig(repo_id="user/repo", episodes=[0, 1, 1, 2])
+
+
+def test_dataset_config_none_episodes_ok():
+    DatasetConfig(repo_id="user/repo", episodes=None)
+
+
+def test_dataset_config_empty_episodes_ok():
+    DatasetConfig(repo_id="user/repo", episodes=[])
diff --git a/lerobot/tests/configs/test_plugin_loading.py b/lerobot/tests/configs/test_plugin_loading.py
new file mode 100644
index 0000000000000000000000000000000000000000..3ec60a4858eaa7151e972587d947fc260326bf53
--- /dev/null
+++ b/lerobot/tests/configs/test_plugin_loading.py
@@ -0,0 +1,105 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import sys
+from collections.abc import Generator
+from dataclasses import dataclass
+from pathlib import Path
+
+import pytest
+
+from lerobot.configs.parser import PluginLoadError, load_plugin, parse_plugin_args, wrap
+from lerobot.envs.configs import EnvConfig
+
+
+def create_plugin_code(*, base_class: str = "EnvConfig", plugin_name: str = "test_env") -> str:
+    """Creates a dummy plugin module that implements its own EnvConfig subclass."""
+    return f"""
+from dataclasses import dataclass
+from lerobot.envs.configs import {base_class}
+
+@{base_class}.register_subclass("{plugin_name}")
+@dataclass
+class TestPluginConfig:
+    value: int = 42
+    """
+
+
+@pytest.fixture
+def plugin_dir(tmp_path: Path) -> Generator[Path, None, None]:
+    """Creates a temporary plugin package structure."""
+    plugin_pkg = tmp_path / "test_plugin"
+    plugin_pkg.mkdir()
+    (plugin_pkg / "__init__.py").touch()
+
+    with open(plugin_pkg / "my_plugin.py", "w") as f:
+        f.write(create_plugin_code())
+
+    # Add tmp_path to Python path so we can import from it
+    sys.path.insert(0, str(tmp_path))
+    yield plugin_pkg
+    sys.path.pop(0)
+
+
+def test_parse_plugin_args():
+    cli_args = [
+        "--env.type=test",
+        "--model.discover_packages_path=some.package",
+        "--env.discover_packages_path=other.package",
+    ]
+    plugin_args = parse_plugin_args("discover_packages_path", cli_args)
+    assert plugin_args == {
+        "model.discover_packages_path": "some.package",
+        "env.discover_packages_path": "other.package",
+    }
+
+
+def test_load_plugin_success(plugin_dir: Path):
+    # Import should work and register the plugin with the real EnvConfig
+    load_plugin("test_plugin")
+
+    assert "test_env" in EnvConfig.get_known_choices()
+    plugin_cls = EnvConfig.get_choice_class("test_env")
+    plugin_instance = plugin_cls()
+    assert plugin_instance.value == 42
+
+
+def test_load_plugin_failure():
+    with pytest.raises(PluginLoadError) as exc_info:
+        load_plugin("nonexistent_plugin")
+    assert "Failed to load plugin 'nonexistent_plugin'" in str(exc_info.value)
+
+
+def test_wrap_with_plugin(plugin_dir: Path):
+    @dataclass
+    class Config:
+        env: EnvConfig
+
+    @wrap()
+    def dummy_func(cfg: Config):
+        return cfg
+
+    # Test loading plugin via CLI args
+    sys.argv = [
+        "dummy_script.py",
+        "--env.discover_packages_path=test_plugin",
+        "--env.type=test_env",
+    ]
+
+    cfg = dummy_func()
+    assert isinstance(cfg, Config)
+    assert isinstance(cfg.env, EnvConfig.get_choice_class("test_env"))
+    assert cfg.env.value == 42
diff --git a/lerobot/tests/conftest.py b/lerobot/tests/conftest.py
new file mode 100644
index 0000000000000000000000000000000000000000..2fcf878abdbe32f9cccd7fa4c0c61b71561c5d45
--- /dev/null
+++ b/lerobot/tests/conftest.py
@@ -0,0 +1,90 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import traceback
+
+import pytest
+from serial import SerialException
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from tests.utils import DEVICE
+
+# Import fixture modules as plugins
+pytest_plugins = [
+    "tests.fixtures.dataset_factories",
+    "tests.fixtures.files",
+    "tests.fixtures.hub",
+    "tests.fixtures.optimizers",
+]
+
+
+def pytest_collection_finish():
+    print(f"\nTesting with {DEVICE=}")
+
+
+def _check_component_availability(component_type, available_components, make_component):
+    """Generic helper to check if a hardware component is available"""
+    if component_type not in available_components:
+        raise ValueError(
+            f"The {component_type} type is not valid. Expected one of these '{available_components}'"
+        )
+
+    try:
+        component = make_component(component_type)
+        component.connect()
+        del component
+        return True
+
+    except Exception as e:
+        print(f"\nA {component_type} is not available.")
+
+        if isinstance(e, ModuleNotFoundError):
+            print(f"\nInstall module '{e.name}'")
+        elif isinstance(e, SerialException):
+            print("\nNo physical device detected.")
+        elif isinstance(e, ValueError) and "camera_index" in str(e):
+            print("\nNo physical camera detected.")
+        else:
+            traceback.print_exc()
+
+        return False
+
+
+@pytest.fixture
+def patch_builtins_input(monkeypatch):
+    def print_text(text=None):
+        if text is not None:
+            print(text)
+
+    monkeypatch.setattr("builtins.input", print_text)
+
+
+@pytest.fixture
+def policy_feature_factory():
+    """PolicyFeature factory"""
+
+    def _pf(ft: FeatureType, shape: tuple[int, ...]) -> PolicyFeature:
+        return PolicyFeature(type=ft, shape=shape)
+
+    return _pf
+
+
+def assert_contract_is_typed(features: dict[PipelineFeatureType, dict[str, PolicyFeature]]) -> None:
+    assert isinstance(features, dict)
+    assert all(isinstance(k, PipelineFeatureType) for k in features)
+    assert all(isinstance(v, dict) for v in features.values())
+    assert all(all(isinstance(nk, str) for nk in v) for v in features.values())
+    assert all(all(isinstance(nv, PolicyFeature) for nv in v.values()) for v in features.values())
diff --git a/lerobot/tests/datasets/test_aggregate.py b/lerobot/tests/datasets/test_aggregate.py
new file mode 100644
index 0000000000000000000000000000000000000000..4ac7e001a4c656cdde1def66dc589ac44909ad98
--- /dev/null
+++ b/lerobot/tests/datasets/test_aggregate.py
@@ -0,0 +1,616 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from unittest.mock import patch
+
+import datasets
+import torch
+
+from lerobot.datasets.aggregate import aggregate_datasets
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from tests.fixtures.constants import DUMMY_REPO_ID
+
+
+def assert_episode_and_frame_counts(aggr_ds, expected_episodes, expected_frames):
+    """Test that total number of episodes and frames are correctly aggregated."""
+    assert aggr_ds.num_episodes == expected_episodes, (
+        f"Expected {expected_episodes} episodes, got {aggr_ds.num_episodes}"
+    )
+    assert aggr_ds.num_frames == expected_frames, (
+        f"Expected {expected_frames} frames, got {aggr_ds.num_frames}"
+    )
+
+
+def assert_dataset_content_integrity(aggr_ds, ds_0, ds_1):
+    """Test that the content of both datasets is preserved correctly in the aggregated dataset."""
+    keys_to_ignore = ["episode_index", "index", "timestamp"]
+
+    # Test first part of dataset corresponds to ds_0, check first item (index 0) matches ds_0[0]
+    aggr_first_item = aggr_ds[0]
+    ds_0_first_item = ds_0[0]
+
+    # Compare all keys except episode_index and index which should be updated
+    for key in ds_0_first_item:
+        if key not in keys_to_ignore:
+            # Handle both tensor and non-tensor data
+            if torch.is_tensor(aggr_first_item[key]) and torch.is_tensor(ds_0_first_item[key]):
+                assert torch.allclose(aggr_first_item[key], ds_0_first_item[key], atol=1e-6), (
+                    f"First item key '{key}' doesn't match between aggregated and ds_0"
+                )
+            else:
+                assert aggr_first_item[key] == ds_0_first_item[key], (
+                    f"First item key '{key}' doesn't match between aggregated and ds_0"
+                )
+
+    # Check last item of ds_0 part (index len(ds_0)-1) matches ds_0[-1]
+    aggr_ds_0_last_item = aggr_ds[len(ds_0) - 1]
+    ds_0_last_item = ds_0[-1]
+
+    for key in ds_0_last_item:
+        if key not in keys_to_ignore:
+            # Handle both tensor and non-tensor data
+            if torch.is_tensor(aggr_ds_0_last_item[key]) and torch.is_tensor(ds_0_last_item[key]):
+                assert torch.allclose(aggr_ds_0_last_item[key], ds_0_last_item[key], atol=1e-6), (
+                    f"Last ds_0 item key '{key}' doesn't match between aggregated and ds_0"
+                )
+            else:
+                assert aggr_ds_0_last_item[key] == ds_0_last_item[key], (
+                    f"Last ds_0 item key '{key}' doesn't match between aggregated and ds_0"
+                )
+
+    # Test second part of dataset corresponds to ds_1
+    # Check first item of ds_1 part (index len(ds_0)) matches ds_1[0]
+    aggr_ds_1_first_item = aggr_ds[len(ds_0)]
+    ds_1_first_item = ds_1[0]
+
+    for key in ds_1_first_item:
+        if key not in keys_to_ignore:
+            # Handle both tensor and non-tensor data
+            if torch.is_tensor(aggr_ds_1_first_item[key]) and torch.is_tensor(ds_1_first_item[key]):
+                assert torch.allclose(aggr_ds_1_first_item[key], ds_1_first_item[key], atol=1e-6), (
+                    f"First ds_1 item key '{key}' doesn't match between aggregated and ds_1"
+                )
+            else:
+                assert aggr_ds_1_first_item[key] == ds_1_first_item[key], (
+                    f"First ds_1 item key '{key}' doesn't match between aggregated and ds_1"
+                )
+
+    # Check last item matches ds_1[-1]
+    aggr_last_item = aggr_ds[-1]
+    ds_1_last_item = ds_1[-1]
+
+    for key in ds_1_last_item:
+        if key not in keys_to_ignore:
+            # Handle both tensor and non-tensor data
+            if torch.is_tensor(aggr_last_item[key]) and torch.is_tensor(ds_1_last_item[key]):
+                assert torch.allclose(aggr_last_item[key], ds_1_last_item[key], atol=1e-6), (
+                    f"Last item key '{key}' doesn't match between aggregated and ds_1"
+                )
+            else:
+                assert aggr_last_item[key] == ds_1_last_item[key], (
+                    f"Last item key '{key}' doesn't match between aggregated and ds_1"
+                )
+
+
+def assert_metadata_consistency(aggr_ds, ds_0, ds_1):
+    """Test that metadata is correctly aggregated."""
+    # Test basic info
+    assert aggr_ds.fps == ds_0.fps == ds_1.fps, "FPS should be the same across all datasets"
+    assert aggr_ds.meta.info["robot_type"] == ds_0.meta.info["robot_type"] == ds_1.meta.info["robot_type"], (
+        "Robot type should be the same"
+    )
+
+    # Test features are the same
+    assert aggr_ds.features == ds_0.features == ds_1.features, "Features should be the same"
+
+    # Test tasks aggregation
+    expected_tasks = set(ds_0.meta.tasks.index) | set(ds_1.meta.tasks.index)
+    actual_tasks = set(aggr_ds.meta.tasks.index)
+    assert actual_tasks == expected_tasks, f"Expected tasks {expected_tasks}, got {actual_tasks}"
+
+
+def assert_episode_indices_updated_correctly(aggr_ds, ds_0, ds_1):
+    """Test that episode indices are correctly updated after aggregation."""
+    # ds_0 episodes should have episode_index 0 to ds_0.num_episodes-1
+    for i in range(len(ds_0)):
+        assert aggr_ds[i]["episode_index"] < ds_0.num_episodes, (
+            f"Episode index {aggr_ds[i]['episode_index']} at position {i} should be < {ds_0.num_episodes}"
+        )
+
+    def ds1_episodes_condition(ep_idx):
+        return (ep_idx >= ds_0.num_episodes) and (ep_idx < ds_0.num_episodes + ds_1.num_episodes)
+
+    # ds_1 episodes should have episode_index ds_0.num_episodes to total_episodes-1
+    for i in range(len(ds_0), len(ds_0) + len(ds_1)):
+        expected_min_episode_idx = ds_0.num_episodes
+        assert ds1_episodes_condition(aggr_ds[i]["episode_index"]), (
+            f"Episode index {aggr_ds[i]['episode_index']} at position {i} should be >= {expected_min_episode_idx}"
+        )
+
+
+def assert_video_frames_integrity(aggr_ds, ds_0, ds_1):
+    """Test that video frames are correctly preserved and frame indices are updated."""
+
+    def visual_frames_equal(frame1, frame2):
+        return torch.allclose(frame1, frame2)
+
+    video_keys = list(
+        filter(
+            lambda key: aggr_ds.meta.info["features"][key]["dtype"] == "video",
+            aggr_ds.meta.info["features"].keys(),
+        )
+    )
+
+    # Test the section corresponding to the first dataset (ds_0)
+    for i in range(len(ds_0)):
+        assert aggr_ds[i]["index"] == i, (
+            f"Frame index at position {i} should be {i}, but got {aggr_ds[i]['index']}"
+        )
+        for key in video_keys:
+            assert visual_frames_equal(aggr_ds[i][key], ds_0[i][key]), (
+                f"Visual frames at position {i} should be equal between aggregated and ds_0"
+            )
+
+    # Test the section corresponding to the second dataset (ds_1)
+    for i in range(len(ds_0), len(ds_0) + len(ds_1)):
+        # The frame index in the aggregated dataset should also match its position.
+        assert aggr_ds[i]["index"] == i, (
+            f"Frame index at position {i} should be {i}, but got {aggr_ds[i]['index']}"
+        )
+        for key in video_keys:
+            assert visual_frames_equal(aggr_ds[i][key], ds_1[i - len(ds_0)][key]), (
+                f"Visual frames at position {i} should be equal between aggregated and ds_1"
+            )
+
+
+def assert_dataset_iteration_works(aggr_ds):
+    """Test that we can iterate through the entire dataset without errors."""
+    for _ in aggr_ds:
+        pass
+
+
+def assert_video_timestamps_within_bounds(aggr_ds):
+    """Test that all video timestamps are within valid bounds for their respective video files.
+
+    This catches bugs where timestamps point to frames beyond the actual video length,
+    which would cause "Invalid frame index" errors during data loading.
+    """
+    try:
+        from torchcodec.decoders import VideoDecoder
+    except ImportError:
+        return
+
+    for ep_idx in range(aggr_ds.num_episodes):
+        ep = aggr_ds.meta.episodes[ep_idx]
+
+        for vid_key in aggr_ds.meta.video_keys:
+            from_ts = ep[f"videos/{vid_key}/from_timestamp"]
+            to_ts = ep[f"videos/{vid_key}/to_timestamp"]
+            video_path = aggr_ds.root / aggr_ds.meta.get_video_file_path(ep_idx, vid_key)
+
+            if not video_path.exists():
+                continue
+
+            from_frame_idx = round(from_ts * aggr_ds.fps)
+            to_frame_idx = round(to_ts * aggr_ds.fps)
+
+            try:
+                decoder = VideoDecoder(str(video_path))
+                num_frames = len(decoder)
+
+                # Verify timestamps don't exceed video bounds
+                assert from_frame_idx >= 0, (
+                    f"Episode {ep_idx}, {vid_key}: from_frame_idx ({from_frame_idx}) < 0"
+                )
+                assert from_frame_idx < num_frames, (
+                    f"Episode {ep_idx}, {vid_key}: from_frame_idx ({from_frame_idx}) >= video frames ({num_frames})"
+                )
+                assert to_frame_idx <= num_frames, (
+                    f"Episode {ep_idx}, {vid_key}: to_frame_idx ({to_frame_idx}) > video frames ({num_frames})"
+                )
+                assert from_frame_idx < to_frame_idx, (
+                    f"Episode {ep_idx}, {vid_key}: from_frame_idx ({from_frame_idx}) >= to_frame_idx ({to_frame_idx})"
+                )
+            except Exception as e:
+                raise AssertionError(
+                    f"Failed to verify timestamps for episode {ep_idx}, {vid_key}: {e}"
+                ) from e
+
+
+def test_aggregate_datasets(tmp_path, lerobot_dataset_factory):
+    """Test basic aggregation functionality with standard parameters."""
+    ds_0_num_frames = 400
+    ds_1_num_frames = 800
+    ds_0_num_episodes = 10
+    ds_1_num_episodes = 25
+
+    # Create two datasets with different number of frames and episodes
+    ds_0 = lerobot_dataset_factory(
+        root=tmp_path / "test_0",
+        repo_id=f"{DUMMY_REPO_ID}_0",
+        total_episodes=ds_0_num_episodes,
+        total_frames=ds_0_num_frames,
+    )
+    ds_1 = lerobot_dataset_factory(
+        root=tmp_path / "test_1",
+        repo_id=f"{DUMMY_REPO_ID}_1",
+        total_episodes=ds_1_num_episodes,
+        total_frames=ds_1_num_frames,
+    )
+
+    aggregate_datasets(
+        repo_ids=[ds_0.repo_id, ds_1.repo_id],
+        roots=[ds_0.root, ds_1.root],
+        aggr_repo_id=f"{DUMMY_REPO_ID}_aggr",
+        aggr_root=tmp_path / "test_aggr",
+    )
+
+    # Mock the revision to prevent Hub calls during dataset loading
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "test_aggr")
+        aggr_ds = LeRobotDataset(f"{DUMMY_REPO_ID}_aggr", root=tmp_path / "test_aggr")
+
+    # Run all assertion functions
+    expected_total_episodes = ds_0.num_episodes + ds_1.num_episodes
+    expected_total_frames = ds_0.num_frames + ds_1.num_frames
+
+    assert_episode_and_frame_counts(aggr_ds, expected_total_episodes, expected_total_frames)
+    assert_dataset_content_integrity(aggr_ds, ds_0, ds_1)
+    assert_metadata_consistency(aggr_ds, ds_0, ds_1)
+    assert_episode_indices_updated_correctly(aggr_ds, ds_0, ds_1)
+    assert_video_frames_integrity(aggr_ds, ds_0, ds_1)
+    assert_video_timestamps_within_bounds(aggr_ds)
+    assert_dataset_iteration_works(aggr_ds)
+
+
+def test_aggregate_with_low_threshold(tmp_path, lerobot_dataset_factory):
+    """Test aggregation with small file size limits to force file rotation/sharding."""
+    ds_0_num_episodes = ds_1_num_episodes = 10
+    ds_0_num_frames = ds_1_num_frames = 400
+
+    ds_0 = lerobot_dataset_factory(
+        root=tmp_path / "small_0",
+        repo_id=f"{DUMMY_REPO_ID}_small_0",
+        total_episodes=ds_0_num_episodes,
+        total_frames=ds_0_num_frames,
+    )
+    ds_1 = lerobot_dataset_factory(
+        root=tmp_path / "small_1",
+        repo_id=f"{DUMMY_REPO_ID}_small_1",
+        total_episodes=ds_1_num_episodes,
+        total_frames=ds_1_num_frames,
+    )
+
+    # Use the new configurable parameters to force file rotation
+    aggregate_datasets(
+        repo_ids=[ds_0.repo_id, ds_1.repo_id],
+        roots=[ds_0.root, ds_1.root],
+        aggr_repo_id=f"{DUMMY_REPO_ID}_small_aggr",
+        aggr_root=tmp_path / "small_aggr",
+        # Tiny file size to trigger new file instantiation
+        data_files_size_in_mb=0.01,
+        video_files_size_in_mb=0.1,
+    )
+
+    # Mock the revision to prevent Hub calls during dataset loading
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "small_aggr")
+        aggr_ds = LeRobotDataset(f"{DUMMY_REPO_ID}_small_aggr", root=tmp_path / "small_aggr")
+
+    # Verify aggregation worked correctly despite file size constraints
+    expected_total_episodes = ds_0_num_episodes + ds_1_num_episodes
+    expected_total_frames = ds_0_num_frames + ds_1_num_frames
+
+    assert_episode_and_frame_counts(aggr_ds, expected_total_episodes, expected_total_frames)
+    assert_dataset_content_integrity(aggr_ds, ds_0, ds_1)
+    assert_metadata_consistency(aggr_ds, ds_0, ds_1)
+    assert_episode_indices_updated_correctly(aggr_ds, ds_0, ds_1)
+    assert_video_frames_integrity(aggr_ds, ds_0, ds_1)
+    assert_video_timestamps_within_bounds(aggr_ds)
+    assert_dataset_iteration_works(aggr_ds)
+
+    # Check that multiple files were actually created due to small size limits
+    data_dir = tmp_path / "small_aggr" / "data"
+    video_dir = tmp_path / "small_aggr" / "videos"
+
+    if data_dir.exists():
+        parquet_files = list(data_dir.rglob("*.parquet"))
+        assert len(parquet_files) > 1, "Small file size limits should create multiple parquet files"
+
+    if video_dir.exists():
+        video_files = list(video_dir.rglob("*.mp4"))
+        assert len(video_files) > 1, "Small file size limits should create multiple video files"
+
+
+def test_video_timestamps_regression(tmp_path, lerobot_dataset_factory):
+    """Regression test for video timestamp bug when merging datasets.
+
+    This test specifically checks that video timestamps are correctly calculated
+    and accumulated when merging multiple datasets.
+    """
+    datasets = []
+    for i in range(3):
+        ds = lerobot_dataset_factory(
+            root=tmp_path / f"regression_{i}",
+            repo_id=f"{DUMMY_REPO_ID}_regression_{i}",
+            total_episodes=2,
+            total_frames=100,
+        )
+        datasets.append(ds)
+
+    aggregate_datasets(
+        repo_ids=[ds.repo_id for ds in datasets],
+        roots=[ds.root for ds in datasets],
+        aggr_repo_id=f"{DUMMY_REPO_ID}_regression_aggr",
+        aggr_root=tmp_path / "regression_aggr",
+    )
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "regression_aggr")
+        aggr_ds = LeRobotDataset(f"{DUMMY_REPO_ID}_regression_aggr", root=tmp_path / "regression_aggr")
+
+    assert_video_timestamps_within_bounds(aggr_ds)
+
+    for i in range(len(aggr_ds)):
+        item = aggr_ds[i]
+        for key in aggr_ds.meta.video_keys:
+            assert key in item, f"Video key {key} missing from item {i}"
+            assert item[key].shape[0] == 3, f"Expected 3 channels for video key {key}"
+
+
+def assert_image_schema_preserved(aggr_ds):
+    """Test that HuggingFace Image feature schema is preserved in aggregated parquet files.
+
+    This verifies the fix for a bug where image columns were written with a generic
+    struct schema {'bytes': Value('binary'), 'path': Value('string')} instead of
+    the proper Image() feature type, causing HuggingFace Hub viewer to display
+    raw dict objects instead of image thumbnails.
+    """
+    image_keys = aggr_ds.meta.image_keys
+    if not image_keys:
+        return
+
+    # Check that parquet files have proper Image schema
+    data_dir = aggr_ds.root / "data"
+    parquet_files = list(data_dir.rglob("*.parquet"))
+    assert len(parquet_files) > 0, "No parquet files found in aggregated dataset"
+
+    for parquet_file in parquet_files:
+        # Load with HuggingFace datasets to check schema
+        ds = datasets.Dataset.from_parquet(str(parquet_file))
+
+        for image_key in image_keys:
+            feature = ds.features.get(image_key)
+            assert feature is not None, f"Image key '{image_key}' not found in parquet schema"
+            assert isinstance(feature, datasets.Image), (
+                f"Image key '{image_key}' should have Image() feature type, "
+                f"but got {type(feature).__name__}: {feature}. "
+                "This indicates image schema was not preserved during aggregation."
+            )
+
+
+def assert_image_frames_integrity(aggr_ds, ds_0, ds_1):
+    """Test that image frames are correctly preserved after aggregation."""
+    image_keys = aggr_ds.meta.image_keys
+    if not image_keys:
+        return
+
+    def images_equal(img1, img2):
+        return torch.allclose(img1, img2)
+
+    # Test the section corresponding to the first dataset (ds_0)
+    for i in range(len(ds_0)):
+        assert aggr_ds[i]["index"] == i, (
+            f"Frame index at position {i} should be {i}, but got {aggr_ds[i]['index']}"
+        )
+        for key in image_keys:
+            assert images_equal(aggr_ds[i][key], ds_0[i][key]), (
+                f"Image frames at position {i} should be equal between aggregated and ds_0"
+            )
+
+    # Test the section corresponding to the second dataset (ds_1)
+    for i in range(len(ds_0), len(ds_0) + len(ds_1)):
+        assert aggr_ds[i]["index"] == i, (
+            f"Frame index at position {i} should be {i}, but got {aggr_ds[i]['index']}"
+        )
+        for key in image_keys:
+            assert images_equal(aggr_ds[i][key], ds_1[i - len(ds_0)][key]), (
+                f"Image frames at position {i} should be equal between aggregated and ds_1"
+            )
+
+
+def test_aggregate_image_datasets(tmp_path, lerobot_dataset_factory):
+    """Test aggregation of image-based datasets preserves HuggingFace Image schema.
+
+    This test specifically verifies that:
+    1. Image-based datasets can be aggregated correctly
+    2. The HuggingFace Image() feature type is preserved in parquet files
+    3. Image data integrity is maintained across aggregation
+    4. Images can be properly decoded after aggregation
+
+    This catches the bug where to_parquet_with_hf_images() was not passing
+    the features schema, causing image columns to be written as generic
+    struct types instead of Image() types.
+    """
+    ds_0_num_frames = 50
+    ds_1_num_frames = 75
+    ds_0_num_episodes = 2
+    ds_1_num_episodes = 3
+
+    # Create two image-based datasets (use_videos=False)
+    ds_0 = lerobot_dataset_factory(
+        root=tmp_path / "image_0",
+        repo_id=f"{DUMMY_REPO_ID}_image_0",
+        total_episodes=ds_0_num_episodes,
+        total_frames=ds_0_num_frames,
+        use_videos=False,  # Image-based dataset
+    )
+    ds_1 = lerobot_dataset_factory(
+        root=tmp_path / "image_1",
+        repo_id=f"{DUMMY_REPO_ID}_image_1",
+        total_episodes=ds_1_num_episodes,
+        total_frames=ds_1_num_frames,
+        use_videos=False,  # Image-based dataset
+    )
+
+    # Verify source datasets have image keys
+    assert len(ds_0.meta.image_keys) > 0, "ds_0 should have image keys"
+    assert len(ds_1.meta.image_keys) > 0, "ds_1 should have image keys"
+
+    # Aggregate the datasets
+    aggregate_datasets(
+        repo_ids=[ds_0.repo_id, ds_1.repo_id],
+        roots=[ds_0.root, ds_1.root],
+        aggr_repo_id=f"{DUMMY_REPO_ID}_image_aggr",
+        aggr_root=tmp_path / "image_aggr",
+    )
+
+    # Load the aggregated dataset
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "image_aggr")
+        aggr_ds = LeRobotDataset(f"{DUMMY_REPO_ID}_image_aggr", root=tmp_path / "image_aggr")
+
+    # Verify aggregated dataset has image keys
+    assert len(aggr_ds.meta.image_keys) > 0, "Aggregated dataset should have image keys"
+    assert aggr_ds.meta.image_keys == ds_0.meta.image_keys, "Image keys should match source datasets"
+
+    # Run standard aggregation assertions
+    expected_total_episodes = ds_0_num_episodes + ds_1_num_episodes
+    expected_total_frames = ds_0_num_frames + ds_1_num_frames
+
+    assert_episode_and_frame_counts(aggr_ds, expected_total_episodes, expected_total_frames)
+    assert_dataset_content_integrity(aggr_ds, ds_0, ds_1)
+    assert_metadata_consistency(aggr_ds, ds_0, ds_1)
+    assert_episode_indices_updated_correctly(aggr_ds, ds_0, ds_1)
+
+    # Image-specific assertions
+    assert_image_schema_preserved(aggr_ds)
+    assert_image_frames_integrity(aggr_ds, ds_0, ds_1)
+
+    # Verify images can be accessed and have correct shape
+    sample_item = aggr_ds[0]
+    for image_key in aggr_ds.meta.image_keys:
+        img = sample_item[image_key]
+        assert isinstance(img, torch.Tensor), f"Image {image_key} should be a tensor"
+        assert img.dim() == 3, f"Image {image_key} should have 3 dimensions (C, H, W)"
+        assert img.shape[0] == 3, f"Image {image_key} should have 3 channels"
+
+    assert_dataset_iteration_works(aggr_ds)
+
+
+def test_aggregate_already_merged_dataset(tmp_path, lerobot_dataset_factory):
+    """Regression test for aggregating a dataset that is itself a result of a previous merge.
+
+    This test reproduces the bug where merging datasets with multiple parquet files
+    (e.g., from a previous merge with file rotation) would cause FileNotFoundError
+    because metadata file indices were incorrectly preserved instead of being mapped
+    to their actual destination files.
+
+    The fix adds src_to_dst tracking in aggregate_data() to correctly map source
+    file indices to destination file indices.
+    """
+    # Step 1: Create datasets A and B
+    ds_a = lerobot_dataset_factory(
+        root=tmp_path / "ds_a",
+        repo_id=f"{DUMMY_REPO_ID}_a",
+        total_episodes=4,
+        total_frames=200,
+    )
+    ds_b = lerobot_dataset_factory(
+        root=tmp_path / "ds_b",
+        repo_id=f"{DUMMY_REPO_ID}_b",
+        total_episodes=4,
+        total_frames=200,
+    )
+
+    # Step 2: Merge A+B into AB with small file size to force multiple files
+    aggregate_datasets(
+        repo_ids=[ds_a.repo_id, ds_b.repo_id],
+        roots=[ds_a.root, ds_b.root],
+        aggr_repo_id=f"{DUMMY_REPO_ID}_ab",
+        aggr_root=tmp_path / "ds_ab",
+        data_files_size_in_mb=0.01,  # Force file rotation
+    )
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "ds_ab")
+        ds_ab = LeRobotDataset(f"{DUMMY_REPO_ID}_ab", root=tmp_path / "ds_ab")
+
+    # Verify AB has multiple data files (file rotation occurred)
+    ab_data_files = list((tmp_path / "ds_ab" / "data").rglob("*.parquet"))
+    assert len(ab_data_files) > 1, "First merge should create multiple parquet files"
+
+    # Step 3: Create dataset C
+    ds_c = lerobot_dataset_factory(
+        root=tmp_path / "ds_c",
+        repo_id=f"{DUMMY_REPO_ID}_c",
+        total_episodes=2,
+        total_frames=100,
+    )
+
+    # Step 4: Merge AB+C into final - THIS IS WHERE THE BUG OCCURRED
+    aggregate_datasets(
+        repo_ids=[ds_ab.repo_id, ds_c.repo_id],
+        roots=[ds_ab.root, ds_c.root],
+        aggr_repo_id=f"{DUMMY_REPO_ID}_abc",
+        aggr_root=tmp_path / "ds_abc",
+    )
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "ds_abc")
+        ds_abc = LeRobotDataset(f"{DUMMY_REPO_ID}_abc", root=tmp_path / "ds_abc")
+
+    # Step 5: Verify all data files referenced in metadata actually exist
+    for ep_idx in range(ds_abc.num_episodes):
+        data_file_path = ds_abc.root / ds_abc.meta.get_data_file_path(ep_idx)
+        assert data_file_path.exists(), (
+            f"Episode {ep_idx} references non-existent file: {data_file_path}\n"
+            "This indicates the src_to_dst mapping fix is not working correctly."
+        )
+
+    # Step 6: Verify we can iterate through the entire dataset without FileNotFoundError
+    expected_episodes = ds_a.num_episodes + ds_b.num_episodes + ds_c.num_episodes
+    expected_frames = ds_a.num_frames + ds_b.num_frames + ds_c.num_frames
+
+    assert ds_abc.num_episodes == expected_episodes
+    assert ds_abc.num_frames == expected_frames
+
+    # This would raise FileNotFoundError before the fix
+    assert_dataset_iteration_works(ds_abc)
diff --git a/lerobot/tests/datasets/test_compute_stats.py b/lerobot/tests/datasets/test_compute_stats.py
new file mode 100644
index 0000000000000000000000000000000000000000..973c80bd891686ec3a601e7fad65847d63bb0d2b
--- /dev/null
+++ b/lerobot/tests/datasets/test_compute_stats.py
@@ -0,0 +1,834 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from unittest.mock import patch
+
+import numpy as np
+import pytest
+
+from lerobot.datasets.compute_stats import (
+    RunningQuantileStats,
+    _assert_type_and_shape,
+    aggregate_feature_stats,
+    aggregate_stats,
+    compute_episode_stats,
+    estimate_num_samples,
+    get_feature_stats,
+    sample_images,
+    sample_indices,
+)
+from lerobot.utils.constants import OBS_IMAGE, OBS_STATE
+
+
+def mock_load_image_as_numpy(path, dtype, channel_first):
+    return np.ones((3, 32, 32), dtype=dtype) if channel_first else np.ones((32, 32, 3), dtype=dtype)
+
+
+@pytest.fixture
+def sample_array():
+    return np.array([[1, 2, 3], [4, 5, 6], [7, 8, 9]])
+
+
+def test_estimate_num_samples():
+    assert estimate_num_samples(1) == 1
+    assert estimate_num_samples(10) == 10
+    assert estimate_num_samples(100) == 100
+    assert estimate_num_samples(200) == 100
+    assert estimate_num_samples(1000) == 177
+    assert estimate_num_samples(2000) == 299
+    assert estimate_num_samples(5000) == 594
+    assert estimate_num_samples(10_000) == 1000
+    assert estimate_num_samples(20_000) == 1681
+    assert estimate_num_samples(50_000) == 3343
+    assert estimate_num_samples(500_000) == 10_000
+
+
+def test_sample_indices():
+    indices = sample_indices(10)
+    assert len(indices) > 0
+    assert indices[0] == 0
+    assert indices[-1] == 9
+    assert len(indices) == estimate_num_samples(10)
+
+
+@patch("lerobot.datasets.compute_stats.load_image_as_numpy", side_effect=mock_load_image_as_numpy)
+def test_sample_images(mock_load):
+    image_paths = [f"image_{i}.jpg" for i in range(100)]
+    images = sample_images(image_paths)
+    assert isinstance(images, np.ndarray)
+    assert images.shape[1:] == (3, 32, 32)
+    assert images.dtype == np.uint8
+    assert len(images) == estimate_num_samples(100)
+
+
+def test_get_feature_stats_images():
+    data = np.random.rand(100, 3, 32, 32)
+    stats = get_feature_stats(data, axis=(0, 2, 3), keepdims=True)
+    assert "min" in stats and "max" in stats and "mean" in stats and "std" in stats and "count" in stats
+    np.testing.assert_equal(stats["count"], np.array([100]))
+    assert stats["min"].shape == stats["max"].shape == stats["mean"].shape == stats["std"].shape
+
+
+def test_get_feature_stats_axis_0_keepdims(sample_array):
+    expected = {
+        "min": np.array([[1, 2, 3]]),
+        "max": np.array([[7, 8, 9]]),
+        "mean": np.array([[4.0, 5.0, 6.0]]),
+        "std": np.array([[2.44948974, 2.44948974, 2.44948974]]),
+        "count": np.array([3]),
+    }
+    result = get_feature_stats(sample_array, axis=(0,), keepdims=True)
+    for key in expected:
+        np.testing.assert_allclose(result[key], expected[key])
+
+
+def test_get_feature_stats_axis_1(sample_array):
+    expected = {
+        "min": np.array([1, 4, 7]),
+        "max": np.array([3, 6, 9]),
+        "mean": np.array([2.0, 5.0, 8.0]),
+        "std": np.array([0.81649658, 0.81649658, 0.81649658]),
+        "count": np.array([3]),
+    }
+    result = get_feature_stats(sample_array, axis=(1,), keepdims=False)
+
+    # Check that basic stats are correct (quantiles are also included now)
+    assert set(expected.keys()).issubset(set(result.keys()))
+    for key in expected:
+        np.testing.assert_allclose(result[key], expected[key])
+
+
+def test_get_feature_stats_no_axis(sample_array):
+    expected = {
+        "min": np.array(1),
+        "max": np.array(9),
+        "mean": np.array(5.0),
+        "std": np.array(2.5819889),
+        "count": np.array([3]),
+    }
+    result = get_feature_stats(sample_array, axis=None, keepdims=False)
+
+    # Check that basic stats are correct (quantiles are also included now)
+    assert set(expected.keys()).issubset(set(result.keys()))
+    for key in expected:
+        np.testing.assert_allclose(result[key], expected[key])
+
+
+def test_get_feature_stats_empty_array():
+    array = np.array([])
+    with pytest.raises(ValueError):
+        get_feature_stats(array, axis=(0,), keepdims=True)
+
+
+def test_get_feature_stats_single_value():
+    array = np.array([[1337]])
+    result = get_feature_stats(array, axis=None, keepdims=True)
+    np.testing.assert_equal(result["min"], np.array(1337))
+    np.testing.assert_equal(result["max"], np.array(1337))
+    np.testing.assert_equal(result["mean"], np.array(1337.0))
+    np.testing.assert_equal(result["std"], np.array(0.0))
+    np.testing.assert_equal(result["count"], np.array([1]))
+
+
+def test_compute_episode_stats():
+    episode_data = {
+        OBS_IMAGE: [f"image_{i}.jpg" for i in range(100)],
+        OBS_STATE: np.random.rand(100, 10),
+    }
+    features = {
+        OBS_IMAGE: {"dtype": "image"},
+        OBS_STATE: {"dtype": "numeric"},
+    }
+
+    with patch("lerobot.datasets.compute_stats.load_image_as_numpy", side_effect=mock_load_image_as_numpy):
+        stats = compute_episode_stats(episode_data, features)
+
+    assert OBS_IMAGE in stats and OBS_STATE in stats
+    assert stats[OBS_IMAGE]["count"].item() == 100
+    assert stats[OBS_STATE]["count"].item() == 100
+    assert stats[OBS_IMAGE]["mean"].shape == (3, 1, 1)
+
+
+def test_assert_type_and_shape_valid():
+    valid_stats = [
+        {
+            "feature1": {
+                "min": np.array([1.0]),
+                "max": np.array([10.0]),
+                "mean": np.array([5.0]),
+                "std": np.array([2.0]),
+                "count": np.array([1]),
+            }
+        }
+    ]
+    _assert_type_and_shape(valid_stats)
+
+
+def test_assert_type_and_shape_invalid_type():
+    invalid_stats = [
+        {
+            "feature1": {
+                "min": [1.0],  # Not a numpy array
+                "max": np.array([10.0]),
+                "mean": np.array([5.0]),
+                "std": np.array([2.0]),
+                "count": np.array([1]),
+            }
+        }
+    ]
+    with pytest.raises(ValueError, match="Stats must be composed of numpy array"):
+        _assert_type_and_shape(invalid_stats)
+
+
+def test_assert_type_and_shape_invalid_shape():
+    invalid_stats = [
+        {
+            "feature1": {
+                "count": np.array([1, 2]),  # Wrong shape
+            }
+        }
+    ]
+    with pytest.raises(ValueError, match=r"Shape of 'count' must be \(1\)"):
+        _assert_type_and_shape(invalid_stats)
+
+
+def test_aggregate_feature_stats():
+    stats_ft_list = [
+        {
+            "min": np.array([1.0]),
+            "max": np.array([10.0]),
+            "mean": np.array([5.0]),
+            "std": np.array([2.0]),
+            "count": np.array([1]),
+        },
+        {
+            "min": np.array([2.0]),
+            "max": np.array([12.0]),
+            "mean": np.array([6.0]),
+            "std": np.array([2.5]),
+            "count": np.array([1]),
+        },
+    ]
+    result = aggregate_feature_stats(stats_ft_list)
+    np.testing.assert_allclose(result["min"], np.array([1.0]))
+    np.testing.assert_allclose(result["max"], np.array([12.0]))
+    np.testing.assert_allclose(result["mean"], np.array([5.5]))
+    np.testing.assert_allclose(result["std"], np.array([2.318405]), atol=1e-6)
+    np.testing.assert_allclose(result["count"], np.array([2]))
+
+
+def test_aggregate_stats():
+    all_stats = [
+        {
+            OBS_IMAGE: {
+                "min": [1, 2, 3],
+                "max": [10, 20, 30],
+                "mean": [5.5, 10.5, 15.5],
+                "std": [2.87, 5.87, 8.87],
+                "count": 10,
+            },
+            OBS_STATE: {"min": 1, "max": 10, "mean": 5.5, "std": 2.87, "count": 10},
+            "extra_key_0": {"min": 5, "max": 25, "mean": 15, "std": 6, "count": 6},
+        },
+        {
+            OBS_IMAGE: {
+                "min": [2, 1, 0],
+                "max": [15, 10, 5],
+                "mean": [8.5, 5.5, 2.5],
+                "std": [3.42, 2.42, 1.42],
+                "count": 15,
+            },
+            OBS_STATE: {"min": 2, "max": 15, "mean": 8.5, "std": 3.42, "count": 15},
+            "extra_key_1": {"min": 0, "max": 20, "mean": 10, "std": 5, "count": 5},
+        },
+    ]
+
+    expected_agg_stats = {
+        OBS_IMAGE: {
+            "min": [1, 1, 0],
+            "max": [15, 20, 30],
+            "mean": [7.3, 7.5, 7.7],
+            "std": [3.5317, 4.8267, 8.5581],
+            "count": 25,
+        },
+        OBS_STATE: {
+            "min": 1,
+            "max": 15,
+            "mean": 7.3,
+            "std": 3.5317,
+            "count": 25,
+        },
+        "extra_key_0": {
+            "min": 5,
+            "max": 25,
+            "mean": 15.0,
+            "std": 6.0,
+            "count": 6,
+        },
+        "extra_key_1": {
+            "min": 0,
+            "max": 20,
+            "mean": 10.0,
+            "std": 5.0,
+            "count": 5,
+        },
+    }
+
+    # cast to numpy
+    for ep_stats in all_stats:
+        for fkey, stats in ep_stats.items():
+            for k in stats:
+                stats[k] = np.array(stats[k], dtype=np.int64 if k == "count" else np.float32)
+                if fkey == OBS_IMAGE and k != "count":
+                    stats[k] = stats[k].reshape(3, 1, 1)  # for normalization on image channels
+                else:
+                    stats[k] = stats[k].reshape(1)
+
+    # cast to numpy
+    for fkey, stats in expected_agg_stats.items():
+        for k in stats:
+            stats[k] = np.array(stats[k], dtype=np.int64 if k == "count" else np.float32)
+            if fkey == OBS_IMAGE and k != "count":
+                stats[k] = stats[k].reshape(3, 1, 1)  # for normalization on image channels
+            else:
+                stats[k] = stats[k].reshape(1)
+
+    results = aggregate_stats(all_stats)
+
+    for fkey in expected_agg_stats:
+        np.testing.assert_allclose(results[fkey]["min"], expected_agg_stats[fkey]["min"])
+        np.testing.assert_allclose(results[fkey]["max"], expected_agg_stats[fkey]["max"])
+        np.testing.assert_allclose(results[fkey]["mean"], expected_agg_stats[fkey]["mean"])
+        np.testing.assert_allclose(
+            results[fkey]["std"], expected_agg_stats[fkey]["std"], atol=1e-04, rtol=1e-04
+        )
+        np.testing.assert_allclose(results[fkey]["count"], expected_agg_stats[fkey]["count"])
+
+
+def test_running_quantile_stats_initialization():
+    """Test proper initialization of RunningQuantileStats."""
+    running_stats = RunningQuantileStats()
+    assert running_stats._count == 0
+    assert running_stats._mean is None
+    assert running_stats._num_quantile_bins == 5000
+
+    # Test custom bin size
+    running_stats_custom = RunningQuantileStats(num_quantile_bins=1000)
+    assert running_stats_custom._num_quantile_bins == 1000
+
+
+def test_running_quantile_stats_single_batch_update():
+    """Test updating with a single batch."""
+    np.random.seed(42)
+    data = np.random.normal(0, 1, (100, 3))
+
+    running_stats = RunningQuantileStats()
+    running_stats.update(data)
+
+    assert running_stats._count == 100
+    assert running_stats._mean.shape == (3,)
+    assert len(running_stats._histograms) == 3
+    assert len(running_stats._bin_edges) == 3
+
+    # Verify basic statistics are reasonable
+    np.testing.assert_allclose(running_stats._mean, np.mean(data, axis=0), atol=1e-10)
+
+
+def test_running_quantile_stats_multiple_batch_updates():
+    """Test updating with multiple batches."""
+    np.random.seed(42)
+    data1 = np.random.normal(0, 1, (100, 2))
+    data2 = np.random.normal(1, 1, (150, 2))
+
+    running_stats = RunningQuantileStats()
+    running_stats.update(data1)
+    running_stats.update(data2)
+
+    assert running_stats._count == 250
+
+    # Verify running mean is correct
+    combined_data = np.vstack([data1, data2])
+    expected_mean = np.mean(combined_data, axis=0)
+    np.testing.assert_allclose(running_stats._mean, expected_mean, atol=1e-10)
+
+
+def test_running_quantile_stats_get_statistics_basic():
+    """Test getting basic statistics without quantiles."""
+    np.random.seed(42)
+    data = np.random.normal(0, 1, (100, 2))
+
+    running_stats = RunningQuantileStats()
+    running_stats.update(data)
+
+    stats = running_stats.get_statistics()
+
+    # Should have basic stats
+    expected_keys = {"min", "max", "mean", "std", "count"}
+    assert expected_keys.issubset(set(stats.keys()))
+
+    # Verify values
+    np.testing.assert_allclose(stats["mean"], np.mean(data, axis=0), atol=1e-10)
+    np.testing.assert_allclose(stats["std"], np.std(data, axis=0), atol=1e-6)
+    np.testing.assert_equal(stats["count"], np.array([100]))
+
+
+def test_running_quantile_stats_get_statistics_with_quantiles():
+    """Test getting statistics with quantiles."""
+    np.random.seed(42)
+    data = np.random.normal(0, 1, (1000, 2))
+
+    running_stats = RunningQuantileStats()
+    running_stats.update(data)
+
+    stats = running_stats.get_statistics()
+
+    # Should have basic stats plus quantiles
+    expected_keys = {"min", "max", "mean", "std", "count", "q01", "q10", "q50", "q90", "q99"}
+    assert expected_keys.issubset(set(stats.keys()))
+
+    # Verify quantile values are reasonable
+    from lerobot.datasets.compute_stats import DEFAULT_QUANTILES
+
+    for i, q in enumerate(DEFAULT_QUANTILES):
+        q_key = f"q{int(q * 100):02d}"
+        assert q_key in stats
+        assert stats[q_key].shape == (2,)
+
+        # Check that quantiles are in reasonable order
+        if i > 0:
+            prev_q_key = f"q{int(DEFAULT_QUANTILES[i - 1] * 100):02d}"
+            assert np.all(stats[prev_q_key] <= stats[q_key])
+
+
+def test_running_quantile_stats_histogram_adjustment():
+    """Test that histograms adjust when min/max change."""
+    running_stats = RunningQuantileStats()
+
+    # Initial data with small range
+    data1 = np.array([[0.0, 1.0], [0.1, 1.1], [0.2, 1.2]])
+    running_stats.update(data1)
+
+    initial_edges_0 = running_stats._bin_edges[0].copy()
+    initial_edges_1 = running_stats._bin_edges[1].copy()
+
+    # Add data with much larger range
+    data2 = np.array([[10.0, -10.0], [11.0, -11.0]])
+    running_stats.update(data2)
+
+    # Bin edges should have changed
+    assert not np.array_equal(initial_edges_0, running_stats._bin_edges[0])
+    assert not np.array_equal(initial_edges_1, running_stats._bin_edges[1])
+
+    # New edges should cover the expanded range
+    # First dimension: min should still be ~0.0, max should be ~11.0
+    assert running_stats._bin_edges[0][0] <= 0.0
+    assert running_stats._bin_edges[0][-1] >= 11.0
+
+    # Second dimension: min should be ~-11.0, max should be ~1.2
+    assert running_stats._bin_edges[1][0] <= -11.0
+    assert running_stats._bin_edges[1][-1] >= 1.2
+
+
+def test_running_quantile_stats_insufficient_data_error():
+    """Test error when trying to get stats with insufficient data."""
+    running_stats = RunningQuantileStats()
+
+    with pytest.raises(ValueError, match="Cannot compute statistics for less than 2 vectors"):
+        running_stats.get_statistics()
+
+    # Single vector should also fail
+    running_stats.update(np.array([[1.0]]))
+    with pytest.raises(ValueError, match="Cannot compute statistics for less than 2 vectors"):
+        running_stats.get_statistics()
+
+
+def test_running_quantile_stats_vector_length_consistency():
+    """Test error when vector lengths don't match."""
+    running_stats = RunningQuantileStats()
+    running_stats.update(np.array([[1.0, 2.0], [3.0, 4.0]]))
+
+    with pytest.raises(ValueError, match="The length of new vectors does not match"):
+        running_stats.update(np.array([[1.0, 2.0, 3.0]]))  # Different length
+
+
+def test_running_quantile_stats_reshape_handling():
+    """Test that various input shapes are handled correctly."""
+    running_stats = RunningQuantileStats()
+
+    # Test 3D input (e.g., images)
+    data_3d = np.random.normal(0, 1, (10, 32, 32))
+    running_stats.update(data_3d)
+
+    assert running_stats._count == 10 * 32
+    assert running_stats._mean.shape == (32,)
+
+    # Test 1D input
+    running_stats_1d = RunningQuantileStats()
+    data_1d = np.array([1, 2, 3, 4, 5]).reshape(-1, 1)
+    running_stats_1d.update(data_1d)
+
+    assert running_stats_1d._count == 5
+    assert running_stats_1d._mean.shape == (1,)
+
+
+def test_get_feature_stats_quantiles_enabled_by_default():
+    """Test that quantiles are computed by default."""
+    data = np.random.normal(0, 1, (100, 5))
+    stats = get_feature_stats(data, axis=0, keepdims=False)
+
+    expected_keys = {"min", "max", "mean", "std", "count", "q01", "q10", "q50", "q90", "q99"}
+    assert set(stats.keys()) == expected_keys
+
+
+def test_get_feature_stats_quantiles_with_vector_data():
+    """Test quantile computation with vector data."""
+    np.random.seed(42)
+    data = np.random.normal(0, 1, (100, 5))
+
+    stats = get_feature_stats(data, axis=0, keepdims=False)
+
+    expected_keys = {"min", "max", "mean", "std", "count", "q01", "q10", "q50", "q90", "q99"}
+    assert set(stats.keys()) == expected_keys
+
+    # Verify shapes
+    assert stats["q01"].shape == (5,)
+    assert stats["q99"].shape == (5,)
+
+    # Verify quantiles are reasonable
+    assert np.all(stats["q01"] < stats["q99"])
+
+
+def test_get_feature_stats_quantiles_with_image_data():
+    """Test quantile computation with image data."""
+    np.random.seed(42)
+    data = np.random.normal(0, 1, (50, 3, 32, 32))  # batch, channels, height, width
+
+    stats = get_feature_stats(data, axis=(0, 2, 3), keepdims=True)
+
+    expected_keys = {"min", "max", "mean", "std", "count", "q01", "q10", "q50", "q90", "q99"}
+    assert set(stats.keys()) == expected_keys
+
+    # Verify shapes for images (should be (1, channels, 1, 1))
+    assert stats["q01"].shape == (1, 3, 1, 1)
+    assert stats["q50"].shape == (1, 3, 1, 1)
+    assert stats["q99"].shape == (1, 3, 1, 1)
+
+
+def test_get_feature_stats_fixed_quantiles():
+    """Test that fixed quantiles are always computed."""
+    data = np.random.normal(0, 1, (200, 3))
+
+    stats = get_feature_stats(data, axis=0, keepdims=False)
+
+    expected_quantile_keys = {"q01", "q10", "q50", "q90", "q99"}
+    assert expected_quantile_keys.issubset(set(stats.keys()))
+
+
+def test_get_feature_stats_unsupported_axis_error():
+    """Test error for unsupported axis configuration."""
+    data = np.random.normal(0, 1, (10, 5))
+
+    with pytest.raises(ValueError, match="Unsupported axis configuration"):
+        get_feature_stats(
+            data,
+            axis=(1, 2),  # Unsupported axis
+            keepdims=False,
+        )
+
+
+def test_compute_episode_stats_backward_compatibility():
+    """Test that existing functionality is preserved."""
+    episode_data = {
+        "action": np.random.normal(0, 1, (100, 7)),
+        "observation.state": np.random.normal(0, 1, (100, 10)),
+    }
+    features = {
+        "action": {"dtype": "float32", "shape": (7,)},
+        "observation.state": {"dtype": "float32", "shape": (10,)},
+    }
+
+    stats = compute_episode_stats(episode_data, features)
+
+    for key in ["action", "observation.state"]:
+        expected_keys = {"min", "max", "mean", "std", "count", "q01", "q10", "q50", "q90", "q99"}
+        assert set(stats[key].keys()) == expected_keys
+
+
+def test_compute_episode_stats_with_custom_quantiles():
+    """Test quantile computation with custom quantile values."""
+    np.random.seed(42)
+    episode_data = {
+        "action": np.random.normal(0, 1, (100, 7)),
+        "observation.state": np.random.normal(2, 1, (100, 10)),
+    }
+    features = {
+        "action": {"dtype": "float32", "shape": (7,)},
+        "observation.state": {"dtype": "float32", "shape": (10,)},
+    }
+
+    stats = compute_episode_stats(episode_data, features)
+
+    # Should have quantiles
+    for key in ["action", "observation.state"]:
+        expected_keys = {"min", "max", "mean", "std", "count", "q01", "q10", "q50", "q90", "q99"}
+        assert set(stats[key].keys()) == expected_keys
+
+        # Verify shapes
+        assert stats[key]["q01"].shape == (features[key]["shape"][0],)
+        assert stats[key]["q99"].shape == (features[key]["shape"][0],)
+
+
+def test_compute_episode_stats_with_image_data():
+    """Test quantile computation with image features."""
+    image_paths = [f"image_{i}.jpg" for i in range(50)]
+    episode_data = {
+        "observation.image": image_paths,
+        "action": np.random.normal(0, 1, (50, 5)),
+    }
+    features = {
+        "observation.image": {"dtype": "image"},
+        "action": {"dtype": "float32", "shape": (5,)},
+    }
+
+    with patch("lerobot.datasets.compute_stats.load_image_as_numpy", side_effect=mock_load_image_as_numpy):
+        stats = compute_episode_stats(episode_data, features)
+
+    # Image quantiles should be normalized and have correct shape
+    assert "q01" in stats["observation.image"]
+    assert "q50" in stats["observation.image"]
+    assert "q99" in stats["observation.image"]
+    assert stats["observation.image"]["q01"].shape == (3, 1, 1)
+    assert stats["observation.image"]["q50"].shape == (3, 1, 1)
+    assert stats["observation.image"]["q99"].shape == (3, 1, 1)
+
+    # Action quantiles should have correct shape
+    assert stats["action"]["q01"].shape == (5,)
+    assert stats["action"]["q50"].shape == (5,)
+    assert stats["action"]["q99"].shape == (5,)
+
+
+def test_compute_episode_stats_string_features_skipped():
+    """Test that string features are properly skipped."""
+    episode_data = {
+        "task": ["pick_apple"] * 100,  # String feature
+        "action": np.random.normal(0, 1, (100, 5)),
+    }
+    features = {
+        "task": {"dtype": "string"},
+        "action": {"dtype": "float32", "shape": (5,)},
+    }
+
+    stats = compute_episode_stats(
+        episode_data,
+        features,
+    )
+
+    # String features should be skipped
+    assert "task" not in stats
+    assert "action" in stats
+    assert "q01" in stats["action"]
+
+
+def test_aggregate_feature_stats_with_quantiles():
+    """Test aggregating feature stats that include quantiles."""
+    stats_ft_list = [
+        {
+            "min": np.array([1.0]),
+            "max": np.array([10.0]),
+            "mean": np.array([5.0]),
+            "std": np.array([2.0]),
+            "count": np.array([100]),
+            "q01": np.array([1.5]),
+            "q99": np.array([9.5]),
+        },
+        {
+            "min": np.array([2.0]),
+            "max": np.array([12.0]),
+            "mean": np.array([6.0]),
+            "std": np.array([2.5]),
+            "count": np.array([150]),
+            "q01": np.array([2.5]),
+            "q99": np.array([11.5]),
+        },
+    ]
+
+    result = aggregate_feature_stats(stats_ft_list)
+
+    # Should preserve quantiles
+    assert "q01" in result
+    assert "q99" in result
+
+    # Verify quantile aggregation (weighted average)
+    expected_q01 = (1.5 * 100 + 2.5 * 150) / 250  # ≈ 2.1
+    expected_q99 = (9.5 * 100 + 11.5 * 150) / 250  # ≈ 10.7
+
+    np.testing.assert_allclose(result["q01"], np.array([expected_q01]), atol=1e-6)
+    np.testing.assert_allclose(result["q99"], np.array([expected_q99]), atol=1e-6)
+
+
+def test_aggregate_stats_mixed_quantiles():
+    """Test aggregating stats where some have quantiles and some don't."""
+    stats_with_quantiles = {
+        "feature1": {
+            "min": np.array([1.0]),
+            "max": np.array([10.0]),
+            "mean": np.array([5.0]),
+            "std": np.array([2.0]),
+            "count": np.array([100]),
+            "q01": np.array([1.5]),
+            "q99": np.array([9.5]),
+        }
+    }
+
+    stats_without_quantiles = {
+        "feature2": {
+            "min": np.array([0.0]),
+            "max": np.array([5.0]),
+            "mean": np.array([2.5]),
+            "std": np.array([1.5]),
+            "count": np.array([50]),
+        }
+    }
+
+    all_stats = [stats_with_quantiles, stats_without_quantiles]
+    result = aggregate_stats(all_stats)
+
+    # Feature1 should keep its quantiles
+    assert "q01" in result["feature1"]
+    assert "q99" in result["feature1"]
+
+    # Feature2 should not have quantiles
+    assert "q01" not in result["feature2"]
+    assert "q99" not in result["feature2"]
+
+
+def test_assert_type_and_shape_with_quantiles():
+    """Test validation works correctly with quantile keys."""
+    # Valid stats with quantiles
+    valid_stats = [
+        {
+            "observation.image": {
+                "min": np.array([0.0, 0.0, 0.0]).reshape(3, 1, 1),
+                "max": np.array([1.0, 1.0, 1.0]).reshape(3, 1, 1),
+                "mean": np.array([0.5, 0.5, 0.5]).reshape(3, 1, 1),
+                "std": np.array([0.2, 0.2, 0.2]).reshape(3, 1, 1),
+                "count": np.array([100]),
+                "q01": np.array([0.1, 0.1, 0.1]).reshape(3, 1, 1),
+                "q99": np.array([0.9, 0.9, 0.9]).reshape(3, 1, 1),
+            }
+        }
+    ]
+
+    # Should not raise error
+    _assert_type_and_shape(valid_stats)
+
+    # Invalid shape for quantile
+    invalid_stats = [
+        {
+            "observation.image": {
+                "count": np.array([100]),
+                "q01": np.array([0.1, 0.2]),  # Wrong shape for image quantile
+            }
+        }
+    ]
+
+    with pytest.raises(ValueError, match="Shape of quantile 'q01' must be \\(3,1,1\\)"):
+        _assert_type_and_shape(invalid_stats)
+
+
+def test_quantile_integration_single_value_quantiles():
+    """Test quantile computation with single repeated value."""
+    data = np.ones((100, 3))  # All ones
+
+    running_stats = RunningQuantileStats()
+    running_stats.update(data)
+
+    stats = running_stats.get_statistics()
+
+    # All quantiles should be approximately 1.0
+    np.testing.assert_allclose(stats["q01"], np.array([1.0, 1.0, 1.0]), atol=1e-6)
+    np.testing.assert_allclose(stats["q50"], np.array([1.0, 1.0, 1.0]), atol=1e-6)
+    np.testing.assert_allclose(stats["q99"], np.array([1.0, 1.0, 1.0]), atol=1e-6)
+
+
+def test_quantile_integration_fixed_quantiles():
+    """Test that fixed quantiles are computed."""
+    np.random.seed(42)
+    data = np.random.normal(0, 1, (1000, 2))
+
+    stats = get_feature_stats(data, axis=0, keepdims=False)
+
+    # Check all fixed quantiles are present
+    assert "q01" in stats
+    assert "q10" in stats
+    assert "q50" in stats
+    assert "q90" in stats
+    assert "q99" in stats
+
+
+def test_quantile_integration_large_dataset_quantiles():
+    """Test quantile computation efficiency with large datasets."""
+    np.random.seed(42)
+    large_data = np.random.normal(0, 1, (10000, 5))
+
+    running_stats = RunningQuantileStats(num_quantile_bins=1000)  # Reduced bins for speed
+    running_stats.update(large_data)
+
+    stats = running_stats.get_statistics()
+
+    # Should complete without issues and produce reasonable results
+    assert stats["count"][0] == 10000
+    assert len(stats["q01"]) == 5
+
+
+def test_fixed_quantiles_always_computed():
+    """Test that the fixed quantiles [0.01, 0.10, 0.50, 0.90, 0.99] are always computed."""
+    np.random.seed(42)
+    # Test with vector data
+    vector_data = np.random.normal(0, 1, (100, 5))
+    vector_stats = get_feature_stats(vector_data, axis=0, keepdims=False)
+
+    # Check all fixed quantiles are present
+    expected_quantiles = ["q01", "q10", "q50", "q90", "q99"]
+    for q_key in expected_quantiles:
+        assert q_key in vector_stats
+        assert vector_stats[q_key].shape == (5,)
+
+    # Test with image data
+    image_data = np.random.randint(0, 256, (50, 3, 32, 32), dtype=np.uint8)
+    image_stats = get_feature_stats(image_data, axis=(0, 2, 3), keepdims=True)
+
+    # Check all fixed quantiles are present for images
+    for q_key in expected_quantiles:
+        assert q_key in image_stats
+        assert image_stats[q_key].shape == (1, 3, 1, 1)
+
+    # Test with episode data
+    episode_data = {
+        "action": np.random.normal(0, 1, (100, 7)),
+        "observation.state": np.random.normal(0, 1, (100, 10)),
+    }
+    features = {
+        "action": {"dtype": "float32", "shape": (7,)},
+        "observation.state": {"dtype": "float32", "shape": (10,)},
+    }
+
+    episode_stats = compute_episode_stats(episode_data, features)
+
+    # Check all fixed quantiles are present in episode stats
+    for key in ["action", "observation.state"]:
+        for q_key in expected_quantiles:
+            assert q_key in episode_stats[key]
+            assert episode_stats[key][q_key].shape == (features[key]["shape"][0],)
diff --git a/lerobot/tests/datasets/test_dataset_tools.py b/lerobot/tests/datasets/test_dataset_tools.py
new file mode 100644
index 0000000000000000000000000000000000000000..5ed7aa1a30dde075adc098f5f7adcaac6cc3f522
--- /dev/null
+++ b/lerobot/tests/datasets/test_dataset_tools.py
@@ -0,0 +1,1323 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tests for dataset tools utilities."""
+
+from unittest.mock import patch
+
+import numpy as np
+import pytest
+import torch
+
+from lerobot.datasets.dataset_tools import (
+    add_features,
+    delete_episodes,
+    merge_datasets,
+    modify_features,
+    modify_tasks,
+    remove_feature,
+    split_dataset,
+)
+from lerobot.scripts.lerobot_edit_dataset import convert_image_to_video_dataset
+
+
+@pytest.fixture
+def sample_dataset(tmp_path, empty_lerobot_dataset_factory):
+    """Create a sample dataset for testing."""
+    features = {
+        "action": {"dtype": "float32", "shape": (6,), "names": None},
+        "observation.state": {"dtype": "float32", "shape": (4,), "names": None},
+        "observation.images.top": {"dtype": "image", "shape": (224, 224, 3), "names": None},
+    }
+
+    dataset = empty_lerobot_dataset_factory(
+        root=tmp_path / "test_dataset",
+        features=features,
+    )
+
+    for ep_idx in range(5):
+        for _ in range(10):
+            frame = {
+                "action": np.random.randn(6).astype(np.float32),
+                "observation.state": np.random.randn(4).astype(np.float32),
+                "observation.images.top": np.random.randint(0, 255, size=(224, 224, 3), dtype=np.uint8),
+                "task": f"task_{ep_idx % 2}",
+            }
+            dataset.add_frame(frame)
+        dataset.save_episode()
+
+    dataset.finalize()
+    return dataset
+
+
+def test_delete_single_episode(sample_dataset, tmp_path):
+    """Test deleting a single episode."""
+    output_dir = tmp_path / "filtered"
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(output_dir)
+
+        new_dataset = delete_episodes(
+            sample_dataset,
+            episode_indices=[2],
+            output_dir=output_dir,
+        )
+
+    assert new_dataset.meta.total_episodes == 4
+    assert new_dataset.meta.total_frames == 40
+
+    episode_indices = {int(idx.item()) for idx in new_dataset.hf_dataset["episode_index"]}
+    assert episode_indices == {0, 1, 2, 3}
+
+    assert len(new_dataset) == 40
+
+
+def test_delete_multiple_episodes(sample_dataset, tmp_path):
+    """Test deleting multiple episodes."""
+    output_dir = tmp_path / "filtered"
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(output_dir)
+
+        new_dataset = delete_episodes(
+            sample_dataset,
+            episode_indices=[1, 3],
+            output_dir=output_dir,
+        )
+
+    assert new_dataset.meta.total_episodes == 3
+    assert new_dataset.meta.total_frames == 30
+
+    episode_indices = {int(idx.item()) for idx in new_dataset.hf_dataset["episode_index"]}
+    assert episode_indices == {0, 1, 2}
+
+
+def test_delete_invalid_episodes(sample_dataset, tmp_path):
+    """Test error handling for invalid episode indices."""
+    with pytest.raises(ValueError, match="Invalid episode indices"):
+        delete_episodes(
+            sample_dataset,
+            episode_indices=[10, 20],
+            output_dir=tmp_path / "filtered",
+        )
+
+
+def test_delete_all_episodes(sample_dataset, tmp_path):
+    """Test error when trying to delete all episodes."""
+    with pytest.raises(ValueError, match="Cannot delete all episodes"):
+        delete_episodes(
+            sample_dataset,
+            episode_indices=list(range(5)),
+            output_dir=tmp_path / "filtered",
+        )
+
+
+def test_delete_empty_list(sample_dataset, tmp_path):
+    """Test error when no episodes specified."""
+    with pytest.raises(ValueError, match="No episodes to delete"):
+        delete_episodes(
+            sample_dataset,
+            episode_indices=[],
+            output_dir=tmp_path / "filtered",
+        )
+
+
+def test_split_by_episodes(sample_dataset, tmp_path):
+    """Test splitting dataset by specific episode indices."""
+    splits = {
+        "train": [0, 1, 2],
+        "val": [3, 4],
+    }
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+
+        def mock_snapshot(repo_id, **kwargs):
+            if "train" in repo_id:
+                return str(tmp_path / f"{sample_dataset.repo_id}_train")
+            elif "val" in repo_id:
+                return str(tmp_path / f"{sample_dataset.repo_id}_val")
+            return str(kwargs.get("local_dir", tmp_path))
+
+        mock_snapshot_download.side_effect = mock_snapshot
+
+        result = split_dataset(
+            sample_dataset,
+            splits=splits,
+            output_dir=tmp_path,
+        )
+
+    assert set(result.keys()) == {"train", "val"}
+
+    assert result["train"].meta.total_episodes == 3
+    assert result["train"].meta.total_frames == 30
+
+    assert result["val"].meta.total_episodes == 2
+    assert result["val"].meta.total_frames == 20
+
+    train_episodes = {int(idx.item()) for idx in result["train"].hf_dataset["episode_index"]}
+    assert train_episodes == {0, 1, 2}
+
+    val_episodes = {int(idx.item()) for idx in result["val"].hf_dataset["episode_index"]}
+    assert val_episodes == {0, 1}
+
+
+def test_split_by_fractions(sample_dataset, tmp_path):
+    """Test splitting dataset by fractions."""
+    splits = {
+        "train": 0.6,
+        "val": 0.4,
+    }
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+
+        def mock_snapshot(repo_id, **kwargs):
+            for split_name in splits:
+                if split_name in repo_id:
+                    return str(tmp_path / f"{sample_dataset.repo_id}_{split_name}")
+            return str(kwargs.get("local_dir", tmp_path))
+
+        mock_snapshot_download.side_effect = mock_snapshot
+
+        result = split_dataset(
+            sample_dataset,
+            splits=splits,
+            output_dir=tmp_path,
+        )
+
+    assert result["train"].meta.total_episodes == 3
+    assert result["val"].meta.total_episodes == 2
+
+
+def test_split_overlapping_episodes(sample_dataset, tmp_path):
+    """Test error when episodes appear in multiple splits."""
+    splits = {
+        "train": [0, 1, 2],
+        "val": [2, 3, 4],
+    }
+
+    with pytest.raises(ValueError, match="Episodes cannot appear in multiple splits"):
+        split_dataset(sample_dataset, splits=splits, output_dir=tmp_path)
+
+
+def test_split_invalid_fractions(sample_dataset, tmp_path):
+    """Test error when fractions sum to more than 1."""
+    splits = {
+        "train": 0.7,
+        "val": 0.5,
+    }
+
+    with pytest.raises(ValueError, match="Split fractions must sum to <= 1.0"):
+        split_dataset(sample_dataset, splits=splits, output_dir=tmp_path)
+
+
+def test_split_empty(sample_dataset, tmp_path):
+    """Test error with empty splits."""
+    with pytest.raises(ValueError, match="No splits provided"):
+        split_dataset(sample_dataset, splits={}, output_dir=tmp_path)
+
+
+def test_merge_two_datasets(sample_dataset, tmp_path, empty_lerobot_dataset_factory):
+    """Test merging two datasets."""
+    features = {
+        "action": {"dtype": "float32", "shape": (6,), "names": None},
+        "observation.state": {"dtype": "float32", "shape": (4,), "names": None},
+        "observation.images.top": {"dtype": "image", "shape": (224, 224, 3), "names": None},
+    }
+
+    dataset2 = empty_lerobot_dataset_factory(
+        root=tmp_path / "test_dataset2",
+        features=features,
+    )
+
+    for ep_idx in range(3):
+        for _ in range(10):
+            frame = {
+                "action": np.random.randn(6).astype(np.float32),
+                "observation.state": np.random.randn(4).astype(np.float32),
+                "observation.images.top": np.random.randint(0, 255, size=(224, 224, 3), dtype=np.uint8),
+                "task": f"task_{ep_idx % 2}",
+            }
+            dataset2.add_frame(frame)
+        dataset2.save_episode()
+    dataset2.finalize()
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "merged_dataset")
+
+        merged = merge_datasets(
+            [sample_dataset, dataset2],
+            output_repo_id="merged_dataset",
+            output_dir=tmp_path / "merged_dataset",
+        )
+
+    assert merged.meta.total_episodes == 8  # 5 + 3
+    assert merged.meta.total_frames == 80  # 50 + 30
+
+    episode_indices = sorted({int(idx.item()) for idx in merged.hf_dataset["episode_index"]})
+    assert episode_indices == list(range(8))
+
+
+def test_merge_empty_list(tmp_path):
+    """Test error when merging empty list."""
+    with pytest.raises(ValueError, match="No datasets to merge"):
+        merge_datasets([], output_repo_id="merged", output_dir=tmp_path)
+
+
+def test_add_features_with_values(sample_dataset, tmp_path):
+    """Test adding a feature with pre-computed values."""
+    num_frames = sample_dataset.meta.total_frames
+    reward_values = np.random.randn(num_frames, 1).astype(np.float32)
+
+    feature_info = {
+        "dtype": "float32",
+        "shape": (1,),
+        "names": None,
+    }
+    features = {
+        "reward": (reward_values, feature_info),
+    }
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "with_reward")
+
+        new_dataset = add_features(
+            dataset=sample_dataset,
+            features=features,
+            output_dir=tmp_path / "with_reward",
+        )
+
+    assert "reward" in new_dataset.meta.features
+    assert new_dataset.meta.features["reward"] == feature_info
+
+    assert len(new_dataset) == num_frames
+    sample_item = new_dataset[0]
+    assert "reward" in sample_item
+    assert isinstance(sample_item["reward"], torch.Tensor)
+
+
+def test_add_features_with_callable(sample_dataset, tmp_path):
+    """Test adding a feature with a callable."""
+
+    def compute_reward(frame_dict, episode_idx, frame_idx):
+        return float(episode_idx * 10 + frame_idx)
+
+    feature_info = {
+        "dtype": "float32",
+        "shape": (1,),
+        "names": None,
+    }
+    features = {
+        "reward": (compute_reward, feature_info),
+    }
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "with_reward")
+
+        new_dataset = add_features(
+            dataset=sample_dataset,
+            features=features,
+            output_dir=tmp_path / "with_reward",
+        )
+
+    assert "reward" in new_dataset.meta.features
+
+    items = [new_dataset[i] for i in range(10)]
+    first_episode_items = [item for item in items if item["episode_index"] == 0]
+    assert len(first_episode_items) == 10
+
+    first_frame = first_episode_items[0]
+    assert first_frame["frame_index"] == 0
+    assert float(first_frame["reward"]) == 0.0
+
+
+def test_add_existing_feature(sample_dataset, tmp_path):
+    """Test error when adding an existing feature."""
+    feature_info = {"dtype": "float32", "shape": (1,)}
+    features = {
+        "action": (np.zeros(50), feature_info),
+    }
+
+    with pytest.raises(ValueError, match="Feature 'action' already exists"):
+        add_features(
+            dataset=sample_dataset,
+            features=features,
+            output_dir=tmp_path / "modified",
+        )
+
+
+def test_add_feature_invalid_info(sample_dataset, tmp_path):
+    """Test error with invalid feature info."""
+    with pytest.raises(ValueError, match="feature_info for 'reward' must contain keys"):
+        add_features(
+            dataset=sample_dataset,
+            features={
+                "reward": (np.zeros(50), {"dtype": "float32"}),
+            },
+            output_dir=tmp_path / "modified",
+        )
+
+
+def test_modify_features_add_and_remove(sample_dataset, tmp_path):
+    """Test modifying features by adding and removing simultaneously."""
+    feature_info = {"dtype": "float32", "shape": (1,), "names": None}
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "modified")
+
+        # First add a feature we'll later remove
+        dataset_with_reward = add_features(
+            sample_dataset,
+            features={"reward": (np.random.randn(50, 1).astype(np.float32), feature_info)},
+            output_dir=tmp_path / "with_reward",
+        )
+
+        # Now use modify_features to add "success" and remove "reward" in one pass
+        modified_dataset = modify_features(
+            dataset_with_reward,
+            add_features={
+                "success": (np.random.randn(50, 1).astype(np.float32), feature_info),
+            },
+            remove_features="reward",
+            output_dir=tmp_path / "modified",
+        )
+
+    assert "success" in modified_dataset.meta.features
+    assert "reward" not in modified_dataset.meta.features
+    assert len(modified_dataset) == 50
+
+
+def test_modify_features_only_add(sample_dataset, tmp_path):
+    """Test that modify_features works with only add_features."""
+    feature_info = {"dtype": "float32", "shape": (1,), "names": None}
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "modified")
+
+        modified_dataset = modify_features(
+            sample_dataset,
+            add_features={
+                "reward": (np.random.randn(50, 1).astype(np.float32), feature_info),
+            },
+            output_dir=tmp_path / "modified",
+        )
+
+    assert "reward" in modified_dataset.meta.features
+    assert len(modified_dataset) == 50
+
+
+def test_modify_features_only_remove(sample_dataset, tmp_path):
+    """Test that modify_features works with only remove_features."""
+    feature_info = {"dtype": "float32", "shape": (1,), "names": None}
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.side_effect = lambda repo_id, **kwargs: str(kwargs.get("local_dir", tmp_path))
+
+        dataset_with_reward = add_features(
+            sample_dataset,
+            features={"reward": (np.random.randn(50, 1).astype(np.float32), feature_info)},
+            output_dir=tmp_path / "with_reward",
+        )
+
+        modified_dataset = modify_features(
+            dataset_with_reward,
+            remove_features="reward",
+            output_dir=tmp_path / "modified",
+        )
+
+    assert "reward" not in modified_dataset.meta.features
+
+
+def test_modify_features_no_changes(sample_dataset, tmp_path):
+    """Test error when modify_features is called with no changes."""
+    with pytest.raises(ValueError, match="Must specify at least one of add_features or remove_features"):
+        modify_features(
+            sample_dataset,
+            output_dir=tmp_path / "modified",
+        )
+
+
+def test_remove_single_feature(sample_dataset, tmp_path):
+    """Test removing a single feature."""
+    feature_info = {"dtype": "float32", "shape": (1,), "names": None}
+    features = {
+        "reward": (np.random.randn(50, 1).astype(np.float32), feature_info),
+    }
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.side_effect = lambda repo_id, **kwargs: str(kwargs.get("local_dir", tmp_path))
+
+        dataset_with_reward = add_features(
+            dataset=sample_dataset,
+            features=features,
+            output_dir=tmp_path / "with_reward",
+        )
+
+        dataset_without_reward = remove_feature(
+            dataset_with_reward,
+            feature_names="reward",
+            output_dir=tmp_path / "without_reward",
+        )
+
+    assert "reward" not in dataset_without_reward.meta.features
+
+    sample_item = dataset_without_reward[0]
+    assert "reward" not in sample_item
+
+
+def test_remove_multiple_features(sample_dataset, tmp_path):
+    """Test removing multiple features at once."""
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.side_effect = lambda repo_id, **kwargs: str(kwargs.get("local_dir", tmp_path))
+
+        dataset = sample_dataset
+        features = {}
+        for feature_name in ["reward", "success"]:
+            feature_info = {"dtype": "float32", "shape": (1,), "names": None}
+            features[feature_name] = (
+                np.random.randn(dataset.meta.total_frames, 1).astype(np.float32),
+                feature_info,
+            )
+
+        dataset_with_features = add_features(
+            dataset, features=features, output_dir=tmp_path / "with_features"
+        )
+        dataset_clean = remove_feature(
+            dataset_with_features, feature_names=["reward", "success"], output_dir=tmp_path / "clean"
+        )
+
+    assert "reward" not in dataset_clean.meta.features
+    assert "success" not in dataset_clean.meta.features
+
+
+def test_remove_nonexistent_feature(sample_dataset, tmp_path):
+    """Test error when removing non-existent feature."""
+    with pytest.raises(ValueError, match="Feature 'nonexistent' not found"):
+        remove_feature(
+            sample_dataset,
+            feature_names="nonexistent",
+            output_dir=tmp_path / "modified",
+        )
+
+
+def test_remove_required_feature(sample_dataset, tmp_path):
+    """Test error when trying to remove required features."""
+    with pytest.raises(ValueError, match="Cannot remove required features"):
+        remove_feature(
+            sample_dataset,
+            feature_names="timestamp",
+            output_dir=tmp_path / "modified",
+        )
+
+
+def test_remove_camera_feature(sample_dataset, tmp_path):
+    """Test removing a camera feature."""
+    camera_keys = sample_dataset.meta.camera_keys
+    if not camera_keys:
+        pytest.skip("No camera keys in dataset")
+
+    camera_to_remove = camera_keys[0]
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "without_camera")
+
+        dataset_without_camera = remove_feature(
+            sample_dataset,
+            feature_names=camera_to_remove,
+            output_dir=tmp_path / "without_camera",
+        )
+
+    assert camera_to_remove not in dataset_without_camera.meta.features
+    assert camera_to_remove not in dataset_without_camera.meta.camera_keys
+
+    sample_item = dataset_without_camera[0]
+    assert camera_to_remove not in sample_item
+
+
+def test_complex_workflow_integration(sample_dataset, tmp_path):
+    """Test a complex workflow combining multiple operations."""
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.side_effect = lambda repo_id, **kwargs: str(kwargs.get("local_dir", tmp_path))
+
+        dataset = add_features(
+            sample_dataset,
+            features={
+                "reward": (
+                    np.random.randn(50, 1).astype(np.float32),
+                    {"dtype": "float32", "shape": (1,), "names": None},
+                )
+            },
+            output_dir=tmp_path / "step1",
+        )
+
+        dataset = delete_episodes(
+            dataset,
+            episode_indices=[2],
+            output_dir=tmp_path / "step2",
+        )
+
+        splits = split_dataset(
+            dataset,
+            splits={"train": 0.75, "val": 0.25},
+            output_dir=tmp_path / "step3",
+        )
+
+        merged = merge_datasets(
+            list(splits.values()),
+            output_repo_id="final_dataset",
+            output_dir=tmp_path / "step4",
+        )
+
+    assert merged.meta.total_episodes == 4
+    assert merged.meta.total_frames == 40
+    assert "reward" in merged.meta.features
+
+    assert len(merged) == 40
+    sample_item = merged[0]
+    assert "reward" in sample_item
+
+
+def test_delete_episodes_preserves_stats(sample_dataset, tmp_path):
+    """Test that deleting episodes preserves statistics correctly."""
+    output_dir = tmp_path / "filtered"
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(output_dir)
+
+        new_dataset = delete_episodes(
+            sample_dataset,
+            episode_indices=[2],
+            output_dir=output_dir,
+        )
+
+    assert new_dataset.meta.stats is not None
+    for feature in ["action", "observation.state"]:
+        assert feature in new_dataset.meta.stats
+        assert "mean" in new_dataset.meta.stats[feature]
+        assert "std" in new_dataset.meta.stats[feature]
+
+
+def test_delete_episodes_preserves_tasks(sample_dataset, tmp_path):
+    """Test that tasks are preserved correctly after deletion."""
+    output_dir = tmp_path / "filtered"
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(output_dir)
+
+        new_dataset = delete_episodes(
+            sample_dataset,
+            episode_indices=[0],
+            output_dir=output_dir,
+        )
+
+    assert new_dataset.meta.tasks is not None
+    assert len(new_dataset.meta.tasks) == 2
+
+    tasks_in_dataset = {str(item["task"]) for item in new_dataset}
+    assert len(tasks_in_dataset) > 0
+
+
+def test_split_three_ways(sample_dataset, tmp_path):
+    """Test splitting dataset into three splits."""
+    splits = {
+        "train": 0.6,
+        "val": 0.2,
+        "test": 0.2,
+    }
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+
+        def mock_snapshot(repo_id, **kwargs):
+            for split_name in splits:
+                if split_name in repo_id:
+                    return str(tmp_path / f"{sample_dataset.repo_id}_{split_name}")
+            return str(kwargs.get("local_dir", tmp_path))
+
+        mock_snapshot_download.side_effect = mock_snapshot
+
+        result = split_dataset(
+            sample_dataset,
+            splits=splits,
+            output_dir=tmp_path,
+        )
+
+    assert set(result.keys()) == {"train", "val", "test"}
+    assert result["train"].meta.total_episodes == 3
+    assert result["val"].meta.total_episodes == 1
+    assert result["test"].meta.total_episodes == 1
+
+    total_frames = sum(ds.meta.total_frames for ds in result.values())
+    assert total_frames == sample_dataset.meta.total_frames
+
+
+def test_split_preserves_stats(sample_dataset, tmp_path):
+    """Test that statistics are preserved when splitting."""
+    splits = {"train": [0, 1, 2], "val": [3, 4]}
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+
+        def mock_snapshot(repo_id, **kwargs):
+            for split_name in splits:
+                if split_name in repo_id:
+                    return str(tmp_path / f"{sample_dataset.repo_id}_{split_name}")
+            return str(kwargs.get("local_dir", tmp_path))
+
+        mock_snapshot_download.side_effect = mock_snapshot
+
+        result = split_dataset(
+            sample_dataset,
+            splits=splits,
+            output_dir=tmp_path,
+        )
+
+    for split_ds in result.values():
+        assert split_ds.meta.stats is not None
+        for feature in ["action", "observation.state"]:
+            assert feature in split_ds.meta.stats
+            assert "mean" in split_ds.meta.stats[feature]
+            assert "std" in split_ds.meta.stats[feature]
+
+
+def test_merge_three_datasets(sample_dataset, tmp_path, empty_lerobot_dataset_factory):
+    """Test merging three datasets."""
+    features = {
+        "action": {"dtype": "float32", "shape": (6,), "names": None},
+        "observation.state": {"dtype": "float32", "shape": (4,), "names": None},
+        "observation.images.top": {"dtype": "image", "shape": (224, 224, 3), "names": None},
+    }
+
+    datasets = [sample_dataset]
+
+    for i in range(2):
+        dataset = empty_lerobot_dataset_factory(
+            root=tmp_path / f"test_dataset{i + 2}",
+            features=features,
+        )
+
+        for ep_idx in range(2):
+            for _ in range(10):
+                frame = {
+                    "action": np.random.randn(6).astype(np.float32),
+                    "observation.state": np.random.randn(4).astype(np.float32),
+                    "observation.images.top": np.random.randint(0, 255, size=(224, 224, 3), dtype=np.uint8),
+                    "task": f"task_{ep_idx}",
+                }
+                dataset.add_frame(frame)
+            dataset.save_episode()
+        dataset.finalize()
+
+        datasets.append(dataset)
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "merged_dataset")
+
+        merged = merge_datasets(
+            datasets,
+            output_repo_id="merged_dataset",
+            output_dir=tmp_path / "merged_dataset",
+        )
+
+    assert merged.meta.total_episodes == 9
+    assert merged.meta.total_frames == 90
+
+
+def test_merge_preserves_stats(sample_dataset, tmp_path, empty_lerobot_dataset_factory):
+    """Test that statistics are computed for merged datasets."""
+    features = {
+        "action": {"dtype": "float32", "shape": (6,), "names": None},
+        "observation.state": {"dtype": "float32", "shape": (4,), "names": None},
+        "observation.images.top": {"dtype": "image", "shape": (224, 224, 3), "names": None},
+    }
+
+    dataset2 = empty_lerobot_dataset_factory(
+        root=tmp_path / "test_dataset2",
+        features=features,
+    )
+
+    for ep_idx in range(3):
+        for _ in range(10):
+            frame = {
+                "action": np.random.randn(6).astype(np.float32),
+                "observation.state": np.random.randn(4).astype(np.float32),
+                "observation.images.top": np.random.randint(0, 255, size=(224, 224, 3), dtype=np.uint8),
+                "task": f"task_{ep_idx % 2}",
+            }
+            dataset2.add_frame(frame)
+        dataset2.save_episode()
+    dataset2.finalize()
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "merged_dataset")
+
+        merged = merge_datasets(
+            [sample_dataset, dataset2],
+            output_repo_id="merged_dataset",
+            output_dir=tmp_path / "merged_dataset",
+        )
+
+    assert merged.meta.stats is not None
+    for feature in ["action", "observation.state"]:
+        assert feature in merged.meta.stats
+        assert "mean" in merged.meta.stats[feature]
+        assert "std" in merged.meta.stats[feature]
+
+
+def test_add_features_preserves_existing_stats(sample_dataset, tmp_path):
+    """Test that adding a feature preserves existing stats."""
+    num_frames = sample_dataset.meta.total_frames
+    reward_values = np.random.randn(num_frames, 1).astype(np.float32)
+
+    feature_info = {
+        "dtype": "float32",
+        "shape": (1,),
+        "names": None,
+    }
+    features = {
+        "reward": (reward_values, feature_info),
+    }
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "with_reward")
+
+        new_dataset = add_features(
+            dataset=sample_dataset,
+            features=features,
+            output_dir=tmp_path / "with_reward",
+        )
+
+    assert new_dataset.meta.stats is not None
+    for feature in ["action", "observation.state"]:
+        assert feature in new_dataset.meta.stats
+        assert "mean" in new_dataset.meta.stats[feature]
+        assert "std" in new_dataset.meta.stats[feature]
+
+
+def test_remove_feature_updates_stats(sample_dataset, tmp_path):
+    """Test that removing a feature removes it from stats."""
+    feature_info = {"dtype": "float32", "shape": (1,), "names": None}
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.side_effect = lambda repo_id, **kwargs: str(kwargs.get("local_dir", tmp_path))
+
+        dataset_with_reward = add_features(
+            sample_dataset,
+            features={
+                "reward": (np.random.randn(50, 1).astype(np.float32), feature_info),
+            },
+            output_dir=tmp_path / "with_reward",
+        )
+
+        dataset_without_reward = remove_feature(
+            dataset_with_reward,
+            feature_names="reward",
+            output_dir=tmp_path / "without_reward",
+        )
+
+    if dataset_without_reward.meta.stats:
+        assert "reward" not in dataset_without_reward.meta.stats
+
+
+def test_delete_consecutive_episodes(sample_dataset, tmp_path):
+    """Test deleting consecutive episodes."""
+    output_dir = tmp_path / "filtered"
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(output_dir)
+
+        new_dataset = delete_episodes(
+            sample_dataset,
+            episode_indices=[1, 2, 3],
+            output_dir=output_dir,
+        )
+
+    assert new_dataset.meta.total_episodes == 2
+    assert new_dataset.meta.total_frames == 20
+
+    episode_indices = sorted({int(idx.item()) for idx in new_dataset.hf_dataset["episode_index"]})
+    assert episode_indices == [0, 1]
+
+
+def test_delete_first_and_last_episodes(sample_dataset, tmp_path):
+    """Test deleting first and last episodes."""
+    output_dir = tmp_path / "filtered"
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(output_dir)
+
+        new_dataset = delete_episodes(
+            sample_dataset,
+            episode_indices=[0, 4],
+            output_dir=output_dir,
+        )
+
+    assert new_dataset.meta.total_episodes == 3
+    assert new_dataset.meta.total_frames == 30
+
+    episode_indices = sorted({int(idx.item()) for idx in new_dataset.hf_dataset["episode_index"]})
+    assert episode_indices == [0, 1, 2]
+
+
+def test_split_all_episodes_assigned(sample_dataset, tmp_path):
+    """Test that all episodes can be explicitly assigned to splits."""
+    splits = {
+        "split1": [0, 1],
+        "split2": [2, 3],
+        "split3": [4],
+    }
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+
+        def mock_snapshot(repo_id, **kwargs):
+            for split_name in splits:
+                if split_name in repo_id:
+                    return str(tmp_path / f"{sample_dataset.repo_id}_{split_name}")
+            return str(kwargs.get("local_dir", tmp_path))
+
+        mock_snapshot_download.side_effect = mock_snapshot
+
+        result = split_dataset(
+            sample_dataset,
+            splits=splits,
+            output_dir=tmp_path,
+        )
+
+    total_episodes = sum(ds.meta.total_episodes for ds in result.values())
+    assert total_episodes == sample_dataset.meta.total_episodes
+
+
+def test_modify_features_preserves_file_structure(sample_dataset, tmp_path):
+    """Test that modifying features preserves chunk_idx and file_idx from source dataset."""
+    feature_info = {"dtype": "float32", "shape": (1,), "names": None}
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+
+        def mock_snapshot(repo_id, **kwargs):
+            return str(kwargs.get("local_dir", tmp_path / repo_id.split("/")[-1]))
+
+        mock_snapshot_download.side_effect = mock_snapshot
+
+        # First split the dataset to create a non-zero starting chunk/file structure
+        splits = split_dataset(
+            sample_dataset,
+            splits={"train": [0, 1, 2], "val": [3, 4]},
+            output_dir=tmp_path / "splits",
+        )
+
+        train_dataset = splits["train"]
+
+        # Get original chunk/file indices from first episode
+        if train_dataset.meta.episodes is None:
+            from lerobot.datasets.io_utils import load_episodes
+
+            train_dataset.meta.episodes = load_episodes(train_dataset.meta.root)
+        original_chunk_indices = [ep["data/chunk_index"] for ep in train_dataset.meta.episodes]
+        original_file_indices = [ep["data/file_index"] for ep in train_dataset.meta.episodes]
+
+        # Now add a feature to the split dataset
+        modified_dataset = add_features(
+            train_dataset,
+            features={
+                "reward": (
+                    np.random.randn(train_dataset.meta.total_frames, 1).astype(np.float32),
+                    feature_info,
+                ),
+            },
+            output_dir=tmp_path / "modified",
+        )
+
+        # Check that chunk/file indices are preserved
+        if modified_dataset.meta.episodes is None:
+            from lerobot.datasets.io_utils import load_episodes
+
+            modified_dataset.meta.episodes = load_episodes(modified_dataset.meta.root)
+        new_chunk_indices = [ep["data/chunk_index"] for ep in modified_dataset.meta.episodes]
+        new_file_indices = [ep["data/file_index"] for ep in modified_dataset.meta.episodes]
+
+        assert new_chunk_indices == original_chunk_indices, "Chunk indices should be preserved"
+        assert new_file_indices == original_file_indices, "File indices should be preserved"
+        assert "reward" in modified_dataset.meta.features
+
+
+def test_modify_tasks_single_task_for_all(sample_dataset):
+    """Test setting a single task for all episodes."""
+    new_task = "Pick up the cube and place it"
+
+    modified_dataset = modify_tasks(sample_dataset, new_task=new_task)
+
+    # Verify all episodes have the new task
+    assert len(modified_dataset.meta.tasks) == 1
+    assert new_task in modified_dataset.meta.tasks.index
+
+    # Verify task_index is 0 for all frames (only one task)
+    for i in range(len(modified_dataset)):
+        item = modified_dataset[i]
+        assert item["task_index"].item() == 0
+        assert item["task"] == new_task
+
+
+def test_modify_tasks_episode_specific(sample_dataset):
+    """Test setting different tasks for specific episodes."""
+    episode_tasks = {
+        0: "Task A",
+        1: "Task B",
+        2: "Task A",
+        3: "Task C",
+        4: "Task B",
+    }
+
+    modified_dataset = modify_tasks(sample_dataset, episode_tasks=episode_tasks)
+
+    # Verify correct number of unique tasks
+    unique_tasks = set(episode_tasks.values())
+    assert len(modified_dataset.meta.tasks) == len(unique_tasks)
+
+    # Verify each episode has the correct task
+    for ep_idx, expected_task in episode_tasks.items():
+        ep_data = modified_dataset.meta.episodes[ep_idx]
+        assert ep_data["tasks"][0] == expected_task
+
+
+def test_modify_tasks_default_with_overrides(sample_dataset):
+    """Test setting a default task with specific overrides."""
+    default_task = "Default task"
+    override_task = "Special task"
+    episode_tasks = {2: override_task, 4: override_task}
+
+    modified_dataset = modify_tasks(
+        sample_dataset,
+        new_task=default_task,
+        episode_tasks=episode_tasks,
+    )
+
+    # Verify correct number of unique tasks
+    assert len(modified_dataset.meta.tasks) == 2
+    assert default_task in modified_dataset.meta.tasks.index
+    assert override_task in modified_dataset.meta.tasks.index
+
+    # Verify episodes have correct tasks
+    for ep_idx in range(5):
+        ep_data = modified_dataset.meta.episodes[ep_idx]
+        if ep_idx in episode_tasks:
+            assert ep_data["tasks"][0] == override_task
+        else:
+            assert ep_data["tasks"][0] == default_task
+
+
+def test_modify_tasks_no_task_specified(sample_dataset):
+    """Test error when no task is specified."""
+    with pytest.raises(ValueError, match="Must specify at least one of new_task or episode_tasks"):
+        modify_tasks(sample_dataset)
+
+
+def test_modify_tasks_invalid_episode_indices(sample_dataset):
+    """Test error with invalid episode indices."""
+    with pytest.raises(ValueError, match="Invalid episode indices"):
+        modify_tasks(sample_dataset, episode_tasks={10: "Task", 20: "Task"})
+
+
+def test_modify_tasks_updates_info_json(sample_dataset):
+    """Test that total_tasks is updated in info.json."""
+    episode_tasks = {0: "Task A", 1: "Task B", 2: "Task C", 3: "Task A", 4: "Task B"}
+
+    modified_dataset = modify_tasks(sample_dataset, episode_tasks=episode_tasks)
+
+    # Verify total_tasks is updated
+    assert modified_dataset.meta.total_tasks == 3
+
+
+def test_modify_tasks_preserves_other_metadata(sample_dataset):
+    """Test that modifying tasks preserves other metadata."""
+    original_frames = sample_dataset.meta.total_frames
+    original_episodes = sample_dataset.meta.total_episodes
+    original_fps = sample_dataset.meta.fps
+
+    modified_dataset = modify_tasks(sample_dataset, new_task="New task")
+
+    # Verify other metadata is preserved
+    assert modified_dataset.meta.total_frames == original_frames
+    assert modified_dataset.meta.total_episodes == original_episodes
+    assert modified_dataset.meta.fps == original_fps
+
+
+def test_modify_tasks_task_index_correct(sample_dataset):
+    """Test that task_index values are correct in data files."""
+    # Create tasks that will have predictable indices (sorted alphabetically)
+    episode_tasks = {
+        0: "Alpha task",  # Will be index 0
+        1: "Beta task",  # Will be index 1
+        2: "Alpha task",  # Will be index 0
+        3: "Gamma task",  # Will be index 2
+        4: "Beta task",  # Will be index 1
+    }
+
+    modified_dataset = modify_tasks(sample_dataset, episode_tasks=episode_tasks)
+
+    # Verify task indices are correct
+    task_to_expected_idx = {
+        "Alpha task": 0,
+        "Beta task": 1,
+        "Gamma task": 2,
+    }
+
+    for i in range(len(modified_dataset)):
+        item = modified_dataset[i]
+        ep_idx = item["episode_index"].item()
+        expected_task = episode_tasks[ep_idx]
+        expected_idx = task_to_expected_idx[expected_task]
+        assert item["task_index"].item() == expected_idx
+        assert item["task"] == expected_task
+
+
+def test_modify_tasks_in_place(sample_dataset):
+    """Test that modify_tasks modifies the dataset in-place."""
+    original_root = sample_dataset.root
+
+    modified_dataset = modify_tasks(sample_dataset, new_task="New task")
+
+    # Verify same instance is returned and root is unchanged
+    assert modified_dataset is sample_dataset
+    assert modified_dataset.root == original_root
+
+
+def test_modify_tasks_keeps_original_when_not_overridden(sample_dataset):
+    """Test that original tasks are kept when using episode_tasks without new_task."""
+    from lerobot.datasets.io_utils import load_episodes
+
+    # Ensure episodes metadata is loaded
+    if sample_dataset.meta.episodes is None:
+        sample_dataset.meta.episodes = load_episodes(sample_dataset.meta.root)
+
+    # Get original tasks for episodes not being overridden
+    original_task_ep0 = sample_dataset.meta.episodes[0]["tasks"][0]
+    original_task_ep1 = sample_dataset.meta.episodes[1]["tasks"][0]
+
+    # Only override episodes 2, 3, 4
+    episode_tasks = {2: "New Task A", 3: "New Task B", 4: "New Task A"}
+
+    modified_dataset = modify_tasks(sample_dataset, episode_tasks=episode_tasks)
+
+    # Verify original tasks are kept for episodes 0 and 1
+    assert modified_dataset.meta.episodes[0]["tasks"][0] == original_task_ep0
+    assert modified_dataset.meta.episodes[1]["tasks"][0] == original_task_ep1
+
+    # Verify new tasks for overridden episodes
+    assert modified_dataset.meta.episodes[2]["tasks"][0] == "New Task A"
+    assert modified_dataset.meta.episodes[3]["tasks"][0] == "New Task B"
+    assert modified_dataset.meta.episodes[4]["tasks"][0] == "New Task A"
+
+
+def test_convert_image_to_video_dataset(tmp_path):
+    """Test converting lerobot/pusht_image dataset to video format."""
+    from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+    # Load the actual lerobot/pusht_image dataset (only first 2 episodes for speed)
+    source_dataset = LeRobotDataset("lerobot/pusht_image", episodes=[0, 1])
+
+    output_dir = tmp_path / "pusht_video"
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(output_dir)
+
+        # Verify source dataset has images, not videos
+        assert len(source_dataset.meta.video_keys) == 0
+        assert "observation.image" in source_dataset.meta.features
+
+        # Convert to video dataset (only first 2 episodes for speed)
+        video_dataset = convert_image_to_video_dataset(
+            dataset=source_dataset,
+            output_dir=output_dir,
+            repo_id="lerobot/pusht_video",
+            vcodec="libsvtav1",
+            pix_fmt="yuv420p",
+            g=2,
+            crf=30,
+            episode_indices=[0, 1],
+            num_workers=2,
+        )
+
+        # Verify new dataset has videos
+        assert len(video_dataset.meta.video_keys) > 0
+        assert "observation.image" in video_dataset.meta.video_keys
+
+        # Verify correct number of episodes and frames (2 episodes)
+        assert video_dataset.meta.total_episodes == 2
+        # Compare against the actual number of frames in the loaded episodes, not metadata total
+        assert len(video_dataset) == len(source_dataset)
+
+        # Verify video files exist
+        for ep_idx in range(video_dataset.meta.total_episodes):
+            for video_key in video_dataset.meta.video_keys:
+                video_path = video_dataset.root / video_dataset.meta.get_video_file_path(ep_idx, video_key)
+                assert video_path.exists(), f"Video file should exist: {video_path}"
+
+        # Verify we can load the dataset and access it
+        assert len(video_dataset) == video_dataset.meta.total_frames
+
+        # Test that we can actually get an item from the video dataset
+        item = video_dataset[0]
+        assert "observation.image" in item
+        assert "action" in item
+
+        # Cleanup
+        import shutil
+
+        if output_dir.exists():
+            shutil.rmtree(output_dir)
+
+
+def test_convert_image_to_video_dataset_subset_episodes(tmp_path):
+    """Test converting only specific episodes from lerobot/pusht_image to video format."""
+    from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+    # Load the actual lerobot/pusht_image dataset (only first 3 episodes)
+    source_dataset = LeRobotDataset("lerobot/pusht_image", episodes=[0, 1, 2])
+
+    output_dir = tmp_path / "pusht_video_subset"
+
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(output_dir)
+
+        # Convert only episode 0 to video (subset of loaded episodes)
+        episode_indices = [0]
+
+        video_dataset = convert_image_to_video_dataset(
+            dataset=source_dataset,
+            output_dir=output_dir,
+            repo_id="lerobot/pusht_video_subset",
+            episode_indices=episode_indices,
+            num_workers=2,
+        )
+
+        # Verify correct number of episodes
+        assert video_dataset.meta.total_episodes == len(episode_indices)
+
+        # Verify video files exist for selected episodes
+        assert len(video_dataset.meta.video_keys) > 0
+        assert "observation.image" in video_dataset.meta.video_keys
+
+        # Cleanup
+        import shutil
+
+        if output_dir.exists():
+            shutil.rmtree(output_dir)
diff --git a/lerobot/tests/datasets/test_dataset_utils.py b/lerobot/tests/datasets/test_dataset_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..874099e2bf0edb35e650690dc58a6eabc8f380d5
--- /dev/null
+++ b/lerobot/tests/datasets/test_dataset_utils.py
@@ -0,0 +1,150 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import pytest
+import torch
+from datasets import Dataset
+from huggingface_hub import DatasetCard
+
+from lerobot.datasets.feature_utils import combine_feature_dicts
+from lerobot.datasets.io_utils import hf_transform_to_torch
+from lerobot.datasets.utils import create_lerobot_dataset_card
+from lerobot.utils.constants import ACTION, OBS_IMAGES
+
+
+def calculate_episode_data_index(hf_dataset: Dataset) -> dict[str, torch.Tensor]:
+    """Calculate episode data index for testing. Returns {"from": Tensor, "to": Tensor}."""
+    episode_data_index: dict[str, list[int]] = {"from": [], "to": []}
+    current_episode = None
+    if len(hf_dataset) == 0:
+        return {"from": torch.tensor([]), "to": torch.tensor([])}
+    for idx, episode_idx in enumerate(hf_dataset["episode_index"]):
+        if episode_idx != current_episode:
+            episode_data_index["from"].append(idx)
+            if current_episode is not None:
+                episode_data_index["to"].append(idx)
+            current_episode = episode_idx
+    episode_data_index["to"].append(idx + 1)
+    return {k: torch.tensor(v) for k, v in episode_data_index.items()}
+
+
+def test_default_parameters():
+    card = create_lerobot_dataset_card()
+    assert isinstance(card, DatasetCard)
+    assert card.data.tags == ["LeRobot"]
+    assert card.data.task_categories == ["robotics"]
+    assert card.data.configs == [
+        {
+            "config_name": "default",
+            "data_files": "data/*/*.parquet",
+        }
+    ]
+
+
+def test_with_tags():
+    tags = ["tag1", "tag2"]
+    card = create_lerobot_dataset_card(tags=tags)
+    assert card.data.tags == ["LeRobot", "tag1", "tag2"]
+
+
+def test_calculate_episode_data_index():
+    dataset = Dataset.from_dict(
+        {
+            "timestamp": [0.1, 0.2, 0.3, 0.4, 0.5, 0.6],
+            "index": [0, 1, 2, 3, 4, 5],
+            "episode_index": [0, 0, 1, 2, 2, 2],
+        },
+    )
+    dataset.set_transform(hf_transform_to_torch)
+    episode_data_index = calculate_episode_data_index(dataset)
+    assert torch.equal(episode_data_index["from"], torch.tensor([0, 2, 3]))
+    assert torch.equal(episode_data_index["to"], torch.tensor([2, 3, 6]))
+
+
+def test_merge_simple_vectors():
+    g1 = {
+        ACTION: {
+            "dtype": "float32",
+            "shape": (2,),
+            "names": ["ee.x", "ee.y"],
+        }
+    }
+    g2 = {
+        ACTION: {
+            "dtype": "float32",
+            "shape": (2,),
+            "names": ["ee.y", "ee.z"],
+        }
+    }
+
+    out = combine_feature_dicts(g1, g2)
+
+    assert ACTION in out
+    assert out[ACTION]["dtype"] == "float32"
+    # Names merged with preserved order and de-dupuplication
+    assert out[ACTION]["names"] == ["ee.x", "ee.y", "ee.z"]
+    # Shape correctly recomputed from names length
+    assert out[ACTION]["shape"] == (3,)
+
+
+def test_merge_multiple_groups_order_and_dedup():
+    g1 = {ACTION: {"dtype": "float32", "shape": (2,), "names": ["a", "b"]}}
+    g2 = {ACTION: {"dtype": "float32", "shape": (2,), "names": ["b", "c"]}}
+    g3 = {ACTION: {"dtype": "float32", "shape": (3,), "names": ["a", "c", "d"]}}
+
+    out = combine_feature_dicts(g1, g2, g3)
+
+    assert out[ACTION]["names"] == ["a", "b", "c", "d"]
+    assert out[ACTION]["shape"] == (4,)
+
+
+def test_non_vector_last_wins_for_images():
+    # Non-vector (images) with same name should be overwritten by the last image specified
+    g1 = {
+        f"{OBS_IMAGES}.front": {
+            "dtype": "image",
+            "shape": (3, 480, 640),
+            "names": ["channels", "height", "width"],
+        }
+    }
+    g2 = {
+        f"{OBS_IMAGES}.front": {
+            "dtype": "image",
+            "shape": (3, 720, 1280),
+            "names": ["channels", "height", "width"],
+        }
+    }
+
+    out = combine_feature_dicts(g1, g2)
+    assert out[f"{OBS_IMAGES}.front"]["shape"] == (3, 720, 1280)
+    assert out[f"{OBS_IMAGES}.front"]["dtype"] == "image"
+
+
+def test_dtype_mismatch_raises():
+    g1 = {ACTION: {"dtype": "float32", "shape": (1,), "names": ["a"]}}
+    g2 = {ACTION: {"dtype": "float64", "shape": (1,), "names": ["b"]}}
+
+    with pytest.raises(ValueError, match="dtype mismatch for 'action'"):
+        _ = combine_feature_dicts(g1, g2)
+
+
+def test_non_dict_passthrough_last_wins():
+    g1 = {"misc": 123}
+    g2 = {"misc": 456}
+
+    out = combine_feature_dicts(g1, g2)
+    # For non-dict entries the last one wins
+    assert out["misc"] == 456
diff --git a/lerobot/tests/datasets/test_datasets.py b/lerobot/tests/datasets/test_datasets.py
new file mode 100644
index 0000000000000000000000000000000000000000..67878d8f6e2654f68ff667907967b5dc2983f556
--- /dev/null
+++ b/lerobot/tests/datasets/test_datasets.py
@@ -0,0 +1,1654 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+import re
+from itertools import chain
+from pathlib import Path
+
+import numpy as np
+import pytest
+import torch
+from huggingface_hub import HfApi
+from PIL import Image
+from safetensors.torch import load_file
+
+import lerobot
+from lerobot.configs.default import DatasetConfig
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.datasets.factory import make_dataset
+from lerobot.datasets.feature_utils import get_hf_features_from_features, hw_to_dataset_features
+from lerobot.datasets.image_writer import image_array_to_pil_image
+from lerobot.datasets.io_utils import hf_transform_to_torch
+from lerobot.datasets.lerobot_dataset import (
+    LeRobotDataset,
+    _encode_video_worker,
+)
+from lerobot.datasets.multi_dataset import MultiLeRobotDataset
+from lerobot.datasets.utils import (
+    DEFAULT_CHUNK_SIZE,
+    DEFAULT_DATA_FILE_SIZE_IN_MB,
+    DEFAULT_VIDEO_FILE_SIZE_IN_MB,
+    create_branch,
+)
+from lerobot.datasets.video_utils import VALID_VIDEO_CODECS
+from lerobot.envs.factory import make_env_config
+from lerobot.policies.factory import make_policy_config
+from lerobot.robots import make_robot_from_config
+from lerobot.utils.constants import ACTION, DONE, OBS_IMAGES, OBS_STATE, OBS_STR, REWARD
+from tests.fixtures.constants import DUMMY_CHW, DUMMY_HWC, DUMMY_REPO_ID
+from tests.mocks.mock_robot import MockRobotConfig
+from tests.utils import require_x86_64_kernel
+
+
+@pytest.fixture
+def image_dataset(tmp_path, empty_lerobot_dataset_factory):
+    features = {
+        "image": {
+            "dtype": "image",
+            "shape": DUMMY_CHW,
+            "names": [
+                "channels",
+                "height",
+                "width",
+            ],
+        }
+    }
+    return empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+
+
+def test_same_attributes_defined(tmp_path, lerobot_dataset_factory):
+    """
+    Instantiate a LeRobotDataset both ways with '__init__()' and 'create()' and verify that instantiated
+    objects have the same sets of attributes defined.
+    """
+    # Instantiate both ways
+    robot = make_robot_from_config(MockRobotConfig())
+    action_features = hw_to_dataset_features(robot.action_features, ACTION, True)
+    obs_features = hw_to_dataset_features(robot.observation_features, OBS_STR, True)
+    dataset_features = {**action_features, **obs_features}
+    root_create = tmp_path / "create"
+    dataset_create = LeRobotDataset.create(
+        repo_id=DUMMY_REPO_ID, fps=30, features=dataset_features, root=root_create
+    )
+
+    root_init = tmp_path / "init"
+    dataset_init = lerobot_dataset_factory(root=root_init, total_episodes=1, total_frames=1)
+
+    init_attr = set(vars(dataset_init).keys())
+    create_attr = set(vars(dataset_create).keys())
+
+    assert init_attr == create_attr
+
+
+def test_dataset_initialization(tmp_path, lerobot_dataset_factory):
+    kwargs = {
+        "repo_id": DUMMY_REPO_ID,
+        "total_episodes": 10,
+        "total_frames": 400,
+        "episodes": [2, 5, 6],
+    }
+    dataset = lerobot_dataset_factory(root=tmp_path / "test", **kwargs)
+
+    assert dataset.repo_id == kwargs["repo_id"]
+    assert dataset.meta.total_episodes == kwargs["total_episodes"]
+    assert dataset.meta.total_frames == kwargs["total_frames"]
+    assert dataset.episodes == kwargs["episodes"]
+    assert dataset.num_episodes == len(kwargs["episodes"])
+    assert dataset.num_frames == len(dataset)
+
+
+# TODO(rcadene, aliberts): do not run LeRobotDataset.create, instead refactor LeRobotDatasetMetadata.create
+# and test the small resulting function that validates the features
+def test_dataset_feature_with_forward_slash_raises_error():
+    # make sure dir does not exist
+    from lerobot.utils.constants import HF_LEROBOT_HOME
+
+    dataset_dir = HF_LEROBOT_HOME / "lerobot/test/with/slash"
+    # make sure does not exist
+    if dataset_dir.exists():
+        dataset_dir.rmdir()
+
+    with pytest.raises(ValueError):
+        LeRobotDataset.create(
+            repo_id="lerobot/test/with/slash",
+            fps=30,
+            features={"a/b": {"dtype": "float32", "shape": 2, "names": None}},
+        )
+
+
+def test_add_frame_missing_task(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    with pytest.raises(
+        ValueError, match="Feature mismatch in `frame` dictionary:\nMissing features: {'task'}\n"
+    ):
+        dataset.add_frame({"state": torch.randn(1)})
+
+
+def test_add_frame_missing_feature(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    with pytest.raises(
+        ValueError, match="Feature mismatch in `frame` dictionary:\nMissing features: {'state'}\n"
+    ):
+        dataset.add_frame({"task": "Dummy task"})
+
+
+def test_add_frame_extra_feature(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    with pytest.raises(
+        ValueError, match="Feature mismatch in `frame` dictionary:\nExtra features: {'extra'}\n"
+    ):
+        dataset.add_frame({"state": torch.randn(1), "task": "Dummy task", "extra": "dummy_extra"})
+
+
+def test_add_frame_wrong_type(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    with pytest.raises(
+        ValueError, match="The feature 'state' of dtype 'float16' is not of the expected dtype 'float32'.\n"
+    ):
+        dataset.add_frame({"state": torch.randn(1, dtype=torch.float16), "task": "Dummy task"})
+
+
+def test_add_frame_wrong_shape(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (2,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    with pytest.raises(
+        ValueError,
+        match=re.escape("The feature 'state' of shape '(1,)' does not have the expected shape '(2,)'.\n"),
+    ):
+        dataset.add_frame({"state": torch.randn(1), "task": "Dummy task"})
+
+
+def test_add_frame_wrong_shape_python_float(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    with pytest.raises(
+        ValueError,
+        match=re.escape(
+            "The feature 'state' is not a 'np.ndarray'. Expected type is 'float32', but type '<class 'float'>' provided instead.\n"
+        ),
+    ):
+        dataset.add_frame({"state": 1.0, "task": "Dummy task"})
+
+
+def test_add_frame_wrong_shape_torch_ndim_0(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    with pytest.raises(
+        ValueError,
+        match=re.escape("The feature 'state' of shape '()' does not have the expected shape '(1,)'.\n"),
+    ):
+        dataset.add_frame({"state": torch.tensor(1.0), "task": "Dummy task"})
+
+
+def test_add_frame_wrong_shape_numpy_ndim_0(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    with pytest.raises(
+        ValueError,
+        match=re.escape(
+            "The feature 'state' is not a 'np.ndarray'. Expected type is 'float32', but type '<class 'numpy.float32'>' provided instead.\n"
+        ),
+    ):
+        dataset.add_frame({"state": np.float32(1.0), "task": "Dummy task"})
+
+
+def test_add_frame(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    dataset.add_frame({"state": torch.randn(1), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert len(dataset) == 1
+    assert dataset[0]["task"] == "Dummy task"
+    assert dataset[0]["task_index"] == 0
+    assert dataset[0]["state"].ndim == 0
+
+
+def test_add_frame_state_1d(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (2,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    dataset.add_frame({"state": torch.randn(2), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["state"].shape == torch.Size([2])
+
+
+def test_add_frame_state_2d(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (2, 4), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    dataset.add_frame({"state": torch.randn(2, 4), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["state"].shape == torch.Size([2, 4])
+
+
+def test_add_frame_state_3d(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (2, 4, 3), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    dataset.add_frame({"state": torch.randn(2, 4, 3), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["state"].shape == torch.Size([2, 4, 3])
+
+
+def test_add_frame_state_4d(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (2, 4, 3, 5), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    dataset.add_frame({"state": torch.randn(2, 4, 3, 5), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["state"].shape == torch.Size([2, 4, 3, 5])
+
+
+def test_add_frame_state_5d(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (2, 4, 3, 5, 1), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    dataset.add_frame({"state": torch.randn(2, 4, 3, 5, 1), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["state"].shape == torch.Size([2, 4, 3, 5, 1])
+
+
+def test_add_frame_state_numpy(tmp_path, empty_lerobot_dataset_factory):
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    dataset.add_frame({"state": np.array([1], dtype=np.float32), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["state"].ndim == 0
+
+
+def test_add_frame_string(tmp_path, empty_lerobot_dataset_factory):
+    features = {"caption": {"dtype": "string", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+    dataset.add_frame({"caption": "Dummy caption", "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["caption"] == "Dummy caption"
+
+
+def test_add_frame_image_wrong_shape(image_dataset):
+    dataset = image_dataset
+    with pytest.raises(
+        ValueError,
+        match=re.escape(
+            "The feature 'image' of shape '(3, 128, 96)' does not have the expected shape '(3, 96, 128)' or '(96, 128, 3)'.\n"
+        ),
+    ):
+        c, h, w = DUMMY_CHW
+        dataset.add_frame({"image": torch.randn(c, w, h), "task": "Dummy task"})
+
+
+def test_add_frame_image_wrong_range(image_dataset):
+    """This test will display the following error message from a thread:
+    ```
+    Error writing image ...test_add_frame_image_wrong_ran0/test/images/image/episode_000000/frame_000000.png:
+    The image data type is float, which requires values in the range [0.0, 1.0]. However, the provided range is [0.009678772038470007, 254.9776492089887].
+    Please adjust the range or provide a uint8 image with values in the range [0, 255]
+    ```
+    Hence the image won't be saved on disk and save_episode will raise `FileNotFoundError`.
+    """
+    dataset = image_dataset
+    dataset.add_frame({"image": np.random.rand(*DUMMY_CHW) * 255, "task": "Dummy task"})
+    with pytest.raises(FileNotFoundError):
+        dataset.save_episode()
+
+
+def test_add_frame_image(image_dataset):
+    dataset = image_dataset
+    dataset.add_frame({"image": np.random.rand(*DUMMY_CHW), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["image"].shape == torch.Size(DUMMY_CHW)
+
+
+def test_add_frame_image_h_w_c(image_dataset):
+    dataset = image_dataset
+    dataset.add_frame({"image": np.random.rand(*DUMMY_HWC), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["image"].shape == torch.Size(DUMMY_CHW)
+
+
+def test_add_frame_image_uint8(image_dataset):
+    dataset = image_dataset
+    image = np.random.randint(0, 256, DUMMY_HWC, dtype=np.uint8)
+    dataset.add_frame({"image": image, "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["image"].shape == torch.Size(DUMMY_CHW)
+
+
+def test_add_frame_image_pil(image_dataset):
+    dataset = image_dataset
+    image = np.random.randint(0, 256, DUMMY_HWC, dtype=np.uint8)
+    dataset.add_frame({"image": Image.fromarray(image), "task": "Dummy task"})
+    dataset.save_episode()
+
+    assert dataset[0]["image"].shape == torch.Size(DUMMY_CHW)
+
+
+def test_image_array_to_pil_image_wrong_range_float_0_255():
+    image = np.random.rand(*DUMMY_HWC) * 255
+    with pytest.raises(ValueError):
+        image_array_to_pil_image(image)
+
+
+def test_tmp_image_deletion(tmp_path, empty_lerobot_dataset_factory):
+    """Verify temporary image directories are removed for image features after saving episode."""
+    # Image feature: images should be deleted after saving episode
+    image_key = "image"
+    features_image = {
+        image_key: {"dtype": "image", "shape": DUMMY_CHW, "names": ["channels", "height", "width"]}
+    }
+    ds_img = empty_lerobot_dataset_factory(root=tmp_path / "img", features=features_image)
+    ds_img.add_frame({"image": np.random.rand(*DUMMY_CHW), "task": "Dummy task"})
+    ds_img.save_episode()
+    img_dir = ds_img._get_image_file_dir(0, image_key)
+    assert not img_dir.exists(), "Temporary image directory should be removed for image features"
+
+
+def test_tmp_video_deletion(tmp_path, empty_lerobot_dataset_factory):
+    """Verify temporary image directories are removed for video encoding when `batch_encoding_size == 1`."""
+    # Video feature: when batch_encoding_size == 1 temporary images should be deleted
+    vid_key = "video"
+    features_video = {
+        vid_key: {"dtype": "video", "shape": DUMMY_CHW, "names": ["channels", "height", "width"]}
+    }
+
+    ds_vid = empty_lerobot_dataset_factory(root=tmp_path / "vid", features=features_video)
+    ds_vid.batch_encoding_size = 1
+    ds_vid.add_frame({vid_key: np.random.rand(*DUMMY_CHW), "task": "Dummy task"})
+    ds_vid.save_episode()
+    vid_img_dir = ds_vid._get_image_file_dir(0, vid_key)
+    assert not vid_img_dir.exists(), (
+        "Temporary image directory should be removed when batch_encoding_size == 1"
+    )
+
+
+def test_tmp_mixed_deletion(tmp_path, empty_lerobot_dataset_factory):
+    """Verify temporary image directories are removed appropriately when both image and video features are present."""
+    image_key = "image"
+    vid_key = "video"
+    features_mixed = {
+        image_key: {"dtype": "image", "shape": DUMMY_CHW, "names": ["channels", "height", "width"]},
+        vid_key: {"dtype": "video", "shape": DUMMY_HWC, "names": ["height", "width", "channels"]},
+    }
+    ds_mixed = empty_lerobot_dataset_factory(
+        root=tmp_path / "mixed", features=features_mixed, batch_encoding_size=2, streaming_encoding=False
+    )
+    ds_mixed.add_frame(
+        {
+            "image": np.random.rand(*DUMMY_CHW),
+            "video": np.random.rand(*DUMMY_HWC),
+            "task": "Dummy task",
+        }
+    )
+    ds_mixed.save_episode()
+    img_dir = ds_mixed._get_image_file_dir(0, image_key)
+    vid_img_dir = ds_mixed._get_image_file_dir(0, vid_key)
+    assert not img_dir.exists(), "Temporary image directory should be removed for image features"
+    assert vid_img_dir.exists(), (
+        "Temporary image directory should not be removed for video features when batch_encoding_size == 2"
+    )
+
+
+# TODO(aliberts):
+# - [ ] test various attributes & state from init and create
+# - [ ] test init with episodes and check num_frames
+# - [ ] test add_episode
+# - [ ] test push_to_hub
+# - [ ] test smaller methods
+
+# TODO(rcadene):
+# - [ ] fix code so that old test_factory + backward pass
+# - [ ] write new unit tests to test save_episode + getitem
+#   - [ ] save_episode : case where new dataset, concatenate same file, write new file (meta/episodes, data, videos)
+#   - [ ]
+# - [ ] remove old tests
+
+
+@pytest.mark.parametrize(
+    "env_name, repo_id, policy_name",
+    # Single dataset
+    lerobot.env_dataset_policy_triplets,
+    # Multi-dataset
+    # TODO after fix multidataset
+    # + [("aloha", ["lerobot/aloha_sim_insertion_human", "lerobot/aloha_sim_transfer_cube_human"], "act")],
+)
+def test_factory(env_name, repo_id, policy_name):
+    """
+    Tests that:
+        - we can create a dataset with the factory.
+        - for a commonly used set of data keys, the data dimensions are correct.
+    """
+    cfg = TrainPipelineConfig(
+        # TODO(rcadene, aliberts): remove dataset download
+        dataset=DatasetConfig(repo_id=repo_id, episodes=[0]),
+        env=make_env_config(env_name),
+        policy=make_policy_config(policy_name),
+    )
+
+    dataset = make_dataset(cfg)
+    delta_timestamps = dataset.delta_timestamps
+    camera_keys = dataset.meta.camera_keys
+
+    item = dataset[0]
+
+    keys_ndim_required = [
+        (ACTION, 1, True),
+        ("episode_index", 0, True),
+        ("frame_index", 0, True),
+        ("timestamp", 0, True),
+        # TODO(rcadene): should we rename it agent_pos?
+        (OBS_STATE, 1, True),
+        (REWARD, 0, False),
+        (DONE, 0, False),
+    ]
+
+    # test number of dimensions
+    for key, ndim, required in keys_ndim_required:
+        if key not in item:
+            if required:
+                assert key in item, f"{key}"
+            else:
+                logging.warning(f'Missing key in dataset: "{key}" not in {dataset}.')
+                continue
+
+        if delta_timestamps is not None and key in delta_timestamps:
+            assert item[key].ndim == ndim + 1, f"{key}"
+            assert item[key].shape[0] == len(delta_timestamps[key]), f"{key}"
+        else:
+            assert item[key].ndim == ndim, f"{key}"
+
+        if key in camera_keys:
+            assert item[key].dtype == torch.float32, f"{key}"
+            # TODO(rcadene): we assume for now that image normalization takes place in the model
+            assert item[key].max() <= 1.0, f"{key}"
+            assert item[key].min() >= 0.0, f"{key}"
+
+            if delta_timestamps is not None and key in delta_timestamps:
+                # test t,c,h,w
+                assert item[key].shape[1] == 3, f"{key}"
+            else:
+                # test c,h,w
+                assert item[key].shape[0] == 3, f"{key}"
+
+    if delta_timestamps is not None:
+        # test missing keys in delta_timestamps
+        for key in delta_timestamps:
+            assert key in item, f"{key}"
+
+
+# TODO(alexander-soare): If you're hunting for savings on testing time, this takes about 5 seconds.
+@pytest.mark.skip("TODO after fix multidataset")
+def test_multidataset_frames():
+    """Check that all dataset frames are incorporated."""
+    # Note: use the image variants of the dataset to make the test approx 3x faster.
+    # Note: We really do need three repo_ids here as at some point this caught an issue with the chaining
+    # logic that wouldn't be caught with two repo IDs.
+    repo_ids = [
+        "lerobot/aloha_sim_insertion_human_image",
+        "lerobot/aloha_sim_transfer_cube_human_image",
+        "lerobot/aloha_sim_insertion_scripted_image",
+    ]
+    sub_datasets = [LeRobotDataset(repo_id) for repo_id in repo_ids]
+    dataset = MultiLeRobotDataset(repo_ids)
+    assert len(dataset) == sum(len(d) for d in sub_datasets)
+    assert dataset.num_frames == sum(d.num_frames for d in sub_datasets)
+    assert dataset.num_episodes == sum(d.num_episodes for d in sub_datasets)
+
+    # Run through all items of the LeRobotDatasets in parallel with the items of the MultiLerobotDataset and
+    # check they match.
+    expected_dataset_indices = []
+    for i, sub_dataset in enumerate(sub_datasets):
+        expected_dataset_indices.extend([i] * len(sub_dataset))
+
+    for expected_dataset_index, sub_dataset_item, dataset_item in zip(
+        expected_dataset_indices, chain(*sub_datasets), dataset, strict=True
+    ):
+        dataset_index = dataset_item.pop("dataset_index")
+        assert dataset_index == expected_dataset_index
+        assert sub_dataset_item.keys() == dataset_item.keys()
+        for k in sub_dataset_item:
+            assert torch.equal(sub_dataset_item[k], dataset_item[k])
+
+
+@pytest.mark.parametrize(
+    "repo_id",
+    [
+        "lerobot/pusht",
+        "lerobot/aloha_sim_insertion_human",
+        "lerobot/xarm_lift_medium",
+        # (michel-aractingi) commenting the two datasets from openx as test is failing
+        # "lerobot/nyu_franka_play_dataset",
+        # "lerobot/cmu_stretch",
+    ],
+)
+@require_x86_64_kernel
+def test_backward_compatibility(repo_id):
+    """The artifacts for this test have been generated by `tests/artifacts/datasets/save_dataset_to_safetensors.py`."""
+
+    # TODO(rcadene, aliberts): remove dataset download
+    dataset = LeRobotDataset(repo_id, episodes=[0])
+
+    test_dir = Path("tests/artifacts/datasets") / repo_id
+
+    def load_and_compare(i):
+        new_frame = dataset[i]  # noqa: B023
+        old_frame = load_file(test_dir / f"frame_{i}.safetensors")  # noqa: B023
+
+        # ignore language instructions (if exists) in language conditioned datasets
+        # TODO (michel-aractingi): transform language obs to language embeddings via tokenizer
+        new_frame.pop("language_instruction", None)
+        old_frame.pop("language_instruction", None)
+        new_frame.pop("task", None)
+        old_frame.pop("task", None)
+
+        # Remove task_index to allow for backward compatibility
+        # TODO(rcadene): remove when new features have been generated
+        if "task_index" not in old_frame:
+            del new_frame["task_index"]
+
+        new_keys = set(new_frame.keys())
+        old_keys = set(old_frame.keys())
+        assert new_keys == old_keys, f"{new_keys=} and {old_keys=} are not the same"
+
+        for key in new_frame:
+            assert torch.isclose(new_frame[key], old_frame[key]).all(), (
+                f"{key=} for index={i} does not contain the same value"
+            )
+
+    # test2 first frames of first episode
+    i = dataset.meta.episodes[0]["dataset_from_index"]
+    load_and_compare(i)
+    load_and_compare(i + 1)
+
+    # test 2 frames at the middle of first episode
+    i = int(
+        (dataset.meta.episodes[0]["dataset_to_index"] - dataset.meta.episodes[0]["dataset_from_index"]) / 2
+    )
+    load_and_compare(i)
+    load_and_compare(i + 1)
+
+    # test 2 last frames of first episode
+    i = dataset.meta.episodes[0]["dataset_to_index"]
+    load_and_compare(i - 2)
+    load_and_compare(i - 1)
+
+
+@pytest.mark.skip("Requires internet access")
+def test_create_branch():
+    api = HfApi()
+
+    repo_id = "cadene/test_create_branch"
+    repo_type = "dataset"
+    branch = "test"
+    ref = f"refs/heads/{branch}"
+
+    # Prepare a repo with a test branch
+    api.delete_repo(repo_id, repo_type=repo_type, missing_ok=True)
+    api.create_repo(repo_id, repo_type=repo_type)
+    create_branch(repo_id, repo_type=repo_type, branch=branch)
+
+    # Make sure the test branch exists
+    branches = api.list_repo_refs(repo_id, repo_type=repo_type).branches
+    refs = [branch.ref for branch in branches]
+    assert ref in refs
+
+    # Overwrite it
+    create_branch(repo_id, repo_type=repo_type, branch=branch)
+
+    # Clean
+    api.delete_repo(repo_id, repo_type=repo_type)
+
+
+def test_check_cached_episodes_sufficient(tmp_path, lerobot_dataset_factory):
+    """Test the _check_cached_episodes_sufficient method of LeRobotDataset."""
+    # Create a dataset with 5 episodes (0-4)
+    dataset = lerobot_dataset_factory(
+        root=tmp_path / "test",
+        total_episodes=5,
+        total_frames=200,
+        use_videos=False,
+    )
+
+    # Test hf_dataset is None
+    dataset.hf_dataset = None
+    assert dataset._check_cached_episodes_sufficient() is False
+
+    # Test hf_dataset is empty
+    import datasets
+
+    empty_features = get_hf_features_from_features(dataset.features)
+    dataset.hf_dataset = datasets.Dataset.from_dict(
+        {key: [] for key in empty_features}, features=empty_features
+    )
+    dataset.hf_dataset.set_transform(hf_transform_to_torch)
+    assert dataset._check_cached_episodes_sufficient() is False
+
+    # Restore the original dataset for remaining tests
+    dataset.hf_dataset = dataset.load_hf_dataset()
+
+    # Test all episodes requested (self.episodes = None) and all are available
+    dataset.episodes = None
+    assert dataset._check_cached_episodes_sufficient() is True
+
+    # Test specific episodes requested that are all available
+    dataset.episodes = [0, 2, 4]
+    assert dataset._check_cached_episodes_sufficient() is True
+
+    # Test request episodes that don't exist in the cached dataset
+    # Create a dataset with only episodes 0, 1, 2
+    limited_dataset = lerobot_dataset_factory(
+        root=tmp_path / "limited",
+        total_episodes=3,
+        total_frames=120,
+        use_videos=False,
+    )
+
+    # Request episodes that include non-existent ones
+    limited_dataset.episodes = [0, 1, 2, 3, 4]
+    assert limited_dataset._check_cached_episodes_sufficient() is False
+
+    # Test create a dataset with sparse episodes (e.g., only episodes 0, 2, 4)
+    # First create the full dataset structure
+    sparse_dataset = lerobot_dataset_factory(
+        root=tmp_path / "sparse",
+        total_episodes=5,
+        total_frames=200,
+        use_videos=False,
+    )
+
+    # Manually filter hf_dataset to only include episodes 0, 2, 4
+    episode_indices = sparse_dataset.hf_dataset["episode_index"]
+    mask = torch.zeros(len(episode_indices), dtype=torch.bool)
+    for ep in [0, 2, 4]:
+        mask |= torch.tensor(episode_indices) == ep
+
+    # Create a filtered dataset
+    filtered_data = {}
+    # Find image keys by checking features
+    image_keys = [key for key, ft in sparse_dataset.features.items() if ft.get("dtype") == "image"]
+
+    for key in sparse_dataset.hf_dataset.column_names:
+        values = sparse_dataset.hf_dataset[key]
+        # Filter values based on mask
+        filtered_values = [val for i, val in enumerate(values) if mask[i]]
+
+        # Convert float32 image tensors back to uint8 numpy arrays for HuggingFace dataset
+        if key in image_keys and len(filtered_values) > 0:
+            # Convert torch tensors (float32, [0, 1], CHW) back to numpy arrays (uint8, [0, 255], HWC)
+            filtered_values = [
+                (val.permute(1, 2, 0).numpy() * 255).astype(np.uint8) for val in filtered_values
+            ]
+
+        filtered_data[key] = filtered_values
+
+    sparse_dataset.hf_dataset = datasets.Dataset.from_dict(
+        filtered_data, features=get_hf_features_from_features(sparse_dataset.features)
+    )
+    sparse_dataset.hf_dataset.set_transform(hf_transform_to_torch)
+
+    # Test requesting all episodes when only some are cached
+    sparse_dataset.episodes = None
+    assert sparse_dataset._check_cached_episodes_sufficient() is False
+
+    # Test requesting only the available episodes
+    sparse_dataset.episodes = [0, 2, 4]
+    assert sparse_dataset._check_cached_episodes_sufficient() is True
+
+    # Test requesting a mix of available and unavailable episodes
+    sparse_dataset.episodes = [0, 1, 2]
+    assert sparse_dataset._check_cached_episodes_sufficient() is False
+
+
+def test_update_chunk_settings(tmp_path, empty_lerobot_dataset_factory):
+    """Test the update_chunk_settings functionality for both LeRobotDataset and LeRobotDatasetMetadata."""
+    features = {
+        OBS_STATE: {
+            "dtype": "float32",
+            "shape": (6,),
+            "names": ["shoulder_pan", "shoulder_lift", "elbow", "wrist_1", "wrist_2", "wrist_3"],
+        },
+        ACTION: {
+            "dtype": "float32",
+            "shape": (6,),
+            "names": ["shoulder_pan", "shoulder_lift", "elbow", "wrist_1", "wrist_2", "wrist_3"],
+        },
+    }
+
+    # Create dataset with default chunk settings
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features)
+
+    # Test initial default values
+    initial_settings = dataset.meta.get_chunk_settings()
+    assert initial_settings["chunks_size"] == DEFAULT_CHUNK_SIZE
+    assert initial_settings["data_files_size_in_mb"] == DEFAULT_DATA_FILE_SIZE_IN_MB
+    assert initial_settings["video_files_size_in_mb"] == DEFAULT_VIDEO_FILE_SIZE_IN_MB
+
+    # Test updating all settings at once
+    new_chunks_size = 2000
+    new_data_size = 200
+    new_video_size = 1000
+
+    dataset.meta.update_chunk_settings(
+        chunks_size=new_chunks_size,
+        data_files_size_in_mb=new_data_size,
+        video_files_size_in_mb=new_video_size,
+    )
+
+    # Verify settings were updated
+    updated_settings = dataset.meta.get_chunk_settings()
+    assert updated_settings["chunks_size"] == new_chunks_size
+    assert updated_settings["data_files_size_in_mb"] == new_data_size
+    assert updated_settings["video_files_size_in_mb"] == new_video_size
+
+    # Test updating individual settings
+    dataset.meta.update_chunk_settings(chunks_size=1500)
+    settings_after_partial = dataset.meta.get_chunk_settings()
+    assert settings_after_partial["chunks_size"] == 1500
+    assert settings_after_partial["data_files_size_in_mb"] == new_data_size
+    assert settings_after_partial["video_files_size_in_mb"] == new_video_size
+
+    # Test updating only data file size
+    dataset.meta.update_chunk_settings(data_files_size_in_mb=150)
+    settings_after_data = dataset.meta.get_chunk_settings()
+    assert settings_after_data["chunks_size"] == 1500
+    assert settings_after_data["data_files_size_in_mb"] == 150
+    assert settings_after_data["video_files_size_in_mb"] == new_video_size
+
+    # Test updating only video file size
+    dataset.meta.update_chunk_settings(video_files_size_in_mb=800)
+    settings_after_video = dataset.meta.get_chunk_settings()
+    assert settings_after_video["chunks_size"] == 1500
+    assert settings_after_video["data_files_size_in_mb"] == 150
+    assert settings_after_video["video_files_size_in_mb"] == 800
+
+    # Test that settings persist in the info file
+    info_path = dataset.root / "meta" / "info.json"
+    assert info_path.exists()
+
+    # Verify the underlying metadata properties
+    assert dataset.meta.chunks_size == 1500
+    assert dataset.meta.data_files_size_in_mb == 150
+    assert dataset.meta.video_files_size_in_mb == 800
+
+    # Test error handling for invalid values
+    with pytest.raises(ValueError, match="chunks_size must be positive"):
+        dataset.meta.update_chunk_settings(chunks_size=0)
+
+    with pytest.raises(ValueError, match="chunks_size must be positive"):
+        dataset.meta.update_chunk_settings(chunks_size=-100)
+
+    with pytest.raises(ValueError, match="data_files_size_in_mb must be positive"):
+        dataset.meta.update_chunk_settings(data_files_size_in_mb=0)
+
+    with pytest.raises(ValueError, match="data_files_size_in_mb must be positive"):
+        dataset.meta.update_chunk_settings(data_files_size_in_mb=-50)
+
+    with pytest.raises(ValueError, match="video_files_size_in_mb must be positive"):
+        dataset.meta.update_chunk_settings(video_files_size_in_mb=0)
+
+    with pytest.raises(ValueError, match="video_files_size_in_mb must be positive"):
+        dataset.meta.update_chunk_settings(video_files_size_in_mb=-200)
+
+    # Test calling with None values (should not change anything)
+    settings_before_none = dataset.meta.get_chunk_settings()
+    dataset.meta.update_chunk_settings(
+        chunks_size=None, data_files_size_in_mb=None, video_files_size_in_mb=None
+    )
+    settings_after_none = dataset.meta.get_chunk_settings()
+    assert settings_before_none == settings_after_none
+
+    # Test metadata direct access
+    meta_settings = dataset.meta.get_chunk_settings()
+    assert meta_settings == dataset.meta.get_chunk_settings()
+
+    # Test updating via metadata directly
+    dataset.meta.update_chunk_settings(chunks_size=3000)
+    assert dataset.meta.get_chunk_settings()["chunks_size"] == 3000
+
+
+def test_update_chunk_settings_video_dataset(tmp_path):
+    """Test update_chunk_settings with a video dataset to ensure video-specific logic works."""
+    features = {
+        f"{OBS_IMAGES}.cam": {
+            "dtype": "video",
+            "shape": (480, 640, 3),
+            "names": ["height", "width", "channels"],
+        },
+        ACTION: {"dtype": "float32", "shape": (6,), "names": ["j1", "j2", "j3", "j4", "j5", "j6"]},
+    }
+
+    # Create video dataset
+    dataset = LeRobotDataset.create(
+        repo_id=DUMMY_REPO_ID, fps=30, features=features, root=tmp_path / "video_test", use_videos=True
+    )
+
+    # Test that video-specific settings work
+    original_video_size = dataset.meta.get_chunk_settings()["video_files_size_in_mb"]
+    new_video_size = original_video_size * 2
+
+    dataset.meta.update_chunk_settings(video_files_size_in_mb=new_video_size)
+    assert dataset.meta.get_chunk_settings()["video_files_size_in_mb"] == new_video_size
+    assert dataset.meta.video_files_size_in_mb == new_video_size
+
+
+def test_episode_index_distribution(tmp_path, empty_lerobot_dataset_factory):
+    """Test that all frames have correct episode indices across multiple episodes."""
+    features = {"state": {"dtype": "float32", "shape": (2,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features, use_videos=False)
+
+    # Create 3 episodes with different lengths
+    num_episodes = 3
+    frames_per_episode = [10, 15, 8]
+
+    for episode_idx in range(num_episodes):
+        for _ in range(frames_per_episode[episode_idx]):
+            dataset.add_frame({"state": torch.randn(2), "task": f"task_{episode_idx}"})
+        dataset.save_episode()
+
+    dataset.finalize()
+
+    # Load the dataset and check episode indices
+    loaded_dataset = LeRobotDataset(dataset.repo_id, root=dataset.root)
+
+    # Check specific frames across episode boundaries
+    cumulative = 0
+    for ep_idx, ep_length in enumerate(frames_per_episode):
+        # Check start, middle, and end of each episode
+        start_frame = cumulative
+        middle_frame = cumulative + ep_length // 2
+        end_frame = cumulative + ep_length - 1
+
+        for frame_idx in [start_frame, middle_frame, end_frame]:
+            frame_data = loaded_dataset[frame_idx]
+            actual_ep_idx = frame_data["episode_index"].item()
+            assert actual_ep_idx == ep_idx, (
+                f"Frame {frame_idx} has episode_index {actual_ep_idx}, should be {ep_idx}"
+            )
+
+        cumulative += ep_length
+
+    # Check episode index distribution
+    all_episode_indices = [loaded_dataset[i]["episode_index"].item() for i in range(len(loaded_dataset))]
+    from collections import Counter
+
+    distribution = Counter(all_episode_indices)
+    expected_dist = {i: frames_per_episode[i] for i in range(num_episodes)}
+
+    assert dict(distribution) == expected_dist, (
+        f"Episode distribution {dict(distribution)} != expected {expected_dist}"
+    )
+
+
+def test_multi_episode_metadata_consistency(tmp_path, empty_lerobot_dataset_factory):
+    """Test episode metadata consistency across multiple episodes."""
+    features = {
+        "state": {"dtype": "float32", "shape": (3,), "names": ["x", "y", "z"]},
+        ACTION: {"dtype": "float32", "shape": (2,), "names": ["v", "w"]},
+    }
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features, use_videos=False)
+
+    num_episodes = 4
+    frames_per_episode = [20, 35, 10, 25]
+    tasks = ["pick", "place", "pick", "place"]
+
+    for episode_idx in range(num_episodes):
+        for _ in range(frames_per_episode[episode_idx]):
+            dataset.add_frame({"state": torch.randn(3), ACTION: torch.randn(2), "task": tasks[episode_idx]})
+        dataset.save_episode()
+
+    dataset.finalize()
+
+    # Load and validate episode metadata
+    loaded_dataset = LeRobotDataset(dataset.repo_id, root=dataset.root)
+
+    assert loaded_dataset.meta.total_episodes == num_episodes
+    assert loaded_dataset.meta.total_frames == sum(frames_per_episode)
+
+    cumulative_frames = 0
+    for episode_idx in range(num_episodes):
+        episode_metadata = loaded_dataset.meta.episodes[episode_idx]
+
+        # Check basic episode properties
+        assert episode_metadata["episode_index"] == episode_idx
+        assert episode_metadata["length"] == frames_per_episode[episode_idx]
+        assert episode_metadata["tasks"] == [tasks[episode_idx]]
+
+        # Check dataset indices
+        expected_from = cumulative_frames
+        expected_to = cumulative_frames + frames_per_episode[episode_idx]
+
+        assert episode_metadata["dataset_from_index"] == expected_from
+        assert episode_metadata["dataset_to_index"] == expected_to
+
+        cumulative_frames += frames_per_episode[episode_idx]
+
+
+def test_data_consistency_across_episodes(tmp_path, empty_lerobot_dataset_factory):
+    """Test that episodes have no gaps or overlaps in their data indices."""
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features, use_videos=False)
+
+    num_episodes = 5
+    frames_per_episode = [12, 8, 20, 15, 5]
+
+    for episode_idx in range(num_episodes):
+        for _ in range(frames_per_episode[episode_idx]):
+            dataset.add_frame({"state": torch.randn(1), "task": "consistency_test"})
+        dataset.save_episode()
+
+    dataset.finalize()
+
+    loaded_dataset = LeRobotDataset(dataset.repo_id, root=dataset.root)
+
+    # Check data consistency - no gaps or overlaps
+    cumulative_check = 0
+    for episode_idx in range(num_episodes):
+        episode_metadata = loaded_dataset.meta.episodes[episode_idx]
+        from_idx = episode_metadata["dataset_from_index"]
+        to_idx = episode_metadata["dataset_to_index"]
+
+        # Check that episode starts exactly where previous ended
+        assert from_idx == cumulative_check, (
+            f"Episode {episode_idx} starts at {from_idx}, expected {cumulative_check}"
+        )
+
+        # Check that episode length matches expected
+        actual_length = to_idx - from_idx
+        expected_length = frames_per_episode[episode_idx]
+        assert actual_length == expected_length, (
+            f"Episode {episode_idx} length {actual_length} != expected {expected_length}"
+        )
+
+        cumulative_check = to_idx
+
+    # Final check: last episode should end at total frames
+    expected_total_frames = sum(frames_per_episode)
+    assert cumulative_check == expected_total_frames, (
+        f"Final frame count {cumulative_check} != expected {expected_total_frames}"
+    )
+
+
+def test_statistics_metadata_validation(tmp_path, empty_lerobot_dataset_factory):
+    """Test that statistics are properly computed and stored for all features."""
+    features = {
+        "state": {"dtype": "float32", "shape": (2,), "names": ["pos", "vel"]},
+        ACTION: {"dtype": "float32", "shape": (1,), "names": ["force"]},
+    }
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features, use_videos=False)
+
+    # Create controlled data to verify statistics
+    num_episodes = 2
+    frames_per_episode = [10, 10]
+
+    # Use deterministic data for predictable statistics
+    torch.manual_seed(42)
+    for episode_idx in range(num_episodes):
+        for frame_idx in range(frames_per_episode[episode_idx]):
+            state_data = torch.tensor([frame_idx * 0.1, frame_idx * 0.2], dtype=torch.float32)
+            action_data = torch.tensor([frame_idx * 0.05], dtype=torch.float32)
+            dataset.add_frame({"state": state_data, ACTION: action_data, "task": "stats_test"})
+        dataset.save_episode()
+
+    dataset.finalize()
+
+    loaded_dataset = LeRobotDataset(dataset.repo_id, root=dataset.root)
+
+    # Check that statistics exist for all features
+    assert loaded_dataset.meta.stats is not None, "No statistics found"
+
+    for feature_name in features:
+        assert feature_name in loaded_dataset.meta.stats, f"No statistics for feature '{feature_name}'"
+
+        feature_stats = loaded_dataset.meta.stats[feature_name]
+        expected_stats = ["min", "max", "mean", "std", "count"]
+
+        for stat_key in expected_stats:
+            assert stat_key in feature_stats, f"Missing '{stat_key}' statistic for '{feature_name}'"
+
+            stat_value = feature_stats[stat_key]
+            # Basic sanity checks
+            if stat_key == "count":
+                assert stat_value == sum(frames_per_episode), f"Wrong count for '{feature_name}'"
+            elif stat_key in ["min", "max", "mean", "std"]:
+                # Check that statistics are reasonable (not NaN, proper shapes)
+                if hasattr(stat_value, "shape"):
+                    expected_shape = features[feature_name]["shape"]
+                    assert stat_value.shape == expected_shape or len(stat_value) == expected_shape[0], (
+                        f"Wrong shape for {stat_key} of '{feature_name}'"
+                    )
+                # Check no NaN values
+                if hasattr(stat_value, "__iter__"):
+                    assert not any(np.isnan(v) for v in stat_value), f"NaN in {stat_key} for '{feature_name}'"
+                else:
+                    assert not np.isnan(stat_value), f"NaN in {stat_key} for '{feature_name}'"
+
+
+def test_episode_boundary_integrity(tmp_path, empty_lerobot_dataset_factory):
+    """Test frame indices and episode transitions at episode boundaries."""
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features, use_videos=False)
+
+    num_episodes = 3
+    frames_per_episode = [7, 12, 5]
+
+    for episode_idx in range(num_episodes):
+        for frame_idx in range(frames_per_episode[episode_idx]):
+            dataset.add_frame({"state": torch.tensor([float(frame_idx)]), "task": f"episode_{episode_idx}"})
+        dataset.save_episode()
+
+    dataset.finalize()
+
+    loaded_dataset = LeRobotDataset(dataset.repo_id, root=dataset.root)
+
+    # Test episode boundaries
+    cumulative = 0
+    for ep_idx, ep_length in enumerate(frames_per_episode):
+        if ep_idx > 0:
+            # Check last frame of previous episode
+            prev_frame = loaded_dataset[cumulative - 1]
+            assert prev_frame["episode_index"].item() == ep_idx - 1
+
+        # Check first frame of current episode
+        if cumulative < len(loaded_dataset):
+            curr_frame = loaded_dataset[cumulative]
+            assert curr_frame["episode_index"].item() == ep_idx
+
+        # Check frame_index within episode
+        for i in range(ep_length):
+            if cumulative + i < len(loaded_dataset):
+                frame = loaded_dataset[cumulative + i]
+                assert frame["frame_index"].item() == i, f"Frame {cumulative + i} has wrong frame_index"
+                assert frame["episode_index"].item() == ep_idx, (
+                    f"Frame {cumulative + i} has wrong episode_index"
+                )
+
+        cumulative += ep_length
+
+
+def test_task_indexing_and_validation(tmp_path, empty_lerobot_dataset_factory):
+    """Test that tasks are properly indexed and retrievable."""
+    features = {"state": {"dtype": "float32", "shape": (1,), "names": None}}
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features, use_videos=False)
+
+    # Use multiple tasks, including repeated ones
+    tasks = ["pick", "place", "pick", "navigate", "place"]
+    unique_tasks = list(set(tasks))  # ["pick", "place", "navigate"]
+    frames_per_episode = [5, 8, 3, 10, 6]
+
+    for episode_idx, task in enumerate(tasks):
+        for _ in range(frames_per_episode[episode_idx]):
+            dataset.add_frame({"state": torch.randn(1), "task": task})
+        dataset.save_episode()
+
+    dataset.finalize()
+
+    loaded_dataset = LeRobotDataset(dataset.repo_id, root=dataset.root)
+
+    # Check that all unique tasks are in the tasks metadata
+    stored_tasks = set(loaded_dataset.meta.tasks.index)
+    assert stored_tasks == set(unique_tasks), f"Stored tasks {stored_tasks} != expected {set(unique_tasks)}"
+
+    # Check that task indices are consistent
+    cumulative = 0
+    for episode_idx, expected_task in enumerate(tasks):
+        episode_metadata = loaded_dataset.meta.episodes[episode_idx]
+        assert episode_metadata["tasks"] == [expected_task]
+
+        # Check frames in this episode have correct task
+        for i in range(frames_per_episode[episode_idx]):
+            frame = loaded_dataset[cumulative + i]
+            assert frame["task"] == expected_task, f"Frame {cumulative + i} has wrong task"
+
+            # Check task_index consistency
+            expected_task_index = loaded_dataset.meta.get_task_index(expected_task)
+            assert frame["task_index"].item() == expected_task_index
+
+        cumulative += frames_per_episode[episode_idx]
+
+    # Check total number of tasks
+    assert loaded_dataset.meta.total_tasks == len(unique_tasks)
+
+
+def test_dataset_resume_recording(tmp_path, empty_lerobot_dataset_factory):
+    """Test that resuming dataset recording preserves previously recorded episodes.
+
+    This test validates the critical resume functionality by:
+    1. Recording initial episodes and finalizing
+    2. Reopening the dataset
+    3. Recording additional episodes
+    4. Verifying all data (old + new) is intact
+
+    This specifically tests the bug fix where parquet files were being overwritten
+    instead of appended to during resume.
+    """
+    features = {
+        "observation.state": {"dtype": "float32", "shape": (2,), "names": ["x", "y"]},
+        "action": {"dtype": "float32", "shape": (2,), "names": ["x", "y"]},
+    }
+
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features, use_videos=False)
+
+    initial_episodes = 2
+    frames_per_episode = 3
+
+    for ep_idx in range(initial_episodes):
+        for frame_idx in range(frames_per_episode):
+            dataset.add_frame(
+                {
+                    "observation.state": torch.tensor([float(ep_idx), float(frame_idx)]),
+                    "action": torch.tensor([0.5, 0.5]),
+                    "task": f"task_{ep_idx}",
+                }
+            )
+        dataset.save_episode()
+
+    assert dataset.meta.total_episodes == initial_episodes
+    assert dataset.meta.total_frames == initial_episodes * frames_per_episode
+
+    dataset.finalize()
+    initial_root = dataset.root
+    initial_repo_id = dataset.repo_id
+    del dataset
+
+    dataset_verify = LeRobotDataset(initial_repo_id, root=initial_root, revision="v3.0")
+    assert dataset_verify.meta.total_episodes == initial_episodes
+    assert dataset_verify.meta.total_frames == initial_episodes * frames_per_episode
+    assert len(dataset_verify.hf_dataset) == initial_episodes * frames_per_episode
+
+    for idx in range(len(dataset_verify.hf_dataset)):
+        item = dataset_verify[idx]
+        expected_ep = idx // frames_per_episode
+        expected_frame = idx % frames_per_episode
+        assert item["episode_index"].item() == expected_ep
+        assert item["frame_index"].item() == expected_frame
+        assert item["index"].item() == idx
+        assert item["observation.state"][0].item() == float(expected_ep)
+        assert item["observation.state"][1].item() == float(expected_frame)
+
+    del dataset_verify
+
+    # Phase 3: Resume recording - add more episodes
+    dataset_resumed = LeRobotDataset(initial_repo_id, root=initial_root, revision="v3.0")
+
+    assert dataset_resumed.meta.total_episodes == initial_episodes
+    assert dataset_resumed.meta.total_frames == initial_episodes * frames_per_episode
+    assert dataset_resumed.latest_episode is None  # Not recording yet
+    assert dataset_resumed.writer is None
+    assert dataset_resumed.meta.writer is None
+
+    additional_episodes = 2
+    for ep_idx in range(initial_episodes, initial_episodes + additional_episodes):
+        for frame_idx in range(frames_per_episode):
+            dataset_resumed.add_frame(
+                {
+                    "observation.state": torch.tensor([float(ep_idx), float(frame_idx)]),
+                    "action": torch.tensor([0.5, 0.5]),
+                    "task": f"task_{ep_idx}",
+                }
+            )
+        dataset_resumed.save_episode()
+
+    total_episodes = initial_episodes + additional_episodes
+    total_frames = total_episodes * frames_per_episode
+    assert dataset_resumed.meta.total_episodes == total_episodes
+    assert dataset_resumed.meta.total_frames == total_frames
+
+    dataset_resumed.finalize()
+    del dataset_resumed
+
+    dataset_final = LeRobotDataset(initial_repo_id, root=initial_root, revision="v3.0")
+
+    assert dataset_final.meta.total_episodes == total_episodes
+    assert dataset_final.meta.total_frames == total_frames
+    assert len(dataset_final.hf_dataset) == total_frames
+
+    for idx in range(total_frames):
+        item = dataset_final[idx]
+        expected_ep = idx // frames_per_episode
+        expected_frame = idx % frames_per_episode
+
+        assert item["episode_index"].item() == expected_ep, (
+            f"Frame {idx}: wrong episode_index. Expected {expected_ep}, got {item['episode_index'].item()}"
+        )
+        assert item["frame_index"].item() == expected_frame, (
+            f"Frame {idx}: wrong frame_index. Expected {expected_frame}, got {item['frame_index'].item()}"
+        )
+        assert item["index"].item() == idx, (
+            f"Frame {idx}: wrong index. Expected {idx}, got {item['index'].item()}"
+        )
+
+        # Verify data integrity
+        assert item["observation.state"][0].item() == float(expected_ep), (
+            f"Frame {idx}: wrong observation.state[0]. Expected {float(expected_ep)}, "
+            f"got {item['observation.state'][0].item()}"
+        )
+        assert item["observation.state"][1].item() == float(expected_frame), (
+            f"Frame {idx}: wrong observation.state[1]. Expected {float(expected_frame)}, "
+            f"got {item['observation.state'][1].item()}"
+        )
+
+    assert len(dataset_final.meta.episodes) == total_episodes
+    for ep_idx in range(total_episodes):
+        ep_metadata = dataset_final.meta.episodes[ep_idx]
+        assert ep_metadata["episode_index"] == ep_idx
+        assert ep_metadata["length"] == frames_per_episode
+        assert ep_metadata["tasks"] == [f"task_{ep_idx}"]
+
+        expected_from = ep_idx * frames_per_episode
+        expected_to = (ep_idx + 1) * frames_per_episode
+        assert ep_metadata["dataset_from_index"] == expected_from
+        assert ep_metadata["dataset_to_index"] == expected_to
+
+
+def test_frames_in_current_file_calculation(tmp_path, empty_lerobot_dataset_factory):
+    """Regression test for bug where frames_in_current_file only counted frames from last episode instead of all frames in current file."""
+    features = {
+        "observation.state": {"dtype": "float32", "shape": (2,), "names": ["x", "y"]},
+        "action": {"dtype": "float32", "shape": (2,), "names": ["vx", "vy"]},
+    }
+
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features, use_videos=False)
+    dataset.meta.update_chunk_settings(data_files_size_in_mb=100)
+
+    assert dataset._current_file_start_frame is None
+
+    frames_per_episode = 10
+    for _ in range(frames_per_episode):
+        dataset.add_frame(
+            {
+                "observation.state": torch.randn(2),
+                "action": torch.randn(2),
+                "task": "task_0",
+            }
+        )
+    dataset.save_episode()
+
+    assert dataset._current_file_start_frame == 0
+    assert dataset.meta.total_episodes == 1
+    assert dataset.meta.total_frames == frames_per_episode
+
+    for _ in range(frames_per_episode):
+        dataset.add_frame(
+            {
+                "observation.state": torch.randn(2),
+                "action": torch.randn(2),
+                "task": "task_1",
+            }
+        )
+    dataset.save_episode()
+
+    assert dataset._current_file_start_frame == 0
+    assert dataset.meta.total_episodes == 2
+    assert dataset.meta.total_frames == 2 * frames_per_episode
+
+    ep1_chunk = dataset.latest_episode["data/chunk_index"]
+    ep1_file = dataset.latest_episode["data/file_index"]
+    assert ep1_chunk == 0
+    assert ep1_file == 0
+
+    for _ in range(frames_per_episode):
+        dataset.add_frame(
+            {
+                "observation.state": torch.randn(2),
+                "action": torch.randn(2),
+                "task": "task_2",
+            }
+        )
+    dataset.save_episode()
+
+    assert dataset._current_file_start_frame == 0
+    assert dataset.meta.total_episodes == 3
+    assert dataset.meta.total_frames == 3 * frames_per_episode
+
+    ep2_chunk = dataset.latest_episode["data/chunk_index"]
+    ep2_file = dataset.latest_episode["data/file_index"]
+    assert ep2_chunk == 0
+    assert ep2_file == 0
+
+    dataset.finalize()
+
+    from lerobot.datasets.io_utils import load_episodes
+
+    dataset.meta.episodes = load_episodes(dataset.root)
+    assert dataset.meta.episodes is not None
+
+    for ep_idx in range(3):
+        ep_metadata = dataset.meta.episodes[ep_idx]
+        assert ep_metadata["data/chunk_index"] == 0
+        assert ep_metadata["data/file_index"] == 0
+
+        expected_from = ep_idx * frames_per_episode
+        expected_to = (ep_idx + 1) * frames_per_episode
+        assert ep_metadata["dataset_from_index"] == expected_from
+        assert ep_metadata["dataset_to_index"] == expected_to
+
+    loaded_dataset = LeRobotDataset(dataset.repo_id, root=dataset.root)
+    assert len(loaded_dataset) == 3 * frames_per_episode
+    assert loaded_dataset.meta.total_episodes == 3
+    assert loaded_dataset.meta.total_frames == 3 * frames_per_episode
+
+    for idx in range(len(loaded_dataset)):
+        frame = loaded_dataset[idx]
+        expected_ep = idx // frames_per_episode
+        assert frame["episode_index"].item() == expected_ep
+
+
+def test_encode_video_worker_forwards_vcodec(tmp_path):
+    """Test that _encode_video_worker correctly forwards the vcodec parameter to encode_video_frames."""
+    from unittest.mock import patch
+
+    from lerobot.datasets.utils import DEFAULT_IMAGE_PATH
+
+    # Create the expected directory structure
+    video_key = "observation.images.laptop"
+    episode_index = 0
+    frame_index = 0
+
+    fpath = DEFAULT_IMAGE_PATH.format(
+        image_key=video_key, episode_index=episode_index, frame_index=frame_index
+    )
+    img_dir = tmp_path / Path(fpath).parent
+    img_dir.mkdir(parents=True, exist_ok=True)
+
+    # Create a dummy image file
+    dummy_img = Image.new("RGB", (64, 64), color="red")
+    dummy_img.save(img_dir / "frame-000000.png")
+
+    # Track what vcodec was passed to encode_video_frames
+    captured_kwargs = {}
+
+    def mock_encode_video_frames(imgs_dir, video_path, fps, **kwargs):
+        captured_kwargs.update(kwargs)
+        # Create a dummy output file so the worker doesn't fail
+        Path(video_path).parent.mkdir(parents=True, exist_ok=True)
+        Path(video_path).touch()
+
+    with patch("lerobot.datasets.lerobot_dataset.encode_video_frames", side_effect=mock_encode_video_frames):
+        # Test with h264 codec
+        _encode_video_worker(video_key, episode_index, tmp_path, fps=30, vcodec="h264")
+
+    assert "vcodec" in captured_kwargs
+    assert captured_kwargs["vcodec"] == "h264"
+
+
+def test_encode_video_worker_default_vcodec(tmp_path):
+    """Test that _encode_video_worker uses libsvtav1 as the default codec."""
+    from unittest.mock import patch
+
+    from lerobot.datasets.utils import DEFAULT_IMAGE_PATH
+
+    # Create the expected directory structure
+    video_key = "observation.images.laptop"
+    episode_index = 0
+    frame_index = 0
+
+    fpath = DEFAULT_IMAGE_PATH.format(
+        image_key=video_key, episode_index=episode_index, frame_index=frame_index
+    )
+    img_dir = tmp_path / Path(fpath).parent
+    img_dir.mkdir(parents=True, exist_ok=True)
+
+    # Create a dummy image file
+    dummy_img = Image.new("RGB", (64, 64), color="red")
+    dummy_img.save(img_dir / "frame-000000.png")
+
+    # Track what vcodec was passed to encode_video_frames
+    captured_kwargs = {}
+
+    def mock_encode_video_frames(imgs_dir, video_path, fps, **kwargs):
+        captured_kwargs.update(kwargs)
+        # Create a dummy output file so the worker doesn't fail
+        Path(video_path).parent.mkdir(parents=True, exist_ok=True)
+        Path(video_path).touch()
+
+    with patch("lerobot.datasets.lerobot_dataset.encode_video_frames", side_effect=mock_encode_video_frames):
+        # Test with default codec (no vcodec specified)
+        _encode_video_worker(video_key, episode_index, tmp_path, fps=30)
+
+    assert "vcodec" in captured_kwargs
+    assert captured_kwargs["vcodec"] == "libsvtav1"
+
+
+def test_lerobot_dataset_vcodec_validation():
+    """Test that LeRobotDataset validates the vcodec parameter."""
+    # Test that invalid vcodec raises ValueError
+    with pytest.raises(ValueError, match="Invalid vcodec"):
+        LeRobotDataset.__new__(LeRobotDataset)  # bypass __init__ to test validation directly
+        # Actually test via create since it's easier
+        LeRobotDataset.create(
+            repo_id="test/invalid_codec",
+            fps=30,
+            features={"observation.state": {"dtype": "float32", "shape": (2,), "names": ["x", "y"]}},
+            vcodec="invalid_codec",
+        )
+
+
+def test_valid_video_codecs_constant():
+    """Test that VALID_VIDEO_CODECS contains the expected codecs."""
+    assert "h264" in VALID_VIDEO_CODECS
+    assert "hevc" in VALID_VIDEO_CODECS
+    assert "libsvtav1" in VALID_VIDEO_CODECS
+    assert "auto" in VALID_VIDEO_CODECS
+    assert "h264_videotoolbox" in VALID_VIDEO_CODECS
+    assert "h264_nvenc" in VALID_VIDEO_CODECS
+    assert len(VALID_VIDEO_CODECS) == 10
+
+
+def test_delta_timestamps_with_episodes_filter(tmp_path, empty_lerobot_dataset_factory):
+    """Regression test for bug where delta_timestamps incorrectly marked all frames as padded when using episodes filter.
+
+    The bug occurred because _get_query_indices was using the relative index (idx) in the filtered dataset
+    instead of the absolute index when comparing against episode boundaries (ep_start, ep_end).
+    """
+    features = {
+        "observation.state": {"dtype": "float32", "shape": (2,), "names": ["x", "y"]},
+        "action": {"dtype": "float32", "shape": (2,), "names": ["vx", "vy"]},
+    }
+
+    dataset = empty_lerobot_dataset_factory(root=tmp_path / "test", features=features, use_videos=False)
+
+    # Create 3 episodes with 10 frames each
+    frames_per_episode = 10
+    for ep_idx in range(3):
+        for frame_idx in range(frames_per_episode):
+            dataset.add_frame(
+                {
+                    "observation.state": torch.tensor([ep_idx, frame_idx], dtype=torch.float32),
+                    "action": torch.randn(2),
+                    "task": f"task_{ep_idx}",
+                }
+            )
+        dataset.save_episode()
+    dataset.finalize()
+
+    # Load only episode 1 (middle episode) with delta_timestamps
+    delta_ts = {"observation.state": [0.0]}  # Just the current frame
+    filtered_dataset = LeRobotDataset(
+        dataset.repo_id,
+        root=dataset.root,
+        episodes=[1],
+        delta_timestamps=delta_ts,
+    )
+
+    # Verify the filtered dataset has the correct length
+    assert len(filtered_dataset) == frames_per_episode
+
+    # Check that no frames are marked as padded (since delta=0 should always be valid)
+    for idx in range(len(filtered_dataset)):
+        frame = filtered_dataset[idx]
+        assert frame["observation.state_is_pad"].item() is False, f"Frame {idx} incorrectly marked as padded"
+        # Verify we're getting data from episode 1
+        assert frame["episode_index"].item() == 1
+
+
+def test_delta_timestamps_padding_at_episode_boundaries(tmp_path, empty_lerobot_dataset_factory):
+    """Test that delta_timestamps correctly marks padding at episode boundaries when using episodes filter."""
+    features = {
+        "observation.state": {"dtype": "float32", "shape": (2,), "names": ["x", "y"]},
+        "action": {"dtype": "float32", "shape": (2,), "names": ["vx", "vy"]},
+    }
+
+    dataset = empty_lerobot_dataset_factory(
+        root=tmp_path / "test", features=features, use_videos=False, fps=10
+    )
+
+    # Create 3 episodes with 5 frames each
+    frames_per_episode = 5
+    for ep_idx in range(3):
+        for frame_idx in range(frames_per_episode):
+            dataset.add_frame(
+                {
+                    "observation.state": torch.tensor([ep_idx, frame_idx], dtype=torch.float32),
+                    "action": torch.randn(2),
+                    "task": f"task_{ep_idx}",
+                }
+            )
+        dataset.save_episode()
+    dataset.finalize()
+
+    # Load only episode 1 with delta_timestamps that go beyond episode boundaries
+    # fps=10, so 0.1s = 1 frame offset
+    delta_ts = {"observation.state": [-0.2, -0.1, 0.0, 0.1, 0.2]}  # -2, -1, 0, +1, +2 frames
+    filtered_dataset = LeRobotDataset(
+        dataset.repo_id,
+        root=dataset.root,
+        episodes=[1],
+        delta_timestamps=delta_ts,
+        tolerance_s=0.04,  # Slightly less than half a frame at 10fps
+    )
+
+    assert len(filtered_dataset) == frames_per_episode
+
+    # Check padding at the start of the episode (first frame)
+    first_frame = filtered_dataset[0]
+    is_pad = first_frame["observation.state_is_pad"].tolist()
+    # At frame 0 of episode 1: delta -2 and -1 should be padded, 0, +1, +2 should not
+    assert is_pad == [True, True, False, False, False], f"First frame padding incorrect: {is_pad}"
+
+    # Check middle frame (no padding expected)
+    mid_frame = filtered_dataset[2]
+    is_pad = mid_frame["observation.state_is_pad"].tolist()
+    assert is_pad == [False, False, False, False, False], f"Middle frame padding incorrect: {is_pad}"
+
+    # Check padding at the end of the episode (last frame)
+    last_frame = filtered_dataset[4]
+    is_pad = last_frame["observation.state_is_pad"].tolist()
+    # At frame 4 of episode 1: delta -2, -1, 0 should not be padded, +1, +2 should be
+    assert is_pad == [False, False, False, True, True], f"Last frame padding incorrect: {is_pad}"
+
+
+def test_delta_timestamps_multiple_episodes_filter(tmp_path, empty_lerobot_dataset_factory):
+    """Test delta_timestamps with multiple non-consecutive episodes selected."""
+    features = {
+        "observation.state": {"dtype": "float32", "shape": (2,), "names": ["x", "y"]},
+    }
+
+    dataset = empty_lerobot_dataset_factory(
+        root=tmp_path / "test", features=features, use_videos=False, fps=10
+    )
+
+    # Create 5 episodes with 5 frames each
+    frames_per_episode = 5
+    for ep_idx in range(5):
+        for frame_idx in range(frames_per_episode):
+            dataset.add_frame(
+                {
+                    "observation.state": torch.tensor([ep_idx, frame_idx], dtype=torch.float32),
+                    "task": f"task_{ep_idx}",
+                }
+            )
+        dataset.save_episode()
+    dataset.finalize()
+
+    # Load episodes 1 and 3 (non-consecutive)
+    delta_ts = {"observation.state": [0.0]}
+    filtered_dataset = LeRobotDataset(
+        dataset.repo_id,
+        root=dataset.root,
+        episodes=[1, 3],
+        delta_timestamps=delta_ts,
+    )
+
+    assert len(filtered_dataset) == 2 * frames_per_episode
+
+    # All frames should have valid (non-padded) data for delta=0
+    for idx in range(len(filtered_dataset)):
+        frame = filtered_dataset[idx]
+        assert frame["observation.state_is_pad"].item() is False
+
+    # Verify we're getting the correct episodes
+    episode_indices = [filtered_dataset[i]["episode_index"].item() for i in range(len(filtered_dataset))]
+    expected_episodes = [1] * frames_per_episode + [3] * frames_per_episode
+    assert episode_indices == expected_episodes
+
+
+def test_delta_timestamps_query_returns_correct_values(tmp_path, empty_lerobot_dataset_factory):
+    """Test that delta_timestamps returns the correct observation values, not just correct padding."""
+    features = {
+        "observation.state": {"dtype": "float32", "shape": (1,), "names": ["x"]},
+    }
+
+    dataset = empty_lerobot_dataset_factory(
+        root=tmp_path / "test", features=features, use_videos=False, fps=10
+    )
+
+    # Create 2 episodes with known values
+    # Episode 0: frames with values 0, 1, 2, 3, 4
+    # Episode 1: frames with values 10, 11, 12, 13, 14
+    frames_per_episode = 5
+    for ep_idx in range(2):
+        for frame_idx in range(frames_per_episode):
+            value = ep_idx * 10 + frame_idx
+            dataset.add_frame(
+                {
+                    "observation.state": torch.tensor([value], dtype=torch.float32),
+                    "task": f"task_{ep_idx}",
+                }
+            )
+        dataset.save_episode()
+    dataset.finalize()
+
+    # Load episode 1 with delta that looks at previous frame
+    delta_ts = {"observation.state": [-0.1, 0.0]}  # Previous frame and current frame
+    filtered_dataset = LeRobotDataset(
+        dataset.repo_id,
+        root=dataset.root,
+        episodes=[1],
+        delta_timestamps=delta_ts,
+        tolerance_s=0.04,
+    )
+
+    # Check frame 2 of episode 1 (which has absolute index 7, value 12)
+    frame = filtered_dataset[2]
+    state_values = frame["observation.state"].tolist()
+    # Should get [11, 12] - the previous and current values within episode 1
+    assert state_values == [11.0, 12.0], f"Expected [11.0, 12.0], got {state_values}"
+
+    # Check first frame - previous frame should be clamped to episode start (padded)
+    first_frame = filtered_dataset[0]
+    state_values = first_frame["observation.state"].tolist()
+    is_pad = first_frame["observation.state_is_pad"].tolist()
+    # Previous frame is outside episode, so it's clamped to first frame and marked as padded
+    assert state_values == [10.0, 10.0], f"Expected [10.0, 10.0], got {state_values}"
+    assert is_pad == [True, False], f"Expected [True, False], got {is_pad}"
diff --git a/lerobot/tests/datasets/test_delta_timestamps.py b/lerobot/tests/datasets/test_delta_timestamps.py
new file mode 100644
index 0000000000000000000000000000000000000000..8d9529f68cc01f01c39cb2195bf857086472ccb2
--- /dev/null
+++ b/lerobot/tests/datasets/test_delta_timestamps.py
@@ -0,0 +1,138 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import pytest
+
+from lerobot.datasets.feature_utils import (
+    check_delta_timestamps,
+    get_delta_indices,
+)
+from tests.fixtures.constants import DUMMY_MOTOR_FEATURES
+
+
+@pytest.fixture(scope="module")
+def valid_delta_timestamps_factory():
+    def _create_valid_delta_timestamps(
+        fps: int = 30, keys: list = DUMMY_MOTOR_FEATURES, min_max_range: tuple[int, int] = (-10, 10)
+    ) -> dict:
+        delta_timestamps = {key: [i * (1 / fps) for i in range(*min_max_range)] for key in keys}
+        return delta_timestamps
+
+    return _create_valid_delta_timestamps
+
+
+@pytest.fixture(scope="module")
+def invalid_delta_timestamps_factory(valid_delta_timestamps_factory):
+    def _create_invalid_delta_timestamps(
+        fps: int = 30, tolerance_s: float = 1e-4, keys: list = DUMMY_MOTOR_FEATURES
+    ) -> dict:
+        delta_timestamps = valid_delta_timestamps_factory(fps, keys)
+        # Modify a single timestamp just outside tolerance
+        for key in keys:
+            delta_timestamps[key][3] += tolerance_s * 1.1
+        return delta_timestamps
+
+    return _create_invalid_delta_timestamps
+
+
+@pytest.fixture(scope="module")
+def slightly_off_delta_timestamps_factory(valid_delta_timestamps_factory):
+    def _create_slightly_off_delta_timestamps(
+        fps: int = 30, tolerance_s: float = 1e-4, keys: list = DUMMY_MOTOR_FEATURES
+    ) -> dict:
+        delta_timestamps = valid_delta_timestamps_factory(fps, keys)
+        # Modify a single timestamp just inside tolerance
+        for key in delta_timestamps:
+            delta_timestamps[key][3] += tolerance_s * 0.9
+            delta_timestamps[key][-3] += tolerance_s * 0.9
+        return delta_timestamps
+
+    return _create_slightly_off_delta_timestamps
+
+
+@pytest.fixture(scope="module")
+def delta_indices_factory():
+    def _delta_indices(keys: list = DUMMY_MOTOR_FEATURES, min_max_range: tuple[int, int] = (-10, 10)) -> dict:
+        return {key: list(range(*min_max_range)) for key in keys}
+
+    return _delta_indices
+
+
+def test_check_delta_timestamps_valid(valid_delta_timestamps_factory):
+    fps = 30
+    tolerance_s = 1e-4
+    valid_delta_timestamps = valid_delta_timestamps_factory(fps)
+    result = check_delta_timestamps(
+        delta_timestamps=valid_delta_timestamps,
+        fps=fps,
+        tolerance_s=tolerance_s,
+    )
+    assert result is True
+
+
+def test_check_delta_timestamps_slightly_off(slightly_off_delta_timestamps_factory):
+    fps = 30
+    tolerance_s = 1e-4
+    slightly_off_delta_timestamps = slightly_off_delta_timestamps_factory(fps, tolerance_s)
+    result = check_delta_timestamps(
+        delta_timestamps=slightly_off_delta_timestamps,
+        fps=fps,
+        tolerance_s=tolerance_s,
+    )
+    assert result is True
+
+
+def test_check_delta_timestamps_invalid(invalid_delta_timestamps_factory):
+    fps = 30
+    tolerance_s = 1e-4
+    invalid_delta_timestamps = invalid_delta_timestamps_factory(fps, tolerance_s)
+    with pytest.raises(ValueError):
+        check_delta_timestamps(
+            delta_timestamps=invalid_delta_timestamps,
+            fps=fps,
+            tolerance_s=tolerance_s,
+        )
+
+
+def test_check_delta_timestamps_invalid_no_exception(invalid_delta_timestamps_factory):
+    fps = 30
+    tolerance_s = 1e-4
+    invalid_delta_timestamps = invalid_delta_timestamps_factory(fps, tolerance_s)
+    result = check_delta_timestamps(
+        delta_timestamps=invalid_delta_timestamps,
+        fps=fps,
+        tolerance_s=tolerance_s,
+        raise_value_error=False,
+    )
+    assert result is False
+
+
+def test_check_delta_timestamps_empty():
+    delta_timestamps = {}
+    fps = 30
+    tolerance_s = 1e-4
+    result = check_delta_timestamps(
+        delta_timestamps=delta_timestamps,
+        fps=fps,
+        tolerance_s=tolerance_s,
+    )
+    assert result is True
+
+
+def test_delta_indices(valid_delta_timestamps_factory, delta_indices_factory):
+    fps = 50
+    min_max_range = (-100, 100)
+    delta_timestamps = valid_delta_timestamps_factory(fps, min_max_range=min_max_range)
+    expected_delta_indices = delta_indices_factory(min_max_range=min_max_range)
+    actual_delta_indices = get_delta_indices(delta_timestamps, fps)
+    assert expected_delta_indices == actual_delta_indices
diff --git a/lerobot/tests/datasets/test_image_transforms.py b/lerobot/tests/datasets/test_image_transforms.py
new file mode 100644
index 0000000000000000000000000000000000000000..ef7e8c3957770513a9ba95efebf2286f77b8cb42
--- /dev/null
+++ b/lerobot/tests/datasets/test_image_transforms.py
@@ -0,0 +1,455 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import pytest
+import torch
+from packaging import version
+from safetensors.torch import load_file
+from torchvision.transforms import v2
+from torchvision.transforms.v2 import functional as F  # noqa: N812
+
+from lerobot.datasets.transforms import (
+    ImageTransformConfig,
+    ImageTransforms,
+    ImageTransformsConfig,
+    RandomSubsetApply,
+    SharpnessJitter,
+    make_transform_from_config,
+)
+from lerobot.scripts.lerobot_imgtransform_viz import (
+    save_all_transforms,
+    save_each_transform,
+)
+from lerobot.utils.random_utils import seeded_context
+from tests.artifacts.image_transforms.save_image_transforms_to_safetensors import ARTIFACT_DIR
+from tests.utils import require_x86_64_kernel
+
+
+@pytest.fixture
+def color_jitters():
+    return [
+        v2.ColorJitter(brightness=0.5),
+        v2.ColorJitter(contrast=0.5),
+        v2.ColorJitter(saturation=0.5),
+    ]
+
+
+@pytest.fixture
+def single_transforms():
+    return load_file(ARTIFACT_DIR / "single_transforms.safetensors")
+
+
+@pytest.fixture
+def img_tensor(single_transforms):
+    return single_transforms["original_frame"]
+
+
+@pytest.fixture
+def default_transforms():
+    return load_file(ARTIFACT_DIR / "default_transforms.safetensors")
+
+
+def test_get_image_transforms_no_transform_enable_false(img_tensor_factory):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig()  # default is enable=False
+    tf_actual = ImageTransforms(tf_cfg)
+    torch.testing.assert_close(tf_actual(img_tensor), img_tensor)
+
+
+def test_get_image_transforms_no_transform_max_num_transforms_0(img_tensor_factory):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(enable=True, max_num_transforms=0)
+    tf_actual = ImageTransforms(tf_cfg)
+    torch.testing.assert_close(tf_actual(img_tensor), img_tensor)
+
+
+@pytest.mark.parametrize("min_max", [(0.5, 0.5), (2.0, 2.0)])
+def test_get_image_transforms_brightness(img_tensor_factory, min_max):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(
+        enable=True,
+        tfs={"brightness": ImageTransformConfig(type="ColorJitter", kwargs={"brightness": min_max})},
+    )
+    tf_actual = ImageTransforms(tf_cfg)
+    tf_expected = v2.ColorJitter(brightness=min_max)
+    torch.testing.assert_close(tf_actual(img_tensor), tf_expected(img_tensor))
+
+
+@pytest.mark.parametrize("min_max", [(0.5, 0.5), (2.0, 2.0)])
+def test_get_image_transforms_contrast(img_tensor_factory, min_max):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(
+        enable=True, tfs={"contrast": ImageTransformConfig(type="ColorJitter", kwargs={"contrast": min_max})}
+    )
+    tf_actual = ImageTransforms(tf_cfg)
+    tf_expected = v2.ColorJitter(contrast=min_max)
+    torch.testing.assert_close(tf_actual(img_tensor), tf_expected(img_tensor))
+
+
+@pytest.mark.parametrize("min_max", [(0.5, 0.5), (2.0, 2.0)])
+def test_get_image_transforms_saturation(img_tensor_factory, min_max):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(
+        enable=True,
+        tfs={"saturation": ImageTransformConfig(type="ColorJitter", kwargs={"saturation": min_max})},
+    )
+    tf_actual = ImageTransforms(tf_cfg)
+    tf_expected = v2.ColorJitter(saturation=min_max)
+    torch.testing.assert_close(tf_actual(img_tensor), tf_expected(img_tensor))
+
+
+@pytest.mark.parametrize("min_max", [(-0.25, -0.25), (0.25, 0.25)])
+def test_get_image_transforms_hue(img_tensor_factory, min_max):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(
+        enable=True, tfs={"hue": ImageTransformConfig(type="ColorJitter", kwargs={"hue": min_max})}
+    )
+    tf_actual = ImageTransforms(tf_cfg)
+    tf_expected = v2.ColorJitter(hue=min_max)
+    torch.testing.assert_close(tf_actual(img_tensor), tf_expected(img_tensor))
+
+
+@pytest.mark.parametrize("min_max", [(0.5, 0.5), (2.0, 2.0)])
+def test_get_image_transforms_sharpness(img_tensor_factory, min_max):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(
+        enable=True,
+        tfs={"sharpness": ImageTransformConfig(type="SharpnessJitter", kwargs={"sharpness": min_max})},
+    )
+    tf_actual = ImageTransforms(tf_cfg)
+    tf_expected = SharpnessJitter(sharpness=min_max)
+    torch.testing.assert_close(tf_actual(img_tensor), tf_expected(img_tensor))
+
+
+@pytest.mark.parametrize("degrees, translate", [((-5.0, 5.0), (0.05, 0.05)), ((10.0, 10.0), (0.1, 0.1))])
+def test_get_image_transforms_affine(img_tensor_factory, degrees, translate):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(
+        enable=True,
+        tfs={
+            "affine": ImageTransformConfig(
+                type="RandomAffine", kwargs={"degrees": degrees, "translate": translate}
+            )
+        },
+    )
+    tf = ImageTransforms(tf_cfg)
+    output = tf(img_tensor)
+    # Verify output shape is preserved
+    assert output.shape == img_tensor.shape
+    # Verify transform is type RandomAffine
+    assert isinstance(tf.transforms["affine"], v2.RandomAffine)
+
+
+def test_get_image_transforms_max_num_transforms(img_tensor_factory):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(
+        enable=True,
+        max_num_transforms=5,
+        tfs={
+            "brightness": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"brightness": (0.5, 0.5)},
+            ),
+            "contrast": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"contrast": (0.5, 0.5)},
+            ),
+            "saturation": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"saturation": (0.5, 0.5)},
+            ),
+            "hue": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"hue": (0.5, 0.5)},
+            ),
+            "sharpness": ImageTransformConfig(
+                weight=1.0,
+                type="SharpnessJitter",
+                kwargs={"sharpness": (0.5, 0.5)},
+            ),
+        },
+    )
+    tf_actual = ImageTransforms(tf_cfg)
+    tf_expected = v2.Compose(
+        [
+            v2.ColorJitter(brightness=(0.5, 0.5)),
+            v2.ColorJitter(contrast=(0.5, 0.5)),
+            v2.ColorJitter(saturation=(0.5, 0.5)),
+            v2.ColorJitter(hue=(0.5, 0.5)),
+            SharpnessJitter(sharpness=(0.5, 0.5)),
+        ]
+    )
+    torch.testing.assert_close(tf_actual(img_tensor), tf_expected(img_tensor))
+
+
+@require_x86_64_kernel
+def test_get_image_transforms_random_order(img_tensor_factory):
+    out_imgs = []
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(
+        enable=True,
+        random_order=True,
+        tfs={
+            "brightness": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"brightness": (0.5, 0.5)},
+            ),
+            "contrast": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"contrast": (0.5, 0.5)},
+            ),
+            "saturation": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"saturation": (0.5, 0.5)},
+            ),
+            "hue": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"hue": (0.5, 0.5)},
+            ),
+            "sharpness": ImageTransformConfig(
+                weight=1.0,
+                type="SharpnessJitter",
+                kwargs={"sharpness": (0.5, 0.5)},
+            ),
+        },
+    )
+    tf = ImageTransforms(tf_cfg)
+
+    with seeded_context(1338):
+        for _ in range(10):
+            out_imgs.append(tf(img_tensor))
+
+            tmp_img_tensor = img_tensor
+            for sub_tf in tf.tf.selected_transforms:
+                tmp_img_tensor = sub_tf(tmp_img_tensor)
+            torch.testing.assert_close(tmp_img_tensor, out_imgs[-1])
+
+    for i in range(1, len(out_imgs)):
+        with pytest.raises(AssertionError):
+            torch.testing.assert_close(out_imgs[0], out_imgs[i])
+
+
+@pytest.mark.parametrize(
+    "tf_type, tf_name, min_max_values",
+    [
+        ("ColorJitter", "brightness", [(0.5, 0.5), (2.0, 2.0)]),
+        ("ColorJitter", "contrast", [(0.5, 0.5), (2.0, 2.0)]),
+        ("ColorJitter", "saturation", [(0.5, 0.5), (2.0, 2.0)]),
+        ("ColorJitter", "hue", [(-0.25, -0.25), (0.25, 0.25)]),
+        ("SharpnessJitter", "sharpness", [(0.5, 0.5), (2.0, 2.0)]),
+    ],
+)
+def test_backward_compatibility_single_transforms(
+    img_tensor, tf_type, tf_name, min_max_values, single_transforms
+):
+    for min_max in min_max_values:
+        tf_cfg = ImageTransformConfig(type=tf_type, kwargs={tf_name: min_max})
+        tf = make_transform_from_config(tf_cfg)
+        actual = tf(img_tensor)
+        key = f"{tf_name}_{min_max[0]}_{min_max[1]}"
+        expected = single_transforms[key]
+        torch.testing.assert_close(actual, expected)
+
+
+@require_x86_64_kernel
+@pytest.mark.skipif(
+    version.parse(torch.__version__) < version.parse("2.7.0"),
+    reason="Test artifacts were generated with PyTorch >= 2.7.0 which has different multinomial behavior",
+)
+def test_backward_compatibility_default_config(img_tensor, default_transforms):
+    # NOTE: PyTorch versions have different randomness, it might break this test.
+    # See this PR: https://github.com/huggingface/lerobot/pull/1127.
+
+    # Use config without affine to match original test artifacts
+    cfg = ImageTransformsConfig(
+        enable=True,
+        tfs={
+            "brightness": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"brightness": (0.8, 1.2)},
+            ),
+            "contrast": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"contrast": (0.8, 1.2)},
+            ),
+            "saturation": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"saturation": (0.5, 1.5)},
+            ),
+            "hue": ImageTransformConfig(
+                weight=1.0,
+                type="ColorJitter",
+                kwargs={"hue": (-0.05, 0.05)},
+            ),
+            "sharpness": ImageTransformConfig(
+                weight=1.0,
+                type="SharpnessJitter",
+                kwargs={"sharpness": (0.5, 1.5)},
+            ),
+        },
+    )
+    default_tf = ImageTransforms(cfg)
+
+    with seeded_context(1337):
+        actual = default_tf(img_tensor)
+
+    expected = default_transforms["default"]
+
+    torch.testing.assert_close(actual, expected)
+
+
+@pytest.mark.parametrize("p", [[0, 1], [1, 0]])
+def test_random_subset_apply_single_choice(img_tensor_factory, p):
+    img_tensor = img_tensor_factory()
+    flips = [v2.RandomHorizontalFlip(p=1), v2.RandomVerticalFlip(p=1)]
+    random_choice = RandomSubsetApply(flips, p=p, n_subset=1, random_order=False)
+    actual = random_choice(img_tensor)
+
+    p_horz, _ = p
+    if p_horz:
+        torch.testing.assert_close(actual, F.horizontal_flip(img_tensor))
+    else:
+        torch.testing.assert_close(actual, F.vertical_flip(img_tensor))
+
+
+def test_random_subset_apply_random_order(img_tensor_factory):
+    img_tensor = img_tensor_factory()
+    flips = [v2.RandomHorizontalFlip(p=1), v2.RandomVerticalFlip(p=1)]
+    random_order = RandomSubsetApply(flips, p=[0.5, 0.5], n_subset=2, random_order=True)
+    # We can't really check whether the transforms are actually applied in random order. However,
+    # horizontal and vertical flip are commutative. Meaning, even under the assumption that the transform
+    # applies them in random order, we can use a fixed order to compute the expected value.
+    actual = random_order(img_tensor)
+    expected = v2.Compose(flips)(img_tensor)
+    torch.testing.assert_close(actual, expected)
+
+
+def test_random_subset_apply_valid_transforms(img_tensor_factory, color_jitters):
+    img_tensor = img_tensor_factory()
+    transform = RandomSubsetApply(color_jitters)
+    output = transform(img_tensor)
+    assert output.shape == img_tensor.shape
+
+
+def test_random_subset_apply_probability_length_mismatch(color_jitters):
+    with pytest.raises(ValueError):
+        RandomSubsetApply(color_jitters, p=[0.5, 0.5])
+
+
+@pytest.mark.parametrize("n_subset", [0, 5])
+def test_random_subset_apply_invalid_n_subset(color_jitters, n_subset):
+    with pytest.raises(ValueError):
+        RandomSubsetApply(color_jitters, n_subset=n_subset)
+
+
+def test_sharpness_jitter_valid_range_tuple(img_tensor_factory):
+    img_tensor = img_tensor_factory()
+    tf = SharpnessJitter((0.1, 2.0))
+    output = tf(img_tensor)
+    assert output.shape == img_tensor.shape
+
+
+def test_sharpness_jitter_valid_range_float(img_tensor_factory):
+    img_tensor = img_tensor_factory()
+    tf = SharpnessJitter(0.5)
+    output = tf(img_tensor)
+    assert output.shape == img_tensor.shape
+
+
+def test_sharpness_jitter_invalid_range_min_negative():
+    with pytest.raises(ValueError):
+        SharpnessJitter((-0.1, 2.0))
+
+
+def test_sharpness_jitter_invalid_range_max_smaller():
+    with pytest.raises(ValueError):
+        SharpnessJitter((2.0, 0.1))
+
+
+def test_make_transform_from_config_with_v2_resize(img_tensor_factory):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformConfig(type="Resize", kwargs={"size": (32, 32)})
+    tf = make_transform_from_config(tf_cfg)
+    assert isinstance(tf, v2.Resize)
+    output = tf(img_tensor)
+    assert output.shape[-2:] == (32, 32)
+
+
+def test_make_transform_from_config_with_v2_identity(img_tensor_factory):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformConfig(type="Identity", kwargs={})
+    tf = make_transform_from_config(tf_cfg)
+    assert isinstance(tf, v2.Identity)
+    output = tf(img_tensor)
+    assert output.shape == img_tensor.shape
+
+
+def test_make_transform_from_config_invalid_type():
+    tf_cfg = ImageTransformConfig(type="NotARealTransform", kwargs={})
+    with pytest.raises(ValueError, match="not valid"):
+        make_transform_from_config(tf_cfg)
+
+
+def test_save_all_transforms(img_tensor_factory, tmp_path):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(enable=True)
+    n_examples = 3
+
+    save_all_transforms(tf_cfg, img_tensor, tmp_path, n_examples)
+
+    # Check if the combined transforms directory exists and contains the right files
+    combined_transforms_dir = tmp_path / "all"
+    assert combined_transforms_dir.exists(), "Combined transforms directory was not created."
+    assert any(combined_transforms_dir.iterdir()), (
+        "No transformed images found in combined transforms directory."
+    )
+    for i in range(1, n_examples + 1):
+        assert (combined_transforms_dir / f"{i}.png").exists(), (
+            f"Combined transform image {i}.png was not found."
+        )
+
+
+def test_save_each_transform(img_tensor_factory, tmp_path):
+    img_tensor = img_tensor_factory()
+    tf_cfg = ImageTransformsConfig(enable=True)
+    n_examples = 3
+
+    save_each_transform(tf_cfg, img_tensor, tmp_path, n_examples)
+
+    # Check if the transformed images exist for each transform type
+    transforms = ["brightness", "contrast", "saturation", "hue", "sharpness", "affine"]
+    for transform in transforms:
+        transform_dir = tmp_path / transform
+        assert transform_dir.exists(), f"{transform} directory was not created."
+        assert any(transform_dir.iterdir()), f"No transformed images found in {transform} directory."
+
+        # Check for specific files within each transform directory
+        expected_files = [f"{i}.png" for i in range(1, n_examples + 1)] + ["min.png", "max.png", "mean.png"]
+        for file_name in expected_files:
+            assert (transform_dir / file_name).exists(), (
+                f"{file_name} was not found in {transform} directory."
+            )
diff --git a/lerobot/tests/datasets/test_image_writer.py b/lerobot/tests/datasets/test_image_writer.py
new file mode 100644
index 0000000000000000000000000000000000000000..e0275517175b49ca4118e082a2a76ed6f22931bb
--- /dev/null
+++ b/lerobot/tests/datasets/test_image_writer.py
@@ -0,0 +1,386 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import queue
+import time
+from multiprocessing import queues
+from unittest.mock import MagicMock, patch
+
+import numpy as np
+import pytest
+from PIL import Image
+
+from lerobot.datasets.image_writer import (
+    AsyncImageWriter,
+    image_array_to_pil_image,
+    safe_stop_image_writer,
+    write_image,
+)
+from tests.fixtures.constants import DUMMY_HWC
+
+DUMMY_IMAGE = "test_image.png"
+
+
+def test_init_threading():
+    writer = AsyncImageWriter(num_processes=0, num_threads=2)
+    try:
+        assert writer.num_processes == 0
+        assert writer.num_threads == 2
+        assert isinstance(writer.queue, queue.Queue)
+        assert len(writer.threads) == 2
+        assert len(writer.processes) == 0
+        assert all(t.is_alive() for t in writer.threads)
+    finally:
+        writer.stop()
+
+
+def test_init_multiprocessing():
+    writer = AsyncImageWriter(num_processes=2, num_threads=2)
+    try:
+        assert writer.num_processes == 2
+        assert writer.num_threads == 2
+        assert isinstance(writer.queue, queues.JoinableQueue)
+        assert len(writer.threads) == 0
+        assert len(writer.processes) == 2
+        assert all(p.is_alive() for p in writer.processes)
+    finally:
+        writer.stop()
+
+
+def test_zero_threads():
+    with pytest.raises(ValueError):
+        AsyncImageWriter(num_processes=0, num_threads=0)
+
+
+def test_image_array_to_pil_image_float_array_wrong_range_0_255():
+    image = np.random.rand(*DUMMY_HWC) * 255
+    with pytest.raises(ValueError):
+        image_array_to_pil_image(image)
+
+
+def test_image_array_to_pil_image_float_array_wrong_range_neg_1_1():
+    image = np.random.rand(*DUMMY_HWC) * 2 - 1
+    with pytest.raises(ValueError):
+        image_array_to_pil_image(image)
+
+
+def test_image_array_to_pil_image_rgb(img_array_factory):
+    img_array = img_array_factory(100, 100)
+    result_image = image_array_to_pil_image(img_array)
+    assert isinstance(result_image, Image.Image)
+    assert result_image.size == (100, 100)
+    assert result_image.mode == "RGB"
+
+
+def test_image_array_to_pil_image_pytorch_format(img_array_factory):
+    img_array = img_array_factory(100, 100).transpose(2, 0, 1)
+    result_image = image_array_to_pil_image(img_array)
+    assert isinstance(result_image, Image.Image)
+    assert result_image.size == (100, 100)
+    assert result_image.mode == "RGB"
+
+
+def test_image_array_to_pil_image_single_channel(img_array_factory):
+    img_array = img_array_factory(channels=1)
+    with pytest.raises(NotImplementedError):
+        image_array_to_pil_image(img_array)
+
+
+def test_image_array_to_pil_image_4_channels(img_array_factory):
+    img_array = img_array_factory(channels=4)
+    with pytest.raises(NotImplementedError):
+        image_array_to_pil_image(img_array)
+
+
+def test_image_array_to_pil_image_float_array(img_array_factory):
+    img_array = img_array_factory(dtype=np.float32)
+    result_image = image_array_to_pil_image(img_array)
+    assert isinstance(result_image, Image.Image)
+    assert result_image.size == (100, 100)
+    assert result_image.mode == "RGB"
+    assert np.array(result_image).dtype == np.uint8
+
+
+def test_image_array_to_pil_image_uint8_array(img_array_factory):
+    img_array = img_array_factory(dtype=np.float32)
+    result_image = image_array_to_pil_image(img_array)
+    assert isinstance(result_image, Image.Image)
+    assert result_image.size == (100, 100)
+    assert result_image.mode == "RGB"
+    assert np.array(result_image).dtype == np.uint8
+
+
+def test_write_image_numpy(tmp_path, img_array_factory):
+    image_array = img_array_factory()
+    fpath = tmp_path / DUMMY_IMAGE
+    write_image(image_array, fpath)
+    assert fpath.exists()
+    saved_image = np.array(Image.open(fpath))
+    assert np.array_equal(image_array, saved_image)
+
+
+def test_write_image_image(tmp_path, img_factory):
+    image_pil = img_factory()
+    fpath = tmp_path / DUMMY_IMAGE
+    write_image(image_pil, fpath)
+    assert fpath.exists()
+    saved_image = Image.open(fpath)
+    assert list(saved_image.getdata()) == list(image_pil.getdata())
+    assert np.array_equal(image_pil, saved_image)
+
+
+def test_write_image_exception(tmp_path):
+    image_array = "invalid data"
+    fpath = tmp_path / DUMMY_IMAGE
+    with patch("lerobot.datasets.image_writer.logger") as mock_logger:
+        write_image(image_array, fpath)
+        mock_logger.error.assert_called()
+        assert not fpath.exists()
+
+
+def test_save_image_numpy(tmp_path, img_array_factory):
+    writer = AsyncImageWriter()
+    try:
+        image_array = img_array_factory()
+        fpath = tmp_path / DUMMY_IMAGE
+        fpath.parent.mkdir(parents=True, exist_ok=True)
+        writer.save_image(image_array, fpath)
+        writer.wait_until_done()
+        assert fpath.exists()
+        saved_image = np.array(Image.open(fpath))
+        assert np.array_equal(image_array, saved_image)
+    finally:
+        writer.stop()
+
+
+def test_save_image_numpy_multiprocessing(tmp_path, img_array_factory):
+    writer = AsyncImageWriter(num_processes=2, num_threads=2)
+    try:
+        image_array = img_array_factory()
+        fpath = tmp_path / DUMMY_IMAGE
+        writer.save_image(image_array, fpath)
+        writer.wait_until_done()
+        assert fpath.exists()
+        saved_image = np.array(Image.open(fpath))
+        assert np.array_equal(image_array, saved_image)
+    finally:
+        writer.stop()
+
+
+def test_save_image_torch(tmp_path, img_tensor_factory):
+    writer = AsyncImageWriter()
+    try:
+        image_tensor = img_tensor_factory()
+        fpath = tmp_path / DUMMY_IMAGE
+        fpath.parent.mkdir(parents=True, exist_ok=True)
+        writer.save_image(image_tensor, fpath)
+        writer.wait_until_done()
+        assert fpath.exists()
+        saved_image = np.array(Image.open(fpath))
+        expected_image = (image_tensor.permute(1, 2, 0).cpu().numpy() * 255).astype(np.uint8)
+        assert np.array_equal(expected_image, saved_image)
+    finally:
+        writer.stop()
+
+
+def test_save_image_torch_multiprocessing(tmp_path, img_tensor_factory):
+    writer = AsyncImageWriter(num_processes=2, num_threads=2)
+    try:
+        image_tensor = img_tensor_factory()
+        fpath = tmp_path / DUMMY_IMAGE
+        writer.save_image(image_tensor, fpath)
+        writer.wait_until_done()
+        assert fpath.exists()
+        saved_image = np.array(Image.open(fpath))
+        expected_image = (image_tensor.permute(1, 2, 0).cpu().numpy() * 255).astype(np.uint8)
+        assert np.array_equal(expected_image, saved_image)
+    finally:
+        writer.stop()
+
+
+def test_save_image_pil(tmp_path, img_factory):
+    writer = AsyncImageWriter()
+    try:
+        image_pil = img_factory()
+        fpath = tmp_path / DUMMY_IMAGE
+        fpath.parent.mkdir(parents=True, exist_ok=True)
+        writer.save_image(image_pil, fpath)
+        writer.wait_until_done()
+        assert fpath.exists()
+        saved_image = Image.open(fpath)
+        assert list(saved_image.getdata()) == list(image_pil.getdata())
+    finally:
+        writer.stop()
+
+
+def test_save_image_pil_multiprocessing(tmp_path, img_factory):
+    writer = AsyncImageWriter(num_processes=2, num_threads=2)
+    try:
+        image_pil = img_factory()
+        fpath = tmp_path / DUMMY_IMAGE
+        writer.save_image(image_pil, fpath)
+        writer.wait_until_done()
+        assert fpath.exists()
+        saved_image = Image.open(fpath)
+        assert list(saved_image.getdata()) == list(image_pil.getdata())
+    finally:
+        writer.stop()
+
+
+def test_save_image_invalid_data(tmp_path):
+    writer = AsyncImageWriter()
+    try:
+        image_array = "invalid data"
+        fpath = tmp_path / DUMMY_IMAGE
+        fpath.parent.mkdir(parents=True, exist_ok=True)
+        with patch("lerobot.datasets.image_writer.logger") as mock_logger:
+            writer.save_image(image_array, fpath)
+            writer.wait_until_done()
+            mock_logger.error.assert_called()
+            assert not fpath.exists()
+    finally:
+        writer.stop()
+
+
+def test_save_image_after_stop(tmp_path, img_array_factory):
+    writer = AsyncImageWriter()
+    writer.stop()
+    image_array = img_array_factory()
+    fpath = tmp_path / DUMMY_IMAGE
+    writer.save_image(image_array, fpath)
+    time.sleep(1)
+    assert not fpath.exists()
+
+
+def test_stop():
+    writer = AsyncImageWriter(num_processes=0, num_threads=2)
+    writer.stop()
+    assert not any(t.is_alive() for t in writer.threads)
+
+
+def test_stop_multiprocessing():
+    writer = AsyncImageWriter(num_processes=2, num_threads=2)
+    writer.stop()
+    assert not any(p.is_alive() for p in writer.processes)
+
+
+def test_multiple_stops():
+    writer = AsyncImageWriter()
+    writer.stop()
+    writer.stop()  # Should not raise an exception
+    assert not any(t.is_alive() for t in writer.threads)
+
+
+def test_multiple_stops_multiprocessing():
+    writer = AsyncImageWriter(num_processes=2, num_threads=2)
+    writer.stop()
+    writer.stop()  # Should not raise an exception
+    assert not any(t.is_alive() for t in writer.threads)
+
+
+def test_wait_until_done(tmp_path, img_array_factory):
+    writer = AsyncImageWriter(num_processes=0, num_threads=4)
+    try:
+        num_images = 100
+        image_arrays = [img_array_factory(height=500, width=500) for _ in range(num_images)]
+        fpaths = [tmp_path / f"frame_{i:06d}.png" for i in range(num_images)]
+        for image_array, fpath in zip(image_arrays, fpaths, strict=True):
+            fpath.parent.mkdir(parents=True, exist_ok=True)
+            writer.save_image(image_array, fpath)
+        writer.wait_until_done()
+        for i, fpath in enumerate(fpaths):
+            assert fpath.exists()
+            saved_image = np.array(Image.open(fpath))
+            assert np.array_equal(saved_image, image_arrays[i])
+    finally:
+        writer.stop()
+
+
+def test_wait_until_done_multiprocessing(tmp_path, img_array_factory):
+    writer = AsyncImageWriter(num_processes=2, num_threads=2)
+    try:
+        num_images = 100
+        image_arrays = [img_array_factory() for _ in range(num_images)]
+        fpaths = [tmp_path / f"frame_{i:06d}.png" for i in range(num_images)]
+        for image_array, fpath in zip(image_arrays, fpaths, strict=True):
+            fpath.parent.mkdir(parents=True, exist_ok=True)
+            writer.save_image(image_array, fpath)
+        writer.wait_until_done()
+        for i, fpath in enumerate(fpaths):
+            assert fpath.exists()
+            saved_image = np.array(Image.open(fpath))
+            assert np.array_equal(saved_image, image_arrays[i])
+    finally:
+        writer.stop()
+
+
+def test_exception_handling(tmp_path, img_array_factory):
+    writer = AsyncImageWriter()
+    try:
+        image_array = img_array_factory()
+        with (
+            patch.object(writer.queue, "put", side_effect=queue.Full("Queue is full")),
+            pytest.raises(queue.Full) as exc_info,
+        ):
+            writer.save_image(image_array, tmp_path / "test.png")
+        assert str(exc_info.value) == "Queue is full"
+    finally:
+        writer.stop()
+
+
+def test_with_different_image_formats(tmp_path, img_array_factory):
+    writer = AsyncImageWriter()
+    try:
+        image_array = img_array_factory()
+        formats = ["png", "jpeg", "bmp"]
+        for fmt in formats:
+            fpath = tmp_path / f"test_image.{fmt}"
+            write_image(image_array, fpath)
+            assert fpath.exists()
+    finally:
+        writer.stop()
+
+
+def test_safe_stop_image_writer_decorator():
+    class MockDataset:
+        def __init__(self):
+            self.image_writer = MagicMock(spec=AsyncImageWriter)
+
+    @safe_stop_image_writer
+    def function_that_raises_exception(dataset=None):
+        raise Exception("Test exception")
+
+    dataset = MockDataset()
+
+    with pytest.raises(Exception) as exc_info:
+        function_that_raises_exception(dataset=dataset)
+
+    assert str(exc_info.value) == "Test exception"
+    dataset.image_writer.stop.assert_called_once()
+
+
+def test_main_process_time(tmp_path, img_tensor_factory):
+    writer = AsyncImageWriter()
+    try:
+        image_tensor = img_tensor_factory()
+        fpath = tmp_path / DUMMY_IMAGE
+        start_time = time.perf_counter()
+        writer.save_image(image_tensor, fpath)
+        end_time = time.perf_counter()
+        time_spent = end_time - start_time
+        # Might need to adjust this threshold depending on hardware
+        assert time_spent < 0.01, f"Main process time exceeded threshold: {time_spent}s"
+        writer.wait_until_done()
+        assert fpath.exists()
+    finally:
+        writer.stop()
diff --git a/lerobot/tests/datasets/test_quantiles_dataset_integration.py b/lerobot/tests/datasets/test_quantiles_dataset_integration.py
new file mode 100644
index 0000000000000000000000000000000000000000..4df7fab068b9891372873f329091d5100cb94d16
--- /dev/null
+++ b/lerobot/tests/datasets/test_quantiles_dataset_integration.py
@@ -0,0 +1,212 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Integration tests for quantile functionality in LeRobotDataset."""
+
+import numpy as np
+import pytest
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+
+def mock_load_image_as_numpy(path, dtype, channel_first):
+    """Mock image loading for consistent test results."""
+    return np.ones((3, 32, 32), dtype=dtype) if channel_first else np.ones((32, 32, 3), dtype=dtype)
+
+
+@pytest.fixture
+def simple_features():
+    """Simple feature configuration for testing."""
+    return {
+        "action": {
+            "dtype": "float32",
+            "shape": (4,),
+            "names": ["arm_x", "arm_y", "arm_z", "gripper"],
+        },
+        "observation.state": {
+            "dtype": "float32",
+            "shape": (10,),
+            "names": [f"joint_{i}" for i in range(10)],
+        },
+    }
+
+
+def test_create_dataset_with_fixed_quantiles(tmp_path, simple_features):
+    """Test creating dataset with fixed quantiles."""
+    dataset = LeRobotDataset.create(
+        repo_id="test_dataset_fixed_quantiles",
+        fps=30,
+        features=simple_features,
+        root=tmp_path / "create_fixed_quantiles",
+    )
+
+    # Dataset should be created successfully
+    assert dataset is not None
+
+
+def test_save_episode_computes_all_quantiles(tmp_path, simple_features):
+    """Test that all fixed quantiles are computed when saving an episode."""
+    dataset = LeRobotDataset.create(
+        repo_id="test_dataset_save_episode",
+        fps=30,
+        features=simple_features,
+        root=tmp_path / "save_episode_quantiles",
+    )
+
+    # Add some frames
+    for _ in range(10):
+        dataset.add_frame(
+            {
+                "action": np.random.randn(4).astype(np.float32),  # Correct shape for action
+                "observation.state": np.random.randn(10).astype(np.float32),
+                "task": "test_task",
+            }
+        )
+
+    dataset.save_episode()
+
+    # Check that all fixed quantiles were computed
+    stats = dataset.meta.stats
+    for key in ["action", "observation.state"]:
+        assert "q01" in stats[key]
+        assert "q10" in stats[key]
+        assert "q50" in stats[key]
+        assert "q90" in stats[key]
+        assert "q99" in stats[key]
+
+
+def test_quantile_values_ordering(tmp_path, simple_features):
+    """Test that quantile values are properly ordered."""
+    dataset = LeRobotDataset.create(
+        repo_id="test_dataset_quantile_ordering",
+        fps=30,
+        features=simple_features,
+        root=tmp_path / "quantile_ordering",
+    )
+
+    # Add data with known distribution
+    np.random.seed(42)
+    for _ in range(100):
+        dataset.add_frame(
+            {
+                "action": np.random.randn(4).astype(np.float32),  # Correct shape for action
+                "observation.state": np.random.randn(10).astype(np.float32),
+                "task": "test_task",
+            }
+        )
+
+    dataset.save_episode()
+    stats = dataset.meta.stats
+
+    # Verify quantile ordering
+    for key in ["action", "observation.state"]:
+        assert np.all(stats[key]["q01"] <= stats[key]["q10"])
+        assert np.all(stats[key]["q10"] <= stats[key]["q50"])
+        assert np.all(stats[key]["q50"] <= stats[key]["q90"])
+        assert np.all(stats[key]["q90"] <= stats[key]["q99"])
+
+
+def test_save_episode_with_fixed_quantiles(tmp_path, simple_features):
+    """Test saving episode always computes fixed quantiles."""
+    dataset = LeRobotDataset.create(
+        repo_id="test_dataset_save_fixed",
+        fps=30,
+        features=simple_features,
+        root=tmp_path / "save_fixed_quantiles",
+    )
+
+    # Add frames to episode
+    np.random.seed(42)
+    for _ in range(50):
+        frame = {
+            "action": np.random.normal(0, 1, (4,)).astype(np.float32),
+            "observation.state": np.random.normal(0, 1, (10,)).astype(np.float32),
+            "task": "test_task",
+        }
+        dataset.add_frame(frame)
+
+    dataset.save_episode()
+
+    # Check that all fixed quantiles are included
+    stats = dataset.meta.stats
+    for key in ["action", "observation.state"]:
+        feature_stats = stats[key]
+        expected_keys = {"min", "max", "mean", "std", "count", "q01", "q10", "q50", "q90", "q99"}
+        assert set(feature_stats.keys()) == expected_keys
+
+
+def test_quantile_aggregation_across_episodes(tmp_path, simple_features):
+    """Test quantile aggregation across multiple episodes."""
+    dataset = LeRobotDataset.create(
+        repo_id="test_dataset_aggregation",
+        fps=30,
+        features=simple_features,
+        root=tmp_path / "quantile_aggregation",
+    )
+
+    # Add frames to episode
+    np.random.seed(42)
+    for _ in range(100):
+        frame = {
+            "action": np.random.normal(0, 1, (4,)).astype(np.float32),
+            "observation.state": np.random.normal(2, 1, (10,)).astype(np.float32),
+            "task": "test_task",
+        }
+        dataset.add_frame(frame)
+
+    dataset.save_episode()
+
+    # Check stats include all fixed quantiles
+    stats = dataset.meta.stats
+    for key in ["action", "observation.state"]:
+        feature_stats = stats[key]
+        expected_keys = {"min", "max", "mean", "std", "count", "q01", "q10", "q50", "q90", "q99"}
+        assert set(feature_stats.keys()) == expected_keys
+        assert feature_stats["q01"].shape == (simple_features[key]["shape"][0],)
+        assert feature_stats["q50"].shape == (simple_features[key]["shape"][0],)
+        assert feature_stats["q99"].shape == (simple_features[key]["shape"][0],)
+        assert np.all(feature_stats["q01"] <= feature_stats["q50"])
+        assert np.all(feature_stats["q50"] <= feature_stats["q99"])
+
+
+def test_save_multiple_episodes_with_quantiles(tmp_path, simple_features):
+    """Test quantile aggregation across multiple episodes."""
+    dataset = LeRobotDataset.create(
+        repo_id="test_dataset_multiple_episodes",
+        fps=30,
+        features=simple_features,
+        root=tmp_path / "multiple_episodes",
+    )
+
+    # Save multiple episodes
+    np.random.seed(42)
+    for episode_idx in range(3):
+        for _ in range(50):
+            frame = {
+                "action": np.random.normal(episode_idx * 2.0, 1, (4,)).astype(np.float32),
+                "observation.state": np.random.normal(-episode_idx * 1.5, 1, (10,)).astype(np.float32),
+                "task": f"task_{episode_idx}",
+            }
+            dataset.add_frame(frame)
+
+        dataset.save_episode()
+
+    # Verify final stats include properly aggregated quantiles
+    stats = dataset.meta.stats
+    for key in ["action", "observation.state"]:
+        feature_stats = stats[key]
+        assert "q01" in feature_stats and "q99" in feature_stats
+        assert feature_stats["count"][0] == 150  # 3 episodes * 50 frames
diff --git a/lerobot/tests/datasets/test_sampler.py b/lerobot/tests/datasets/test_sampler.py
new file mode 100644
index 0000000000000000000000000000000000000000..18fb1c8ac6301686ec4107df8b617ce0fccb8cc0
--- /dev/null
+++ b/lerobot/tests/datasets/test_sampler.py
@@ -0,0 +1,136 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+
+import pytest
+import torch
+from datasets import Dataset
+
+from lerobot.datasets.io_utils import (
+    hf_transform_to_torch,
+)
+from lerobot.datasets.sampler import EpisodeAwareSampler
+
+
+def calculate_episode_data_index(hf_dataset: Dataset) -> dict[str, torch.Tensor]:
+    """Calculate episode data index for testing. Returns {"from": Tensor, "to": Tensor}."""
+    episode_data_index: dict[str, list[int]] = {"from": [], "to": []}
+    current_episode = None
+    if len(hf_dataset) == 0:
+        return {"from": torch.tensor([]), "to": torch.tensor([])}
+    for idx, episode_idx in enumerate(hf_dataset["episode_index"]):
+        if episode_idx != current_episode:
+            episode_data_index["from"].append(idx)
+            if current_episode is not None:
+                episode_data_index["to"].append(idx)
+            current_episode = episode_idx
+    episode_data_index["to"].append(idx + 1)
+    return {k: torch.tensor(v) for k, v in episode_data_index.items()}
+
+
+def test_drop_n_first_frames():
+    dataset = Dataset.from_dict(
+        {
+            "timestamp": [0.1, 0.2, 0.3, 0.4, 0.5, 0.6],
+            "index": [0, 1, 2, 3, 4, 5],
+            "episode_index": [0, 0, 1, 2, 2, 2],
+        },
+    )
+    dataset.set_transform(hf_transform_to_torch)
+    episode_data_index = calculate_episode_data_index(dataset)
+    sampler = EpisodeAwareSampler(episode_data_index["from"], episode_data_index["to"], drop_n_first_frames=1)
+    assert sampler.indices == [1, 4, 5]
+    assert len(sampler) == 3
+    assert list(sampler) == [1, 4, 5]
+
+
+def test_drop_n_last_frames():
+    dataset = Dataset.from_dict(
+        {
+            "timestamp": [0.1, 0.2, 0.3, 0.4, 0.5, 0.6],
+            "index": [0, 1, 2, 3, 4, 5],
+            "episode_index": [0, 0, 1, 2, 2, 2],
+        },
+    )
+    dataset.set_transform(hf_transform_to_torch)
+    episode_data_index = calculate_episode_data_index(dataset)
+    sampler = EpisodeAwareSampler(episode_data_index["from"], episode_data_index["to"], drop_n_last_frames=1)
+    assert sampler.indices == [0, 3, 4]
+    assert len(sampler) == 3
+    assert list(sampler) == [0, 3, 4]
+
+
+def test_episode_indices_to_use():
+    dataset = Dataset.from_dict(
+        {
+            "timestamp": [0.1, 0.2, 0.3, 0.4, 0.5, 0.6],
+            "index": [0, 1, 2, 3, 4, 5],
+            "episode_index": [0, 0, 1, 2, 2, 2],
+        },
+    )
+    dataset.set_transform(hf_transform_to_torch)
+    episode_data_index = calculate_episode_data_index(dataset)
+    sampler = EpisodeAwareSampler(
+        episode_data_index["from"], episode_data_index["to"], episode_indices_to_use=[0, 2]
+    )
+    assert sampler.indices == [0, 1, 3, 4, 5]
+    assert len(sampler) == 5
+    assert list(sampler) == [0, 1, 3, 4, 5]
+
+
+def test_shuffle():
+    dataset = Dataset.from_dict(
+        {
+            "timestamp": [0.1, 0.2, 0.3, 0.4, 0.5, 0.6],
+            "index": [0, 1, 2, 3, 4, 5],
+            "episode_index": [0, 0, 1, 2, 2, 2],
+        },
+    )
+    dataset.set_transform(hf_transform_to_torch)
+    episode_data_index = calculate_episode_data_index(dataset)
+    sampler = EpisodeAwareSampler(episode_data_index["from"], episode_data_index["to"], shuffle=False)
+    assert sampler.indices == [0, 1, 2, 3, 4, 5]
+    assert len(sampler) == 6
+    assert list(sampler) == [0, 1, 2, 3, 4, 5]
+    sampler = EpisodeAwareSampler(episode_data_index["from"], episode_data_index["to"], shuffle=True)
+    assert sampler.indices == [0, 1, 2, 3, 4, 5]
+    assert len(sampler) == 6
+    assert set(sampler) == {0, 1, 2, 3, 4, 5}
+
+
+def test_negative_drop_first_frames_raises():
+    with pytest.raises(ValueError, match="drop_n_first_frames must be >= 0"):
+        EpisodeAwareSampler([0], [10], drop_n_first_frames=-1)
+
+
+def test_negative_drop_last_frames_raises():
+    with pytest.raises(ValueError, match="drop_n_last_frames must be >= 0"):
+        EpisodeAwareSampler([0], [10], drop_n_last_frames=-1)
+
+
+def test_all_episodes_dropped_raises():
+    # All episodes have 1 frame, drop_n_first_frames=1 removes all
+    with pytest.raises(ValueError, match="No valid frames remain"):
+        EpisodeAwareSampler([0, 1, 2], [1, 2, 3], drop_n_first_frames=1)
+
+
+def test_partial_episode_drop_warns(caplog):
+    # Episode 0: 1 frame (dropped), Episode 1: 5 frames (kept)
+    with caplog.at_level(logging.WARNING, logger="lerobot.datasets.sampler"):
+        sampler = EpisodeAwareSampler([0, 1], [1, 6], drop_n_first_frames=1)
+    # Episode 0 is skipped (1 frame, drop 1), Episode 1 keeps frames 2-5
+    assert sampler.indices == [2, 3, 4, 5]
+    assert "Episode 0" in caplog.text
diff --git a/lerobot/tests/datasets/test_streaming.py b/lerobot/tests/datasets/test_streaming.py
new file mode 100644
index 0000000000000000000000000000000000000000..1bd4c1787c48329b53700c202560cc3dcd58c6e3
--- /dev/null
+++ b/lerobot/tests/datasets/test_streaming.py
@@ -0,0 +1,392 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import numpy as np
+import pytest
+import torch
+
+from lerobot.datasets.streaming_dataset import StreamingLeRobotDataset
+from lerobot.datasets.utils import safe_shard
+from lerobot.utils.constants import ACTION
+from tests.fixtures.constants import DUMMY_REPO_ID
+
+
+def get_frames_expected_order(streaming_ds: StreamingLeRobotDataset) -> list[int]:
+    """Replicates the shuffling logic of StreamingLeRobotDataset to get the expected order of indices."""
+    rng = np.random.default_rng(streaming_ds.seed)
+    buffer_size = streaming_ds.buffer_size
+    num_shards = streaming_ds.num_shards
+
+    shards_indices = []
+    for shard_idx in range(num_shards):
+        shard = streaming_ds.hf_dataset.shard(num_shards, index=shard_idx)
+        shard_indices = [item["index"] for item in shard]
+        shards_indices.append(shard_indices)
+
+    shard_iterators = {i: iter(s) for i, s in enumerate(shards_indices)}
+
+    buffer_indices_generator = streaming_ds._iter_random_indices(rng, buffer_size)
+
+    frames_buffer = []
+    expected_indices = []
+
+    while shard_iterators:  # While there are still available shards
+        available_shard_keys = list(shard_iterators.keys())
+        if not available_shard_keys:
+            break
+
+        # Call _infinite_generator_over_elements with current available shards (key difference!)
+        shard_key = next(streaming_ds._infinite_generator_over_elements(rng, available_shard_keys))
+
+        try:
+            frame_index = next(shard_iterators[shard_key])
+
+            if len(frames_buffer) == buffer_size:
+                i = next(buffer_indices_generator)
+                expected_indices.append(frames_buffer[i])
+                frames_buffer[i] = frame_index
+            else:
+                frames_buffer.append(frame_index)
+
+        except StopIteration:
+            del shard_iterators[shard_key]  # Remove exhausted shard
+
+    rng.shuffle(frames_buffer)
+    expected_indices.extend(frames_buffer)
+
+    return expected_indices
+
+
+def test_single_frame_consistency(tmp_path, lerobot_dataset_factory):
+    """Test if are correctly accessed"""
+    ds_num_frames = 400
+    ds_num_episodes = 10
+    buffer_size = 100
+
+    local_path = tmp_path / "test"
+    repo_id = f"{DUMMY_REPO_ID}"
+
+    ds = lerobot_dataset_factory(
+        root=local_path,
+        repo_id=repo_id,
+        total_episodes=ds_num_episodes,
+        total_frames=ds_num_frames,
+    )
+
+    streaming_ds = iter(StreamingLeRobotDataset(repo_id=repo_id, root=local_path, buffer_size=buffer_size))
+
+    key_checks = []
+    for _ in range(ds_num_frames):
+        streaming_frame = next(streaming_ds)
+        frame_idx = streaming_frame["index"]
+        target_frame = ds[frame_idx]
+
+        for key in streaming_frame:
+            left = streaming_frame[key]
+            right = target_frame[key]
+
+            if isinstance(left, str):
+                check = left == right
+
+            elif isinstance(left, torch.Tensor):
+                check = torch.allclose(left, right) and left.shape == right.shape
+
+            elif isinstance(left, float):
+                check = left == right.item()  # right is a torch.Tensor
+
+            key_checks.append((key, check))
+
+        assert all(t[1] for t in key_checks), (
+            f"Checking {list(filter(lambda t: not t[1], key_checks))[0][0]} left and right were found different (frame_idx: {frame_idx})"
+        )
+
+
+@pytest.mark.parametrize(
+    "shuffle",
+    [False, True],
+)
+def test_frames_order_over_epochs(tmp_path, lerobot_dataset_factory, shuffle):
+    """Test if streamed frames correspond to shuffling operations over in-memory dataset."""
+    ds_num_frames = 400
+    ds_num_episodes = 10
+    buffer_size = 100
+    seed = 42
+    n_epochs = 3
+
+    local_path = tmp_path / "test"
+    repo_id = f"{DUMMY_REPO_ID}"
+
+    lerobot_dataset_factory(
+        root=local_path,
+        repo_id=repo_id,
+        total_episodes=ds_num_episodes,
+        total_frames=ds_num_frames,
+    )
+
+    streaming_ds = StreamingLeRobotDataset(
+        repo_id=repo_id, root=local_path, buffer_size=buffer_size, seed=seed, shuffle=shuffle
+    )
+
+    first_epoch_indices = [frame["index"] for frame in streaming_ds]
+    expected_indices = get_frames_expected_order(streaming_ds)
+
+    assert first_epoch_indices == expected_indices, "First epoch indices do not match expected indices"
+
+    expected_indices = get_frames_expected_order(streaming_ds)
+    for _ in range(n_epochs):
+        streaming_indices = [frame["index"] for frame in streaming_ds]
+        frames_match = all(
+            s_index == e_index for s_index, e_index in zip(streaming_indices, expected_indices, strict=True)
+        )
+
+        if shuffle:
+            assert not frames_match
+        else:
+            assert frames_match
+
+
+@pytest.mark.parametrize(
+    "shuffle",
+    [False, True],
+)
+def test_frames_order_with_shards(tmp_path, lerobot_dataset_factory, shuffle):
+    """Test if streamed frames correspond to shuffling operations over in-memory dataset with multiple shards."""
+    ds_num_frames = 100
+    ds_num_episodes = 10
+    buffer_size = 10
+
+    seed = 42
+    n_epochs = 3
+    data_file_size_mb = 0.001
+
+    chunks_size = 1
+
+    local_path = tmp_path / "test"
+    repo_id = f"{DUMMY_REPO_ID}-ciao"
+
+    lerobot_dataset_factory(
+        root=local_path,
+        repo_id=repo_id,
+        total_episodes=ds_num_episodes,
+        total_frames=ds_num_frames,
+        data_files_size_in_mb=data_file_size_mb,
+        chunks_size=chunks_size,
+    )
+
+    streaming_ds = StreamingLeRobotDataset(
+        repo_id=repo_id,
+        root=local_path,
+        buffer_size=buffer_size,
+        seed=seed,
+        shuffle=shuffle,
+        max_num_shards=4,
+    )
+
+    first_epoch_indices = [frame["index"] for frame in streaming_ds]
+    expected_indices = get_frames_expected_order(streaming_ds)
+
+    assert first_epoch_indices == expected_indices, "First epoch indices do not match expected indices"
+
+    for _ in range(n_epochs):
+        streaming_indices = [
+            frame["index"] for frame in streaming_ds
+        ]  # NOTE: this is the same as first_epoch_indices
+        frames_match = all(
+            s_index == e_index for s_index, e_index in zip(streaming_indices, expected_indices, strict=True)
+        )
+        if shuffle:
+            assert not frames_match
+        else:
+            assert frames_match
+
+
+@pytest.mark.parametrize(
+    "state_deltas, action_deltas",
+    [
+        ([-1, -0.5, -0.20, 0], [0, 1, 2, 3]),
+        ([-1, -0.5, -0.20, 0], [-1.5, -1, -0.5, -0.20, -0.10, 0]),
+        ([-2, -1, -0.5, 0], [0, 1, 2, 3]),
+        ([-2, -1, -0.5, 0], [-1.5, -1, -0.5, -0.20, -0.10, 0]),
+    ],
+)
+def test_frames_with_delta_consistency(tmp_path, lerobot_dataset_factory, state_deltas, action_deltas):
+    ds_num_frames = 500
+    ds_num_episodes = 10
+    buffer_size = 100
+
+    seed = 42
+
+    local_path = tmp_path / "test"
+    repo_id = f"{DUMMY_REPO_ID}-ciao"
+    camera_key = "phone"
+
+    delta_timestamps = {
+        camera_key: state_deltas,
+        "state": state_deltas,
+        ACTION: action_deltas,
+    }
+
+    ds = lerobot_dataset_factory(
+        root=local_path,
+        repo_id=repo_id,
+        total_episodes=ds_num_episodes,
+        total_frames=ds_num_frames,
+        delta_timestamps=delta_timestamps,
+    )
+
+    streaming_ds = iter(
+        StreamingLeRobotDataset(
+            repo_id=repo_id,
+            root=local_path,
+            buffer_size=buffer_size,
+            seed=seed,
+            shuffle=False,
+            delta_timestamps=delta_timestamps,
+        )
+    )
+
+    for i in range(ds_num_frames):
+        streaming_frame = next(streaming_ds)
+        frame_idx = streaming_frame["index"]
+        target_frame = ds[frame_idx]
+
+        assert set(streaming_frame.keys()) == set(target_frame.keys()), (
+            f"Keys differ between streaming frame and target one. Differ at: {set(streaming_frame.keys()) - set(target_frame.keys())}"
+        )
+
+        key_checks = []
+        for key in streaming_frame:
+            left = streaming_frame[key]
+            right = target_frame[key]
+
+            if isinstance(left, str):
+                check = left == right
+
+            elif isinstance(left, torch.Tensor):
+                if (
+                    key not in ds.meta.camera_keys
+                    and "is_pad" not in key
+                    and f"{key}_is_pad" in streaming_frame
+                ):
+                    # comparing frames only on non-padded regions. Padding is applied to last-valid broadcasting
+                    left = left[~streaming_frame[f"{key}_is_pad"]]
+                    right = right[~target_frame[f"{key}_is_pad"]]
+
+                check = torch.allclose(left, right) and left.shape == right.shape
+
+            key_checks.append((key, check))
+
+        assert all(t[1] for t in key_checks), (
+            f"Checking {list(filter(lambda t: not t[1], key_checks))[0][0]} left and right were found different (i: {i}, frame_idx: {frame_idx})"
+        )
+
+
+@pytest.mark.parametrize(
+    "state_deltas, action_deltas",
+    [
+        ([-1, -0.5, -0.20, 0], [0, 1, 2, 3, 10, 20]),
+        ([-1, -0.5, -0.20, 0], [-20, -1.5, -1, -0.5, -0.20, -0.10, 0]),
+        ([-2, -1, -0.5, 0], [0, 1, 2, 3, 10, 20]),
+        ([-2, -1, -0.5, 0], [-20, -1.5, -1, -0.5, -0.20, -0.10, 0]),
+    ],
+)
+def test_frames_with_delta_consistency_with_shards(
+    tmp_path, lerobot_dataset_factory, state_deltas, action_deltas
+):
+    ds_num_frames = 100
+    ds_num_episodes = 10
+    buffer_size = 10
+    data_file_size_mb = 0.001
+    chunks_size = 1
+
+    seed = 42
+
+    local_path = tmp_path / "test"
+    repo_id = f"{DUMMY_REPO_ID}-ciao"
+    camera_key = "phone"
+
+    delta_timestamps = {
+        camera_key: state_deltas,
+        "state": state_deltas,
+        ACTION: action_deltas,
+    }
+
+    ds = lerobot_dataset_factory(
+        root=local_path,
+        repo_id=repo_id,
+        total_episodes=ds_num_episodes,
+        total_frames=ds_num_frames,
+        delta_timestamps=delta_timestamps,
+        data_files_size_in_mb=data_file_size_mb,
+        chunks_size=chunks_size,
+    )
+    streaming_ds = StreamingLeRobotDataset(
+        repo_id=repo_id,
+        root=local_path,
+        buffer_size=buffer_size,
+        seed=seed,
+        shuffle=False,
+        delta_timestamps=delta_timestamps,
+        max_num_shards=4,
+    )
+
+    iter(streaming_ds)
+
+    num_shards = 4
+    shards_indices = []
+    for shard_idx in range(num_shards):
+        shard = safe_shard(streaming_ds.hf_dataset, shard_idx, num_shards)
+        shard_indices = [item["index"] for item in shard]
+        shards_indices.append(shard_indices)
+
+    streaming_ds = iter(streaming_ds)
+
+    for i in range(ds_num_frames):
+        streaming_frame = next(streaming_ds)
+        frame_idx = streaming_frame["index"]
+        target_frame = ds[frame_idx]
+
+        assert set(streaming_frame.keys()) == set(target_frame.keys()), (
+            f"Keys differ between streaming frame and target one. Differ at: {set(streaming_frame.keys()) - set(target_frame.keys())}"
+        )
+
+        key_checks = []
+        for key in streaming_frame:
+            left = streaming_frame[key]
+            right = target_frame[key]
+
+            if isinstance(left, str):
+                check = left == right
+
+            elif isinstance(left, torch.Tensor):
+                if (
+                    key not in ds.meta.camera_keys
+                    and "is_pad" not in key
+                    and f"{key}_is_pad" in streaming_frame
+                ):
+                    # comparing frames only on non-padded regions. Padding is applied to last-valid broadcasting
+                    left = left[~streaming_frame[f"{key}_is_pad"]]
+                    right = right[~target_frame[f"{key}_is_pad"]]
+
+                check = torch.allclose(left, right) and left.shape == right.shape
+
+            elif isinstance(left, float):
+                check = left == right.item()  # right is a torch.Tensor
+
+            key_checks.append((key, check))
+
+        assert all(t[1] for t in key_checks), (
+            f"Checking {list(filter(lambda t: not t[1], key_checks))[0][0]} left and right were found different (i: {i}, frame_idx: {frame_idx})"
+        )
diff --git a/lerobot/tests/datasets/test_streaming_video_encoder.py b/lerobot/tests/datasets/test_streaming_video_encoder.py
new file mode 100644
index 0000000000000000000000000000000000000000..a85db6a8dedf4f0fb3292c0d3f4b791f4d462e2b
--- /dev/null
+++ b/lerobot/tests/datasets/test_streaming_video_encoder.py
@@ -0,0 +1,730 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Tests for streaming video encoding and hardware-accelerated encoding."""
+
+import queue
+import threading
+from unittest.mock import patch
+
+import av
+import numpy as np
+import pytest
+
+from lerobot.datasets.video_utils import (
+    VALID_VIDEO_CODECS,
+    StreamingVideoEncoder,
+    _CameraEncoderThread,
+    _get_codec_options,
+    detect_available_hw_encoders,
+    resolve_vcodec,
+)
+from lerobot.utils.constants import OBS_IMAGES
+
+# ─── _get_codec_options tests ───
+
+
+class TestGetCodecOptions:
+    def test_libsvtav1_defaults(self):
+        opts = _get_codec_options("libsvtav1")
+        assert opts["g"] == "2"
+        assert opts["crf"] == "30"
+        assert opts["preset"] == "12"
+
+    def test_libsvtav1_custom_preset(self):
+        opts = _get_codec_options("libsvtav1", preset=8)
+        assert opts["preset"] == "8"
+
+    def test_h264_options(self):
+        opts = _get_codec_options("h264", g=10, crf=23)
+        assert opts["g"] == "10"
+        assert opts["crf"] == "23"
+        assert "preset" not in opts
+
+    def test_videotoolbox_options(self):
+        opts = _get_codec_options("h264_videotoolbox", g=2, crf=30)
+        assert opts["g"] == "2"
+        # CRF 30 maps to quality = max(1, min(100, 100 - 30*2)) = 40
+        assert opts["q:v"] == "40"
+        assert "crf" not in opts
+
+    def test_nvenc_options(self):
+        opts = _get_codec_options("h264_nvenc", g=2, crf=25)
+        assert opts["rc"] == "constqp"
+        assert opts["qp"] == "25"
+        assert "crf" not in opts
+        # NVENC doesn't support g
+        assert "g" not in opts
+
+    def test_vaapi_options(self):
+        opts = _get_codec_options("h264_vaapi", crf=28)
+        assert opts["qp"] == "28"
+
+    def test_qsv_options(self):
+        opts = _get_codec_options("h264_qsv", crf=25)
+        assert opts["global_quality"] == "25"
+
+    def test_no_g_no_crf(self):
+        opts = _get_codec_options("h264", g=None, crf=None)
+        assert "g" not in opts
+        assert "crf" not in opts
+
+
+# ─── HW encoder detection tests ───
+
+
+class TestHWEncoderDetection:
+    def test_detect_available_hw_encoders_returns_list(self):
+        result = detect_available_hw_encoders()
+        assert isinstance(result, list)
+
+    def test_detect_available_hw_encoders_only_valid(self):
+        from lerobot.datasets.video_utils import HW_ENCODERS
+
+        result = detect_available_hw_encoders()
+        for encoder in result:
+            assert encoder in HW_ENCODERS
+
+    def test_resolve_vcodec_passthrough(self):
+        assert resolve_vcodec("libsvtav1") == "libsvtav1"
+        assert resolve_vcodec("h264") == "h264"
+
+    def test_resolve_vcodec_auto_fallback(self):
+        """When no HW encoders are available, auto should fall back to libsvtav1."""
+        with patch("lerobot.datasets.video_utils.detect_available_hw_encoders", return_value=[]):
+            assert resolve_vcodec("auto") == "libsvtav1"
+
+    def test_resolve_vcodec_auto_picks_hw(self):
+        """When a HW encoder is available, auto should pick it."""
+        with patch(
+            "lerobot.datasets.video_utils.detect_available_hw_encoders",
+            return_value=["h264_videotoolbox"],
+        ):
+            assert resolve_vcodec("auto") == "h264_videotoolbox"
+
+    def test_resolve_vcodec_auto_returns_valid(self):
+        """Test that resolve_vcodec('auto') returns a known valid codec."""
+        result = resolve_vcodec("auto")
+        assert result in VALID_VIDEO_CODECS
+
+    def test_hw_encoder_names_accepted_in_validation(self):
+        """Test that HW encoder names pass validation in VALID_VIDEO_CODECS."""
+        assert "auto" in VALID_VIDEO_CODECS
+        assert "h264_videotoolbox" in VALID_VIDEO_CODECS
+        assert "h264_nvenc" in VALID_VIDEO_CODECS
+
+    def test_resolve_vcodec_invalid_raises(self):
+        """Test that resolve_vcodec raises ValueError for invalid codecs."""
+        with pytest.raises(ValueError, match="Invalid vcodec"):
+            resolve_vcodec("not_a_real_codec")
+
+
+# ─── _CameraEncoderThread tests ───
+
+
+class TestCameraEncoderThread:
+    def test_encodes_valid_mp4(self, tmp_path):
+        """Test that the encoder thread creates a valid MP4 file with correct frame count."""
+        num_frames = 30
+        height, width = 64, 96
+        fps = 30
+        video_path = tmp_path / "test_output" / "test.mp4"
+
+        frame_queue: queue.Queue = queue.Queue(maxsize=60)
+        result_queue: queue.Queue = queue.Queue(maxsize=1)
+        stop_event = threading.Event()
+
+        encoder_thread = _CameraEncoderThread(
+            video_path=video_path,
+            fps=fps,
+            vcodec="libsvtav1",
+            pix_fmt="yuv420p",
+            g=2,
+            crf=30,
+            preset=13,
+            frame_queue=frame_queue,
+            result_queue=result_queue,
+            stop_event=stop_event,
+        )
+        encoder_thread.start()
+
+        # Feed frames (HWC uint8)
+        for _ in range(num_frames):
+            frame = np.random.randint(0, 255, (height, width, 3), dtype=np.uint8)
+            frame_queue.put(frame)
+
+        # Send sentinel
+        frame_queue.put(None)
+        encoder_thread.join(timeout=60)
+        assert not encoder_thread.is_alive()
+
+        # Check result
+        status, data = result_queue.get(timeout=5)
+        assert status == "ok"
+        assert data is not None  # Stats should be returned
+        assert "mean" in data
+        assert "std" in data
+        assert "min" in data
+        assert "max" in data
+        assert "count" in data
+
+        # Verify the MP4 file is valid
+        assert video_path.exists()
+        with av.open(str(video_path)) as container:
+            stream = container.streams.video[0]
+            # The frame count should match
+            total_frames = sum(1 for _ in container.decode(stream))
+        assert total_frames == num_frames
+
+    def test_handles_chw_input(self, tmp_path):
+        """Test that CHW format input is handled correctly."""
+        num_frames = 5
+        fps = 30
+        video_path = tmp_path / "test_chw" / "test.mp4"
+
+        frame_queue: queue.Queue = queue.Queue(maxsize=60)
+        result_queue: queue.Queue = queue.Queue(maxsize=1)
+        stop_event = threading.Event()
+
+        encoder_thread = _CameraEncoderThread(
+            video_path=video_path,
+            fps=fps,
+            vcodec="libsvtav1",
+            pix_fmt="yuv420p",
+            g=2,
+            crf=30,
+            preset=13,
+            frame_queue=frame_queue,
+            result_queue=result_queue,
+            stop_event=stop_event,
+        )
+        encoder_thread.start()
+
+        # Feed CHW frames
+        for _ in range(num_frames):
+            frame = np.random.randint(0, 255, (3, 64, 96), dtype=np.uint8)
+            frame_queue.put(frame)
+
+        frame_queue.put(None)
+        encoder_thread.join(timeout=60)
+
+        status, _ = result_queue.get(timeout=5)
+        assert status == "ok"
+        assert video_path.exists()
+
+    def test_stop_event_cancellation(self, tmp_path):
+        """Test that setting the stop event causes the thread to exit."""
+        fps = 30
+        video_path = tmp_path / "test_cancel" / "test.mp4"
+
+        frame_queue: queue.Queue = queue.Queue(maxsize=60)
+        result_queue: queue.Queue = queue.Queue(maxsize=1)
+        stop_event = threading.Event()
+
+        encoder_thread = _CameraEncoderThread(
+            video_path=video_path,
+            fps=fps,
+            vcodec="libsvtav1",
+            pix_fmt="yuv420p",
+            g=2,
+            crf=30,
+            preset=13,
+            frame_queue=frame_queue,
+            result_queue=result_queue,
+            stop_event=stop_event,
+        )
+        encoder_thread.start()
+
+        # Feed a few frames
+        for _ in range(3):
+            frame = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            frame_queue.put(frame)
+
+        # Signal stop instead of sending sentinel
+        stop_event.set()
+        encoder_thread.join(timeout=10)
+        assert not encoder_thread.is_alive()
+
+
+# ─── StreamingVideoEncoder tests ───
+
+
+class TestStreamingVideoEncoder:
+    def test_single_camera_episode(self, tmp_path):
+        """Test encoding a single camera episode."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p", g=2, crf=30, preset=13)
+
+        video_keys = [f"{OBS_IMAGES}.laptop"]
+        encoder.start_episode(video_keys, tmp_path)
+
+        num_frames = 20
+        for _ in range(num_frames):
+            frame = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            encoder.feed_frame(f"{OBS_IMAGES}.laptop", frame)
+
+        results = encoder.finish_episode()
+        assert f"{OBS_IMAGES}.laptop" in results
+
+        mp4_path, stats = results[f"{OBS_IMAGES}.laptop"]
+        assert mp4_path.exists()
+        assert stats is not None
+
+        # Verify frame count
+        with av.open(str(mp4_path)) as container:
+            stream = container.streams.video[0]
+            total_frames = sum(1 for _ in container.decode(stream))
+        assert total_frames == num_frames
+
+        encoder.close()
+
+    def test_multi_camera_episode(self, tmp_path):
+        """Test encoding multiple cameras simultaneously."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p", g=2, crf=30)
+
+        video_keys = [f"{OBS_IMAGES}.laptop", f"{OBS_IMAGES}.phone"]
+        encoder.start_episode(video_keys, tmp_path)
+
+        num_frames = 15
+        for _ in range(num_frames):
+            frame0 = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            frame1 = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            encoder.feed_frame(video_keys[0], frame0)
+            encoder.feed_frame(video_keys[1], frame1)
+
+        results = encoder.finish_episode()
+
+        for key in video_keys:
+            assert key in results
+            mp4_path, stats = results[key]
+            assert mp4_path.exists()
+            assert stats is not None
+
+        encoder.close()
+
+    def test_sequential_episodes(self, tmp_path):
+        """Test that multiple sequential episodes work correctly."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p", g=2, crf=30)
+        video_keys = [f"{OBS_IMAGES}.cam"]
+
+        for ep in range(3):
+            encoder.start_episode(video_keys, tmp_path)
+            num_frames = 10 + ep * 5
+            for _ in range(num_frames):
+                frame = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+                encoder.feed_frame(f"{OBS_IMAGES}.cam", frame)
+            results = encoder.finish_episode()
+
+            mp4_path, stats = results[f"{OBS_IMAGES}.cam"]
+            assert mp4_path.exists()
+
+            with av.open(str(mp4_path)) as container:
+                stream = container.streams.video[0]
+                total_frames = sum(1 for _ in container.decode(stream))
+            assert total_frames == num_frames
+
+        encoder.close()
+
+    def test_cancel_episode(self, tmp_path):
+        """Test that canceling an episode cleans up properly."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p", g=2, crf=30)
+        video_keys = [f"{OBS_IMAGES}.cam"]
+
+        encoder.start_episode(video_keys, tmp_path)
+
+        for _ in range(5):
+            frame = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            encoder.feed_frame(f"{OBS_IMAGES}.cam", frame)
+
+        encoder.cancel_episode()
+
+        # Should be able to start a new episode after cancel
+        encoder.start_episode(video_keys, tmp_path)
+        for _ in range(5):
+            frame = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            encoder.feed_frame(f"{OBS_IMAGES}.cam", frame)
+        results = encoder.finish_episode()
+
+        assert f"{OBS_IMAGES}.cam" in results
+        encoder.close()
+
+    def test_feed_without_start_raises(self, tmp_path):
+        """Test that feeding frames without starting an episode raises."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p")
+        with pytest.raises(RuntimeError, match="No active episode"):
+            encoder.feed_frame("cam", np.zeros((64, 96, 3), dtype=np.uint8))
+        encoder.close()
+
+    def test_finish_without_start_raises(self, tmp_path):
+        """Test that finishing without starting raises."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p")
+        with pytest.raises(RuntimeError, match="No active episode"):
+            encoder.finish_episode()
+        encoder.close()
+
+    def test_close_is_idempotent(self, tmp_path):
+        """Test that close() can be called multiple times safely."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p")
+        encoder.close()
+        encoder.close()  # Should not raise
+
+    def test_video_duration_matches_frame_count(self, tmp_path):
+        """Test that encoded video duration matches num_frames / fps."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p", g=2, crf=30, preset=13)
+        video_keys = [f"{OBS_IMAGES}.cam"]
+        encoder.start_episode(video_keys, tmp_path)
+
+        num_frames = 90  # 3 seconds at 30fps
+        for _ in range(num_frames):
+            frame = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            encoder.feed_frame(f"{OBS_IMAGES}.cam", frame)
+
+        results = encoder.finish_episode()
+        mp4_path, _ = results[f"{OBS_IMAGES}.cam"]
+
+        expected_duration = num_frames / 30.0  # 3.0 seconds
+
+        with av.open(str(mp4_path)) as container:
+            stream = container.streams.video[0]
+            total_frames = sum(1 for _ in container.decode(stream))
+            if stream.duration is not None:
+                actual_duration = float(stream.duration * stream.time_base)
+            else:
+                actual_duration = float(container.duration / av.time_base)
+
+        assert total_frames == num_frames
+        # Allow small tolerance for duration due to codec framing
+        assert abs(actual_duration - expected_duration) < 0.5, (
+            f"Video duration {actual_duration:.2f}s != expected {expected_duration:.2f}s"
+        )
+
+        encoder.close()
+
+    def test_multi_camera_start_episode_called_once(self, tmp_path):
+        """Test that with multiple cameras, no frames are lost due to double start_episode."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p", g=2, crf=30)
+
+        video_keys = [f"{OBS_IMAGES}.cam1", f"{OBS_IMAGES}.cam2"]
+        encoder.start_episode(video_keys, tmp_path)
+
+        num_frames = 30
+        for _ in range(num_frames):
+            frame0 = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            frame1 = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            encoder.feed_frame(video_keys[0], frame0)
+            encoder.feed_frame(video_keys[1], frame1)
+
+        results = encoder.finish_episode()
+
+        # Both cameras should have all frames
+        for key in video_keys:
+            mp4_path, stats = results[key]
+            assert mp4_path.exists()
+            with av.open(str(mp4_path)) as container:
+                stream = container.streams.video[0]
+                total_frames = sum(1 for _ in container.decode(stream))
+            assert total_frames == num_frames, (
+                f"Camera {key}: expected {num_frames} frames, got {total_frames}"
+            )
+
+        encoder.close()
+
+    def test_encoder_threads_passed_to_thread(self, tmp_path):
+        """Test that encoder_threads is stored and passed through to encoder threads."""
+        encoder = StreamingVideoEncoder(
+            fps=30, vcodec="libsvtav1", pix_fmt="yuv420p", g=2, crf=30, encoder_threads=2
+        )
+        assert encoder.encoder_threads == 2
+
+        video_keys = [f"{OBS_IMAGES}.cam"]
+        encoder.start_episode(video_keys, tmp_path)
+
+        # Verify the thread received the encoder_threads value
+        thread = encoder._threads[f"{OBS_IMAGES}.cam"]
+        assert thread.encoder_threads == 2
+
+        # Feed some frames and finish to ensure it works end-to-end
+        num_frames = 10
+        for _ in range(num_frames):
+            frame = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            encoder.feed_frame(f"{OBS_IMAGES}.cam", frame)
+
+        results = encoder.finish_episode()
+        mp4_path, stats = results[f"{OBS_IMAGES}.cam"]
+        assert mp4_path.exists()
+        assert stats is not None
+
+        with av.open(str(mp4_path)) as container:
+            stream = container.streams.video[0]
+            total_frames = sum(1 for _ in container.decode(stream))
+        assert total_frames == num_frames
+
+        encoder.close()
+
+    def test_encoder_threads_none_by_default(self, tmp_path):
+        """Test that encoder_threads defaults to None (codec auto-detect)."""
+        encoder = StreamingVideoEncoder(fps=30, vcodec="libsvtav1", pix_fmt="yuv420p")
+        assert encoder.encoder_threads is None
+        encoder.close()
+
+    def test_graceful_frame_dropping(self, tmp_path):
+        """Test that full queue drops frames instead of crashing."""
+        encoder = StreamingVideoEncoder(
+            fps=30, vcodec="libsvtav1", pix_fmt="yuv420p", g=2, crf=30, preset=13, queue_maxsize=1
+        )
+        video_keys = [f"{OBS_IMAGES}.cam"]
+        encoder.start_episode(video_keys, tmp_path)
+
+        # Feed many frames quickly - with queue_maxsize=1, some will be dropped
+        num_frames = 50
+        for _ in range(num_frames):
+            frame = np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8)
+            encoder.feed_frame(f"{OBS_IMAGES}.cam", frame)
+
+        # Should not raise - frames are dropped gracefully
+        results = encoder.finish_episode()
+        assert f"{OBS_IMAGES}.cam" in results
+
+        mp4_path, _ = results[f"{OBS_IMAGES}.cam"]
+        assert mp4_path.exists()
+
+        # Some frames should have been dropped (queue was tiny)
+        dropped = encoder._dropped_frames.get(f"{OBS_IMAGES}.cam", 0)
+        # We can't guarantee drops but can verify no crash occurred
+        assert dropped >= 0
+
+        encoder.close()
+
+
+# ─── Integration tests with LeRobotDataset ───
+
+
+class TestStreamingEncoderIntegration:
+    def test_add_frame_save_episode_streaming(self, tmp_path):
+        """Full integration test: add_frame -> save_episode with streaming encoding."""
+        from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+        features = {
+            "observation.images.cam": {
+                "dtype": "video",
+                "shape": (64, 96, 3),
+                "names": ["height", "width", "channels"],
+            },
+            "action": {"dtype": "float32", "shape": (6,), "names": ["j1", "j2", "j3", "j4", "j5", "j6"]},
+        }
+
+        dataset = LeRobotDataset.create(
+            repo_id="test/streaming",
+            fps=30,
+            features=features,
+            root=tmp_path / "streaming_test",
+            use_videos=True,
+            streaming_encoding=True,
+        )
+
+        assert dataset._streaming_encoder is not None
+
+        num_frames = 20
+        for _ in range(num_frames):
+            frame = {
+                "observation.images.cam": np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8),
+                "action": np.random.randn(6).astype(np.float32),
+                "task": "test task",
+            }
+            dataset.add_frame(frame)
+
+        dataset.save_episode()
+
+        # Verify dataset metadata
+        assert dataset.meta.total_episodes == 1
+        assert dataset.meta.total_frames == num_frames
+
+        # Verify stats exist for the video key
+        assert dataset.meta.stats is not None
+        assert "observation.images.cam" in dataset.meta.stats
+        assert "action" in dataset.meta.stats
+
+        dataset.finalize()
+
+    def test_streaming_disabled_creates_pngs(self, tmp_path):
+        """Test that disabling streaming encoding falls back to PNG path."""
+        from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+        features = {
+            "observation.images.cam": {
+                "dtype": "video",
+                "shape": (64, 96, 3),
+                "names": ["height", "width", "channels"],
+            },
+            "action": {"dtype": "float32", "shape": (6,), "names": ["j1", "j2", "j3", "j4", "j5", "j6"]},
+        }
+
+        dataset = LeRobotDataset.create(
+            repo_id="test/no_streaming",
+            fps=30,
+            features=features,
+            root=tmp_path / "no_streaming_test",
+            use_videos=True,
+            streaming_encoding=False,
+        )
+
+        assert dataset._streaming_encoder is None
+
+        num_frames = 5
+        for _ in range(num_frames):
+            frame = {
+                "observation.images.cam": np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8),
+                "action": np.random.randn(6).astype(np.float32),
+                "task": "test task",
+            }
+            dataset.add_frame(frame)
+
+        # With streaming disabled, PNG files should be written
+        images_dir = dataset.root / "images"
+        assert images_dir.exists()
+
+        dataset.save_episode()
+        dataset.finalize()
+
+    def test_multi_episode_streaming(self, tmp_path):
+        """Test recording multiple episodes with streaming encoding."""
+        from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+        features = {
+            "observation.images.cam": {
+                "dtype": "video",
+                "shape": (64, 96, 3),
+                "names": ["height", "width", "channels"],
+            },
+            "action": {"dtype": "float32", "shape": (2,), "names": ["j1", "j2"]},
+        }
+
+        dataset = LeRobotDataset.create(
+            repo_id="test/multi_ep",
+            fps=30,
+            features=features,
+            root=tmp_path / "multi_ep_test",
+            use_videos=True,
+            streaming_encoding=True,
+        )
+
+        for ep in range(3):
+            num_frames = 10 + ep * 5
+            for _ in range(num_frames):
+                frame = {
+                    "observation.images.cam": np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8),
+                    "action": np.random.randn(2).astype(np.float32),
+                    "task": f"task_{ep}",
+                }
+                dataset.add_frame(frame)
+            dataset.save_episode()
+
+        assert dataset.meta.total_episodes == 3
+        assert dataset.meta.total_frames == 10 + 15 + 20
+
+        dataset.finalize()
+
+    def test_clear_episode_buffer_cancels_streaming(self, tmp_path):
+        """Test that clearing episode buffer cancels streaming encoding."""
+        from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+        features = {
+            "observation.images.cam": {
+                "dtype": "video",
+                "shape": (64, 96, 3),
+                "names": ["height", "width", "channels"],
+            },
+            "action": {"dtype": "float32", "shape": (2,), "names": ["j1", "j2"]},
+        }
+
+        dataset = LeRobotDataset.create(
+            repo_id="test/cancel",
+            fps=30,
+            features=features,
+            root=tmp_path / "cancel_test",
+            use_videos=True,
+            streaming_encoding=True,
+        )
+
+        # Add some frames
+        for _ in range(5):
+            frame = {
+                "observation.images.cam": np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8),
+                "action": np.random.randn(2).astype(np.float32),
+                "task": "task",
+            }
+            dataset.add_frame(frame)
+
+        # Cancel and re-record
+        dataset.clear_episode_buffer()
+
+        # Record a new episode
+        for _ in range(10):
+            frame = {
+                "observation.images.cam": np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8),
+                "action": np.random.randn(2).astype(np.float32),
+                "task": "task",
+            }
+            dataset.add_frame(frame)
+        dataset.save_episode()
+
+        assert dataset.meta.total_episodes == 1
+        assert dataset.meta.total_frames == 10
+
+        dataset.finalize()
+
+    def test_multi_camera_add_frame_streaming(self, tmp_path):
+        """Test that start_episode is called once with multiple video keys."""
+        from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+        features = {
+            "observation.images.cam1": {
+                "dtype": "video",
+                "shape": (64, 96, 3),
+                "names": ["height", "width", "channels"],
+            },
+            "observation.images.cam2": {
+                "dtype": "video",
+                "shape": (64, 96, 3),
+                "names": ["height", "width", "channels"],
+            },
+            "action": {"dtype": "float32", "shape": (2,), "names": ["j1", "j2"]},
+        }
+
+        dataset = LeRobotDataset.create(
+            repo_id="test/multi_cam",
+            fps=30,
+            features=features,
+            root=tmp_path / "multi_cam_test",
+            use_videos=True,
+            streaming_encoding=True,
+        )
+
+        num_frames = 15
+        for _ in range(num_frames):
+            frame = {
+                "observation.images.cam1": np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8),
+                "observation.images.cam2": np.random.randint(0, 255, (64, 96, 3), dtype=np.uint8),
+                "action": np.random.randn(2).astype(np.float32),
+                "task": "test task",
+            }
+            dataset.add_frame(frame)
+
+        dataset.save_episode()
+
+        assert dataset.meta.total_episodes == 1
+        assert dataset.meta.total_frames == num_frames
+
+        dataset.finalize()
diff --git a/lerobot/tests/datasets/test_subtask_dataset.py b/lerobot/tests/datasets/test_subtask_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..f80a6c72dd5383fb97efb89c25a8488c3e4b7933
--- /dev/null
+++ b/lerobot/tests/datasets/test_subtask_dataset.py
@@ -0,0 +1,190 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Tests for subtask functionality in LeRobotDataset.
+
+These tests verify that:
+- Subtask information is correctly loaded from datasets that have subtask data
+- The __getitem__ method correctly adds subtask strings to returned items
+- Subtask handling gracefully handles missing data
+"""
+
+import pandas as pd
+import pytest
+import torch
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+
+class TestSubtaskDataset:
+    """Tests for subtask handling in LeRobotDataset."""
+
+    @pytest.fixture
+    def subtask_dataset(self):
+        """Load the test subtask dataset from the hub."""
+        # Use lerobot/pusht-subtask dataset with episode 1
+        return LeRobotDataset(
+            repo_id="lerobot/pusht-subtask",
+            episodes=[1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11],
+        )
+
+    def test_subtask_dataset_loads(self, subtask_dataset):
+        """Test that the subtask dataset loads successfully."""
+        assert subtask_dataset is not None
+        assert len(subtask_dataset) > 0
+
+    def test_subtask_metadata_loaded(self, subtask_dataset):
+        """Test that subtask metadata is loaded when present in dataset."""
+        # The dataset should have subtasks metadata loaded
+        assert subtask_dataset.meta.subtasks is not None
+        assert isinstance(subtask_dataset.meta.subtasks, pd.DataFrame)
+
+    def test_subtask_index_in_features(self, subtask_dataset):
+        """Test that subtask_index is a feature when dataset has subtasks."""
+        assert "subtask_index" in subtask_dataset.features
+
+    def test_getitem_returns_subtask_string(self, subtask_dataset):
+        """Test that __getitem__ correctly adds subtask string to returned item."""
+        item = subtask_dataset[0]
+
+        # Subtask should be present in the returned item
+        assert "subtask" in item
+        assert isinstance(item["subtask"], str)
+        assert len(item["subtask"]) > 0  # Should not be empty
+
+    def test_getitem_has_subtask_index(self, subtask_dataset):
+        """Test that __getitem__ includes subtask_index."""
+        item = subtask_dataset[0]
+
+        assert "subtask_index" in item
+        assert isinstance(item["subtask_index"], torch.Tensor)
+
+    def test_subtask_index_maps_to_valid_subtask(self, subtask_dataset):
+        """Test that subtask_index correctly maps to a subtask in metadata."""
+        item = subtask_dataset[0]
+
+        subtask_idx = item["subtask_index"].item()
+        subtask_from_metadata = subtask_dataset.meta.subtasks.iloc[subtask_idx].name
+
+        assert item["subtask"] == subtask_from_metadata
+
+    def test_all_items_have_subtask(self, subtask_dataset):
+        """Test that all items in the dataset have subtask information."""
+        for i in range(min(len(subtask_dataset), 5)):  # Check first 5 items
+            item = subtask_dataset[i]
+            assert "subtask" in item
+            assert isinstance(item["subtask"], str)
+
+    def test_task_and_subtask_coexist(self, subtask_dataset):
+        """Test that both task and subtask are present in returned items."""
+        item = subtask_dataset[0]
+
+        # Both task and subtask should be present
+        assert "task" in item
+        assert "subtask" in item
+        assert isinstance(item["task"], str)
+        assert isinstance(item["subtask"], str)
+
+
+class TestSubtaskDatasetMissing:
+    """Tests for graceful handling when subtask data is missing."""
+
+    @pytest.fixture
+    def dataset_without_subtasks(self, tmp_path, empty_lerobot_dataset_factory):
+        """Create a dataset without subtask information."""
+        features = {"state": {"dtype": "float32", "shape": (2,), "names": None}}
+        dataset = empty_lerobot_dataset_factory(root=tmp_path / "no_subtask", features=features)
+
+        # Add some frames and save
+        for _ in range(5):
+            dataset.add_frame({"state": torch.randn(2), "task": "Test task"})
+        dataset.save_episode()
+        dataset.finalize()
+
+        # Reload the dataset
+        return LeRobotDataset(dataset.repo_id, root=dataset.root)
+
+    def test_no_subtask_in_features(self, dataset_without_subtasks):
+        """Test that subtask_index is not in features when not provided."""
+        assert "subtask_index" not in dataset_without_subtasks.features
+
+    def test_getitem_without_subtask(self, dataset_without_subtasks):
+        """Test that __getitem__ works when subtask is not present."""
+        item = dataset_without_subtasks[0]
+
+        # Item should still be retrievable
+        assert item is not None
+        assert "state" in item
+        assert "task" in item
+
+        # Subtask should NOT be present
+        assert "subtask" not in item
+
+    def test_subtasks_metadata_is_none(self, dataset_without_subtasks):
+        """Test that subtasks metadata is None when not present."""
+        assert dataset_without_subtasks.meta.subtasks is None
+
+
+class TestSubtaskEdgeCases:
+    """Edge case tests for subtask handling."""
+
+    def test_subtask_with_multiple_episodes(self):
+        """Test subtask handling with multiple episodes if available."""
+        try:
+            dataset = LeRobotDataset(
+                repo_id="lerobot/pusht-subtask",
+                episodes=[1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11],
+            )
+        except Exception:
+            pytest.skip("Could not load test-subtask dataset")
+
+        # Check first and last items have valid subtasks
+        first_item = dataset[0]
+        last_item = dataset[len(dataset) - 1]
+
+        assert "subtask" in first_item
+        assert "subtask" in last_item
+        assert isinstance(first_item["subtask"], str)
+        assert isinstance(last_item["subtask"], str)
+
+    def test_subtask_index_consistency(self):
+        """Test that same subtask_index returns same subtask string."""
+        try:
+            dataset = LeRobotDataset(
+                repo_id="lerobot/pusht-subtask",
+                episodes=[1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11],
+            )
+        except Exception:
+            pytest.skip("Could not load test-subtask dataset")
+
+        if len(dataset) < 2:
+            pytest.skip("Dataset too small for this test")
+
+        # Collect subtask_index to subtask mappings
+        subtask_map = {}
+        for i in range(min(len(dataset), 10)):
+            item = dataset[i]
+            idx = item["subtask_index"].item()
+            subtask = item["subtask"]
+
+            if idx in subtask_map:
+                # Same index should always return same subtask
+                assert subtask_map[idx] == subtask, (
+                    f"Inconsistent subtask for index {idx}: '{subtask_map[idx]}' vs '{subtask}'"
+                )
+            else:
+                subtask_map[idx] = subtask
diff --git a/lerobot/tests/datasets/test_visualize_dataset.py b/lerobot/tests/datasets/test_visualize_dataset.py
new file mode 100644
index 0000000000000000000000000000000000000000..8e92ec82e85b3808d8ab457eb88cb4d7974ce075
--- /dev/null
+++ b/lerobot/tests/datasets/test_visualize_dataset.py
@@ -0,0 +1,33 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import pytest
+
+from lerobot.scripts.lerobot_dataset_viz import visualize_dataset
+
+
+@pytest.mark.skip("TODO: add dummy videos")
+def test_visualize_local_dataset(tmp_path, lerobot_dataset_factory):
+    root = tmp_path / "dataset"
+    output_dir = tmp_path / "outputs"
+    dataset = lerobot_dataset_factory(root=root)
+    rrd_path = visualize_dataset(
+        dataset,
+        episode_index=0,
+        batch_size=32,
+        save=True,
+        output_dir=output_dir,
+    )
+    assert rrd_path.exists()
diff --git a/lerobot/tests/envs/test_envs.py b/lerobot/tests/envs/test_envs.py
new file mode 100644
index 0000000000000000000000000000000000000000..910c275eb4df023cf40f873007da4fb1812e6f85
--- /dev/null
+++ b/lerobot/tests/envs/test_envs.py
@@ -0,0 +1,268 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import importlib
+from dataclasses import dataclass, field
+
+import gymnasium as gym
+import numpy as np
+import pytest
+import torch
+from gymnasium.envs.registration import register, registry as gym_registry
+from gymnasium.utils.env_checker import check_env
+
+import lerobot
+from lerobot.configs.types import PolicyFeature
+from lerobot.envs.configs import EnvConfig
+from lerobot.envs.factory import make_env, make_env_config
+from lerobot.envs.utils import (
+    _normalize_hub_result,
+    _parse_hub_url,
+    preprocess_observation,
+)
+from tests.utils import require_env
+
+OBS_TYPES = ["state", "pixels", "pixels_agent_pos"]
+
+
+@pytest.mark.parametrize("obs_type", OBS_TYPES)
+@pytest.mark.parametrize("env_name, env_task", lerobot.env_task_pairs)
+@require_env
+def test_env(env_name, env_task, obs_type):
+    if env_name == "aloha" and obs_type == "state":
+        pytest.skip("`state` observations not available for aloha")
+
+    package_name = f"gym_{env_name}"
+    importlib.import_module(package_name)
+    env = gym.make(f"{package_name}/{env_task}", obs_type=obs_type)
+    check_env(env.unwrapped, skip_render_check=True)
+    env.close()
+
+
+@pytest.mark.parametrize("env_name", lerobot.available_envs)
+@require_env
+def test_factory(env_name):
+    cfg = make_env_config(env_name)
+    envs = make_env(cfg, n_envs=1)
+    suite_name = next(iter(envs))
+    task_id = next(iter(envs[suite_name]))
+    env = envs[suite_name][task_id]
+    obs, _ = env.reset()
+    obs = preprocess_observation(obs)
+
+    # test image keys are float32 in range [0,1]
+    for key in obs:
+        if "image" not in key:
+            continue
+        img = obs[key]
+        assert img.dtype == torch.float32
+        # TODO(rcadene): we assume for now that image normalization takes place in the model
+        assert img.max() <= 1.0
+        assert img.min() >= 0.0
+
+    env.close()
+
+
+def test_factory_custom_gym_id():
+    gym_id = "dummy_gym_pkg/DummyTask-v0"
+    if gym_id in gym_registry:
+        pytest.skip(f"Environment ID {gym_id} is already registered")
+
+    @EnvConfig.register_subclass("dummy")
+    @dataclass
+    class DummyEnv(EnvConfig):
+        task: str = "DummyTask-v0"
+        fps: int = 10
+        features: dict[str, PolicyFeature] = field(default_factory=dict)
+
+        @property
+        def package_name(self) -> str:
+            return "dummy_gym_pkg"
+
+        @property
+        def gym_id(self) -> str:
+            return gym_id
+
+        @property
+        def gym_kwargs(self) -> dict:
+            return {}
+
+    try:
+        register(id=gym_id, entry_point="gymnasium.envs.classic_control:CartPoleEnv")
+
+        cfg = DummyEnv()
+        envs_dict = make_env(cfg, n_envs=1)
+        dummy_envs = envs_dict["dummy"]
+        assert len(dummy_envs) == 1
+        env = next(iter(dummy_envs.values()))
+        assert env is not None and isinstance(env, gym.vector.VectorEnv)
+        env.close()
+
+    finally:
+        if gym_id in gym_registry:
+            del gym_registry[gym_id]
+
+
+# Hub environment loading tests
+
+
+def test_make_env_hub_url_parsing():
+    """Test URL parsing for hub environment references."""
+    # simple repo_id
+    repo_id, revision, file_path = _parse_hub_url("user/repo")
+    assert repo_id == "user/repo"
+    assert revision is None
+    assert file_path == "env.py"
+
+    # repo with revision
+    repo_id, revision, file_path = _parse_hub_url("user/repo@main")
+    assert repo_id == "user/repo"
+    assert revision == "main"
+    assert file_path == "env.py"
+
+    # repo with custom file path
+    repo_id, revision, file_path = _parse_hub_url("user/repo:custom_env.py")
+    assert repo_id == "user/repo"
+    assert revision is None
+    assert file_path == "custom_env.py"
+
+    # repo with revision and custom file path
+    repo_id, revision, file_path = _parse_hub_url("user/repo@v1.0:envs/my_env.py")
+    assert repo_id == "user/repo"
+    assert revision == "v1.0"
+    assert file_path == "envs/my_env.py"
+
+    # repo with commit hash
+    repo_id, revision, file_path = _parse_hub_url("org/repo@abc123def456")
+    assert repo_id == "org/repo"
+    assert revision == "abc123def456"
+    assert file_path == "env.py"
+
+
+def test_normalize_hub_result():
+    """Test normalization of different return types from hub make_env."""
+    # test with VectorEnv (most common case)
+    mock_vec_env = gym.vector.SyncVectorEnv([lambda: gym.make("CartPole-v1")])
+    result = _normalize_hub_result(mock_vec_env)
+    assert isinstance(result, dict)
+    assert len(result) == 1
+    suite_name = next(iter(result))
+    assert 0 in result[suite_name]
+    assert isinstance(result[suite_name][0], gym.vector.VectorEnv)
+    mock_vec_env.close()
+
+    # test with single Env
+    mock_env = gym.make("CartPole-v1")
+    result = _normalize_hub_result(mock_env)
+    assert isinstance(result, dict)
+    suite_name = next(iter(result))
+    assert 0 in result[suite_name]
+    assert isinstance(result[suite_name][0], gym.vector.VectorEnv)
+    result[suite_name][0].close()
+
+    # test with dict (already normalized)
+    mock_vec_env = gym.vector.SyncVectorEnv([lambda: gym.make("CartPole-v1")])
+    input_dict = {"my_suite": {0: mock_vec_env}}
+    result = _normalize_hub_result(input_dict)
+    assert result == input_dict
+    assert "my_suite" in result
+    assert 0 in result["my_suite"]
+    mock_vec_env.close()
+
+    # test with invalid type
+    with pytest.raises(ValueError, match="Hub `make_env` must return"):
+        _normalize_hub_result("invalid_type")
+
+
+def test_make_env_from_hub_requires_trust_remote_code():
+    """Test that loading from hub requires explicit trust_remote_code=True."""
+    hub_id = "lerobot/cartpole-env"
+
+    # Should raise RuntimeError when trust_remote_code=False (default)
+    with pytest.raises(RuntimeError, match="Refusing to execute remote code"):
+        make_env(hub_id, trust_remote_code=False)
+
+    # Should also raise when not specified (defaults to False)
+    with pytest.raises(RuntimeError, match="Refusing to execute remote code"):
+        make_env(hub_id)
+
+
+@pytest.mark.parametrize(
+    "hub_id",
+    [
+        "lerobot/cartpole-env",
+        "lerobot/cartpole-env@main",
+        "lerobot/cartpole-env:env.py",
+    ],
+)
+def test_make_env_from_hub_with_trust(hub_id):
+    """Test loading environment from Hugging Face Hub with trust_remote_code=True."""
+    # load environment from hub
+    envs_dict = make_env(hub_id, n_envs=2, trust_remote_code=True)
+
+    # verify structure
+    assert isinstance(envs_dict, dict)
+    assert len(envs_dict) >= 1
+
+    # get the first suite and task
+    suite_name = next(iter(envs_dict))
+    task_id = next(iter(envs_dict[suite_name]))
+    env = envs_dict[suite_name][task_id]
+
+    # verify it's a vector environment
+    assert isinstance(env, gym.vector.VectorEnv)
+    assert env.num_envs == 2
+
+    # test basic environment interaction
+    obs, info = env.reset()
+    assert obs is not None
+    assert isinstance(obs, (dict, np.ndarray))
+
+    # take a random action
+    action = env.action_space.sample()
+    obs, reward, terminated, truncated, info = env.step(action)
+    assert obs is not None
+    assert isinstance(reward, np.ndarray)
+    assert len(reward) == 2
+
+    # clean up
+    env.close()
+
+
+def test_make_env_from_hub_async():
+    """Test loading hub environment with async vector environments."""
+    hub_id = "lerobot/cartpole-env"
+
+    # load with async envs
+    envs_dict = make_env(hub_id, n_envs=2, use_async_envs=True, trust_remote_code=True)
+
+    suite_name = next(iter(envs_dict))
+    task_id = next(iter(envs_dict[suite_name]))
+    env = envs_dict[suite_name][task_id]
+
+    # verify it's an async vector environment
+    assert isinstance(env, gym.vector.AsyncVectorEnv)
+    assert env.num_envs == 2
+
+    # test basic interaction
+    obs, info = env.reset()
+    assert obs is not None
+
+    action = env.action_space.sample()
+    obs, reward, terminated, truncated, info = env.step(action)
+    assert len(reward) == 2
+
+    # clean up
+    env.close()
diff --git a/lerobot/tests/fixtures/constants.py b/lerobot/tests/fixtures/constants.py
new file mode 100644
index 0000000000000000000000000000000000000000..35d8776ce88956defe8787267f76d17c703959b3
--- /dev/null
+++ b/lerobot/tests/fixtures/constants.py
@@ -0,0 +1,44 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from lerobot.utils.constants import ACTION, HF_LEROBOT_HOME
+
+LEROBOT_TEST_DIR = HF_LEROBOT_HOME / "_testing"
+DUMMY_REPO_ID = "dummy/repo"
+DUMMY_ROBOT_TYPE = "dummy_robot"
+DUMMY_MOTOR_FEATURES = {
+    ACTION: {
+        "dtype": "float32",
+        "shape": (6,),
+        "names": ["shoulder_pan", "shoulder_lift", "elbow_flex", "wrist_flex", "wrist_roll", "gripper"],
+    },
+    "state": {
+        "dtype": "float32",
+        "shape": (6,),
+        "names": ["shoulder_pan", "shoulder_lift", "elbow_flex", "wrist_flex", "wrist_roll", "gripper"],
+    },
+}
+DUMMY_CAMERA_FEATURES = {
+    "laptop": {"shape": (64, 96, 3), "names": ["height", "width", "channels"], "info": None},
+    "phone": {"shape": (64, 96, 3), "names": ["height", "width", "channels"], "info": None},
+}
+DEFAULT_FPS = 30
+DUMMY_VIDEO_INFO = {
+    "video.fps": DEFAULT_FPS,
+    "video.codec": "av1",
+    "video.pix_fmt": "yuv420p",
+    "video.is_depth_map": False,
+    "has_audio": False,
+}
+DUMMY_CHW = (3, 96, 128)
+DUMMY_HWC = (96, 128, 3)
diff --git a/lerobot/tests/fixtures/dataset_factories.py b/lerobot/tests/fixtures/dataset_factories.py
new file mode 100644
index 0000000000000000000000000000000000000000..5ecb5214571a5ec650f0ddcdd1a2bf1d61617f95
--- /dev/null
+++ b/lerobot/tests/fixtures/dataset_factories.py
@@ -0,0 +1,559 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import random
+import shutil
+from functools import partial
+from pathlib import Path
+from typing import Protocol
+from unittest.mock import patch
+
+import datasets
+import numpy as np
+import pandas as pd
+import PIL.Image
+import pytest
+import torch
+from datasets import Dataset
+
+from lerobot.datasets.dataset_metadata import CODEBASE_VERSION, LeRobotDatasetMetadata
+from lerobot.datasets.feature_utils import get_hf_features_from_features
+from lerobot.datasets.io_utils import hf_transform_to_torch
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.utils import (
+    DEFAULT_CHUNK_SIZE,
+    DEFAULT_DATA_FILE_SIZE_IN_MB,
+    DEFAULT_DATA_PATH,
+    DEFAULT_FEATURES,
+    DEFAULT_VIDEO_FILE_SIZE_IN_MB,
+    DEFAULT_VIDEO_PATH,
+    flatten_dict,
+)
+from lerobot.datasets.video_utils import encode_video_frames
+from tests.fixtures.constants import (
+    DEFAULT_FPS,
+    DUMMY_CAMERA_FEATURES,
+    DUMMY_MOTOR_FEATURES,
+    DUMMY_REPO_ID,
+    DUMMY_ROBOT_TYPE,
+    DUMMY_VIDEO_INFO,
+)
+
+
+class LeRobotDatasetFactory(Protocol):
+    def __call__(self, *args, **kwargs) -> LeRobotDataset: ...
+
+
+def get_task_index(tasks: datasets.Dataset, task: str) -> int:
+    task_idx = tasks.loc[task].task_index.item()
+    return task_idx
+
+
+@pytest.fixture(scope="session")
+def img_tensor_factory():
+    def _create_img_tensor(height=100, width=100, channels=3, dtype=torch.float32) -> torch.Tensor:
+        return torch.rand((channels, height, width), dtype=dtype)
+
+    return _create_img_tensor
+
+
+@pytest.fixture(scope="session")
+def img_array_factory():
+    def _create_img_array(height=100, width=100, channels=3, dtype=np.uint8, content=None) -> np.ndarray:
+        if content is None:
+            # Original random noise behavior
+            if np.issubdtype(dtype, np.unsignedinteger):
+                # Int array in [0, 255] range
+                img_array = np.random.randint(0, 256, size=(height, width, channels), dtype=dtype)
+            elif np.issubdtype(dtype, np.floating):
+                # Float array in [0, 1] range
+                img_array = np.random.rand(height, width, channels).astype(dtype)
+            else:
+                raise ValueError(dtype)
+        else:
+            # Create image with text content using OpenCV
+            import cv2
+
+            # Create white background
+            img_array = np.ones((height, width, channels), dtype=np.uint8) * 255
+
+            # Font settings
+            font = cv2.FONT_HERSHEY_SIMPLEX
+            font_scale = max(0.5, height / 200)  # Scale font with image size
+            font_color = (0, 0, 0)  # Black text
+            thickness = max(1, int(height / 100))
+
+            # Get text size to center it
+            text_size = cv2.getTextSize(content, font, font_scale, thickness)[0]
+            text_x = (width - text_size[0]) // 2
+            text_y = (height + text_size[1]) // 2
+
+            # Put text on image
+            cv2.putText(img_array, content, (text_x, text_y), font, font_scale, font_color, thickness)
+
+            # Handle single channel case
+            if channels == 1:
+                img_array = cv2.cvtColor(img_array, cv2.COLOR_BGR2GRAY)
+                img_array = img_array[:, :, np.newaxis]
+
+            # Convert to target dtype
+            if np.issubdtype(dtype, np.floating):
+                img_array = img_array.astype(dtype) / 255.0
+            else:
+                img_array = img_array.astype(dtype)
+
+        return img_array
+
+    return _create_img_array
+
+
+@pytest.fixture(scope="session")
+def img_factory(img_array_factory):
+    def _create_img(height=100, width=100) -> PIL.Image.Image:
+        img_array = img_array_factory(height=height, width=width)
+        return PIL.Image.fromarray(img_array)
+
+    return _create_img
+
+
+@pytest.fixture(scope="session")
+def features_factory():
+    def _create_features(
+        motor_features: dict = DUMMY_MOTOR_FEATURES,
+        camera_features: dict = DUMMY_CAMERA_FEATURES,
+        use_videos: bool = True,
+    ) -> dict:
+        if use_videos:
+            camera_ft = {
+                key: {"dtype": "video", **ft, **DUMMY_VIDEO_INFO} for key, ft in camera_features.items()
+            }
+        else:
+            camera_ft = {key: {"dtype": "image", **ft} for key, ft in camera_features.items()}
+        return {
+            **motor_features,
+            **camera_ft,
+            **DEFAULT_FEATURES,
+        }
+
+    return _create_features
+
+
+@pytest.fixture(scope="session")
+def info_factory(features_factory):
+    def _create_info(
+        codebase_version: str = CODEBASE_VERSION,
+        fps: int = DEFAULT_FPS,
+        robot_type: str = DUMMY_ROBOT_TYPE,
+        total_episodes: int = 0,
+        total_frames: int = 0,
+        total_tasks: int = 0,
+        total_videos: int = 0,
+        chunks_size: int = DEFAULT_CHUNK_SIZE,
+        data_files_size_in_mb: float = DEFAULT_DATA_FILE_SIZE_IN_MB,
+        video_files_size_in_mb: float = DEFAULT_VIDEO_FILE_SIZE_IN_MB,
+        data_path: str = DEFAULT_DATA_PATH,
+        video_path: str = DEFAULT_VIDEO_PATH,
+        motor_features: dict = DUMMY_MOTOR_FEATURES,
+        camera_features: dict = DUMMY_CAMERA_FEATURES,
+        use_videos: bool = True,
+    ) -> dict:
+        features = features_factory(motor_features, camera_features, use_videos)
+        return {
+            "codebase_version": codebase_version,
+            "robot_type": robot_type,
+            "total_episodes": total_episodes,
+            "total_frames": total_frames,
+            "total_tasks": total_tasks,
+            "total_videos": total_videos,
+            "chunks_size": chunks_size,
+            "data_files_size_in_mb": data_files_size_in_mb,
+            "video_files_size_in_mb": video_files_size_in_mb,
+            "fps": fps,
+            "splits": {},
+            "data_path": data_path,
+            "video_path": video_path if use_videos else None,
+            "features": features,
+        }
+
+    return _create_info
+
+
+@pytest.fixture(scope="session")
+def stats_factory():
+    def _create_stats(
+        features: dict[str] | None = None,
+    ) -> dict:
+        stats = {}
+        for key, ft in features.items():
+            shape = ft["shape"]
+            dtype = ft["dtype"]
+            if dtype in ["image", "video"]:
+                stats[key] = {
+                    "max": np.full((3, 1, 1), 1, dtype=np.float32).tolist(),
+                    "mean": np.full((3, 1, 1), 0.5, dtype=np.float32).tolist(),
+                    "min": np.full((3, 1, 1), 0, dtype=np.float32).tolist(),
+                    "std": np.full((3, 1, 1), 0.25, dtype=np.float32).tolist(),
+                    "count": [10],
+                }
+            else:
+                stats[key] = {
+                    "max": np.full(shape, 1, dtype=dtype).tolist(),
+                    "mean": np.full(shape, 0.5, dtype=dtype).tolist(),
+                    "min": np.full(shape, 0, dtype=dtype).tolist(),
+                    "std": np.full(shape, 0.25, dtype=dtype).tolist(),
+                    "count": [10],
+                }
+        return stats
+
+    return _create_stats
+
+
+@pytest.fixture(scope="session")
+def tasks_factory():
+    def _create_tasks(total_tasks: int = 3) -> pd.DataFrame:
+        ids = list(range(total_tasks))
+        tasks = [f"Perform action {i}." for i in ids]
+        df = pd.DataFrame({"task_index": ids}, index=pd.Index(tasks, name="task"))
+        return df
+
+    return _create_tasks
+
+
+@pytest.fixture(scope="session")
+def episodes_factory(tasks_factory, stats_factory):
+    def _create_episodes(
+        features: dict[str],
+        fps: int = DEFAULT_FPS,
+        total_episodes: int = 3,
+        total_frames: int = 400,
+        video_keys: list[str] | None = None,
+        tasks: pd.DataFrame | None = None,
+        multi_task: bool = False,
+    ):
+        if total_episodes <= 0 or total_frames <= 0:
+            raise ValueError("num_episodes and total_length must be positive integers.")
+        if total_frames < total_episodes:
+            raise ValueError("total_length must be greater than or equal to num_episodes.")
+
+        if tasks is None:
+            min_tasks = 2 if multi_task else 1
+            total_tasks = random.randint(min_tasks, total_episodes)
+            tasks = tasks_factory(total_tasks)
+
+        num_tasks_available = len(tasks)
+
+        if total_episodes < num_tasks_available and not multi_task:
+            raise ValueError("The number of tasks should be less than the number of episodes.")
+
+        # Generate random lengths that sum up to total_length
+        lengths = np.random.multinomial(total_frames, [1 / total_episodes] * total_episodes).tolist()
+
+        # Create empty dictionaries with all keys
+        d = {
+            "episode_index": [],
+            "meta/episodes/chunk_index": [],
+            "meta/episodes/file_index": [],
+            "data/chunk_index": [],
+            "data/file_index": [],
+            "dataset_from_index": [],
+            "dataset_to_index": [],
+            "tasks": [],
+            "length": [],
+        }
+        if video_keys is not None:
+            for video_key in video_keys:
+                d[f"videos/{video_key}/chunk_index"] = []
+                d[f"videos/{video_key}/file_index"] = []
+                d[f"videos/{video_key}/from_timestamp"] = []
+                d[f"videos/{video_key}/to_timestamp"] = []
+
+        for stats_key in flatten_dict({"stats": stats_factory(features)}):
+            d[stats_key] = []
+
+        num_frames = 0
+        remaining_tasks = list(tasks.index)
+        for ep_idx in range(total_episodes):
+            num_tasks_in_episode = random.randint(1, min(3, num_tasks_available)) if multi_task else 1
+            tasks_to_sample = remaining_tasks if len(remaining_tasks) > 0 else list(tasks.index)
+            episode_tasks = random.sample(tasks_to_sample, min(num_tasks_in_episode, len(tasks_to_sample)))
+            if remaining_tasks:
+                for task in episode_tasks:
+                    remaining_tasks.remove(task)
+
+            d["episode_index"].append(ep_idx)
+            # TODO(rcadene): remove heuristic of only one file
+            d["meta/episodes/chunk_index"].append(0)
+            d["meta/episodes/file_index"].append(0)
+            d["data/chunk_index"].append(0)
+            d["data/file_index"].append(0)
+            d["dataset_from_index"].append(num_frames)
+            d["dataset_to_index"].append(num_frames + lengths[ep_idx])
+            d["tasks"].append(episode_tasks)
+            d["length"].append(lengths[ep_idx])
+
+            if video_keys is not None:
+                for video_key in video_keys:
+                    d[f"videos/{video_key}/chunk_index"].append(0)
+                    d[f"videos/{video_key}/file_index"].append(0)
+                    d[f"videos/{video_key}/from_timestamp"].append(num_frames / fps)
+                    d[f"videos/{video_key}/to_timestamp"].append((num_frames + lengths[ep_idx]) / fps)
+
+            # Add stats columns like "stats/action/max"
+            for stats_key, stats in flatten_dict({"stats": stats_factory(features)}).items():
+                d[stats_key].append(stats)
+
+            num_frames += lengths[ep_idx]
+
+        return Dataset.from_dict(d)
+
+    return _create_episodes
+
+
+@pytest.fixture(scope="session")
+def create_videos(info_factory, img_array_factory):
+    def _create_video_directory(
+        root: Path,
+        info: dict | None = None,
+        total_episodes: int = 3,
+        total_frames: int = 150,
+        total_tasks: int = 1,
+    ):
+        if info is None:
+            info = info_factory(
+                total_episodes=total_episodes, total_frames=total_frames, total_tasks=total_tasks
+            )
+
+        video_feats = {key: feats for key, feats in info["features"].items() if feats["dtype"] == "video"}
+        for key, ft in video_feats.items():
+            # create and save images with identifiable content
+            tmp_dir = root / "tmp_images"
+            tmp_dir.mkdir(parents=True, exist_ok=True)
+            for frame_index in range(info["total_frames"]):
+                content = f"{key}-{frame_index}"
+                img = img_array_factory(height=ft["shape"][0], width=ft["shape"][1], content=content)
+                pil_img = PIL.Image.fromarray(img)
+                path = tmp_dir / f"frame-{frame_index:06d}.png"
+                pil_img.save(path)
+
+            video_path = root / DEFAULT_VIDEO_PATH.format(video_key=key, chunk_index=0, file_index=0)
+            video_path.parent.mkdir(parents=True, exist_ok=True)
+            # Use the global fps from info, not video-specific fps which might not exist
+            encode_video_frames(tmp_dir, video_path, fps=info["fps"])
+            shutil.rmtree(tmp_dir)
+
+    return _create_video_directory
+
+
+@pytest.fixture(scope="session")
+def hf_dataset_factory(features_factory, tasks_factory, episodes_factory, img_array_factory):
+    def _create_hf_dataset(
+        features: dict | None = None,
+        tasks: pd.DataFrame | None = None,
+        episodes: datasets.Dataset | None = None,
+        fps: int = DEFAULT_FPS,
+    ) -> datasets.Dataset:
+        if tasks is None:
+            tasks = tasks_factory()
+        if features is None:
+            features = features_factory()
+        if episodes is None:
+            episodes = episodes_factory(features, fps)
+
+        timestamp_col = np.array([], dtype=np.float32)
+        frame_index_col = np.array([], dtype=np.int64)
+        episode_index_col = np.array([], dtype=np.int64)
+        task_index = np.array([], dtype=np.int64)
+        for ep_dict in episodes:
+            timestamp_col = np.concatenate((timestamp_col, np.arange(ep_dict["length"]) / fps))
+            frame_index_col = np.concatenate((frame_index_col, np.arange(ep_dict["length"], dtype=int)))
+            episode_index_col = np.concatenate(
+                (episode_index_col, np.full(ep_dict["length"], ep_dict["episode_index"], dtype=int))
+            )
+            # Slightly incorrect, but for simplicity, we assign to all frames the first task defined in the episode metadata.
+            # TODO(rcadene): assign the tasks of the episode per chunks of frames
+            ep_task_index = get_task_index(tasks, ep_dict["tasks"][0])
+            task_index = np.concatenate((task_index, np.full(ep_dict["length"], ep_task_index, dtype=int)))
+
+        index_col = np.arange(len(episode_index_col))
+
+        robot_cols = {}
+        for key, ft in features.items():
+            if ft["dtype"] == "image":
+                robot_cols[key] = [
+                    img_array_factory(height=ft["shape"][1], width=ft["shape"][0], content=f"{key}-{i}")
+                    for i in range(len(index_col))
+                ]
+            elif ft["shape"][0] > 1 and ft["dtype"] != "video":
+                robot_cols[key] = np.random.random((len(index_col), ft["shape"][0])).astype(ft["dtype"])
+
+        hf_features = get_hf_features_from_features(features)
+        dataset = datasets.Dataset.from_dict(
+            {
+                **robot_cols,
+                "timestamp": timestamp_col,
+                "frame_index": frame_index_col,
+                "episode_index": episode_index_col,
+                "index": index_col,
+                "task_index": task_index,
+            },
+            features=hf_features,
+        )
+        dataset.set_transform(hf_transform_to_torch)
+        return dataset
+
+    return _create_hf_dataset
+
+
+@pytest.fixture(scope="session")
+def lerobot_dataset_metadata_factory(
+    info_factory,
+    stats_factory,
+    tasks_factory,
+    episodes_factory,
+    mock_snapshot_download_factory,
+):
+    def _create_lerobot_dataset_metadata(
+        root: Path,
+        repo_id: str = DUMMY_REPO_ID,
+        info: dict | None = None,
+        stats: dict | None = None,
+        tasks: pd.DataFrame | None = None,
+        episodes: datasets.Dataset | None = None,
+    ) -> LeRobotDatasetMetadata:
+        if info is None:
+            info = info_factory()
+        if stats is None:
+            stats = stats_factory(features=info["features"])
+        if tasks is None:
+            tasks = tasks_factory(total_tasks=info["total_tasks"])
+        if episodes is None:
+            video_keys = [key for key, ft in info["features"].items() if ft["dtype"] == "video"]
+            episodes = episodes_factory(
+                features=info["features"],
+                fps=info["fps"],
+                total_episodes=info["total_episodes"],
+                total_frames=info["total_frames"],
+                video_keys=video_keys,
+                tasks=tasks,
+            )
+
+        mock_snapshot_download = mock_snapshot_download_factory(
+            info=info,
+            stats=stats,
+            tasks=tasks,
+            episodes=episodes,
+        )
+        with (
+            patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version_patch,
+            patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download_patch,
+        ):
+            mock_get_safe_version_patch.side_effect = lambda repo_id, version: version
+            mock_snapshot_download_patch.side_effect = mock_snapshot_download
+
+            return LeRobotDatasetMetadata(repo_id=repo_id, root=root)
+
+    return _create_lerobot_dataset_metadata
+
+
+@pytest.fixture(scope="session")
+def lerobot_dataset_factory(
+    info_factory,
+    stats_factory,
+    tasks_factory,
+    episodes_factory,
+    hf_dataset_factory,
+    mock_snapshot_download_factory,
+    lerobot_dataset_metadata_factory,
+) -> LeRobotDatasetFactory:
+    def _create_lerobot_dataset(
+        root: Path,
+        repo_id: str = DUMMY_REPO_ID,
+        total_episodes: int = 3,
+        total_frames: int = 150,
+        total_tasks: int = 1,
+        multi_task: bool = False,
+        use_videos: bool = True,
+        info: dict | None = None,
+        stats: dict | None = None,
+        tasks: pd.DataFrame | None = None,
+        episodes_metadata: datasets.Dataset | None = None,
+        hf_dataset: datasets.Dataset | None = None,
+        data_files_size_in_mb: float = DEFAULT_DATA_FILE_SIZE_IN_MB,
+        chunks_size: int = DEFAULT_CHUNK_SIZE,
+        **kwargs,
+    ) -> LeRobotDataset:
+        # Instantiate objects
+        if info is None:
+            info = info_factory(
+                total_episodes=total_episodes,
+                total_frames=total_frames,
+                total_tasks=total_tasks,
+                use_videos=use_videos,
+                data_files_size_in_mb=data_files_size_in_mb,
+                chunks_size=chunks_size,
+            )
+        if stats is None:
+            stats = stats_factory(features=info["features"])
+        if tasks is None:
+            tasks = tasks_factory(total_tasks=info["total_tasks"])
+        if episodes_metadata is None:
+            video_keys = [key for key, ft in info["features"].items() if ft["dtype"] == "video"]
+            episodes_metadata = episodes_factory(
+                features=info["features"],
+                fps=info["fps"],
+                total_episodes=info["total_episodes"],
+                total_frames=info["total_frames"],
+                video_keys=video_keys,
+                tasks=tasks,
+                multi_task=multi_task,
+            )
+        if hf_dataset is None:
+            hf_dataset = hf_dataset_factory(
+                features=info["features"], tasks=tasks, episodes=episodes_metadata, fps=info["fps"]
+            )
+
+        # Write data on disk
+        mock_snapshot_download = mock_snapshot_download_factory(
+            info=info,
+            stats=stats,
+            tasks=tasks,
+            episodes=episodes_metadata,
+            hf_dataset=hf_dataset,
+            data_files_size_in_mb=data_files_size_in_mb,
+            chunks_size=chunks_size,
+        )
+        mock_metadata = lerobot_dataset_metadata_factory(
+            root=root,
+            repo_id=repo_id,
+            info=info,
+            stats=stats,
+            tasks=tasks,
+            episodes=episodes_metadata,
+        )
+        with (
+            patch("lerobot.datasets.lerobot_dataset.LeRobotDatasetMetadata") as mock_metadata_patch,
+            patch("lerobot.datasets.lerobot_dataset.get_safe_version") as mock_get_safe_version_patch,
+            patch("lerobot.datasets.lerobot_dataset.snapshot_download") as mock_snapshot_download_patch,
+        ):
+            mock_metadata_patch.return_value = mock_metadata
+            mock_get_safe_version_patch.side_effect = lambda repo_id, version: version
+            mock_snapshot_download_patch.side_effect = mock_snapshot_download
+
+            return LeRobotDataset(repo_id=repo_id, root=root, **kwargs)
+
+    return _create_lerobot_dataset
+
+
+@pytest.fixture(scope="session")
+def empty_lerobot_dataset_factory() -> LeRobotDatasetFactory:
+    return partial(LeRobotDataset.create, repo_id=DUMMY_REPO_ID, fps=DEFAULT_FPS)
diff --git a/lerobot/tests/fixtures/files.py b/lerobot/tests/fixtures/files.py
new file mode 100644
index 0000000000000000000000000000000000000000..92d9ca1e246d225275538a4db67112a70377d2b0
--- /dev/null
+++ b/lerobot/tests/fixtures/files.py
@@ -0,0 +1,178 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+from pathlib import Path
+
+import datasets
+import numpy as np
+import pandas as pd
+import pytest
+from datasets import Dataset
+
+from lerobot.datasets.io_utils import (
+    get_hf_dataset_size_in_mb,
+    write_episodes,
+    write_info,
+    write_stats,
+    write_tasks,
+)
+from lerobot.datasets.utils import (
+    DEFAULT_CHUNK_SIZE,
+    DEFAULT_DATA_FILE_SIZE_IN_MB,
+    DEFAULT_DATA_PATH,
+    update_chunk_file_indices,
+)
+
+
+def write_hf_dataset(
+    hf_dataset: Dataset,
+    local_dir: Path,
+    data_file_size_mb: float | None = None,
+    chunk_size: int | None = None,
+):
+    """
+    Writes a Hugging Face Dataset to one or more Parquet files in a structured directory format.
+
+    If the dataset size is within `DEFAULT_DATA_FILE_SIZE_IN_MB`, it's saved as a single file.
+    Otherwise, the dataset is split into multiple smaller Parquet files, each not exceeding the size limit.
+    The file and chunk indices are managed to organize the output files in a hierarchical structure,
+    e.g., `data/chunk-000/file-000.parquet`, `data/chunk-000/file-001.parquet`, etc.
+    This function ensures that episodes are not split across multiple files.
+
+    Args:
+        hf_dataset (Dataset): The Hugging Face Dataset to be written to disk.
+        local_dir (Path): The root directory where the dataset files will be stored.
+        data_file_size_mb (float, optional): Maximal size for the parquet data file, in MB. Defaults to DEFAULT_DATA_FILE_SIZE_IN_MB.
+        chunk_size (int, optional): Maximal number of files within a chunk folder before creating another one. Defaults to DEFAULT_CHUNK_SIZE.
+    """
+    if data_file_size_mb is None:
+        data_file_size_mb = DEFAULT_DATA_FILE_SIZE_IN_MB
+    if chunk_size is None:
+        chunk_size = DEFAULT_CHUNK_SIZE
+
+    dataset_size_in_mb = get_hf_dataset_size_in_mb(hf_dataset)
+
+    if dataset_size_in_mb <= data_file_size_mb:
+        # If the dataset is small enough, write it to a single file.
+        path = local_dir / DEFAULT_DATA_PATH.format(chunk_index=0, file_index=0)
+        path.parent.mkdir(parents=True, exist_ok=True)
+        hf_dataset.to_parquet(path)
+        return
+
+    # If the dataset is too large, split it into smaller chunks, keeping episodes whole.
+    episode_indices = np.array(hf_dataset["episode_index"])
+    episode_boundaries = np.where(np.diff(episode_indices) != 0)[0] + 1
+    episode_starts = np.concatenate(([0], episode_boundaries))
+    episode_ends = np.concatenate((episode_boundaries, [len(hf_dataset)]))
+
+    num_episodes = len(episode_starts)
+    current_episode_idx = 0
+    chunk_idx, file_idx = 0, 0
+
+    while current_episode_idx < num_episodes:
+        shard_start_row = episode_starts[current_episode_idx]
+        shard_end_row = episode_ends[current_episode_idx]
+        next_episode_to_try_idx = current_episode_idx + 1
+
+        while next_episode_to_try_idx < num_episodes:
+            potential_shard_end_row = episode_ends[next_episode_to_try_idx]
+            dataset_shard_candidate = hf_dataset.select(range(shard_start_row, potential_shard_end_row))
+            shard_size_mb = get_hf_dataset_size_in_mb(dataset_shard_candidate)
+
+            if shard_size_mb > data_file_size_mb:
+                break
+            else:
+                shard_end_row = potential_shard_end_row
+                next_episode_to_try_idx += 1
+
+        dataset_shard = hf_dataset.select(range(shard_start_row, shard_end_row))
+
+        if (
+            shard_start_row == episode_starts[current_episode_idx]
+            and shard_end_row == episode_ends[current_episode_idx]
+        ):
+            shard_size_mb = get_hf_dataset_size_in_mb(dataset_shard)
+            if shard_size_mb > data_file_size_mb:
+                logging.warning(
+                    f"Episode with index {hf_dataset[shard_start_row.item()]['episode_index']} has size {shard_size_mb:.2f}MB, "
+                    f"which is larger than data_file_size_mb ({data_file_size_mb}MB). "
+                    "Writing it to a separate shard anyway to preserve episode integrity."
+                )
+
+        # Define the path for the current shard and ensure the directory exists.
+        path = local_dir / DEFAULT_DATA_PATH.format(chunk_index=chunk_idx, file_index=file_idx)
+        path.parent.mkdir(parents=True, exist_ok=True)
+
+        # Write the shard to a Parquet file.
+        dataset_shard.to_parquet(path)
+
+        # Update chunk and file indices for the next iteration.
+        chunk_idx, file_idx = update_chunk_file_indices(chunk_idx, file_idx, chunk_size)
+        current_episode_idx = next_episode_to_try_idx
+
+
+@pytest.fixture(scope="session")
+def create_info(info_factory):
+    def _create_info(dir: Path, info: dict | None = None):
+        if info is None:
+            info = info_factory()
+        write_info(info, dir)
+
+    return _create_info
+
+
+@pytest.fixture(scope="session")
+def create_stats(stats_factory):
+    def _create_stats(dir: Path, stats: dict | None = None):
+        if stats is None:
+            stats = stats_factory()
+        write_stats(stats, dir)
+
+    return _create_stats
+
+
+@pytest.fixture(scope="session")
+def create_tasks(tasks_factory):
+    def _create_tasks(dir: Path, tasks: pd.DataFrame | None = None):
+        if tasks is None:
+            tasks = tasks_factory()
+        write_tasks(tasks, dir)
+
+    return _create_tasks
+
+
+@pytest.fixture(scope="session")
+def create_episodes(episodes_factory):
+    def _create_episodes(dir: Path, episodes: datasets.Dataset | None = None):
+        if episodes is None:
+            # TODO(rcadene): add features, fps as arguments
+            episodes = episodes_factory()
+        write_episodes(episodes, dir)
+
+    return _create_episodes
+
+
+@pytest.fixture(scope="session")
+def create_hf_dataset(hf_dataset_factory):
+    def _create_hf_dataset(
+        dir: Path,
+        hf_dataset: datasets.Dataset | None = None,
+        data_file_size_in_mb: float | None = None,
+        chunk_size: int | None = None,
+    ):
+        if hf_dataset is None:
+            hf_dataset = hf_dataset_factory()
+        write_hf_dataset(hf_dataset, dir, data_file_size_in_mb, chunk_size)
+
+    return _create_hf_dataset
diff --git a/lerobot/tests/fixtures/hub.py b/lerobot/tests/fixtures/hub.py
new file mode 100644
index 0000000000000000000000000000000000000000..4333b91a3e9bd45c0d851767f998a71031b10c26
--- /dev/null
+++ b/lerobot/tests/fixtures/hub.py
@@ -0,0 +1,147 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from pathlib import Path
+
+import datasets
+import pandas as pd
+import pytest
+from huggingface_hub.utils import filter_repo_objects
+
+from lerobot.datasets.utils import (
+    DEFAULT_CHUNK_SIZE,
+    DEFAULT_DATA_FILE_SIZE_IN_MB,
+    DEFAULT_DATA_PATH,
+    DEFAULT_EPISODES_PATH,
+    DEFAULT_TASKS_PATH,
+    DEFAULT_VIDEO_PATH,
+    INFO_PATH,
+    STATS_PATH,
+)
+from tests.fixtures.constants import LEROBOT_TEST_DIR
+
+
+@pytest.fixture(scope="session")
+def mock_snapshot_download_factory(
+    info_factory,
+    create_info,
+    stats_factory,
+    create_stats,
+    tasks_factory,
+    create_tasks,
+    episodes_factory,
+    create_episodes,
+    hf_dataset_factory,
+    create_hf_dataset,
+    create_videos,
+):
+    """
+    This factory allows to patch snapshot_download such that when called, it will create expected files rather
+    than making calls to the hub api. Its design allows to pass explicitly files which you want to be created.
+    """
+
+    def _mock_snapshot_download_func(
+        info: dict | None = None,
+        stats: dict | None = None,
+        tasks: pd.DataFrame | None = None,
+        episodes: datasets.Dataset | None = None,
+        hf_dataset: datasets.Dataset | None = None,
+        data_files_size_in_mb: float = DEFAULT_DATA_FILE_SIZE_IN_MB,
+        chunks_size: int = DEFAULT_CHUNK_SIZE,
+    ):
+        if info is None:
+            info = info_factory(data_files_size_in_mb=data_files_size_in_mb, chunks_size=chunks_size)
+        if stats is None:
+            stats = stats_factory(features=info["features"])
+        if tasks is None:
+            tasks = tasks_factory(total_tasks=info["total_tasks"])
+        if episodes is None:
+            episodes = episodes_factory(
+                features=info["features"],
+                fps=info["fps"],
+                total_episodes=info["total_episodes"],
+                total_frames=info["total_frames"],
+                tasks=tasks,
+            )
+        if hf_dataset is None:
+            hf_dataset = hf_dataset_factory(tasks=tasks, episodes=episodes, fps=info["fps"])
+
+        def _mock_snapshot_download(
+            repo_id: str,  # TODO(rcadene): repo_id should be used no?
+            local_dir: str | Path | None = None,
+            allow_patterns: str | list[str] | None = None,
+            ignore_patterns: str | list[str] | None = None,
+            *args,
+            **kwargs,
+        ) -> str:
+            if local_dir is None:
+                local_dir = LEROBOT_TEST_DIR
+
+            # List all possible files
+            all_files = [
+                INFO_PATH,
+                STATS_PATH,
+                # TODO(rcadene): remove naive chunk 0 file 0 ?
+                DEFAULT_TASKS_PATH.format(chunk_index=0, file_index=0),
+                DEFAULT_EPISODES_PATH.format(chunk_index=0, file_index=0),
+                DEFAULT_DATA_PATH.format(chunk_index=0, file_index=0),
+            ]
+
+            video_keys = [key for key, feats in info["features"].items() if feats["dtype"] == "video"]
+            for key in video_keys:
+                all_files.append(DEFAULT_VIDEO_PATH.format(video_key=key, chunk_index=0, file_index=0))
+
+            allowed_files = filter_repo_objects(
+                all_files, allow_patterns=allow_patterns, ignore_patterns=ignore_patterns
+            )
+
+            request_info = False
+            request_tasks = False
+            request_episodes = False
+            request_stats = False
+            request_data = False
+            request_videos = False
+            for rel_path in allowed_files:
+                if rel_path.startswith("meta/info.json"):
+                    request_info = True
+                elif rel_path.startswith("meta/stats"):
+                    request_stats = True
+                elif rel_path.startswith("meta/tasks"):
+                    request_tasks = True
+                elif rel_path.startswith("meta/episodes"):
+                    request_episodes = True
+                elif rel_path.startswith("data/"):
+                    request_data = True
+                elif rel_path.startswith("videos/"):
+                    request_videos = True
+                else:
+                    raise ValueError(f"{rel_path} not supported.")
+
+            if request_info:
+                create_info(local_dir, info)
+            if request_stats:
+                create_stats(local_dir, stats)
+            if request_tasks:
+                create_tasks(local_dir, tasks)
+            if request_episodes:
+                create_episodes(local_dir, episodes)
+            if request_data:
+                create_hf_dataset(local_dir, hf_dataset, data_files_size_in_mb, chunks_size)
+            if request_videos:
+                create_videos(root=local_dir, info=info)
+
+            return str(local_dir)
+
+        return _mock_snapshot_download
+
+    return _mock_snapshot_download_func
diff --git a/lerobot/tests/fixtures/optimizers.py b/lerobot/tests/fixtures/optimizers.py
new file mode 100644
index 0000000000000000000000000000000000000000..a1b4a9da06df4f78e43d17c8f5717737b1e74c2c
--- /dev/null
+++ b/lerobot/tests/fixtures/optimizers.py
@@ -0,0 +1,39 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import pytest
+import torch
+
+from lerobot.optim.optimizers import AdamConfig
+from lerobot.optim.schedulers import VQBeTSchedulerConfig
+
+
+@pytest.fixture
+def model_params():
+    return [torch.nn.Parameter(torch.randn(10, 10))]
+
+
+@pytest.fixture
+def optimizer(model_params):
+    optimizer = AdamConfig().build(model_params)
+    # Dummy step to populate state
+    loss = sum(param.sum() for param in model_params)
+    loss.backward()
+    optimizer.step()
+    return optimizer
+
+
+@pytest.fixture
+def scheduler(optimizer):
+    config = VQBeTSchedulerConfig(num_warmup_steps=10, num_vqvae_training_steps=20, num_cycles=0.5)
+    return config.build(optimizer, num_training_steps=100)
diff --git a/lerobot/tests/mocks/mock_dynamixel.py b/lerobot/tests/mocks/mock_dynamixel.py
new file mode 100644
index 0000000000000000000000000000000000000000..84026fc344d3fc1ecd6e3babc18d3fcc40de0d5b
--- /dev/null
+++ b/lerobot/tests/mocks/mock_dynamixel.py
@@ -0,0 +1,596 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import abc
+from collections.abc import Callable
+
+import dynamixel_sdk as dxl
+import serial
+from mock_serial.mock_serial import MockSerial
+
+from lerobot.motors.dynamixel.dynamixel import _split_into_byte_chunks
+
+from .mock_serial_patch import WaitableStub
+
+# https://emanual.robotis.com/docs/en/dxl/crc/
+DXL_CRC_TABLE = [
+    0x0000, 0x8005, 0x800F, 0x000A, 0x801B, 0x001E, 0x0014, 0x8011,
+    0x8033, 0x0036, 0x003C, 0x8039, 0x0028, 0x802D, 0x8027, 0x0022,
+    0x8063, 0x0066, 0x006C, 0x8069, 0x0078, 0x807D, 0x8077, 0x0072,
+    0x0050, 0x8055, 0x805F, 0x005A, 0x804B, 0x004E, 0x0044, 0x8041,
+    0x80C3, 0x00C6, 0x00CC, 0x80C9, 0x00D8, 0x80DD, 0x80D7, 0x00D2,
+    0x00F0, 0x80F5, 0x80FF, 0x00FA, 0x80EB, 0x00EE, 0x00E4, 0x80E1,
+    0x00A0, 0x80A5, 0x80AF, 0x00AA, 0x80BB, 0x00BE, 0x00B4, 0x80B1,
+    0x8093, 0x0096, 0x009C, 0x8099, 0x0088, 0x808D, 0x8087, 0x0082,
+    0x8183, 0x0186, 0x018C, 0x8189, 0x0198, 0x819D, 0x8197, 0x0192,
+    0x01B0, 0x81B5, 0x81BF, 0x01BA, 0x81AB, 0x01AE, 0x01A4, 0x81A1,
+    0x01E0, 0x81E5, 0x81EF, 0x01EA, 0x81FB, 0x01FE, 0x01F4, 0x81F1,
+    0x81D3, 0x01D6, 0x01DC, 0x81D9, 0x01C8, 0x81CD, 0x81C7, 0x01C2,
+    0x0140, 0x8145, 0x814F, 0x014A, 0x815B, 0x015E, 0x0154, 0x8151,
+    0x8173, 0x0176, 0x017C, 0x8179, 0x0168, 0x816D, 0x8167, 0x0162,
+    0x8123, 0x0126, 0x012C, 0x8129, 0x0138, 0x813D, 0x8137, 0x0132,
+    0x0110, 0x8115, 0x811F, 0x011A, 0x810B, 0x010E, 0x0104, 0x8101,
+    0x8303, 0x0306, 0x030C, 0x8309, 0x0318, 0x831D, 0x8317, 0x0312,
+    0x0330, 0x8335, 0x833F, 0x033A, 0x832B, 0x032E, 0x0324, 0x8321,
+    0x0360, 0x8365, 0x836F, 0x036A, 0x837B, 0x037E, 0x0374, 0x8371,
+    0x8353, 0x0356, 0x035C, 0x8359, 0x0348, 0x834D, 0x8347, 0x0342,
+    0x03C0, 0x83C5, 0x83CF, 0x03CA, 0x83DB, 0x03DE, 0x03D4, 0x83D1,
+    0x83F3, 0x03F6, 0x03FC, 0x83F9, 0x03E8, 0x83ED, 0x83E7, 0x03E2,
+    0x83A3, 0x03A6, 0x03AC, 0x83A9, 0x03B8, 0x83BD, 0x83B7, 0x03B2,
+    0x0390, 0x8395, 0x839F, 0x039A, 0x838B, 0x038E, 0x0384, 0x8381,
+    0x0280, 0x8285, 0x828F, 0x028A, 0x829B, 0x029E, 0x0294, 0x8291,
+    0x82B3, 0x02B6, 0x02BC, 0x82B9, 0x02A8, 0x82AD, 0x82A7, 0x02A2,
+    0x82E3, 0x02E6, 0x02EC, 0x82E9, 0x02F8, 0x82FD, 0x82F7, 0x02F2,
+    0x02D0, 0x82D5, 0x82DF, 0x02DA, 0x82CB, 0x02CE, 0x02C4, 0x82C1,
+    0x8243, 0x0246, 0x024C, 0x8249, 0x0258, 0x825D, 0x8257, 0x0252,
+    0x0270, 0x8275, 0x827F, 0x027A, 0x826B, 0x026E, 0x0264, 0x8261,
+    0x0220, 0x8225, 0x822F, 0x022A, 0x823B, 0x023E, 0x0234, 0x8231,
+    0x8213, 0x0216, 0x021C, 0x8219, 0x0208, 0x820D, 0x8207, 0x0202
+]  # fmt: skip
+
+
+class MockDynamixelPacketv2(abc.ABC):
+    @classmethod
+    def build(cls, dxl_id: int, params: list[int], length: int, *args, **kwargs) -> bytes:
+        packet = cls._build(dxl_id, params, length, *args, **kwargs)
+        packet = cls._add_stuffing(packet)
+        packet = cls._add_crc(packet)
+        return bytes(packet)
+
+    @abc.abstractclassmethod
+    def _build(cls, dxl_id: int, params: list[int], length: int, *args, **kwargs) -> list[int]:
+        pass
+
+    @staticmethod
+    def _add_stuffing(packet: list[int]) -> list[int]:
+        """
+        Byte stuffing is a method of adding additional data to generated instruction packets to ensure that
+        the packets are processed successfully. When the byte pattern "0xFF 0xFF 0xFD" appears in a packet,
+        byte stuffing adds 0xFD to the end of the pattern to convert it to “0xFF 0xFF 0xFD 0xFD” to ensure
+        that it is not interpreted as the header at the start of another packet.
+
+        Source: https://emanual.robotis.com/docs/en/dxl/protocol2/#transmission-process
+
+        Args:
+            packet (list[int]): The raw packet without stuffing.
+
+        Returns:
+            list[int]: The packet stuffed if it contained a "0xFF 0xFF 0xFD" byte sequence in its data bytes.
+        """
+        packet_length_in = dxl.DXL_MAKEWORD(packet[dxl.PKT_LENGTH_L], packet[dxl.PKT_LENGTH_H])
+        packet_length_out = packet_length_in
+
+        temp = [0] * dxl.TXPACKET_MAX_LEN
+
+        # FF FF FD XX ID LEN_L LEN_H
+        temp[dxl.PKT_HEADER0 : dxl.PKT_HEADER0 + dxl.PKT_LENGTH_H + 1] = packet[
+            dxl.PKT_HEADER0 : dxl.PKT_HEADER0 + dxl.PKT_LENGTH_H + 1
+        ]
+
+        index = dxl.PKT_INSTRUCTION
+
+        for i in range(0, packet_length_in - 2):  # except CRC
+            temp[index] = packet[i + dxl.PKT_INSTRUCTION]
+            index = index + 1
+            if (
+                packet[i + dxl.PKT_INSTRUCTION] == 0xFD
+                and packet[i + dxl.PKT_INSTRUCTION - 1] == 0xFF
+                and packet[i + dxl.PKT_INSTRUCTION - 2] == 0xFF
+            ):
+                # FF FF FD
+                temp[index] = 0xFD
+                index = index + 1
+                packet_length_out = packet_length_out + 1
+
+        temp[index] = packet[dxl.PKT_INSTRUCTION + packet_length_in - 2]
+        temp[index + 1] = packet[dxl.PKT_INSTRUCTION + packet_length_in - 1]
+        index = index + 2
+
+        if packet_length_in != packet_length_out:
+            packet = [0] * index
+
+        packet[0:index] = temp[0:index]
+
+        packet[dxl.PKT_LENGTH_L] = dxl.DXL_LOBYTE(packet_length_out)
+        packet[dxl.PKT_LENGTH_H] = dxl.DXL_HIBYTE(packet_length_out)
+
+        return packet
+
+    @staticmethod
+    def _add_crc(packet: list[int]) -> list[int]:
+        """Computes and add CRC to the packet.
+
+        https://emanual.robotis.com/docs/en/dxl/crc/
+        https://en.wikipedia.org/wiki/Cyclic_redundancy_check
+
+        Args:
+            packet (list[int]): The raw packet without CRC (but with placeholders for it).
+
+        Returns:
+            list[int]: The raw packet with a valid CRC.
+        """
+        crc = 0
+        for j in range(len(packet) - 2):
+            i = ((crc >> 8) ^ packet[j]) & 0xFF
+            crc = ((crc << 8) ^ DXL_CRC_TABLE[i]) & 0xFFFF
+
+        packet[-2] = dxl.DXL_LOBYTE(crc)
+        packet[-1] = dxl.DXL_HIBYTE(crc)
+
+        return packet
+
+
+class MockInstructionPacket(MockDynamixelPacketv2):
+    """
+    Helper class to build valid Dynamixel Protocol 2.0 Instruction Packets.
+
+    Protocol 2.0 Instruction Packet structure
+    https://emanual.robotis.com/docs/en/dxl/protocol2/#instruction-packet
+
+    | Header              | Packet ID | Length      | Instruction | Params            | CRC         |
+    | ------------------- | --------- | ----------- | ----------- | ----------------- | ----------- |
+    | 0xFF 0xFF 0xFD 0x00 | ID        | Len_L Len_H | Instr       | Param 1 … Param N | CRC_L CRC_H |
+
+    """
+
+    @classmethod
+    def _build(cls, dxl_id: int, params: list[int], length: int, instruction: int) -> list[int]:
+        length = len(params) + 3
+        return [
+            0xFF, 0xFF, 0xFD, 0x00,  # header
+            dxl_id,                  # servo id
+            dxl.DXL_LOBYTE(length),  # length_l
+            dxl.DXL_HIBYTE(length),  # length_h
+            instruction,             # instruction type
+            *params,                 # data bytes
+            0x00, 0x00               # placeholder for CRC
+        ]  # fmt: skip
+
+    @classmethod
+    def ping(
+        cls,
+        dxl_id: int,
+    ) -> bytes:
+        """
+        Builds a "Ping" broadcast instruction.
+        https://emanual.robotis.com/docs/en/dxl/protocol2/#ping-0x01
+
+        No parameters required.
+        """
+        return cls.build(dxl_id=dxl_id, params=[], length=3, instruction=dxl.INST_PING)
+
+    @classmethod
+    def read(
+        cls,
+        dxl_id: int,
+        start_address: int,
+        data_length: int,
+    ) -> bytes:
+        """
+        Builds a "Read" instruction.
+        https://emanual.robotis.com/docs/en/dxl/protocol2/#read-0x02
+
+        The parameters for Read (Protocol 2.0) are:
+            param[0]   = start_address L
+            param[1]   = start_address H
+            param[2]   = data_length L
+            param[3]   = data_length H
+
+        And 'length' = data_length + 5, where:
+            +1 is for instruction byte,
+            +2 is for the length bytes,
+            +2 is for the CRC at the end.
+        """
+        params = [
+            dxl.DXL_LOBYTE(start_address),
+            dxl.DXL_HIBYTE(start_address),
+            dxl.DXL_LOBYTE(data_length),
+            dxl.DXL_HIBYTE(data_length),
+        ]
+        length = len(params) + 3
+        # length = data_length + 5
+        return cls.build(dxl_id=dxl_id, params=params, length=length, instruction=dxl.INST_READ)
+
+    @classmethod
+    def write(
+        cls,
+        dxl_id: int,
+        value: int,
+        start_address: int,
+        data_length: int,
+    ) -> bytes:
+        """
+        Builds a "Write" instruction.
+        https://emanual.robotis.com/docs/en/dxl/protocol2/#write-0x03
+
+        The parameters for Write (Protocol 2.0) are:
+            param[0]   = start_address L
+            param[1]   = start_address H
+            param[2]   = 1st Byte
+            param[3]   = 2nd Byte
+            ...
+            param[1+X] = X-th Byte
+
+        And 'length' = data_length + 5, where:
+            +1 is for instruction byte,
+            +2 is for the length bytes,
+            +2 is for the CRC at the end.
+        """
+        data = _split_into_byte_chunks(value, data_length)
+        params = [
+            dxl.DXL_LOBYTE(start_address),
+            dxl.DXL_HIBYTE(start_address),
+            *data,
+        ]
+        length = data_length + 5
+        return cls.build(dxl_id=dxl_id, params=params, length=length, instruction=dxl.INST_WRITE)
+
+    @classmethod
+    def sync_read(
+        cls,
+        dxl_ids: list[int],
+        start_address: int,
+        data_length: int,
+    ) -> bytes:
+        """
+        Builds a "Sync_Read" broadcast instruction.
+        https://emanual.robotis.com/docs/en/dxl/protocol2/#sync-read-0x82
+
+        The parameters for Sync_Read (Protocol 2.0) are:
+            param[0]   = start_address L
+            param[1]   = start_address H
+            param[2]   = data_length L
+            param[3]   = data_length H
+            param[4+]  = motor IDs to read from
+
+        And 'length' = (number_of_params + 7), where:
+            +1 is for instruction byte,
+            +2 is for the address bytes,
+            +2 is for the length bytes,
+            +2 is for the CRC at the end.
+        """
+        params = [
+            dxl.DXL_LOBYTE(start_address),
+            dxl.DXL_HIBYTE(start_address),
+            dxl.DXL_LOBYTE(data_length),
+            dxl.DXL_HIBYTE(data_length),
+            *dxl_ids,
+        ]
+        length = len(dxl_ids) + 7
+        return cls.build(
+            dxl_id=dxl.BROADCAST_ID, params=params, length=length, instruction=dxl.INST_SYNC_READ
+        )
+
+    @classmethod
+    def sync_write(
+        cls,
+        ids_values: dict[int, int],
+        start_address: int,
+        data_length: int,
+    ) -> bytes:
+        """
+        Builds a "Sync_Write" broadcast instruction.
+        https://emanual.robotis.com/docs/en/dxl/protocol2/#sync-write-0x83
+
+        The parameters for Sync_Write (Protocol 2.0) are:
+            param[0]   = start_address L
+            param[1]   = start_address H
+            param[2]   = data_length L
+            param[3]   = data_length H
+            param[5]   = [1st motor] ID
+            param[5+1] = [1st motor] 1st Byte
+            param[5+2] = [1st motor] 2nd Byte
+            ...
+            param[5+X] = [1st motor] X-th Byte
+            param[6]   = [2nd motor] ID
+            param[6+1] = [2nd motor] 1st Byte
+            param[6+2] = [2nd motor] 2nd Byte
+            ...
+            param[6+X] = [2nd motor] X-th Byte
+
+        And 'length' = ((number_of_params * 1 + data_length) + 7), where:
+            +1 is for instruction byte,
+            +2 is for the address bytes,
+            +2 is for the length bytes,
+            +2 is for the CRC at the end.
+        """
+        data = []
+        for id_, value in ids_values.items():
+            split_value = _split_into_byte_chunks(value, data_length)
+            data += [id_, *split_value]
+        params = [
+            dxl.DXL_LOBYTE(start_address),
+            dxl.DXL_HIBYTE(start_address),
+            dxl.DXL_LOBYTE(data_length),
+            dxl.DXL_HIBYTE(data_length),
+            *data,
+        ]
+        length = len(ids_values) * (1 + data_length) + 7
+        return cls.build(
+            dxl_id=dxl.BROADCAST_ID, params=params, length=length, instruction=dxl.INST_SYNC_WRITE
+        )
+
+
+class MockStatusPacket(MockDynamixelPacketv2):
+    """
+    Helper class to build valid Dynamixel Protocol 2.0 Status Packets.
+
+    Protocol 2.0 Status Packet structure
+    https://emanual.robotis.com/docs/en/dxl/protocol2/#status-packet
+
+    | Header              | Packet ID | Length      | Instruction | Error | Params            | CRC         |
+    | ------------------- | --------- | ----------- | ----------- | ----- | ----------------- | ----------- |
+    | 0xFF 0xFF 0xFD 0x00 | ID        | Len_L Len_H | 0x55        | Err   | Param 1 … Param N | CRC_L CRC_H |
+    """
+
+    @classmethod
+    def _build(cls, dxl_id: int, params: list[int], length: int, error: int = 0) -> list[int]:
+        return [
+            0xFF, 0xFF, 0xFD, 0x00,  # header
+            dxl_id,                  # servo id
+            dxl.DXL_LOBYTE(length),  # length_l
+            dxl.DXL_HIBYTE(length),  # length_h
+            0x55,                    # instruction = 'status'
+            error,                   # error
+            *params,                 # data bytes
+            0x00, 0x00               # placeholder for CRC
+        ]  # fmt: skip
+
+    @classmethod
+    def ping(cls, dxl_id: int, model_nb: int = 1190, firm_ver: int = 50, error: int = 0) -> bytes:
+        """
+        Builds a 'Ping' status packet.
+        https://emanual.robotis.com/docs/en/dxl/protocol2/#ping-0x01
+
+        Args:
+            dxl_id (int): ID of the servo responding.
+            model_nb (int, optional): Desired 'model number' to be returned in the packet. Defaults to 1190
+                which corresponds to a XL330-M077-T.
+            firm_ver (int, optional): Desired 'firmware version' to be returned in the packet.
+                Defaults to 50.
+
+        Returns:
+            bytes: The raw 'Ping' status packet ready to be sent through serial.
+        """
+        params = [dxl.DXL_LOBYTE(model_nb), dxl.DXL_HIBYTE(model_nb), firm_ver]
+        length = 7
+        return cls.build(dxl_id, params=params, length=length, error=error)
+
+    @classmethod
+    def read(cls, dxl_id: int, value: int, param_length: int, error: int = 0) -> bytes:
+        """
+        Builds a 'Read' status packet (also works for 'Sync Read')
+        https://emanual.robotis.com/docs/en/dxl/protocol2/#read-0x02
+        https://emanual.robotis.com/docs/en/dxl/protocol2/#sync-read-0x82
+
+        Args:
+            dxl_id (int): ID of the servo responding.
+            value (int): Desired value to be returned in the packet.
+            param_length (int): The address length as reported in the control table.
+
+        Returns:
+            bytes: The raw 'Present_Position' status packet ready to be sent through serial.
+        """
+        params = _split_into_byte_chunks(value, param_length)
+        length = param_length + 4
+        return cls.build(dxl_id, params=params, length=length, error=error)
+
+
+class MockPortHandler(dxl.PortHandler):
+    """
+    This class overwrite the 'setupPort' method of the Dynamixel PortHandler because it can specify
+    baudrates that are not supported with a serial port on MacOS.
+    """
+
+    def setupPort(self, cflag_baud):  # noqa: N802
+        if self.is_open:
+            self.closePort()
+
+        self.ser = serial.Serial(
+            port=self.port_name,
+            # baudrate=self.baudrate,  <- This will fail on MacOS
+            # parity = serial.PARITY_ODD,
+            # stopbits = serial.STOPBITS_TWO,
+            bytesize=serial.EIGHTBITS,
+            timeout=0,
+        )
+        self.is_open = True
+        self.ser.reset_input_buffer()
+        self.tx_time_per_byte = (1000.0 / self.baudrate) * 10.0
+
+        return True
+
+
+class MockMotors(MockSerial):
+    """
+    This class will simulate physical motors by responding with valid status packets upon receiving some
+    instruction packets. It is meant to test MotorsBus classes.
+    """
+
+    def __init__(self):
+        super().__init__()
+
+    @property
+    def stubs(self) -> dict[str, WaitableStub]:
+        return super().stubs
+
+    def stub(self, *, name=None, **kwargs):
+        new_stub = WaitableStub(**kwargs)
+        self._MockSerial__stubs[name or new_stub.receive_bytes] = new_stub
+        return new_stub
+
+    def build_broadcast_ping_stub(
+        self, ids_models: dict[int, list[int]] | None = None, num_invalid_try: int = 0
+    ) -> str:
+        ping_request = MockInstructionPacket.ping(dxl.BROADCAST_ID)
+        return_packets = b"".join(MockStatusPacket.ping(id_, model) for id_, model in ids_models.items())
+        ping_response = self._build_send_fn(return_packets, num_invalid_try)
+
+        stub_name = "Ping_" + "_".join([str(id_) for id_ in ids_models])
+        self.stub(
+            name=stub_name,
+            receive_bytes=ping_request,
+            send_fn=ping_response,
+        )
+        return stub_name
+
+    def build_ping_stub(
+        self, dxl_id: int, model_nb: int, firm_ver: int = 50, num_invalid_try: int = 0, error: int = 0
+    ) -> str:
+        ping_request = MockInstructionPacket.ping(dxl_id)
+        return_packet = MockStatusPacket.ping(dxl_id, model_nb, firm_ver, error)
+        ping_response = self._build_send_fn(return_packet, num_invalid_try)
+        stub_name = f"Ping_{dxl_id}"
+        self.stub(
+            name=stub_name,
+            receive_bytes=ping_request,
+            send_fn=ping_response,
+        )
+        return stub_name
+
+    def build_read_stub(
+        self,
+        address: int,
+        length: int,
+        dxl_id: int,
+        value: int,
+        reply: bool = True,
+        error: int = 0,
+        num_invalid_try: int = 0,
+    ) -> str:
+        read_request = MockInstructionPacket.read(dxl_id, address, length)
+        return_packet = MockStatusPacket.read(dxl_id, value, length, error) if reply else b""
+        read_response = self._build_send_fn(return_packet, num_invalid_try)
+        stub_name = f"Read_{address}_{length}_{dxl_id}_{value}_{error}"
+        self.stub(
+            name=stub_name,
+            receive_bytes=read_request,
+            send_fn=read_response,
+        )
+        return stub_name
+
+    def build_write_stub(
+        self,
+        address: int,
+        length: int,
+        dxl_id: int,
+        value: int,
+        reply: bool = True,
+        error: int = 0,
+        num_invalid_try: int = 0,
+    ) -> str:
+        sync_read_request = MockInstructionPacket.write(dxl_id, value, address, length)
+        return_packet = MockStatusPacket.build(dxl_id, params=[], length=4, error=error) if reply else b""
+        stub_name = f"Write_{address}_{length}_{dxl_id}"
+        self.stub(
+            name=stub_name,
+            receive_bytes=sync_read_request,
+            send_fn=self._build_send_fn(return_packet, num_invalid_try),
+        )
+        return stub_name
+
+    def build_sync_read_stub(
+        self,
+        address: int,
+        length: int,
+        ids_values: dict[int, int],
+        reply: bool = True,
+        num_invalid_try: int = 0,
+    ) -> str:
+        sync_read_request = MockInstructionPacket.sync_read(list(ids_values), address, length)
+        return_packets = (
+            b"".join(MockStatusPacket.read(id_, pos, length) for id_, pos in ids_values.items())
+            if reply
+            else b""
+        )
+        sync_read_response = self._build_send_fn(return_packets, num_invalid_try)
+        stub_name = f"Sync_Read_{address}_{length}_" + "_".join([str(id_) for id_ in ids_values])
+        self.stub(
+            name=stub_name,
+            receive_bytes=sync_read_request,
+            send_fn=sync_read_response,
+        )
+        return stub_name
+
+    def build_sequential_sync_read_stub(
+        self, address: int, length: int, ids_values: dict[int, list[int]] | None = None
+    ) -> str:
+        sequence_length = len(next(iter(ids_values.values())))
+        assert all(len(positions) == sequence_length for positions in ids_values.values())
+        sync_read_request = MockInstructionPacket.sync_read(list(ids_values), address, length)
+        sequential_packets = []
+        for count in range(sequence_length):
+            return_packets = b"".join(
+                MockStatusPacket.read(id_, positions[count], length) for id_, positions in ids_values.items()
+            )
+            sequential_packets.append(return_packets)
+
+        sync_read_response = self._build_sequential_send_fn(sequential_packets)
+        stub_name = f"Seq_Sync_Read_{address}_{length}_" + "_".join([str(id_) for id_ in ids_values])
+        self.stub(
+            name=stub_name,
+            receive_bytes=sync_read_request,
+            send_fn=sync_read_response,
+        )
+        return stub_name
+
+    def build_sync_write_stub(
+        self, address: int, length: int, ids_values: dict[int, int], num_invalid_try: int = 0
+    ) -> str:
+        sync_read_request = MockInstructionPacket.sync_write(ids_values, address, length)
+        stub_name = f"Sync_Write_{address}_{length}_" + "_".join([str(id_) for id_ in ids_values])
+        self.stub(
+            name=stub_name,
+            receive_bytes=sync_read_request,
+            send_fn=self._build_send_fn(b"", num_invalid_try),
+        )
+        return stub_name
+
+    @staticmethod
+    def _build_send_fn(packet: bytes, num_invalid_try: int = 0) -> Callable[[int], bytes]:
+        def send_fn(_call_count: int) -> bytes:
+            if num_invalid_try >= _call_count:
+                return b""
+            return packet
+
+        return send_fn
+
+    @staticmethod
+    def _build_sequential_send_fn(packets: list[bytes]) -> Callable[[int], bytes]:
+        def send_fn(_call_count: int) -> bytes:
+            return packets[_call_count - 1]
+
+        return send_fn
diff --git a/lerobot/tests/mocks/mock_feetech.py b/lerobot/tests/mocks/mock_feetech.py
new file mode 100644
index 0000000000000000000000000000000000000000..33cbc41d6f1be340ce56cd4732cb3239cc51d103
--- /dev/null
+++ b/lerobot/tests/mocks/mock_feetech.py
@@ -0,0 +1,444 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import abc
+from collections.abc import Callable
+
+import scservo_sdk as scs
+import serial
+from mock_serial import MockSerial
+
+from lerobot.motors.feetech.feetech import _split_into_byte_chunks, patch_setPacketTimeout
+
+from .mock_serial_patch import WaitableStub
+
+
+class MockFeetechPacket(abc.ABC):
+    @classmethod
+    def build(cls, scs_id: int, params: list[int], length: int, *args, **kwargs) -> bytes:
+        packet = cls._build(scs_id, params, length, *args, **kwargs)
+        packet = cls._add_checksum(packet)
+        return bytes(packet)
+
+    @abc.abstractclassmethod
+    def _build(cls, scs_id: int, params: list[int], length: int, *args, **kwargs) -> list[int]:
+        pass
+
+    @staticmethod
+    def _add_checksum(packet: list[int]) -> list[int]:
+        checksum = 0
+        for id_ in range(2, len(packet) - 1):  # except header & checksum
+            checksum += packet[id_]
+
+        packet[-1] = ~checksum & 0xFF
+
+        return packet
+
+
+class MockInstructionPacket(MockFeetechPacket):
+    """
+    Helper class to build valid Feetech Instruction Packets.
+
+    Instruction Packet structure
+    (from https://files.waveshare.com/upload/2/27/Communication_Protocol_User_Manual-EN%28191218-0923%29.pdf)
+
+    | Header    | Packet ID | Length | Instruction | Params            | Checksum |
+    | --------- | --------- | ------ | ----------- | ----------------- | -------- |
+    | 0xFF 0xFF | ID        | Len    | Instr       | Param 1 … Param N | Sum      |
+
+    """
+
+    @classmethod
+    def _build(cls, scs_id: int, params: list[int], length: int, instruction: int) -> list[int]:
+        return [
+            0xFF, 0xFF,   # header
+            scs_id,       # servo id
+            length,       # length
+            instruction,  # instruction type
+            *params,      # data bytes
+            0x00,         # placeholder for checksum
+        ]  # fmt: skip
+
+    @classmethod
+    def ping(
+        cls,
+        scs_id: int,
+    ) -> bytes:
+        """
+        Builds a "Ping" broadcast instruction.
+
+        No parameters required.
+        """
+        return cls.build(scs_id=scs_id, params=[], length=2, instruction=scs.INST_PING)
+
+    @classmethod
+    def read(
+        cls,
+        scs_id: int,
+        start_address: int,
+        data_length: int,
+    ) -> bytes:
+        """
+        Builds a "Read" instruction.
+
+        The parameters for Read are:
+            param[0]   = start_address
+            param[1]   = data_length
+
+        And 'length' = 4, where:
+            +1 is for instruction byte,
+            +1 is for the address byte,
+            +1 is for the length bytes,
+            +1 is for the checksum at the end.
+        """
+        params = [start_address, data_length]
+        length = 4
+        return cls.build(scs_id=scs_id, params=params, length=length, instruction=scs.INST_READ)
+
+    @classmethod
+    def write(
+        cls,
+        scs_id: int,
+        value: int,
+        start_address: int,
+        data_length: int,
+    ) -> bytes:
+        """
+        Builds a "Write" instruction.
+
+        The parameters for Write are:
+            param[0]   = start_address L
+            param[1]   = start_address H
+            param[2]   = 1st Byte
+            param[3]   = 2nd Byte
+            ...
+            param[1+X] = X-th Byte
+
+        And 'length' = data_length + 3, where:
+            +1 is for instruction byte,
+            +1 is for the length bytes,
+            +1 is for the checksum at the end.
+        """
+        data = _split_into_byte_chunks(value, data_length)
+        params = [start_address, *data]
+        length = data_length + 3
+        return cls.build(scs_id=scs_id, params=params, length=length, instruction=scs.INST_WRITE)
+
+    @classmethod
+    def sync_read(
+        cls,
+        scs_ids: list[int],
+        start_address: int,
+        data_length: int,
+    ) -> bytes:
+        """
+        Builds a "Sync_Read" broadcast instruction.
+
+        The parameters for Sync Read are:
+            param[0]   = start_address
+            param[1]   = data_length
+            param[2+]  = motor IDs to read from
+
+        And 'length' = (number_of_params + 4), where:
+            +1 is for instruction byte,
+            +1 is for the address byte,
+            +1 is for the length bytes,
+            +1 is for the checksum at the end.
+        """
+        params = [start_address, data_length, *scs_ids]
+        length = len(scs_ids) + 4
+        return cls.build(
+            scs_id=scs.BROADCAST_ID, params=params, length=length, instruction=scs.INST_SYNC_READ
+        )
+
+    @classmethod
+    def sync_write(
+        cls,
+        ids_values: dict[int, int],
+        start_address: int,
+        data_length: int,
+    ) -> bytes:
+        """
+        Builds a "Sync_Write" broadcast instruction.
+
+        The parameters for Sync_Write are:
+            param[0]   = start_address
+            param[1]   = data_length
+            param[2]   = [1st motor] ID
+            param[2+1] = [1st motor] 1st Byte
+            param[2+2] = [1st motor] 2nd Byte
+            ...
+            param[5+X] = [1st motor] X-th Byte
+            param[6]   = [2nd motor] ID
+            param[6+1] = [2nd motor] 1st Byte
+            param[6+2] = [2nd motor] 2nd Byte
+            ...
+            param[6+X] = [2nd motor] X-th Byte
+
+        And 'length' = ((number_of_params * 1 + data_length) + 4), where:
+            +1 is for instruction byte,
+            +1 is for the address byte,
+            +1 is for the length bytes,
+            +1 is for the checksum at the end.
+        """
+        data = []
+        for id_, value in ids_values.items():
+            split_value = _split_into_byte_chunks(value, data_length)
+            data += [id_, *split_value]
+        params = [start_address, data_length, *data]
+        length = len(ids_values) * (1 + data_length) + 4
+        return cls.build(
+            scs_id=scs.BROADCAST_ID, params=params, length=length, instruction=scs.INST_SYNC_WRITE
+        )
+
+
+class MockStatusPacket(MockFeetechPacket):
+    """
+    Helper class to build valid Feetech Status Packets.
+
+    Status Packet structure
+    (from https://files.waveshare.com/upload/2/27/Communication_Protocol_User_Manual-EN%28191218-0923%29.pdf)
+
+    | Header    | Packet ID | Length | Error | Params            | Checksum |
+    | --------- | --------- | ------ | ----- | ----------------- | -------- |
+    | 0xFF 0xFF | ID        | Len    | Err   | Param 1 … Param N | Sum      |
+
+    """
+
+    @classmethod
+    def _build(cls, scs_id: int, params: list[int], length: int, error: int = 0) -> list[int]:
+        return [
+            0xFF, 0xFF,  # header
+            scs_id,      # servo id
+            length,      # length
+            error,       # status
+            *params,     # data bytes
+            0x00,        # placeholder for checksum
+        ]  # fmt: skip
+
+    @classmethod
+    def ping(cls, scs_id: int, error: int = 0) -> bytes:
+        """Builds a 'Ping' status packet.
+
+        Args:
+            scs_id (int): ID of the servo responding.
+            error (int, optional): Error to be returned. Defaults to 0 (success).
+
+        Returns:
+            bytes: The raw 'Ping' status packet ready to be sent through serial.
+        """
+        return cls.build(scs_id, params=[], length=2, error=error)
+
+    @classmethod
+    def read(cls, scs_id: int, value: int, param_length: int, error: int = 0) -> bytes:
+        """Builds a 'Read' status packet.
+
+        Args:
+            scs_id (int): ID of the servo responding.
+            value (int): Desired value to be returned in the packet.
+            param_length (int): The address length as reported in the control table.
+
+        Returns:
+            bytes: The raw 'Sync Read' status packet ready to be sent through serial.
+        """
+        params = _split_into_byte_chunks(value, param_length)
+        length = param_length + 2
+        return cls.build(scs_id, params=params, length=length, error=error)
+
+
+class MockPortHandler(scs.PortHandler):
+    """
+    This class overwrite the 'setupPort' method of the Feetech PortHandler because it can specify
+    baudrates that are not supported with a serial port on MacOS.
+    """
+
+    def setupPort(self, cflag_baud):  # noqa: N802
+        if self.is_open:
+            self.closePort()
+
+        self.ser = serial.Serial(
+            port=self.port_name,
+            # baudrate=self.baudrate,  <- This will fail on MacOS
+            # parity = serial.PARITY_ODD,
+            # stopbits = serial.STOPBITS_TWO,
+            bytesize=serial.EIGHTBITS,
+            timeout=0,
+        )
+        self.is_open = True
+        self.ser.reset_input_buffer()
+        self.tx_time_per_byte = (1000.0 / self.baudrate) * 10.0
+
+        return True
+
+    def setPacketTimeout(self, packet_length):  # noqa: N802
+        return patch_setPacketTimeout(self, packet_length)
+
+
+class MockMotors(MockSerial):
+    """
+    This class will simulate physical motors by responding with valid status packets upon receiving some
+    instruction packets. It is meant to test MotorsBus classes.
+    """
+
+    def __init__(self):
+        super().__init__()
+
+    @property
+    def stubs(self) -> dict[str, WaitableStub]:
+        return super().stubs
+
+    def stub(self, *, name=None, **kwargs):
+        new_stub = WaitableStub(**kwargs)
+        self._MockSerial__stubs[name or new_stub.receive_bytes] = new_stub
+        return new_stub
+
+    def build_broadcast_ping_stub(self, ids: list[int] | None = None, num_invalid_try: int = 0) -> str:
+        ping_request = MockInstructionPacket.ping(scs.BROADCAST_ID)
+        return_packets = b"".join(MockStatusPacket.ping(id_) for id_ in ids)
+        ping_response = self._build_send_fn(return_packets, num_invalid_try)
+        stub_name = "Ping_" + "_".join([str(id_) for id_ in ids])
+        self.stub(
+            name=stub_name,
+            receive_bytes=ping_request,
+            send_fn=ping_response,
+        )
+        return stub_name
+
+    def build_ping_stub(self, scs_id: int, num_invalid_try: int = 0, error: int = 0) -> str:
+        ping_request = MockInstructionPacket.ping(scs_id)
+        return_packet = MockStatusPacket.ping(scs_id, error)
+        ping_response = self._build_send_fn(return_packet, num_invalid_try)
+        stub_name = f"Ping_{scs_id}_{error}"
+        self.stub(
+            name=stub_name,
+            receive_bytes=ping_request,
+            send_fn=ping_response,
+        )
+        return stub_name
+
+    def build_read_stub(
+        self,
+        address: int,
+        length: int,
+        scs_id: int,
+        value: int,
+        reply: bool = True,
+        error: int = 0,
+        num_invalid_try: int = 0,
+    ) -> str:
+        read_request = MockInstructionPacket.read(scs_id, address, length)
+        return_packet = MockStatusPacket.read(scs_id, value, length, error) if reply else b""
+        read_response = self._build_send_fn(return_packet, num_invalid_try)
+        stub_name = f"Read_{address}_{length}_{scs_id}_{value}_{error}"
+        self.stub(
+            name=stub_name,
+            receive_bytes=read_request,
+            send_fn=read_response,
+        )
+        return stub_name
+
+    def build_write_stub(
+        self,
+        address: int,
+        length: int,
+        scs_id: int,
+        value: int,
+        reply: bool = True,
+        error: int = 0,
+        num_invalid_try: int = 0,
+    ) -> str:
+        sync_read_request = MockInstructionPacket.write(scs_id, value, address, length)
+        return_packet = MockStatusPacket.build(scs_id, params=[], length=2, error=error) if reply else b""
+        stub_name = f"Write_{address}_{length}_{scs_id}"
+        self.stub(
+            name=stub_name,
+            receive_bytes=sync_read_request,
+            send_fn=self._build_send_fn(return_packet, num_invalid_try),
+        )
+        return stub_name
+
+    def build_sync_read_stub(
+        self,
+        address: int,
+        length: int,
+        ids_values: dict[int, int],
+        reply: bool = True,
+        num_invalid_try: int = 0,
+    ) -> str:
+        sync_read_request = MockInstructionPacket.sync_read(list(ids_values), address, length)
+        return_packets = (
+            b"".join(MockStatusPacket.read(id_, pos, length) for id_, pos in ids_values.items())
+            if reply
+            else b""
+        )
+        sync_read_response = self._build_send_fn(return_packets, num_invalid_try)
+        stub_name = f"Sync_Read_{address}_{length}_" + "_".join([str(id_) for id_ in ids_values])
+        self.stub(
+            name=stub_name,
+            receive_bytes=sync_read_request,
+            send_fn=sync_read_response,
+        )
+        return stub_name
+
+    def build_sequential_sync_read_stub(
+        self, address: int, length: int, ids_values: dict[int, list[int]] | None = None
+    ) -> str:
+        sequence_length = len(next(iter(ids_values.values())))
+        assert all(len(positions) == sequence_length for positions in ids_values.values())
+        sync_read_request = MockInstructionPacket.sync_read(list(ids_values), address, length)
+        sequential_packets = []
+        for count in range(sequence_length):
+            return_packets = b"".join(
+                MockStatusPacket.read(id_, positions[count], length) for id_, positions in ids_values.items()
+            )
+            sequential_packets.append(return_packets)
+
+        sync_read_response = self._build_sequential_send_fn(sequential_packets)
+        stub_name = f"Seq_Sync_Read_{address}_{length}_" + "_".join([str(id_) for id_ in ids_values])
+        self.stub(
+            name=stub_name,
+            receive_bytes=sync_read_request,
+            send_fn=sync_read_response,
+        )
+        return stub_name
+
+    def build_sync_write_stub(
+        self, address: int, length: int, ids_values: dict[int, int], num_invalid_try: int = 0
+    ) -> str:
+        sync_read_request = MockInstructionPacket.sync_write(ids_values, address, length)
+        stub_name = f"Sync_Write_{address}_{length}_" + "_".join([str(id_) for id_ in ids_values])
+        self.stub(
+            name=stub_name,
+            receive_bytes=sync_read_request,
+            send_fn=self._build_send_fn(b"", num_invalid_try),
+        )
+        return stub_name
+
+    @staticmethod
+    def _build_send_fn(packet: bytes, num_invalid_try: int = 0) -> Callable[[int], bytes]:
+        def send_fn(_call_count: int) -> bytes:
+            if num_invalid_try >= _call_count:
+                return b""
+            return packet
+
+        return send_fn
+
+    @staticmethod
+    def _build_sequential_send_fn(packets: list[bytes]) -> Callable[[int], bytes]:
+        def send_fn(_call_count: int) -> bytes:
+            return packets[_call_count - 1]
+
+        return send_fn
diff --git a/lerobot/tests/mocks/mock_motors_bus.py b/lerobot/tests/mocks/mock_motors_bus.py
new file mode 100644
index 0000000000000000000000000000000000000000..a499dbfee340e4624ceecf8aef19ade7aa944616
--- /dev/null
+++ b/lerobot/tests/mocks/mock_motors_bus.py
@@ -0,0 +1,152 @@
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# ruff: noqa: N802
+
+from lerobot.motors.motors_bus import (
+    Motor,
+    MotorsBus,
+)
+
+DUMMY_CTRL_TABLE_1 = {
+    "Firmware_Version": (0, 1),
+    "Model_Number": (1, 2),
+    "Present_Position": (3, 4),
+    "Goal_Position": (11, 2),
+}
+
+DUMMY_CTRL_TABLE_2 = {
+    "Model_Number": (0, 2),
+    "Firmware_Version": (2, 1),
+    "Present_Position": (3, 4),
+    "Present_Velocity": (7, 4),
+    "Goal_Position": (11, 4),
+    "Goal_Velocity": (15, 4),
+    "Lock": (19, 1),
+}
+
+DUMMY_MODEL_CTRL_TABLE = {
+    "model_1": DUMMY_CTRL_TABLE_1,
+    "model_2": DUMMY_CTRL_TABLE_2,
+    "model_3": DUMMY_CTRL_TABLE_2,
+}
+
+DUMMY_BAUDRATE_TABLE = {
+    0: 1_000_000,
+    1: 500_000,
+    2: 250_000,
+}
+
+DUMMY_MODEL_BAUDRATE_TABLE = {
+    "model_1": DUMMY_BAUDRATE_TABLE,
+    "model_2": DUMMY_BAUDRATE_TABLE,
+    "model_3": DUMMY_BAUDRATE_TABLE,
+}
+
+DUMMY_ENCODING_TABLE = {
+    "Present_Position": 8,
+    "Goal_Position": 10,
+}
+
+DUMMY_MODEL_ENCODING_TABLE = {
+    "model_1": DUMMY_ENCODING_TABLE,
+    "model_2": DUMMY_ENCODING_TABLE,
+    "model_3": DUMMY_ENCODING_TABLE,
+}
+
+DUMMY_MODEL_NUMBER_TABLE = {
+    "model_1": 1234,
+    "model_2": 5678,
+    "model_3": 5799,
+}
+
+DUMMY_MODEL_RESOLUTION_TABLE = {
+    "model_1": 4096,
+    "model_2": 1024,
+    "model_3": 4096,
+}
+
+
+class MockPortHandler:
+    def __init__(self, port_name):
+        self.is_open: bool = False
+        self.baudrate: int
+        self.packet_start_time: float
+        self.packet_timeout: float
+        self.tx_time_per_byte: float
+        self.is_using: bool = False
+        self.port_name: str = port_name
+        self.ser = None
+
+    def openPort(self):
+        self.is_open = True
+        return self.is_open
+
+    def closePort(self):
+        self.is_open = False
+
+    def clearPort(self): ...
+    def setPortName(self, port_name):
+        self.port_name = port_name
+
+    def getPortName(self):
+        return self.port_name
+
+    def setBaudRate(self, baudrate):
+        self.baudrate: baudrate
+
+    def getBaudRate(self):
+        return self.baudrate
+
+    def getBytesAvailable(self): ...
+    def readPort(self, length): ...
+    def writePort(self, packet): ...
+    def setPacketTimeout(self, packet_length): ...
+    def setPacketTimeoutMillis(self, msec): ...
+    def isPacketTimeout(self): ...
+    def getCurrentTime(self): ...
+    def getTimeSinceStart(self): ...
+    def setupPort(self, cflag_baud): ...
+    def getCFlagBaud(self, baudrate): ...
+
+
+class MockMotorsBus(MotorsBus):
+    available_baudrates = [500_000, 1_000_000]
+    default_timeout = 1000
+    model_baudrate_table = DUMMY_MODEL_BAUDRATE_TABLE
+    model_ctrl_table = DUMMY_MODEL_CTRL_TABLE
+    model_encoding_table = DUMMY_MODEL_ENCODING_TABLE
+    model_number_table = DUMMY_MODEL_NUMBER_TABLE
+    model_resolution_table = DUMMY_MODEL_RESOLUTION_TABLE
+    normalized_data = ["Present_Position", "Goal_Position"]
+
+    def __init__(self, port: str, motors: dict[str, Motor]):
+        super().__init__(port, motors)
+        self.port_handler = MockPortHandler(port)
+
+    def _assert_protocol_is_compatible(self, instruction_name): ...
+    def _handshake(self): ...
+    def _find_single_motor(self, motor, initial_baudrate): ...
+    def configure_motors(self): ...
+    def is_calibrated(self): ...
+    def read_calibration(self): ...
+    def write_calibration(self, calibration_dict): ...
+    def disable_torque(self, motors, num_retry): ...
+    def _disable_torque(self, motor, model, num_retry): ...
+    def enable_torque(self, motors, num_retry): ...
+    def _get_half_turn_homings(self, positions): ...
+    def _encode_sign(self, data_name, ids_values): ...
+    def _decode_sign(self, data_name, ids_values): ...
+    def _split_into_byte_chunks(self, value, length): ...
+    def broadcast_ping(self, num_retry, raise_on_error): ...
diff --git a/lerobot/tests/mocks/mock_robot.py b/lerobot/tests/mocks/mock_robot.py
new file mode 100644
index 0000000000000000000000000000000000000000..5504b30bf0851f4fc12e4bd4631c9d4789940f45
--- /dev/null
+++ b/lerobot/tests/mocks/mock_robot.py
@@ -0,0 +1,133 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import random
+from dataclasses import dataclass, field
+from functools import cached_property
+
+from lerobot.cameras import CameraConfig, make_cameras_from_configs
+from lerobot.motors.motors_bus import Motor, MotorNormMode
+from lerobot.robots import Robot, RobotConfig
+from lerobot.types import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+from tests.mocks.mock_motors_bus import MockMotorsBus
+
+
+@RobotConfig.register_subclass("mock_robot")
+@dataclass
+class MockRobotConfig(RobotConfig):
+    n_motors: int = 3
+    cameras: dict[str, CameraConfig] = field(default_factory=dict)
+    random_values: bool = True
+    static_values: list[float] | None = None
+    calibrated: bool = True
+
+    def __post_init__(self):
+        if self.n_motors < 1:
+            raise ValueError(self.n_motors)
+
+        if self.random_values and self.static_values is not None:
+            raise ValueError("Choose either random values or static values")
+
+        if self.static_values is not None and len(self.static_values) != self.n_motors:
+            raise ValueError("Specify the same number of static values as motors")
+
+        if len(self.cameras) > 0:
+            raise NotImplementedError  # TODO with the cameras refactor
+
+
+class MockRobot(Robot):
+    """Mock Robot to be used for testing."""
+
+    config_class = MockRobotConfig
+    name = "mock_robot"
+
+    def __init__(self, config: MockRobotConfig):
+        super().__init__(config)
+        self.config = config
+        self._is_connected = False
+        self._is_calibrated = config.calibrated
+        self.cameras = make_cameras_from_configs(config.cameras)
+
+        mock_motors = {}
+        for i in range(config.n_motors):
+            motor_name = f"motor_{i + 1}"
+            mock_motors[motor_name] = Motor(
+                id=i + 1,
+                model="model_1",  # Use model_1 which exists in MockMotorsBus tables
+                norm_mode=MotorNormMode.RANGE_M100_100,
+            )
+
+        self.bus = MockMotorsBus("/dev/dummy-port", mock_motors)
+
+        # NOTE(fracapuano): The .motors attribute was used from the previous interface
+        self.motors = [f"motor_{i + 1}" for i in range(config.n_motors)]
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.motors}
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self._is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self._is_connected = True
+        if calibrate:
+            self.calibrate()
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self._is_calibrated
+
+    @check_if_not_connected
+    def calibrate(self) -> None:
+        self._is_calibrated = True
+
+    def configure(self) -> None:
+        pass
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        if self.config.random_values:
+            return {f"{motor}.pos": random.uniform(-100, 100) for motor in self.motors}
+        else:
+            return {
+                f"{motor}.pos": val for motor, val in zip(self.motors, self.config.static_values, strict=True)
+            }
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        return action
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self._is_connected = False
diff --git a/lerobot/tests/mocks/mock_serial_patch.py b/lerobot/tests/mocks/mock_serial_patch.py
new file mode 100644
index 0000000000000000000000000000000000000000..bde0efae2c9bd24ccf083aac091dadfedeea0204
--- /dev/null
+++ b/lerobot/tests/mocks/mock_serial_patch.py
@@ -0,0 +1,51 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import threading
+import time
+
+from mock_serial.mock_serial import Stub
+
+
+class WaitableStub(Stub):
+    """
+    In some situations, a test might be checking if a stub has been called before `MockSerial` thread had time
+    to read, match, and call the stub. In these situations, the test can fail randomly.
+
+    Use `wait_called()` or `wait_calls()` to block until the stub is called, avoiding race conditions.
+
+    Proposed fix:
+    https://github.com/benthorner/mock_serial/pull/3
+    """
+
+    def __init__(self, **kwargs):
+        super().__init__(**kwargs)
+        self._event = threading.Event()
+
+    def call(self):
+        self._event.set()
+        return super().call()
+
+    def wait_called(self, timeout: float = 1.0):
+        return self._event.wait(timeout)
+
+    def wait_calls(self, min_calls: int = 1, timeout: float = 1.0):
+        start = time.perf_counter()
+        while time.perf_counter() - start < timeout:
+            if self.calls >= min_calls:
+                return self.calls
+            time.sleep(0.005)
+        raise TimeoutError(f"Stub not called {min_calls} times within {timeout} seconds.")
diff --git a/lerobot/tests/mocks/mock_teleop.py b/lerobot/tests/mocks/mock_teleop.py
new file mode 100644
index 0000000000000000000000000000000000000000..b84b2b8918158ecfbee774b3de50a4dae5b4b6a8
--- /dev/null
+++ b/lerobot/tests/mocks/mock_teleop.py
@@ -0,0 +1,102 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import random
+from dataclasses import dataclass
+from functools import cached_property
+from typing import Any
+
+from lerobot.teleoperators import Teleoperator, TeleoperatorConfig
+from lerobot.types import RobotAction
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+
+@TeleoperatorConfig.register_subclass("mock_teleop")
+@dataclass
+class MockTeleopConfig(TeleoperatorConfig):
+    n_motors: int = 3
+    random_values: bool = True
+    static_values: list[float] | None = None
+    calibrated: bool = True
+
+    def __post_init__(self):
+        if self.n_motors < 1:
+            raise ValueError(self.n_motors)
+
+        if self.random_values and self.static_values is not None:
+            raise ValueError("Choose either random values or static values")
+
+        if self.static_values is not None and len(self.static_values) != self.n_motors:
+            raise ValueError("Specify the same number of static values as motors")
+
+
+class MockTeleop(Teleoperator):
+    """Mock Teleoperator to be used for testing."""
+
+    config_class = MockTeleopConfig
+    name = "mock_teleop"
+
+    def __init__(self, config: MockTeleopConfig):
+        super().__init__(config)
+        self.config = config
+        self._is_connected = False
+        self._is_calibrated = config.calibrated
+        self.motors = [f"motor_{i + 1}" for i in range(config.n_motors)]
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.motors}
+
+    @cached_property
+    def feedback_features(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.motors}
+
+    @property
+    def is_connected(self) -> bool:
+        return self._is_connected
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        self._is_connected = True
+        if calibrate:
+            self.calibrate()
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self._is_calibrated
+
+    @check_if_not_connected
+    def calibrate(self) -> None:
+        self._is_calibrated = True
+
+    def configure(self) -> None:
+        pass
+
+    @check_if_not_connected
+    def get_action(self) -> RobotAction:
+        if self.config.random_values:
+            return {f"{motor}.pos": random.uniform(-100, 100) for motor in self.motors}
+        else:
+            return {
+                f"{motor}.pos": val for motor, val in zip(self.motors, self.config.static_values, strict=True)
+            }
+
+    @check_if_not_connected
+    def send_feedback(self, feedback: dict[str, Any]) -> None: ...
+
+    @check_if_not_connected
+    def disconnect(self) -> None:
+        self._is_connected = False
diff --git a/lerobot/tests/motors/test_damiao.py b/lerobot/tests/motors/test_damiao.py
new file mode 100644
index 0000000000000000000000000000000000000000..7ce1af34fc1d9ce9a7676b025d35718a335072c7
--- /dev/null
+++ b/lerobot/tests/motors/test_damiao.py
@@ -0,0 +1,66 @@
+"""Minimal test script for Damiao motor with ID 3."""
+
+import pytest
+
+from lerobot.utils.import_utils import _can_available
+
+if not _can_available:
+    pytest.skip("python-can not available", allow_module_level=True)
+
+from lerobot.motors import Motor
+from lerobot.motors.damiao import DamiaoMotorsBus
+
+
+@pytest.mark.skip(reason="Requires physical Damiao motor and CAN interface")
+def test_damiao_motor():
+    motors = {
+        "joint_3": Motor(
+            id=0x03,
+            model="damiao",
+            norm_mode="degrees",
+            motor_type_str="dm4310",
+            recv_id=0x13,
+        ),
+    }
+
+    bus = DamiaoMotorsBus(port="can0", motors=motors)
+
+    try:
+        print("Connecting...")
+        bus.connect()
+        print("✓ Connected")
+
+        print("Enabling torque...")
+        bus.enable_torque()
+        print("✓ Torque enabled")
+
+        print("Reading all states...")
+        states = bus.sync_read_all_states()
+        print(f"✓ States: {states}")
+
+        print("Reading position...")
+        positions = bus.sync_read("Present_Position")
+        print(f"✓ Position: {positions}")
+
+        print("Testing MIT control batch...")
+        current_pos = states["joint_3"]["position"]
+        commands = {"joint_3": (10.0, 0.5, current_pos, 0.0, 0.0)}
+        bus._mit_control_batch(commands)
+        print("✓ MIT control batch sent")
+
+        print("Disabling torque...")
+        bus.disable_torque()
+        print("✓ Torque disabled")
+
+        print("Setting zero position...")
+        bus.set_zero_position()
+        print("✓ Zero position set")
+
+    finally:
+        print("Disconnecting...")
+        bus.disconnect(disable_torque=True)
+        print("✓ Disconnected")
+
+
+if __name__ == "__main__":
+    test_damiao_motor()
diff --git a/lerobot/tests/motors/test_dynamixel.py b/lerobot/tests/motors/test_dynamixel.py
new file mode 100644
index 0000000000000000000000000000000000000000..8b02d433088d1a9210eec5de18169f990def9226
--- /dev/null
+++ b/lerobot/tests/motors/test_dynamixel.py
@@ -0,0 +1,416 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import re
+import sys
+from collections.abc import Generator
+from unittest.mock import MagicMock, patch
+
+import pytest
+
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.dynamixel import MODEL_NUMBER_TABLE, DynamixelMotorsBus
+from lerobot.motors.dynamixel.tables import X_SERIES_CONTROL_TABLE
+from lerobot.motors.encoding_utils import encode_twos_complement
+
+try:
+    import dynamixel_sdk as dxl
+
+    from tests.mocks.mock_dynamixel import MockMotors, MockPortHandler
+except (ImportError, ModuleNotFoundError):
+    pytest.skip("dynamixel_sdk not available", allow_module_level=True)
+
+
+@pytest.fixture(autouse=True)
+def patch_port_handler():
+    if sys.platform == "darwin":
+        with patch.object(dxl, "PortHandler", MockPortHandler):
+            yield
+    else:
+        yield
+
+
+@pytest.fixture
+def mock_motors() -> Generator[MockMotors, None, None]:
+    motors = MockMotors()
+    motors.open()
+    yield motors
+    motors.close()
+
+
+@pytest.fixture
+def dummy_motors() -> dict[str, Motor]:
+    return {
+        "dummy_1": Motor(1, "xl430-w250", MotorNormMode.RANGE_M100_100),
+        "dummy_2": Motor(2, "xm540-w270", MotorNormMode.RANGE_M100_100),
+        "dummy_3": Motor(3, "xl330-m077", MotorNormMode.RANGE_M100_100),
+    }
+
+
+@pytest.fixture
+def dummy_calibration(dummy_motors) -> dict[str, MotorCalibration]:
+    drive_modes = [0, 1, 0]
+    homings = [-709, -2006, 1624]
+    mins = [43, 27, 145]
+    maxes = [1335, 3608, 3999]
+    calibration = {}
+    for motor, m in dummy_motors.items():
+        calibration[motor] = MotorCalibration(
+            id=m.id,
+            drive_mode=drive_modes[m.id - 1],
+            homing_offset=homings[m.id - 1],
+            range_min=mins[m.id - 1],
+            range_max=maxes[m.id - 1],
+        )
+    return calibration
+
+
+@pytest.mark.skipif(sys.platform != "darwin", reason=f"No patching needed on {sys.platform=}")
+def test_autouse_patch():
+    """Ensures that the autouse fixture correctly patches dxl.PortHandler with MockPortHandler."""
+    assert dxl.PortHandler is MockPortHandler
+
+
+@pytest.mark.parametrize(
+    "value, length, expected",
+    [
+        (0x12,       1, [0x12]),
+        (0x1234,     2, [0x34, 0x12]),
+        (0x12345678, 4, [0x78, 0x56, 0x34, 0x12]),
+    ],
+    ids=[
+        "1 byte",
+        "2 bytes",
+        "4 bytes",
+    ],
+)  # fmt: skip
+def test__split_into_byte_chunks(value, length, expected):
+    bus = DynamixelMotorsBus("", {})
+    assert bus._split_into_byte_chunks(value, length) == expected
+
+
+def test_abc_implementation(dummy_motors):
+    """Instantiation should raise an error if the class doesn't implement abstract methods/properties."""
+    DynamixelMotorsBus(port="/dev/dummy-port", motors=dummy_motors)
+
+
+@pytest.mark.parametrize("id_", [1, 2, 3])
+def test_ping(id_, mock_motors, dummy_motors):
+    expected_model_nb = MODEL_NUMBER_TABLE[dummy_motors[f"dummy_{id_}"].model]
+    stub = mock_motors.build_ping_stub(id_, expected_model_nb)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    ping_model_nb = bus.ping(id_)
+
+    assert ping_model_nb == expected_model_nb
+    assert mock_motors.stubs[stub].called
+
+
+def test_broadcast_ping(mock_motors, dummy_motors):
+    models = {m.id: m.model for m in dummy_motors.values()}
+    expected_model_nbs = {id_: MODEL_NUMBER_TABLE[model] for id_, model in models.items()}
+    stub = mock_motors.build_broadcast_ping_stub(expected_model_nbs)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    ping_model_nbs = bus.broadcast_ping()
+
+    assert ping_model_nbs == expected_model_nbs
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize(
+    "addr, length, id_, value",
+    [
+        (0, 1, 1, 2),
+        (10, 2, 2, 999),
+        (42, 4, 3, 1337),
+    ],
+)
+def test__read(addr, length, id_, value, mock_motors, dummy_motors):
+    stub = mock_motors.build_read_stub(addr, length, id_, value)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    read_value, _, _ = bus._read(addr, length, id_)
+
+    assert mock_motors.stubs[stub].called
+    assert read_value == value
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__read_error(raise_on_error, mock_motors, dummy_motors):
+    addr, length, id_, value, error = (10, 4, 1, 1337, dxl.ERRNUM_DATA_LIMIT)
+    stub = mock_motors.build_read_stub(addr, length, id_, value, error=error)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(
+            RuntimeError, match=re.escape("[RxPacketError] The data value exceeds the limit value!")
+        ):
+            bus._read(addr, length, id_, raise_on_error=raise_on_error)
+    else:
+        _, _, read_error = bus._read(addr, length, id_, raise_on_error=raise_on_error)
+        assert read_error == error
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__read_comm(raise_on_error, mock_motors, dummy_motors):
+    addr, length, id_, value = (10, 4, 1, 1337)
+    stub = mock_motors.build_read_stub(addr, length, id_, value, reply=False)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(ConnectionError, match=re.escape("[TxRxResult] There is no status packet!")):
+            bus._read(addr, length, id_, raise_on_error=raise_on_error)
+    else:
+        _, read_comm, _ = bus._read(addr, length, id_, raise_on_error=raise_on_error)
+        assert read_comm == dxl.COMM_RX_TIMEOUT
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize(
+    "addr, length, id_, value",
+    [
+        (0, 1, 1, 2),
+        (10, 2, 2, 999),
+        (42, 4, 3, 1337),
+    ],
+)
+def test__write(addr, length, id_, value, mock_motors, dummy_motors):
+    stub = mock_motors.build_write_stub(addr, length, id_, value)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    comm, error = bus._write(addr, length, id_, value)
+
+    assert mock_motors.stubs[stub].called
+    assert comm == dxl.COMM_SUCCESS
+    assert error == 0
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__write_error(raise_on_error, mock_motors, dummy_motors):
+    addr, length, id_, value, error = (10, 4, 1, 1337, dxl.ERRNUM_DATA_LIMIT)
+    stub = mock_motors.build_write_stub(addr, length, id_, value, error=error)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(
+            RuntimeError, match=re.escape("[RxPacketError] The data value exceeds the limit value!")
+        ):
+            bus._write(addr, length, id_, value, raise_on_error=raise_on_error)
+    else:
+        _, write_error = bus._write(addr, length, id_, value, raise_on_error=raise_on_error)
+        assert write_error == error
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__write_comm(raise_on_error, mock_motors, dummy_motors):
+    addr, length, id_, value = (10, 4, 1, 1337)
+    stub = mock_motors.build_write_stub(addr, length, id_, value, reply=False)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(ConnectionError, match=re.escape("[TxRxResult] There is no status packet!")):
+            bus._write(addr, length, id_, value, raise_on_error=raise_on_error)
+    else:
+        write_comm, _ = bus._write(addr, length, id_, value, raise_on_error=raise_on_error)
+        assert write_comm == dxl.COMM_RX_TIMEOUT
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize(
+    "addr, length, ids_values",
+    [
+        (0, 1, {1: 4}),
+        (10, 2, {1: 1337, 2: 42}),
+        (42, 4, {1: 1337, 2: 42, 3: 4016}),
+    ],
+    ids=["1 motor", "2 motors", "3 motors"],
+)
+def test__sync_read(addr, length, ids_values, mock_motors, dummy_motors):
+    stub = mock_motors.build_sync_read_stub(addr, length, ids_values)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    read_values, _ = bus._sync_read(addr, length, list(ids_values))
+
+    assert mock_motors.stubs[stub].called
+    assert read_values == ids_values
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__sync_read_comm(raise_on_error, mock_motors, dummy_motors):
+    addr, length, ids_values = (10, 4, {1: 1337})
+    stub = mock_motors.build_sync_read_stub(addr, length, ids_values, reply=False)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(ConnectionError, match=re.escape("[TxRxResult] There is no status packet!")):
+            bus._sync_read(addr, length, list(ids_values), raise_on_error=raise_on_error)
+    else:
+        _, read_comm = bus._sync_read(addr, length, list(ids_values), raise_on_error=raise_on_error)
+        assert read_comm == dxl.COMM_RX_TIMEOUT
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize(
+    "addr, length, ids_values",
+    [
+        (0, 1, {1: 4}),
+        (10, 2, {1: 1337, 2: 42}),
+        (42, 4, {1: 1337, 2: 42, 3: 4016}),
+    ],
+    ids=["1 motor", "2 motors", "3 motors"],
+)
+def test__sync_write(addr, length, ids_values, mock_motors, dummy_motors):
+    stub = mock_motors.build_sync_write_stub(addr, length, ids_values)
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    comm = bus._sync_write(addr, length, ids_values)
+
+    assert mock_motors.stubs[stub].wait_called()
+    assert comm == dxl.COMM_SUCCESS
+
+
+def test_is_calibrated(mock_motors, dummy_motors, dummy_calibration):
+    drive_modes = {m.id: m.drive_mode for m in dummy_calibration.values()}
+    encoded_homings = {m.id: encode_twos_complement(m.homing_offset, 4) for m in dummy_calibration.values()}
+    mins = {m.id: m.range_min for m in dummy_calibration.values()}
+    maxes = {m.id: m.range_max for m in dummy_calibration.values()}
+    drive_modes_stub = mock_motors.build_sync_read_stub(*X_SERIES_CONTROL_TABLE["Drive_Mode"], drive_modes)
+    offsets_stub = mock_motors.build_sync_read_stub(*X_SERIES_CONTROL_TABLE["Homing_Offset"], encoded_homings)
+    mins_stub = mock_motors.build_sync_read_stub(*X_SERIES_CONTROL_TABLE["Min_Position_Limit"], mins)
+    maxes_stub = mock_motors.build_sync_read_stub(*X_SERIES_CONTROL_TABLE["Max_Position_Limit"], maxes)
+    bus = DynamixelMotorsBus(
+        port=mock_motors.port,
+        motors=dummy_motors,
+        calibration=dummy_calibration,
+    )
+    bus.connect(handshake=False)
+
+    is_calibrated = bus.is_calibrated
+
+    assert is_calibrated
+    assert mock_motors.stubs[drive_modes_stub].called
+    assert mock_motors.stubs[offsets_stub].called
+    assert mock_motors.stubs[mins_stub].called
+    assert mock_motors.stubs[maxes_stub].called
+
+
+def test_reset_calibration(mock_motors, dummy_motors):
+    write_homing_stubs = []
+    write_mins_stubs = []
+    write_maxes_stubs = []
+    for motor in dummy_motors.values():
+        write_homing_stubs.append(
+            mock_motors.build_write_stub(*X_SERIES_CONTROL_TABLE["Homing_Offset"], motor.id, 0)
+        )
+        write_mins_stubs.append(
+            mock_motors.build_write_stub(*X_SERIES_CONTROL_TABLE["Min_Position_Limit"], motor.id, 0)
+        )
+        write_maxes_stubs.append(
+            mock_motors.build_write_stub(*X_SERIES_CONTROL_TABLE["Max_Position_Limit"], motor.id, 4095)
+        )
+
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    bus.reset_calibration()
+
+    assert all(mock_motors.stubs[stub].called for stub in write_homing_stubs)
+    assert all(mock_motors.stubs[stub].called for stub in write_mins_stubs)
+    assert all(mock_motors.stubs[stub].called for stub in write_maxes_stubs)
+
+
+def test_set_half_turn_homings(mock_motors, dummy_motors):
+    """
+    For this test, we assume that the homing offsets are already 0 such that
+    Present_Position == Actual_Position
+    """
+    current_positions = {
+        1: 1337,
+        2: 42,
+        3: 3672,
+    }
+    expected_homings = {
+        1: 710,  # 2047 - 1337
+        2: 2005,  # 2047 - 42
+        3: -1625,  # 2047 - 3672
+    }
+    read_pos_stub = mock_motors.build_sync_read_stub(
+        *X_SERIES_CONTROL_TABLE["Present_Position"], current_positions
+    )
+    write_homing_stubs = []
+    for id_, homing in expected_homings.items():
+        encoded_homing = encode_twos_complement(homing, 4)
+        stub = mock_motors.build_write_stub(*X_SERIES_CONTROL_TABLE["Homing_Offset"], id_, encoded_homing)
+        write_homing_stubs.append(stub)
+
+    bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+    bus.reset_calibration = MagicMock()
+
+    bus.set_half_turn_homings()
+
+    bus.reset_calibration.assert_called_once()
+    assert mock_motors.stubs[read_pos_stub].called
+    assert all(mock_motors.stubs[stub].called for stub in write_homing_stubs)
+
+
+def test_record_ranges_of_motion(mock_motors, dummy_motors):
+    positions = {
+        1: [351, 42, 1337],
+        2: [28, 3600, 2444],
+        3: [4002, 2999, 146],
+    }
+    expected_mins = {
+        "dummy_1": 42,
+        "dummy_2": 28,
+        "dummy_3": 146,
+    }
+    expected_maxes = {
+        "dummy_1": 1337,
+        "dummy_2": 3600,
+        "dummy_3": 4002,
+    }
+    read_pos_stub = mock_motors.build_sequential_sync_read_stub(
+        *X_SERIES_CONTROL_TABLE["Present_Position"], positions
+    )
+    with patch("lerobot.motors.motors_bus.enter_pressed", side_effect=[False, True]):
+        bus = DynamixelMotorsBus(port=mock_motors.port, motors=dummy_motors)
+        bus.connect(handshake=False)
+
+        mins, maxes = bus.record_ranges_of_motion(display_values=False)
+
+    assert mock_motors.stubs[read_pos_stub].calls == 3
+    assert mins == expected_mins
+    assert maxes == expected_maxes
diff --git a/lerobot/tests/motors/test_feetech.py b/lerobot/tests/motors/test_feetech.py
new file mode 100644
index 0000000000000000000000000000000000000000..673276e0519c0a33c17edc57741ca12ff3acb41f
--- /dev/null
+++ b/lerobot/tests/motors/test_feetech.py
@@ -0,0 +1,459 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import re
+import sys
+from collections.abc import Generator
+from unittest.mock import MagicMock, patch
+
+import pytest
+
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.encoding_utils import encode_sign_magnitude
+from lerobot.motors.feetech import MODEL_NUMBER, MODEL_NUMBER_TABLE, FeetechMotorsBus
+from lerobot.motors.feetech.tables import STS_SMS_SERIES_CONTROL_TABLE
+
+try:
+    import scservo_sdk as scs
+
+    from tests.mocks.mock_feetech import MockMotors, MockPortHandler
+except (ImportError, ModuleNotFoundError):
+    pytest.skip("scservo_sdk not available", allow_module_level=True)
+
+
+@pytest.fixture(autouse=True)
+def patch_port_handler():
+    if sys.platform == "darwin":
+        with patch.object(scs, "PortHandler", MockPortHandler):
+            yield
+    else:
+        yield
+
+
+@pytest.fixture
+def mock_motors() -> Generator[MockMotors, None, None]:
+    motors = MockMotors()
+    motors.open()
+    yield motors
+    motors.close()
+
+
+@pytest.fixture
+def dummy_motors() -> dict[str, Motor]:
+    return {
+        "dummy_1": Motor(1, "sts3215", MotorNormMode.RANGE_M100_100),
+        "dummy_2": Motor(2, "sts3215", MotorNormMode.RANGE_M100_100),
+        "dummy_3": Motor(3, "sts3215", MotorNormMode.RANGE_M100_100),
+    }
+
+
+@pytest.fixture
+def dummy_calibration(dummy_motors) -> dict[str, MotorCalibration]:
+    homings = [-709, -2006, 1624]
+    mins = [43, 27, 145]
+    maxes = [1335, 3608, 3999]
+    calibration = {}
+    for motor, m in dummy_motors.items():
+        calibration[motor] = MotorCalibration(
+            id=m.id,
+            drive_mode=0,
+            homing_offset=homings[m.id - 1],
+            range_min=mins[m.id - 1],
+            range_max=maxes[m.id - 1],
+        )
+    return calibration
+
+
+@pytest.mark.skipif(sys.platform != "darwin", reason=f"No patching needed on {sys.platform=}")
+def test_autouse_patch():
+    """Ensures that the autouse fixture correctly patches scs.PortHandler with MockPortHandler."""
+    assert scs.PortHandler is MockPortHandler
+
+
+@pytest.mark.parametrize(
+    "protocol, value, length, expected",
+    [
+        (0, 0x12,       1, [0x12]),
+        (1, 0x12,       1, [0x12]),
+        (0, 0x1234,     2, [0x34, 0x12]),
+        (1, 0x1234,     2, [0x12, 0x34]),
+        (0, 0x12345678, 4, [0x78, 0x56, 0x34, 0x12]),
+        (1, 0x12345678, 4, [0x56, 0x78, 0x12, 0x34]),
+    ],
+    ids=[
+        "P0: 1 byte",
+        "P1: 1 byte",
+        "P0: 2 bytes",
+        "P1: 2 bytes",
+        "P0: 4 bytes",
+        "P1: 4 bytes",
+    ],
+)  # fmt: skip
+def test__split_into_byte_chunks(protocol, value, length, expected):
+    bus = FeetechMotorsBus("", {}, protocol_version=protocol)
+    assert bus._split_into_byte_chunks(value, length) == expected
+
+
+def test_abc_implementation(dummy_motors):
+    """Instantiation should raise an error if the class doesn't implement abstract methods/properties."""
+    FeetechMotorsBus(port="/dev/dummy-port", motors=dummy_motors)
+
+
+@pytest.mark.parametrize("id_", [1, 2, 3])
+def test_ping(id_, mock_motors, dummy_motors):
+    expected_model_nb = MODEL_NUMBER_TABLE[dummy_motors[f"dummy_{id_}"].model]
+    addr, length = MODEL_NUMBER
+    ping_stub = mock_motors.build_ping_stub(id_)
+    mobel_nb_stub = mock_motors.build_read_stub(addr, length, id_, expected_model_nb)
+    bus = FeetechMotorsBus(
+        port=mock_motors.port,
+        motors=dummy_motors,
+    )
+    bus.connect(handshake=False)
+
+    ping_model_nb = bus.ping(id_)
+
+    assert ping_model_nb == expected_model_nb
+    assert mock_motors.stubs[ping_stub].called
+    assert mock_motors.stubs[mobel_nb_stub].called
+
+
+def test_broadcast_ping(mock_motors, dummy_motors):
+    models = {m.id: m.model for m in dummy_motors.values()}
+    addr, length = MODEL_NUMBER
+    ping_stub = mock_motors.build_broadcast_ping_stub(list(models))
+    mobel_nb_stubs = []
+    expected_model_nbs = {}
+    for id_, model in models.items():
+        model_nb = MODEL_NUMBER_TABLE[model]
+        stub = mock_motors.build_read_stub(addr, length, id_, model_nb)
+        expected_model_nbs[id_] = model_nb
+        mobel_nb_stubs.append(stub)
+    bus = FeetechMotorsBus(
+        port=mock_motors.port,
+        motors=dummy_motors,
+    )
+    bus.connect(handshake=False)
+
+    ping_model_nbs = bus.broadcast_ping()
+
+    assert ping_model_nbs == expected_model_nbs
+    assert mock_motors.stubs[ping_stub].called
+    assert all(mock_motors.stubs[stub].called for stub in mobel_nb_stubs)
+
+
+@pytest.mark.parametrize(
+    "addr, length, id_, value",
+    [
+        (0, 1, 1, 2),
+        (10, 2, 2, 999),
+        (42, 4, 3, 1337),
+    ],
+)
+def test__read(addr, length, id_, value, mock_motors, dummy_motors):
+    stub = mock_motors.build_read_stub(addr, length, id_, value)
+    bus = FeetechMotorsBus(
+        port=mock_motors.port,
+        motors=dummy_motors,
+    )
+    bus.connect(handshake=False)
+
+    read_value, _, _ = bus._read(addr, length, id_)
+
+    assert mock_motors.stubs[stub].called
+    assert read_value == value
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__read_error(raise_on_error, mock_motors, dummy_motors):
+    addr, length, id_, value, error = (10, 4, 1, 1337, scs.ERRBIT_VOLTAGE)
+    stub = mock_motors.build_read_stub(addr, length, id_, value, error=error)
+    bus = FeetechMotorsBus(
+        port=mock_motors.port,
+        motors=dummy_motors,
+    )
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(RuntimeError, match=re.escape("[RxPacketError] Input voltage error!")):
+            bus._read(addr, length, id_, raise_on_error=raise_on_error)
+    else:
+        _, _, read_error = bus._read(addr, length, id_, raise_on_error=raise_on_error)
+        assert read_error == error
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__read_comm(raise_on_error, mock_motors, dummy_motors):
+    addr, length, id_, value = (10, 4, 1, 1337)
+    stub = mock_motors.build_read_stub(addr, length, id_, value, reply=False)
+    bus = FeetechMotorsBus(
+        port=mock_motors.port,
+        motors=dummy_motors,
+    )
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(ConnectionError, match=re.escape("[TxRxResult] There is no status packet!")):
+            bus._read(addr, length, id_, raise_on_error=raise_on_error)
+    else:
+        _, read_comm, _ = bus._read(addr, length, id_, raise_on_error=raise_on_error)
+        assert read_comm == scs.COMM_RX_TIMEOUT
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize(
+    "addr, length, id_, value",
+    [
+        (0, 1, 1, 2),
+        (10, 2, 2, 999),
+        (42, 4, 3, 1337),
+    ],
+)
+def test__write(addr, length, id_, value, mock_motors, dummy_motors):
+    stub = mock_motors.build_write_stub(addr, length, id_, value)
+    bus = FeetechMotorsBus(
+        port=mock_motors.port,
+        motors=dummy_motors,
+    )
+    bus.connect(handshake=False)
+
+    comm, error = bus._write(addr, length, id_, value)
+
+    assert mock_motors.stubs[stub].wait_called()
+    assert comm == scs.COMM_SUCCESS
+    assert error == 0
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__write_error(raise_on_error, mock_motors, dummy_motors):
+    addr, length, id_, value, error = (10, 4, 1, 1337, scs.ERRBIT_VOLTAGE)
+    stub = mock_motors.build_write_stub(addr, length, id_, value, error=error)
+    bus = FeetechMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(RuntimeError, match=re.escape("[RxPacketError] Input voltage error!")):
+            bus._write(addr, length, id_, value, raise_on_error=raise_on_error)
+    else:
+        _, write_error = bus._write(addr, length, id_, value, raise_on_error=raise_on_error)
+        assert write_error == error
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__write_comm(raise_on_error, mock_motors, dummy_motors):
+    addr, length, id_, value = (10, 4, 1, 1337)
+    stub = mock_motors.build_write_stub(addr, length, id_, value, reply=False)
+    bus = FeetechMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(ConnectionError, match=re.escape("[TxRxResult] There is no status packet!")):
+            bus._write(addr, length, id_, value, raise_on_error=raise_on_error)
+    else:
+        write_comm, _ = bus._write(addr, length, id_, value, raise_on_error=raise_on_error)
+        assert write_comm == scs.COMM_RX_TIMEOUT
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize(
+    "addr, length, ids_values",
+    [
+        (0, 1, {1: 4}),
+        (10, 2, {1: 1337, 2: 42}),
+        (42, 4, {1: 1337, 2: 42, 3: 4016}),
+    ],
+    ids=["1 motor", "2 motors", "3 motors"],
+)
+def test__sync_read(addr, length, ids_values, mock_motors, dummy_motors):
+    stub = mock_motors.build_sync_read_stub(addr, length, ids_values)
+    bus = FeetechMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    read_values, _ = bus._sync_read(addr, length, list(ids_values))
+
+    assert mock_motors.stubs[stub].called
+    assert read_values == ids_values
+
+
+@pytest.mark.parametrize("raise_on_error", (True, False))
+def test__sync_read_comm(raise_on_error, mock_motors, dummy_motors):
+    addr, length, ids_values = (10, 4, {1: 1337})
+    stub = mock_motors.build_sync_read_stub(addr, length, ids_values, reply=False)
+    bus = FeetechMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    if raise_on_error:
+        with pytest.raises(ConnectionError, match=re.escape("[TxRxResult] There is no status packet!")):
+            bus._sync_read(addr, length, list(ids_values), raise_on_error=raise_on_error)
+    else:
+        _, read_comm = bus._sync_read(addr, length, list(ids_values), raise_on_error=raise_on_error)
+        assert read_comm == scs.COMM_RX_TIMEOUT
+
+    assert mock_motors.stubs[stub].called
+
+
+@pytest.mark.parametrize(
+    "addr, length, ids_values",
+    [
+        (0, 1, {1: 4}),
+        (10, 2, {1: 1337, 2: 42}),
+        (42, 4, {1: 1337, 2: 42, 3: 4016}),
+    ],
+    ids=["1 motor", "2 motors", "3 motors"],
+)
+def test__sync_write(addr, length, ids_values, mock_motors, dummy_motors):
+    stub = mock_motors.build_sync_write_stub(addr, length, ids_values)
+    bus = FeetechMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    comm = bus._sync_write(addr, length, ids_values)
+
+    assert mock_motors.stubs[stub].wait_called()
+    assert comm == scs.COMM_SUCCESS
+
+
+def test_is_calibrated(mock_motors, dummy_motors, dummy_calibration):
+    mins_stubs, maxes_stubs, homings_stubs = [], [], []
+    for cal in dummy_calibration.values():
+        mins_stubs.append(
+            mock_motors.build_read_stub(
+                *STS_SMS_SERIES_CONTROL_TABLE["Min_Position_Limit"], cal.id, cal.range_min
+            )
+        )
+        maxes_stubs.append(
+            mock_motors.build_read_stub(
+                *STS_SMS_SERIES_CONTROL_TABLE["Max_Position_Limit"], cal.id, cal.range_max
+            )
+        )
+        homings_stubs.append(
+            mock_motors.build_read_stub(
+                *STS_SMS_SERIES_CONTROL_TABLE["Homing_Offset"],
+                cal.id,
+                encode_sign_magnitude(cal.homing_offset, 11),
+            )
+        )
+
+    bus = FeetechMotorsBus(
+        port=mock_motors.port,
+        motors=dummy_motors,
+        calibration=dummy_calibration,
+    )
+    bus.connect(handshake=False)
+
+    is_calibrated = bus.is_calibrated
+
+    assert is_calibrated
+    assert all(mock_motors.stubs[stub].called for stub in mins_stubs)
+    assert all(mock_motors.stubs[stub].called for stub in maxes_stubs)
+    assert all(mock_motors.stubs[stub].called for stub in homings_stubs)
+
+
+def test_reset_calibration(mock_motors, dummy_motors):
+    write_homing_stubs = []
+    write_mins_stubs = []
+    write_maxes_stubs = []
+    for motor in dummy_motors.values():
+        write_homing_stubs.append(
+            mock_motors.build_write_stub(*STS_SMS_SERIES_CONTROL_TABLE["Homing_Offset"], motor.id, 0)
+        )
+        write_mins_stubs.append(
+            mock_motors.build_write_stub(*STS_SMS_SERIES_CONTROL_TABLE["Min_Position_Limit"], motor.id, 0)
+        )
+        write_maxes_stubs.append(
+            mock_motors.build_write_stub(*STS_SMS_SERIES_CONTROL_TABLE["Max_Position_Limit"], motor.id, 4095)
+        )
+
+    bus = FeetechMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+
+    bus.reset_calibration()
+
+    assert all(mock_motors.stubs[stub].wait_called() for stub in write_homing_stubs)
+    assert all(mock_motors.stubs[stub].wait_called() for stub in write_mins_stubs)
+    assert all(mock_motors.stubs[stub].wait_called() for stub in write_maxes_stubs)
+
+
+def test_set_half_turn_homings(mock_motors, dummy_motors):
+    """
+    For this test, we assume that the homing offsets are already 0 such that
+    Present_Position == Actual_Position
+    """
+    current_positions = {
+        1: 1337,
+        2: 42,
+        3: 3672,
+    }
+    expected_homings = {
+        1: -710,  # 1337 - 2047
+        2: -2005,  # 42 - 2047
+        3: 1625,  # 3672 - 2047
+    }
+    read_pos_stub = mock_motors.build_sync_read_stub(
+        *STS_SMS_SERIES_CONTROL_TABLE["Present_Position"], current_positions
+    )
+    write_homing_stubs = []
+    for id_, homing in expected_homings.items():
+        encoded_homing = encode_sign_magnitude(homing, 11)
+        stub = mock_motors.build_write_stub(
+            *STS_SMS_SERIES_CONTROL_TABLE["Homing_Offset"], id_, encoded_homing
+        )
+        write_homing_stubs.append(stub)
+
+    bus = FeetechMotorsBus(port=mock_motors.port, motors=dummy_motors)
+    bus.connect(handshake=False)
+    bus.reset_calibration = MagicMock()
+
+    bus.set_half_turn_homings()
+
+    bus.reset_calibration.assert_called_once()
+    assert mock_motors.stubs[read_pos_stub].called
+    assert all(mock_motors.stubs[stub].wait_called() for stub in write_homing_stubs)
+
+
+def test_record_ranges_of_motion(mock_motors, dummy_motors):
+    positions = {
+        1: [351, 42, 1337],
+        2: [28, 3600, 2444],
+        3: [4002, 2999, 146],
+    }
+    expected_mins = {
+        "dummy_1": 42,
+        "dummy_2": 28,
+        "dummy_3": 146,
+    }
+    expected_maxes = {
+        "dummy_1": 1337,
+        "dummy_2": 3600,
+        "dummy_3": 4002,
+    }
+    stub = mock_motors.build_sequential_sync_read_stub(
+        *STS_SMS_SERIES_CONTROL_TABLE["Present_Position"], positions
+    )
+    with patch("lerobot.motors.motors_bus.enter_pressed", side_effect=[False, True]):
+        bus = FeetechMotorsBus(port=mock_motors.port, motors=dummy_motors)
+        bus.connect(handshake=False)
+
+        mins, maxes = bus.record_ranges_of_motion(display_values=False)
+
+    assert mock_motors.stubs[stub].calls == 3
+    assert mins == expected_mins
+    assert maxes == expected_maxes
diff --git a/lerobot/tests/motors/test_motors_bus.py b/lerobot/tests/motors/test_motors_bus.py
new file mode 100644
index 0000000000000000000000000000000000000000..27650ef1ba64ffadfbbea096e7b05b9b4f9e4566
--- /dev/null
+++ b/lerobot/tests/motors/test_motors_bus.py
@@ -0,0 +1,358 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import re
+from unittest.mock import patch
+
+import pytest
+
+from lerobot.motors.motors_bus import (
+    Motor,
+    MotorNormMode,
+    assert_same_address,
+    get_address,
+    get_ctrl_table,
+)
+from tests.mocks.mock_motors_bus import (
+    DUMMY_CTRL_TABLE_1,
+    DUMMY_CTRL_TABLE_2,
+    DUMMY_MODEL_CTRL_TABLE,
+    MockMotorsBus,
+)
+
+
+@pytest.fixture
+def dummy_motors() -> dict[str, Motor]:
+    return {
+        "dummy_1": Motor(1, "model_2", MotorNormMode.RANGE_M100_100),
+        "dummy_2": Motor(2, "model_3", MotorNormMode.RANGE_M100_100),
+        "dummy_3": Motor(3, "model_2", MotorNormMode.RANGE_0_100),
+    }
+
+
+def test_get_ctrl_table():
+    model = "model_1"
+    ctrl_table = get_ctrl_table(DUMMY_MODEL_CTRL_TABLE, model)
+    assert ctrl_table == DUMMY_CTRL_TABLE_1
+
+
+def test_get_ctrl_table_error():
+    model = "model_99"
+    with pytest.raises(KeyError, match=f"Control table for {model=} not found."):
+        get_ctrl_table(DUMMY_MODEL_CTRL_TABLE, model)
+
+
+def test_get_address():
+    addr, n_bytes = get_address(DUMMY_MODEL_CTRL_TABLE, "model_1", "Firmware_Version")
+    assert addr == 0
+    assert n_bytes == 1
+
+
+def test_get_address_error():
+    model = "model_1"
+    data_name = "Lock"
+    with pytest.raises(KeyError, match=f"Address for '{data_name}' not found in {model} control table."):
+        get_address(DUMMY_MODEL_CTRL_TABLE, "model_1", data_name)
+
+
+def test_assert_same_address():
+    models = ["model_1", "model_2"]
+    assert_same_address(DUMMY_MODEL_CTRL_TABLE, models, "Present_Position")
+
+
+def test_assert_same_length_different_addresses():
+    models = ["model_1", "model_2"]
+    with pytest.raises(
+        NotImplementedError,
+        match=re.escape("At least two motor models use a different address"),
+    ):
+        assert_same_address(DUMMY_MODEL_CTRL_TABLE, models, "Model_Number")
+
+
+def test_assert_same_address_different_length():
+    models = ["model_1", "model_2"]
+    with pytest.raises(
+        NotImplementedError,
+        match=re.escape("At least two motor models use a different bytes representation"),
+    ):
+        assert_same_address(DUMMY_MODEL_CTRL_TABLE, models, "Goal_Position")
+
+
+def test__serialize_data_invalid_length():
+    bus = MockMotorsBus("", {})
+    with pytest.raises(NotImplementedError):
+        bus._serialize_data(100, 3)
+
+
+def test__serialize_data_negative_numbers():
+    bus = MockMotorsBus("", {})
+    with pytest.raises(ValueError):
+        bus._serialize_data(-1, 1)
+
+
+def test__serialize_data_large_number():
+    bus = MockMotorsBus("", {})
+    with pytest.raises(ValueError):
+        bus._serialize_data(2**32, 4)  # 4-byte max is 0xFFFFFFFF
+
+
+@pytest.mark.parametrize(
+    "data_name, id_, value",
+    [
+        ("Firmware_Version", 1, 14),
+        ("Model_Number", 1, 5678),
+        ("Present_Position", 2, 1337),
+        ("Present_Velocity", 3, 42),
+    ],
+)
+def test_read(data_name, id_, value, dummy_motors):
+    bus = MockMotorsBus("/dev/dummy-port", dummy_motors)
+    bus.connect(handshake=False)
+    addr, length = DUMMY_CTRL_TABLE_2[data_name]
+
+    with (
+        patch.object(MockMotorsBus, "_read", return_value=(value, 0, 0)) as mock__read,
+        patch.object(MockMotorsBus, "_decode_sign", return_value={id_: value}) as mock__decode_sign,
+        patch.object(MockMotorsBus, "_normalize", return_value={id_: value}) as mock__normalize,
+    ):
+        returned_value = bus.read(data_name, f"dummy_{id_}")
+
+    assert returned_value == value
+    mock__read.assert_called_once_with(
+        addr,
+        length,
+        id_,
+        num_retry=0,
+        raise_on_error=True,
+        err_msg=f"Failed to read '{data_name}' on {id_=} after 1 tries.",
+    )
+    mock__decode_sign.assert_called_once_with(data_name, {id_: value})
+    if data_name in bus.normalized_data:
+        mock__normalize.assert_called_once_with({id_: value})
+
+
+@pytest.mark.parametrize(
+    "data_name, id_, value",
+    [
+        ("Goal_Position", 1, 1337),
+        ("Goal_Velocity", 2, 3682),
+        ("Lock", 3, 1),
+    ],
+)
+def test_write(data_name, id_, value, dummy_motors):
+    bus = MockMotorsBus("/dev/dummy-port", dummy_motors)
+    bus.connect(handshake=False)
+    addr, length = DUMMY_CTRL_TABLE_2[data_name]
+
+    with (
+        patch.object(MockMotorsBus, "_write", return_value=(0, 0)) as mock__write,
+        patch.object(MockMotorsBus, "_encode_sign", return_value={id_: value}) as mock__encode_sign,
+        patch.object(MockMotorsBus, "_unnormalize", return_value={id_: value}) as mock__unnormalize,
+    ):
+        bus.write(data_name, f"dummy_{id_}", value)
+
+    mock__write.assert_called_once_with(
+        addr,
+        length,
+        id_,
+        value,
+        num_retry=0,
+        raise_on_error=True,
+        err_msg=f"Failed to write '{data_name}' on {id_=} with '{value}' after 1 tries.",
+    )
+    mock__encode_sign.assert_called_once_with(data_name, {id_: value})
+    if data_name in bus.normalized_data:
+        mock__unnormalize.assert_called_once_with({id_: value})
+
+
+@pytest.mark.parametrize(
+    "data_name, id_, value",
+    [
+        ("Firmware_Version", 1, 14),
+        ("Model_Number", 1, 5678),
+        ("Present_Position", 2, 1337),
+        ("Present_Velocity", 3, 42),
+    ],
+)
+def test_sync_read_by_str(data_name, id_, value, dummy_motors):
+    bus = MockMotorsBus("/dev/dummy-port", dummy_motors)
+    bus.connect(handshake=False)
+    addr, length = DUMMY_CTRL_TABLE_2[data_name]
+    ids = [id_]
+    expected_value = {f"dummy_{id_}": value}
+
+    with (
+        patch.object(MockMotorsBus, "_sync_read", return_value=({id_: value}, 0)) as mock__sync_read,
+        patch.object(MockMotorsBus, "_decode_sign", return_value={id_: value}) as mock__decode_sign,
+        patch.object(MockMotorsBus, "_normalize", return_value={id_: value}) as mock__normalize,
+    ):
+        returned_dict = bus.sync_read(data_name, f"dummy_{id_}")
+
+    assert returned_dict == expected_value
+    mock__sync_read.assert_called_once_with(
+        addr,
+        length,
+        ids,
+        num_retry=0,
+        raise_on_error=True,
+        err_msg=f"Failed to sync read '{data_name}' on {ids=} after 1 tries.",
+    )
+    mock__decode_sign.assert_called_once_with(data_name, {id_: value})
+    if data_name in bus.normalized_data:
+        mock__normalize.assert_called_once_with({id_: value})
+
+
+@pytest.mark.parametrize(
+    "data_name, ids_values",
+    [
+        ("Model_Number", {1: 5678}),
+        ("Present_Position", {1: 1337, 2: 42}),
+        ("Present_Velocity", {1: 1337, 2: 42, 3: 4016}),
+    ],
+    ids=["1 motor", "2 motors", "3 motors"],
+)
+def test_sync_read_by_list(data_name, ids_values, dummy_motors):
+    bus = MockMotorsBus("/dev/dummy-port", dummy_motors)
+    bus.connect(handshake=False)
+    addr, length = DUMMY_CTRL_TABLE_2[data_name]
+    ids = list(ids_values)
+    expected_values = {f"dummy_{id_}": val for id_, val in ids_values.items()}
+
+    with (
+        patch.object(MockMotorsBus, "_sync_read", return_value=(ids_values, 0)) as mock__sync_read,
+        patch.object(MockMotorsBus, "_decode_sign", return_value=ids_values) as mock__decode_sign,
+        patch.object(MockMotorsBus, "_normalize", return_value=ids_values) as mock__normalize,
+    ):
+        returned_dict = bus.sync_read(data_name, [f"dummy_{id_}" for id_ in ids])
+
+    assert returned_dict == expected_values
+    mock__sync_read.assert_called_once_with(
+        addr,
+        length,
+        ids,
+        num_retry=0,
+        raise_on_error=True,
+        err_msg=f"Failed to sync read '{data_name}' on {ids=} after 1 tries.",
+    )
+    mock__decode_sign.assert_called_once_with(data_name, ids_values)
+    if data_name in bus.normalized_data:
+        mock__normalize.assert_called_once_with(ids_values)
+
+
+@pytest.mark.parametrize(
+    "data_name, ids_values",
+    [
+        ("Model_Number", {1: 5678, 2: 5799, 3: 5678}),
+        ("Present_Position", {1: 1337, 2: 42, 3: 4016}),
+        ("Goal_Position", {1: 4008, 2: 199, 3: 3446}),
+    ],
+    ids=["Model_Number", "Present_Position", "Goal_Position"],
+)
+def test_sync_read_by_none(data_name, ids_values, dummy_motors):
+    bus = MockMotorsBus("/dev/dummy-port", dummy_motors)
+    bus.connect(handshake=False)
+    addr, length = DUMMY_CTRL_TABLE_2[data_name]
+    ids = list(ids_values)
+    expected_values = {f"dummy_{id_}": val for id_, val in ids_values.items()}
+
+    with (
+        patch.object(MockMotorsBus, "_sync_read", return_value=(ids_values, 0)) as mock__sync_read,
+        patch.object(MockMotorsBus, "_decode_sign", return_value=ids_values) as mock__decode_sign,
+        patch.object(MockMotorsBus, "_normalize", return_value=ids_values) as mock__normalize,
+    ):
+        returned_dict = bus.sync_read(data_name)
+
+    assert returned_dict == expected_values
+    mock__sync_read.assert_called_once_with(
+        addr,
+        length,
+        ids,
+        num_retry=0,
+        raise_on_error=True,
+        err_msg=f"Failed to sync read '{data_name}' on {ids=} after 1 tries.",
+    )
+    mock__decode_sign.assert_called_once_with(data_name, ids_values)
+    if data_name in bus.normalized_data:
+        mock__normalize.assert_called_once_with(ids_values)
+
+
+@pytest.mark.parametrize(
+    "data_name, value",
+    [
+        ("Goal_Position", 500),
+        ("Goal_Velocity", 4010),
+        ("Lock", 0),
+    ],
+)
+def test_sync_write_by_single_value(data_name, value, dummy_motors):
+    bus = MockMotorsBus("/dev/dummy-port", dummy_motors)
+    bus.connect(handshake=False)
+    addr, length = DUMMY_CTRL_TABLE_2[data_name]
+    ids_values = {m.id: value for m in dummy_motors.values()}
+
+    with (
+        patch.object(MockMotorsBus, "_sync_write", return_value=(ids_values, 0)) as mock__sync_write,
+        patch.object(MockMotorsBus, "_encode_sign", return_value=ids_values) as mock__encode_sign,
+        patch.object(MockMotorsBus, "_unnormalize", return_value=ids_values) as mock__unnormalize,
+    ):
+        bus.sync_write(data_name, value)
+
+    mock__sync_write.assert_called_once_with(
+        addr,
+        length,
+        ids_values,
+        num_retry=0,
+        raise_on_error=True,
+        err_msg=f"Failed to sync write '{data_name}' with {ids_values=} after 1 tries.",
+    )
+    mock__encode_sign.assert_called_once_with(data_name, ids_values)
+    if data_name in bus.normalized_data:
+        mock__unnormalize.assert_called_once_with(ids_values)
+
+
+@pytest.mark.parametrize(
+    "data_name, ids_values",
+    [
+        ("Goal_Position", {1: 1337, 2: 42, 3: 4016}),
+        ("Goal_Velocity", {1: 50, 2: 83, 3: 2777}),
+        ("Lock", {1: 0, 2: 0, 3: 1}),
+    ],
+    ids=["Goal_Position", "Goal_Velocity", "Lock"],
+)
+def test_sync_write_by_value_dict(data_name, ids_values, dummy_motors):
+    bus = MockMotorsBus("/dev/dummy-port", dummy_motors)
+    bus.connect(handshake=False)
+    addr, length = DUMMY_CTRL_TABLE_2[data_name]
+    values = {f"dummy_{id_}": val for id_, val in ids_values.items()}
+
+    with (
+        patch.object(MockMotorsBus, "_sync_write", return_value=(ids_values, 0)) as mock__sync_write,
+        patch.object(MockMotorsBus, "_encode_sign", return_value=ids_values) as mock__encode_sign,
+        patch.object(MockMotorsBus, "_unnormalize", return_value=ids_values) as mock__unnormalize,
+    ):
+        bus.sync_write(data_name, values)
+
+    mock__sync_write.assert_called_once_with(
+        addr,
+        length,
+        ids_values,
+        num_retry=0,
+        raise_on_error=True,
+        err_msg=f"Failed to sync write '{data_name}' with {ids_values=} after 1 tries.",
+    )
+    mock__encode_sign.assert_called_once_with(data_name, ids_values)
+    if data_name in bus.normalized_data:
+        mock__unnormalize.assert_called_once_with(ids_values)
diff --git a/lerobot/tests/optim/test_optimizers.py b/lerobot/tests/optim/test_optimizers.py
new file mode 100644
index 0000000000000000000000000000000000000000..d1856556235b173683dd4c6eb6bcf8bd5848f9a6
--- /dev/null
+++ b/lerobot/tests/optim/test_optimizers.py
@@ -0,0 +1,242 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import pytest
+import torch
+
+from lerobot.optim.optimizers import (
+    AdamConfig,
+    AdamWConfig,
+    MultiAdamConfig,
+    SGDConfig,
+    load_optimizer_state,
+    save_optimizer_state,
+)
+from lerobot.utils.constants import (
+    OPTIMIZER_PARAM_GROUPS,
+    OPTIMIZER_STATE,
+)
+
+
+@pytest.mark.parametrize(
+    "config_cls, expected_class",
+    [
+        (AdamConfig, torch.optim.Adam),
+        (AdamWConfig, torch.optim.AdamW),
+        (SGDConfig, torch.optim.SGD),
+        (MultiAdamConfig, dict),
+    ],
+)
+def test_optimizer_build(config_cls, expected_class, model_params):
+    config = config_cls()
+    if config_cls == MultiAdamConfig:
+        params_dict = {"default": model_params}
+        optimizer = config.build(params_dict)
+        assert isinstance(optimizer, expected_class)
+        assert isinstance(optimizer["default"], torch.optim.Adam)
+        assert optimizer["default"].defaults["lr"] == config.lr
+    else:
+        optimizer = config.build(model_params)
+        assert isinstance(optimizer, expected_class)
+        assert optimizer.defaults["lr"] == config.lr
+
+
+def test_save_optimizer_state(optimizer, tmp_path):
+    save_optimizer_state(optimizer, tmp_path)
+    assert (tmp_path / OPTIMIZER_STATE).is_file()
+    assert (tmp_path / OPTIMIZER_PARAM_GROUPS).is_file()
+
+
+def test_save_and_load_optimizer_state(model_params, optimizer, tmp_path):
+    save_optimizer_state(optimizer, tmp_path)
+    loaded_optimizer = AdamConfig().build(model_params)
+    loaded_optimizer = load_optimizer_state(loaded_optimizer, tmp_path)
+
+    torch.testing.assert_close(optimizer.state_dict(), loaded_optimizer.state_dict())
+
+
+@pytest.fixture
+def base_params_dict():
+    return {
+        "actor": [torch.nn.Parameter(torch.randn(10, 10))],
+        "critic": [torch.nn.Parameter(torch.randn(5, 5))],
+        "temperature": [torch.nn.Parameter(torch.randn(3, 3))],
+    }
+
+
+@pytest.mark.parametrize(
+    "config_params, expected_values",
+    [
+        # Test 1: Basic configuration with different learning rates
+        (
+            {
+                "lr": 1e-3,
+                "weight_decay": 1e-4,
+                "optimizer_groups": {
+                    "actor": {"lr": 1e-4},
+                    "critic": {"lr": 5e-4},
+                    "temperature": {"lr": 2e-3},
+                },
+            },
+            {
+                "actor": {"lr": 1e-4, "weight_decay": 1e-4, "betas": (0.9, 0.999)},
+                "critic": {"lr": 5e-4, "weight_decay": 1e-4, "betas": (0.9, 0.999)},
+                "temperature": {"lr": 2e-3, "weight_decay": 1e-4, "betas": (0.9, 0.999)},
+            },
+        ),
+        # Test 2: Different weight decays and beta values
+        (
+            {
+                "lr": 1e-3,
+                "weight_decay": 1e-4,
+                "optimizer_groups": {
+                    "actor": {"lr": 1e-4, "weight_decay": 1e-5},
+                    "critic": {"lr": 5e-4, "weight_decay": 1e-6},
+                    "temperature": {"lr": 2e-3, "betas": (0.95, 0.999)},
+                },
+            },
+            {
+                "actor": {"lr": 1e-4, "weight_decay": 1e-5, "betas": (0.9, 0.999)},
+                "critic": {"lr": 5e-4, "weight_decay": 1e-6, "betas": (0.9, 0.999)},
+                "temperature": {"lr": 2e-3, "weight_decay": 1e-4, "betas": (0.95, 0.999)},
+            },
+        ),
+        # Test 3: Epsilon parameter customization
+        (
+            {
+                "lr": 1e-3,
+                "weight_decay": 1e-4,
+                "optimizer_groups": {
+                    "actor": {"lr": 1e-4, "eps": 1e-6},
+                    "critic": {"lr": 5e-4, "eps": 1e-7},
+                    "temperature": {"lr": 2e-3, "eps": 1e-8},
+                },
+            },
+            {
+                "actor": {"lr": 1e-4, "weight_decay": 1e-4, "betas": (0.9, 0.999), "eps": 1e-6},
+                "critic": {"lr": 5e-4, "weight_decay": 1e-4, "betas": (0.9, 0.999), "eps": 1e-7},
+                "temperature": {"lr": 2e-3, "weight_decay": 1e-4, "betas": (0.9, 0.999), "eps": 1e-8},
+            },
+        ),
+    ],
+)
+def test_multi_adam_configuration(base_params_dict, config_params, expected_values):
+    # Create config with the given parameters
+    config = MultiAdamConfig(**config_params)
+    optimizers = config.build(base_params_dict)
+
+    # Verify optimizer count and keys
+    assert len(optimizers) == len(expected_values)
+    assert set(optimizers.keys()) == set(expected_values.keys())
+
+    # Check that all optimizers are Adam instances
+    for opt in optimizers.values():
+        assert isinstance(opt, torch.optim.Adam)
+
+    # Verify hyperparameters for each optimizer
+    for name, expected in expected_values.items():
+        optimizer = optimizers[name]
+        for param, value in expected.items():
+            assert optimizer.defaults[param] == value
+
+
+@pytest.fixture
+def multi_optimizers(base_params_dict):
+    config = MultiAdamConfig(
+        lr=1e-3,
+        optimizer_groups={
+            "actor": {"lr": 1e-4},
+            "critic": {"lr": 5e-4},
+            "temperature": {"lr": 2e-3},
+        },
+    )
+    return config.build(base_params_dict)
+
+
+def test_save_multi_optimizer_state(multi_optimizers, tmp_path):
+    # Save optimizer states
+    save_optimizer_state(multi_optimizers, tmp_path)
+
+    # Verify that directories were created for each optimizer
+    for name in multi_optimizers:
+        assert (tmp_path / name).is_dir()
+        assert (tmp_path / name / OPTIMIZER_STATE).is_file()
+        assert (tmp_path / name / OPTIMIZER_PARAM_GROUPS).is_file()
+
+
+def test_save_and_load_multi_optimizer_state(base_params_dict, multi_optimizers, tmp_path):
+    # Option 1: Add a minimal backward pass to populate optimizer states
+    for name, params in base_params_dict.items():
+        if name in multi_optimizers:
+            # Create a dummy loss and do backward
+            dummy_loss = params[0].sum()
+            dummy_loss.backward()
+            # Perform an optimization step
+            multi_optimizers[name].step()
+            # Zero gradients for next steps
+            multi_optimizers[name].zero_grad()
+
+    # Save optimizer states
+    save_optimizer_state(multi_optimizers, tmp_path)
+
+    # Create new optimizers with the same config
+    config = MultiAdamConfig(
+        lr=1e-3,
+        optimizer_groups={
+            "actor": {"lr": 1e-4},
+            "critic": {"lr": 5e-4},
+            "temperature": {"lr": 2e-3},
+        },
+    )
+    new_optimizers = config.build(base_params_dict)
+
+    # Load optimizer states
+    loaded_optimizers = load_optimizer_state(new_optimizers, tmp_path)
+
+    # Verify state dictionaries match
+    for name in multi_optimizers:
+        torch.testing.assert_close(multi_optimizers[name].state_dict(), loaded_optimizers[name].state_dict())
+
+
+def test_save_and_load_empty_multi_optimizer_state(base_params_dict, tmp_path):
+    """Test saving and loading optimizer states even when the state is empty (no backward pass)."""
+    # Create config and build optimizers
+    config = MultiAdamConfig(
+        lr=1e-3,
+        optimizer_groups={
+            "actor": {"lr": 1e-4},
+            "critic": {"lr": 5e-4},
+            "temperature": {"lr": 2e-3},
+        },
+    )
+    optimizers = config.build(base_params_dict)
+
+    # Save optimizer states without any backward pass (empty state)
+    save_optimizer_state(optimizers, tmp_path)
+
+    # Create new optimizers with the same config
+    new_optimizers = config.build(base_params_dict)
+
+    # Load optimizer states
+    loaded_optimizers = load_optimizer_state(new_optimizers, tmp_path)
+
+    # Verify hyperparameters match even with empty state
+    for name, optimizer in optimizers.items():
+        assert optimizer.defaults["lr"] == loaded_optimizers[name].defaults["lr"]
+        assert optimizer.defaults["weight_decay"] == loaded_optimizers[name].defaults["weight_decay"]
+        assert optimizer.defaults["betas"] == loaded_optimizers[name].defaults["betas"]
+
+        # Verify state dictionaries match (they will be empty)
+        torch.testing.assert_close(
+            optimizer.state_dict()["param_groups"], loaded_optimizers[name].state_dict()["param_groups"]
+        )
diff --git a/lerobot/tests/optim/test_schedulers.py b/lerobot/tests/optim/test_schedulers.py
new file mode 100644
index 0000000000000000000000000000000000000000..224613416851e0576adaa827a6d0ec73c41afd81
--- /dev/null
+++ b/lerobot/tests/optim/test_schedulers.py
@@ -0,0 +1,105 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import torch
+from packaging.version import Version
+from torch.optim.lr_scheduler import LambdaLR
+
+from lerobot.optim.schedulers import (
+    CosineDecayWithWarmupSchedulerConfig,
+    DiffuserSchedulerConfig,
+    VQBeTSchedulerConfig,
+    load_scheduler_state,
+    save_scheduler_state,
+)
+from lerobot.utils.constants import SCHEDULER_STATE
+
+
+def test_diffuser_scheduler(optimizer):
+    config = DiffuserSchedulerConfig(name="cosine", num_warmup_steps=5)
+    scheduler = config.build(optimizer, num_training_steps=100)
+    assert isinstance(scheduler, LambdaLR)
+
+    optimizer.step()  # so that we don't get torch warning
+    scheduler.step()
+    expected_state_dict = {
+        "_get_lr_called_within_step": False,
+        "_last_lr": [0.0002],
+        "_step_count": 2,
+        "base_lrs": [0.001],
+        "last_epoch": 1,
+        "lr_lambdas": [None],
+    }
+
+    if Version(torch.__version__) >= Version("2.8"):
+        expected_state_dict["_is_initial"] = False
+
+    assert scheduler.state_dict() == expected_state_dict
+
+
+def test_vqbet_scheduler(optimizer):
+    config = VQBeTSchedulerConfig(num_warmup_steps=10, num_vqvae_training_steps=20, num_cycles=0.5)
+    scheduler = config.build(optimizer, num_training_steps=100)
+    assert isinstance(scheduler, LambdaLR)
+
+    optimizer.step()
+    scheduler.step()
+    expected_state_dict = {
+        "_get_lr_called_within_step": False,
+        "_last_lr": [0.001],
+        "_step_count": 2,
+        "base_lrs": [0.001],
+        "last_epoch": 1,
+        "lr_lambdas": [None],
+    }
+
+    if Version(torch.__version__) >= Version("2.8"):
+        expected_state_dict["_is_initial"] = False
+
+    assert scheduler.state_dict() == expected_state_dict
+
+
+def test_cosine_decay_with_warmup_scheduler(optimizer):
+    config = CosineDecayWithWarmupSchedulerConfig(
+        num_warmup_steps=10, num_decay_steps=90, peak_lr=0.01, decay_lr=0.001
+    )
+    scheduler = config.build(optimizer, num_training_steps=100)
+    assert isinstance(scheduler, LambdaLR)
+
+    optimizer.step()
+    scheduler.step()
+    expected_state_dict = {
+        "_get_lr_called_within_step": False,
+        "_last_lr": [0.0001818181818181819],
+        "_step_count": 2,
+        "base_lrs": [0.001],
+        "last_epoch": 1,
+        "lr_lambdas": [None],
+    }
+
+    if Version(torch.__version__) >= Version("2.8"):
+        expected_state_dict["_is_initial"] = False
+
+    assert scheduler.state_dict() == expected_state_dict
+
+
+def test_save_scheduler_state(scheduler, tmp_path):
+    save_scheduler_state(scheduler, tmp_path)
+    assert (tmp_path / SCHEDULER_STATE).is_file()
+
+
+def test_save_load_scheduler_state(scheduler, tmp_path):
+    save_scheduler_state(scheduler, tmp_path)
+    loaded_scheduler = load_scheduler_state(scheduler, tmp_path)
+
+    assert scheduler.state_dict() == loaded_scheduler.state_dict()
diff --git a/lerobot/tests/policies/groot/test_groot_lerobot.py b/lerobot/tests/policies/groot/test_groot_lerobot.py
new file mode 100644
index 0000000000000000000000000000000000000000..e299a34e2daff90822ae9e3beb7f95d074d26e2a
--- /dev/null
+++ b/lerobot/tests/policies/groot/test_groot_lerobot.py
@@ -0,0 +1,208 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test script for LeRobot's Groot policy forward and inference passes."""
+
+import gc
+import os
+from copy import deepcopy
+from typing import Any
+
+import numpy as np
+import pytest
+import torch
+
+from lerobot.policies.groot.configuration_groot import GrootConfig
+from lerobot.policies.groot.modeling_groot import GrootPolicy
+from lerobot.policies.groot.processor_groot import make_groot_pre_post_processors
+from lerobot.processor import PolicyProcessorPipeline
+from lerobot.types import PolicyAction
+from lerobot.utils.device_utils import auto_select_torch_device
+from tests.utils import require_cuda  # noqa: E402
+
+pytest.importorskip("transformers")
+
+pytestmark = pytest.mark.skipif(
+    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
+    reason="This test requires local Groot installation and is not meant for CI",
+)
+
+
+# Define constants for dummy data
+DUMMY_STATE_DIM = 44
+DUMMY_ACTION_DIM = 44
+DUMMY_ACTION_HORIZON = 16
+IMAGE_SIZE = 256
+DEVICE = auto_select_torch_device()
+MODEL_PATH = "aractingi/bimanual-handover-groot-10k"
+
+
+def cleanup_memory():
+    """Clean up GPU/MPS memory to prevent OOM errors between tests."""
+    print("\nCleaning up memory...")
+    gc.collect()
+    if torch.cuda.is_available():
+        torch.cuda.empty_cache()
+        torch.cuda.synchronize()
+    if torch.backends.mps.is_available():
+        torch.mps.empty_cache()
+    print("Memory cleanup complete.")
+
+
+def set_seed_all(seed: int):
+    """Set random seed for all RNG sources to ensure reproducibility."""
+    import random
+
+    random.seed(seed)
+    np.random.seed(seed)
+    torch.manual_seed(seed)
+
+    if torch.cuda.is_available():
+        torch.cuda.manual_seed(seed)
+        torch.cuda.manual_seed_all(seed)
+
+    # Set deterministic behavior
+    torch.backends.cudnn.deterministic = True
+    torch.backends.cudnn.benchmark = False
+    torch.use_deterministic_algorithms(True, warn_only=True)
+
+
+def instantiate_lerobot_groot(
+    from_pretrained: bool = False,
+    model_path: str = MODEL_PATH,
+) -> tuple[
+    GrootPolicy,
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """Instantiate LeRobot Groot policy with preprocessor and postprocessor."""
+    if from_pretrained:
+        policy = GrootPolicy.from_pretrained(
+            pretrained_name_or_path=model_path,
+            strict=False,
+        )
+        policy.config.embodiment_tag = "gr1"
+    else:
+        config = GrootConfig(
+            base_model_path=model_path,
+            n_action_steps=DUMMY_ACTION_HORIZON,
+            chunk_size=DUMMY_ACTION_HORIZON,
+            image_size=[IMAGE_SIZE, IMAGE_SIZE],
+            device=DEVICE,
+            embodiment_tag="gr1",
+        )
+        policy = GrootPolicy(config)
+
+    policy.to(DEVICE)
+    policy.config.device = DEVICE
+
+    preprocessor, postprocessor = make_groot_pre_post_processors(
+        config=policy.config,
+        dataset_stats=None,  # Pass None for dataset_stats to disable normalization (original GR00T doesn't normalize)
+    )
+
+    return (policy, preprocessor, postprocessor)
+
+
+def create_dummy_data(device=DEVICE):
+    """Create a dummy data batch for testing."""
+    batch_size = 2
+    prompt = "Pick up the red cube and place it in the bin"
+    state = torch.randn(batch_size, DUMMY_STATE_DIM, dtype=torch.float32, device=device)
+
+    batch = {
+        "observation.state": state,
+        "action": torch.randn(
+            batch_size,
+            DUMMY_ACTION_HORIZON,
+            DUMMY_ACTION_DIM,
+            dtype=torch.float32,
+            device=device,  # Action ground truth (for training)
+        ),
+        "observation.images.ego_view": torch.rand(
+            batch_size,
+            3,
+            IMAGE_SIZE,
+            IMAGE_SIZE,
+            dtype=torch.float32,
+            device=device,  # Images in [0, 1] range as expected by LeRobot
+        ),
+        "task": [prompt for _ in range(batch_size)],
+    }
+
+    return batch
+
+
+@require_cuda
+def test_lerobot_groot_inference():
+    """Test the inference pass (select_action) of LeRobot's Groot policy."""
+    print("Test: LeRobot Groot Inference Pass")
+
+    set_seed_all(42)
+
+    # Instantiate policy and processors
+    lerobot_policy, lerobot_preprocessor, lerobot_postprocessor = instantiate_lerobot_groot(
+        from_pretrained=True
+    )
+    batch = create_dummy_data()
+
+    print("\n[LeRobot] Running inference...")
+    lerobot_policy.eval()
+    batch_lerobot_processed = lerobot_preprocessor(deepcopy(batch))
+
+    # Ensure identical RNG state before inference
+    torch.manual_seed(42)
+
+    with torch.no_grad():
+        lerobot_action = lerobot_policy.select_action(batch_lerobot_processed)
+
+    print(f"\nInference successful. Output action shape: {lerobot_action.shape}")
+    print("Output actions (first 5 dims):")
+    print(lerobot_action[:, :5])
+
+    lerobot_action = lerobot_postprocessor(lerobot_action)
+
+    del lerobot_policy, lerobot_preprocessor, lerobot_postprocessor, batch
+    cleanup_memory()
+
+
+@require_cuda
+def test_lerobot_groot_forward_pass():
+    """Test the forward pass of LeRobot's Groot policy."""
+    print("\n" + "=" * 50)
+    print("Test: LeRobot Groot Forward Pass (Training Mode)")
+
+    set_seed_all(42)
+
+    # Instantiate policy and processors
+    lerobot_policy, lerobot_preprocessor, _ = instantiate_lerobot_groot(from_pretrained=True)
+    batch = create_dummy_data()
+
+    lerobot_policy.eval()
+
+    print("\n[LeRobot] Running forward pass...")
+    batch_lerobot_processed = lerobot_preprocessor(deepcopy(batch))
+
+    set_seed_all(42)
+    with torch.no_grad():
+        lerobot_loss, lerobot_metrics = lerobot_policy.forward(batch_lerobot_processed)
+
+    print("\nForward pass successful.")
+    print(f"  - Loss: {lerobot_loss.item():.6f}")
+    print(f"  - Metrics: {lerobot_metrics}")
+
+    del lerobot_policy, lerobot_preprocessor, batch
+    cleanup_memory()
diff --git a/lerobot/tests/policies/groot/test_groot_vs_original.py b/lerobot/tests/policies/groot/test_groot_vs_original.py
new file mode 100644
index 0000000000000000000000000000000000000000..0adad96ca1cb5e32044c48a6381697568ac9174f
--- /dev/null
+++ b/lerobot/tests/policies/groot/test_groot_vs_original.py
@@ -0,0 +1,444 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test script to verify Groot policy integration with LeRobot vs the original implementation, only meant to be run locally!"""
+
+import gc
+import os
+from copy import deepcopy
+from typing import Any
+
+import numpy as np
+import pytest
+import torch
+
+from lerobot.policies.groot.configuration_groot import GrootConfig
+from lerobot.policies.groot.modeling_groot import GrootPolicy
+from lerobot.policies.groot.processor_groot import make_groot_pre_post_processors
+from lerobot.processor import PolicyProcessorPipeline
+from lerobot.types import PolicyAction
+
+pytest.importorskip("gr00t")
+pytest.importorskip("transformers")
+
+pytestmark = pytest.mark.skipif(
+    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
+    reason="This test requires local Groot installation and is not meant for CI",
+)
+
+
+from gr00t.data.dataset import ModalityConfig  # noqa: E402
+from gr00t.data.embodiment_tags import EmbodimentTag  # noqa: E402
+from gr00t.data.transform.base import ComposedModalityTransform  # noqa: E402
+from gr00t.model.policy import Gr00tPolicy  # noqa: E402
+
+# GR1 humanoid dimensions (from pretrained model metadata)
+# The actual GR1 robot has 44 dimensions for both state and action
+# GR00TTransform will pad state to 64 and truncate action to 32
+DUMMY_STATE_DIM = 44
+DUMMY_ACTION_DIM = 44
+DUMMY_ACTION_HORIZON = 16
+IMAGE_SIZE = 256
+DEVICE = "cpu"
+MODEL_PATH = "nvidia/GR00T-N1.5-3B"
+
+GR1_BODY_PARTS = {
+    "left_arm": 7,
+    "left_hand": 6,
+    "left_leg": 6,
+    "neck": 3,
+    "right_arm": 7,
+    "right_hand": 6,
+    "right_leg": 6,
+    "waist": 3,
+}
+
+
+def cleanup_memory():
+    """Clean up GPU/MPS memory to prevent OOM errors between tests."""
+    print("\nCleaning up memory...")
+    gc.collect()
+    if torch.cuda.is_available():
+        torch.cuda.empty_cache()
+        torch.cuda.synchronize()
+    if torch.backends.mps.is_available():
+        torch.mps.empty_cache()
+    print("Memory cleanup complete.")
+
+
+def set_seed_all(seed: int):
+    """Set random seed for all RNG sources to ensure reproducibility."""
+    import random
+
+    random.seed(seed)
+    np.random.seed(seed)
+    torch.manual_seed(seed)
+
+    if torch.cuda.is_available():
+        torch.cuda.manual_seed(seed)
+        torch.cuda.manual_seed_all(seed)
+
+    # Set deterministic behavior
+    torch.backends.cudnn.deterministic = True
+    torch.backends.cudnn.benchmark = False
+    torch.use_deterministic_algorithms(True, warn_only=True)
+
+
+def instantiate_lerobot_groot(
+    from_pretrained: bool = False,
+    model_path: str = MODEL_PATH,
+) -> tuple[
+    GrootPolicy,
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """Instantiate LeRobot Groot policy with preprocessor and postprocessor."""
+    if from_pretrained:
+        policy = GrootPolicy.from_pretrained(
+            pretrained_name_or_path=model_path,
+            strict=False,
+        )
+        policy.config.embodiment_tag = "gr1"
+    else:
+        config = GrootConfig(
+            base_model_path=model_path,
+            n_action_steps=DUMMY_ACTION_HORIZON,
+            chunk_size=DUMMY_ACTION_HORIZON,
+            image_size=[IMAGE_SIZE, IMAGE_SIZE],
+            device=DEVICE,
+            embodiment_tag="gr1",
+        )
+        policy = GrootPolicy(config)
+
+    policy.to(DEVICE)
+    policy.config.device = DEVICE
+
+    preprocessor, postprocessor = make_groot_pre_post_processors(
+        config=policy.config,
+        dataset_stats=None,  # Pass None for dataset_stats to disable normalization (original GR00T doesn't normalize)
+    )
+
+    return (policy, preprocessor, postprocessor)
+
+
+def instantiate_original_groot(
+    from_pretrained: bool = False,
+    model_path: str = MODEL_PATH,
+):
+    """Instantiate original Groot policy from NVIDIA's implementation."""
+    from gr00t.data.transform.concat import ConcatTransform
+    from gr00t.data.transform.state_action import StateActionToTensor
+    from gr00t.data.transform.video import VideoToNumpy, VideoToTensor
+    from gr00t.model.transforms import GR00TTransform
+
+    video_keys = ["video.ego_view"]
+    state_keys = [
+        "state"
+    ]  # Important: Use single concatenated "state" key (not split body parts) to match preprocessing
+    action_keys = [
+        "action.left_arm",
+        "action.right_arm",
+        "action.left_hand",
+        "action.right_hand",
+        "action.left_leg",
+        "action.right_leg",
+        "action.neck",
+        "action.waist",
+    ]
+    language_keys = ["annotation.human.action.task_description"]
+
+    modality_config = {
+        "video": ModalityConfig(
+            delta_indices=[0],  # Current frame only
+            modality_keys=video_keys,
+        ),
+        "state": ModalityConfig(
+            delta_indices=[0],
+            modality_keys=state_keys,
+        ),
+        "action": ModalityConfig(
+            delta_indices=list(range(DUMMY_ACTION_HORIZON)),
+            modality_keys=action_keys,
+        ),
+        "language": ModalityConfig(
+            delta_indices=[0],
+            modality_keys=language_keys,
+        ),
+    }
+
+    modality_transform = ComposedModalityTransform(
+        transforms=[
+            VideoToTensor(apply_to=video_keys),
+            VideoToNumpy(apply_to=video_keys),  # Convert to numpy (GR00TTransform expects numpy arrays)
+            # State is already a single concatenated key, so no StateActionToTensor needed
+            # Convert action from numpy to tensor
+            StateActionToTensor(apply_to=action_keys),
+            # Concatenate only video and actions (state is already single key)
+            ConcatTransform(
+                video_concat_order=video_keys,
+                state_concat_order=[],  # Empty:state is already single key
+                action_concat_order=action_keys,
+            ),
+            GR00TTransform(
+                max_state_dim=64,
+                max_action_dim=32,
+                state_horizon=1,
+                action_horizon=DUMMY_ACTION_HORIZON,
+                training=False,
+            ),
+        ]
+    )
+
+    policy = Gr00tPolicy(
+        model_path=model_path,
+        embodiment_tag=EmbodimentTag.GR1,
+        modality_config=modality_config,
+        modality_transform=modality_transform,
+        device=DEVICE,
+    )
+
+    return policy, modality_config, modality_transform
+
+
+def create_dummy_data(device=DEVICE):
+    """Create dummy data for testing both implementations."""
+    batch_size = 2
+    prompt = "Pick up the red cube and place it in the bin"
+    state = torch.randn(batch_size, DUMMY_STATE_DIM, dtype=torch.float32, device=device)
+
+    batch = {
+        "observation.state": state,
+        "action": torch.randn(
+            batch_size,
+            DUMMY_ACTION_HORIZON,
+            DUMMY_ACTION_DIM,
+            dtype=torch.float32,
+            device=device,  # Action ground truth (for training)
+        ),
+        "observation.images.ego_view": torch.rand(
+            batch_size,
+            3,
+            IMAGE_SIZE,
+            IMAGE_SIZE,
+            dtype=torch.float32,
+            device=device,  # Images in [0, 1] range as expected by LeRobot
+        ),
+        "task": [prompt for _ in range(batch_size)],
+    }
+
+    return batch
+
+
+def convert_lerobot_to_original_format(batch, modality_config):
+    """Convert LeRobot batch format to original Groot format.
+
+    The original Groot expects observations in this format:
+    {
+        "video.<camera_name>": np.ndarray (T, H, W, C) or (B, T, H, W, C)
+        "state.<state_component>": np.ndarray (T, D) or (B, T, D)
+        "action.<action_component>": np.ndarray (T, D) or (B, T, D)
+        "annotation.<annotation_type>": str or list[str]
+    }
+    """
+    # Original Groot expects (T, H, W, C) format for images
+    # LeRobot has (B, C, H, W) format, so we need to convert
+    observation = {}
+
+    for img_key in ["ego_view"]:
+        lerobot_key = f"observation.images.{img_key}"
+        if lerobot_key in batch:
+            img = batch[lerobot_key]
+            # Convert from (B, C, H, W) to (B, T=1, H, W, C)
+            img_np = img.permute(0, 2, 3, 1).unsqueeze(1).cpu().numpy()
+            # Convert [0, 1] to [0, 255] uint8 as expected by original
+            img_np = (img_np * 255).astype(np.uint8)
+            observation[f"video.{img_key}"] = img_np
+
+    # Important: The Original's GR00TTransform expects "state" as (B, T, D), not split body parts
+    if "observation.state" in batch:
+        state = batch["observation.state"]
+        state_np = state.unsqueeze(1).cpu().numpy()  # (B, 1, D)
+        observation["state"] = state_np
+
+    if "action" in batch:
+        action = batch["action"]
+        action_np = action.cpu().numpy()
+
+        start_idx = 0
+        for part_name, part_dim in GR1_BODY_PARTS.items():
+            end_idx = start_idx + part_dim
+            observation[f"action.{part_name}"] = action_np[:, :, start_idx:end_idx]
+            start_idx = end_idx
+
+    if "task" in batch:
+        task_list = batch["task"]
+        # GR00TTransform expects language with (B, T) shape for batched data
+        # Create a (B, T=1) array where each element is the string directly
+        bsz = len(task_list)
+        task_array = np.empty((bsz, 1), dtype=object)
+        for i in range(bsz):
+            task_array[i, 0] = task_list[i]  # Assign string directly to each (i, 0) position
+        observation["annotation.human.action.task_description"] = task_array
+
+    return observation
+
+
+def test_groot_original_vs_lerobot_pretrained():
+    """Test Groot original implementation vs LeRobot implementation with pretrained weights."""
+    print("Test: Groot Original vs LeRobot with Pretrained Weights (Inference)")
+
+    set_seed_all(42)
+
+    lerobot_policy, lerobot_preprocessor, lerobot_postprocessor = instantiate_lerobot_groot(
+        from_pretrained=True
+    )
+    original_policy, modality_config, modality_transform = instantiate_original_groot(from_pretrained=True)
+
+    batch = create_dummy_data()
+    batch_lerobot = deepcopy(batch)
+
+    print("\n[LeRobot] Running inference...")
+    lerobot_policy.eval()
+    batch_lerobot_processed = lerobot_preprocessor(batch_lerobot)
+
+    # Important: Reset seed immediately before inference to ensure identical RNG state
+    torch.manual_seed(42)
+
+    with torch.no_grad():
+        lerobot_actions = lerobot_policy.select_action(batch_lerobot_processed)
+
+    print("\n[Original] Running inference...")
+    original_policy.model.eval()
+    observation = convert_lerobot_to_original_format(batch, modality_config)
+    original_obs_transformed = modality_transform(deepcopy(observation))
+
+    # Important: Reset seed immediately before inference to ensure identical RNG state
+    torch.manual_seed(42)
+
+    with torch.no_grad():
+        original_model_output = original_policy.model.get_action(original_obs_transformed)
+        original_actions_raw = original_model_output["action_pred"]  # [2, 16, 32]
+    # Take first timestep
+    original_actions = original_actions_raw[:, 0, :].to(lerobot_actions.device).to(lerobot_actions.dtype)
+
+    print("Action Comparison:")
+    diff = lerobot_actions - original_actions
+    abs_diff = torch.abs(diff)
+
+    for batch_idx in range(lerobot_actions.shape[0]):
+        print(f"\n{'=' * 60}")
+        print(f"Batch {batch_idx}")
+        print(f"{'=' * 60}")
+        print(f"{'Idx':<5} {'LeRobot':<14} {'Original':<14} {'Difference':<14}")
+        print("-" * 60)
+        for action_idx in range(lerobot_actions.shape[1]):
+            lr_val = lerobot_actions[batch_idx, action_idx].item()
+            orig_val = original_actions[batch_idx, action_idx].item()
+            diff_val = abs(lr_val - orig_val)
+            sign = "+" if (lr_val - orig_val) > 0 else "-"
+            print(f"{action_idx:<5} {lr_val:>13.6f} {orig_val:>13.6f} {sign}{diff_val:>12.6f}")
+
+    max_diff = abs_diff.max().item()
+    tolerance = 0.001
+    assert torch.allclose(lerobot_actions, original_actions, atol=tolerance), (
+        f"Actions differ by more than tolerance ({tolerance}): max diff = {max_diff:.6f}"
+    )
+    print(f"\nSuccess: Actions match within tolerance ({tolerance})!")
+
+    del lerobot_policy, lerobot_preprocessor, lerobot_postprocessor
+    del original_policy, modality_config, modality_transform
+    del batch, batch_lerobot, observation
+    cleanup_memory()
+
+
+def test_groot_forward_pass_comparison():
+    """Test forward pass comparison between LeRobot and Original Groot implementations."""
+    print("Test: Forward Pass Comparison (Training Mode)")
+
+    set_seed_all(42)
+
+    lerobot_policy, lerobot_preprocessor, lerobot_postprocessor = instantiate_lerobot_groot(
+        from_pretrained=True
+    )
+    original_policy, modality_config, modality_transform = instantiate_original_groot(from_pretrained=True)
+
+    batch = create_dummy_data()
+    lerobot_policy.eval()
+    original_policy.model.eval()
+
+    print("\n[LeRobot] Running forward pass...")
+    batch_lerobot = deepcopy(batch)
+    batch_lerobot_processed = lerobot_preprocessor(batch_lerobot)
+
+    set_seed_all(42)
+    with torch.no_grad():
+        lerobot_loss, lerobot_metrics = lerobot_policy.forward(batch_lerobot_processed)
+
+    print(f"  Loss: {lerobot_loss.item():.6f}")
+
+    print("\n[Original] Running forward pass...")
+    observation = convert_lerobot_to_original_format(batch, modality_config)
+    transformed_obs = modality_transform(observation)
+
+    if "action" not in transformed_obs:
+        action_for_forward = batch_lerobot_processed["action"]
+        action_mask_for_forward = batch_lerobot_processed["action_mask"]
+
+        # Match action horizon if needed
+        if action_for_forward.shape[1] != original_policy.model.action_horizon:
+            if action_for_forward.shape[1] < original_policy.model.action_horizon:
+                pad_size = original_policy.model.action_horizon - action_for_forward.shape[1]
+                last_action = action_for_forward[:, -1:, :]
+                padding = last_action.repeat(1, pad_size, 1)
+                action_for_forward = torch.cat([action_for_forward, padding], dim=1)
+
+                mask_padding = torch.zeros(
+                    action_mask_for_forward.shape[0],
+                    pad_size,
+                    action_mask_for_forward.shape[2],
+                    dtype=action_mask_for_forward.dtype,
+                    device=action_mask_for_forward.device,
+                )
+                action_mask_for_forward = torch.cat([action_mask_for_forward, mask_padding], dim=1)
+            else:
+                action_for_forward = action_for_forward[:, : original_policy.model.action_horizon, :]
+                action_mask_for_forward = action_mask_for_forward[
+                    :, : original_policy.model.action_horizon, :
+                ]
+
+        transformed_obs["action"] = action_for_forward
+        transformed_obs["action_mask"] = action_mask_for_forward
+
+    set_seed_all(42)
+    with torch.no_grad():
+        original_outputs = original_policy.model.forward(transformed_obs)
+
+    original_loss = original_outputs["loss"]
+    print(f"  Loss: {original_loss.item():.6f}")
+
+    loss_diff = abs(lerobot_loss.item() - original_loss.item())
+    loss_rel_diff = loss_diff / (abs(original_loss.item()) + 1e-8) * 100
+
+    print("\nLoss Values:")
+    print(f"  LeRobot: {lerobot_loss.item():.6f}")
+    print(f"  Original: {original_loss.item():.6f}")
+    print(f"  Absolute difference: {loss_diff:.6f}")
+    print(f"  Relative difference: {loss_rel_diff:.2f}%")
+
+    del lerobot_policy, lerobot_preprocessor, lerobot_postprocessor
+    del original_policy, modality_config, modality_transform
+    del batch, batch_lerobot, observation, transformed_obs
+    cleanup_memory()
diff --git a/lerobot/tests/policies/hilserl/test_modeling_classifier.py b/lerobot/tests/policies/hilserl/test_modeling_classifier.py
new file mode 100644
index 0000000000000000000000000000000000000000..a62ef3ebb4910204003e8babd1eccc0069207b7d
--- /dev/null
+++ b/lerobot/tests/policies/hilserl/test_modeling_classifier.py
@@ -0,0 +1,153 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.policies.sac.reward_model.configuration_classifier import RewardClassifierConfig
+from lerobot.policies.sac.reward_model.modeling_classifier import ClassifierOutput
+from lerobot.utils.constants import OBS_IMAGE, REWARD
+from tests.utils import require_package
+
+
+def test_classifier_output():
+    output = ClassifierOutput(
+        logits=torch.tensor([1, 2, 3]),
+        probabilities=torch.tensor([0.1, 0.2, 0.3]),
+        hidden_states=None,
+    )
+
+    assert (
+        f"{output}"
+        == "ClassifierOutput(logits=tensor([1, 2, 3]), probabilities=tensor([0.1000, 0.2000, 0.3000]), hidden_states=None)"
+    )
+
+
+@require_package("transformers")
+@pytest.mark.skip(
+    reason="helper2424/resnet10 needs to be updated to work with the latest version of transformers"
+)
+def test_binary_classifier_with_default_params():
+    from lerobot.policies.sac.reward_model.modeling_classifier import Classifier
+
+    config = RewardClassifierConfig()
+    config.input_features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        REWARD: PolicyFeature(type=FeatureType.REWARD, shape=(1,)),
+    }
+    config.normalization_mapping = {
+        "VISUAL": NormalizationMode.IDENTITY,
+        "REWARD": NormalizationMode.IDENTITY,
+    }
+    config.num_cameras = 1
+    classifier = Classifier(config)
+
+    batch_size = 10
+
+    input = {
+        OBS_IMAGE: torch.rand((batch_size, 3, 128, 128)),
+        REWARD: torch.randint(low=0, high=2, size=(batch_size,)).float(),
+    }
+
+    images, labels = classifier.extract_images_and_labels(input)
+    assert len(images) == 1
+    assert images[0].shape == torch.Size([batch_size, 3, 128, 128])
+    assert labels.shape == torch.Size([batch_size])
+
+    output = classifier.predict(images)
+
+    assert output is not None
+    assert output.logits.size() == torch.Size([batch_size])
+    assert not torch.isnan(output.logits).any(), "Tensor contains NaN values"
+    assert output.probabilities.shape == torch.Size([batch_size])
+    assert not torch.isnan(output.probabilities).any(), "Tensor contains NaN values"
+    assert output.hidden_states.shape == torch.Size([batch_size, 256])
+    assert not torch.isnan(output.hidden_states).any(), "Tensor contains NaN values"
+
+
+@require_package("transformers")
+@pytest.mark.skip(
+    reason="helper2424/resnet10 needs to be updated to work with the latest version of transformers"
+)
+def test_multiclass_classifier():
+    from lerobot.policies.sac.reward_model.modeling_classifier import Classifier
+
+    num_classes = 5
+    config = RewardClassifierConfig()
+    config.input_features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        REWARD: PolicyFeature(type=FeatureType.REWARD, shape=(num_classes,)),
+    }
+    config.num_cameras = 1
+    config.num_classes = num_classes
+    classifier = Classifier(config)
+
+    batch_size = 10
+
+    input = {
+        OBS_IMAGE: torch.rand((batch_size, 3, 128, 128)),
+        REWARD: torch.rand((batch_size, num_classes)),
+    }
+
+    images, labels = classifier.extract_images_and_labels(input)
+    assert len(images) == 1
+    assert images[0].shape == torch.Size([batch_size, 3, 128, 128])
+    assert labels.shape == torch.Size([batch_size, num_classes])
+
+    output = classifier.predict(images)
+
+    assert output is not None
+    assert output.logits.shape == torch.Size([batch_size, num_classes])
+    assert not torch.isnan(output.logits).any(), "Tensor contains NaN values"
+    assert output.probabilities.shape == torch.Size([batch_size, num_classes])
+    assert not torch.isnan(output.probabilities).any(), "Tensor contains NaN values"
+    assert output.hidden_states.shape == torch.Size([batch_size, 256])
+    assert not torch.isnan(output.hidden_states).any(), "Tensor contains NaN values"
+
+
+@require_package("transformers")
+@pytest.mark.skip(
+    reason="helper2424/resnet10 needs to be updated to work with the latest version of transformers"
+)
+def test_default_device():
+    from lerobot.policies.sac.reward_model.modeling_classifier import Classifier
+
+    config = RewardClassifierConfig()
+    assert config.device == "cpu"
+
+    classifier = Classifier(config)
+    for p in classifier.parameters():
+        assert p.device == torch.device("cpu")
+
+
+@require_package("transformers")
+@pytest.mark.skip(
+    reason="helper2424/resnet10 needs to be updated to work with the latest version of transformers"
+)
+def test_explicit_device_setup():
+    from lerobot.policies.sac.reward_model.modeling_classifier import Classifier
+
+    config = RewardClassifierConfig(device="cpu")
+    assert config.device == "cpu"
+
+    classifier = Classifier(config)
+    for p in classifier.parameters():
+        assert p.device == torch.device("cpu")
diff --git a/lerobot/tests/policies/pi0_fast/test_pi0_fast_original_vs_lerobot.py b/lerobot/tests/policies/pi0_fast/test_pi0_fast_original_vs_lerobot.py
new file mode 100644
index 0000000000000000000000000000000000000000..b757d5a94334178797ba922f1fe5992fb2e02c4f
--- /dev/null
+++ b/lerobot/tests/policies/pi0_fast/test_pi0_fast_original_vs_lerobot.py
@@ -0,0 +1,518 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test script to verify PI0Fast policy integration with LeRobot vs the original implementation"""
+# ruff: noqa: E402
+
+import random
+from copy import deepcopy
+from typing import Any
+
+import numpy as np
+import pytest
+import torch
+
+pytest.importorskip("transformers")
+pytest.importorskip("scipy")
+
+from lerobot.policies.pi0_fast.configuration_pi0_fast import PI0FastConfig
+from lerobot.policies.pi0_fast.modeling_pi0_fast import PI0FastPolicy
+from lerobot.policies.pi0_fast.processor_pi0_fast import make_pi0_fast_pre_post_processors
+from lerobot.processor import PolicyProcessorPipeline  # noqa: E402
+from lerobot.types import PolicyAction  # noqa: E402
+from lerobot.utils.constants import (
+    ACTION_TOKEN_MASK,
+    ACTION_TOKENS,
+    OBS_IMAGES,
+    OBS_LANGUAGE_ATTENTION_MASK,
+    OBS_LANGUAGE_TOKENS,
+    OBS_STATE,
+)  # noqa: E402
+from tests.utils import require_cuda, require_hf_token  # noqa: E402
+
+# Constants
+DUMMY_ACTION_DIM = 7
+DUMMY_STATE_DIM = 20
+IMAGE_HEIGHT = 224
+IMAGE_WIDTH = 224
+NUM_VIEWS = 2  # Number of camera views
+DEVICE = "cuda"
+MODEL_PATH_LEROBOT = "lerobot/pi0fast-base"
+
+# Expected action token shape: (batch_size, max_decoding_steps)
+EXPECTED_ACTION_TOKENS_SHAPE = (1, 2)
+
+# Expected first 5 action tokens (for reproducibility check)
+EXPECTED_ACTION_TOKENS_FIRST_5 = torch.tensor([255020, 255589])
+
+# Expected actions after detokenization
+EXPECTED_ACTIONS_SHAPE = (1, 2, 32)  # (batch_size, n_action_steps, action_dim)
+EXPECTED_ACTIONS_MEAN = 0.046403881162405014
+EXPECTED_ACTIONS_STD = 0.2607129216194153
+EXPECTED_ACTIONS_FIRST_5 = torch.tensor([0.0000, 0.3536, 0.0707, 0.0000, 0.0000])
+
+
+@require_cuda
+@require_hf_token
+def set_seed_all(seed: int):
+    """Set random seed for all RNG sources to ensure reproducibility."""
+    random.seed(seed)
+    np.random.seed(seed)
+    torch.manual_seed(seed)
+
+    if torch.cuda.is_available():
+        torch.cuda.manual_seed(seed)
+        torch.cuda.manual_seed_all(seed)
+
+    # Set deterministic behavior
+    torch.backends.cudnn.deterministic = True
+    torch.backends.cudnn.benchmark = False
+    torch.use_deterministic_algorithms(True, warn_only=True)
+
+
+@require_cuda
+@require_hf_token
+def instantiate_lerobot_pi0_fast(
+    from_pretrained: bool = False,
+    model_path: str = MODEL_PATH_LEROBOT,
+) -> tuple[
+    Any,  # Policy
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """Instantiate LeRobot PI0Fast policy with preprocessor and postprocessor."""
+    if from_pretrained:
+        policy = PI0FastPolicy.from_pretrained(
+            pretrained_name_or_path=model_path,
+            strict=True,
+        )
+        policy.config.validate_action_token_prefix = False
+        policy.config.max_action_tokens = 2
+        policy.config.max_decoding_steps = 2
+        policy.config.chunk_size = 2
+        policy.config.n_action_steps = 2
+    else:
+        config = PI0FastConfig(
+            n_action_steps=2,
+            max_action_dim=DUMMY_ACTION_DIM,
+            max_state_dim=DUMMY_STATE_DIM,
+            device=DEVICE,
+            validate_action_token_prefix=False,
+            max_action_tokens=2,
+            max_decoding_steps=2,
+            chunk_size=2,
+        )
+        policy = PI0FastPolicy(config)
+
+    policy.to(DEVICE)
+    policy.config.device = DEVICE
+    preprocessor, postprocessor = make_pi0_fast_pre_post_processors(
+        config=policy.config,
+        dataset_stats=None,  # Pass None for dataset_stats to disable normalization
+    )
+
+    return policy, preprocessor, postprocessor
+
+
+@require_cuda
+@require_hf_token
+def create_dummy_data(device=DEVICE):
+    """Create dummy data for testing both implementations."""
+    batch_size = 1
+    prompt = "Pick up the red block and place it in the bin"
+
+    # Create random RGB images in [0, 255] uint8 range (as PIL images would be)
+    # Then convert to [0, 1] float32 range for LeRobot
+    def fake_rgb(h, w):
+        arr = np.random.randint(0, 255, (h, w, 3), dtype=np.uint8)
+        t = torch.from_numpy(arr).permute(2, 0, 1)  # CHW
+        return t
+
+    batch = {
+        f"{OBS_IMAGES}.base_0_rgb": torch.stack(
+            [fake_rgb(IMAGE_HEIGHT, IMAGE_WIDTH) for _ in range(batch_size)]
+        ).to(device),
+        f"{OBS_IMAGES}.left_wrist_0_rgb": torch.stack(
+            [fake_rgb(IMAGE_HEIGHT, IMAGE_WIDTH) for _ in range(batch_size)]
+        ).to(device),
+        f"{OBS_IMAGES}.right_wrist_0_rgb": torch.stack(
+            [fake_rgb(IMAGE_HEIGHT, IMAGE_WIDTH) for _ in range(batch_size)]
+        ).to(device),
+        OBS_STATE: torch.randn(batch_size, DUMMY_STATE_DIM, dtype=torch.float32, device=device),
+        "task": [prompt for _ in range(batch_size)],
+    }
+
+    return batch
+
+
+# Pytest fixtures
+@pytest.fixture(scope="module")
+@require_cuda
+@require_hf_token
+def pi0_fast_components():
+    """Fixture to instantiate and provide all PI0Fast components for tests."""
+    print(f"\nTesting with DEVICE='{DEVICE}'")
+    print("\n[Setup] Instantiating LeRobot PI0Fast policy...")
+    policy_obj, preprocessor_obj, postprocessor_obj = instantiate_lerobot_pi0_fast(from_pretrained=True)
+    print("Model loaded successfully")
+    return policy_obj, preprocessor_obj, postprocessor_obj
+
+
+@pytest.fixture(scope="module")
+@require_cuda
+@require_hf_token
+def policy(pi0_fast_components):
+    """Fixture to provide the PI0Fast policy for tests."""
+    return pi0_fast_components[0]
+
+
+@pytest.fixture(scope="module")
+@require_cuda
+@require_hf_token
+def preprocessor(pi0_fast_components):
+    """Fixture to provide the PI0Fast preprocessor for tests."""
+    return pi0_fast_components[1]
+
+
+@require_cuda
+@require_hf_token
+def test_pi0_fast_preprocessor_alignment(policy, preprocessor):
+    """Test that LeRobot PI0Fast preprocessor produces expected outputs."""
+    print("\n" + "=" * 80)
+    print("Test: PI0Fast Preprocessor Outputs")
+    print("=" * 80)
+
+    set_seed_all(42)
+
+    print("\nCreating dummy data...")
+    batch = create_dummy_data()
+
+    print("\n[LeRobot] Preprocessing...")
+    lerobot_observation = preprocessor(deepcopy(batch))
+
+    print("\nVerifying preprocessor outputs:")
+    print("-" * 80)
+
+    # Expected keys from PI0Fast preprocessing
+    expected_keys = [
+        "observation.images.base_0_rgb",
+        "observation.images.left_wrist_0_rgb",
+        "observation.images.right_wrist_0_rgb",
+        "observation.state",
+        "observation.language_tokens",
+        "observation.language_attention_mask",
+    ]
+
+    for key in expected_keys:
+        if key in lerobot_observation:
+            shape = tuple(lerobot_observation[key].shape)
+            print(f"\nKey: {key}")
+            print(f"Shape: {shape}")
+            print(f"Dtype: {lerobot_observation[key].dtype}")
+        else:
+            print(f"\nKey '{key}' not found in inputs!")
+
+    # Check language tokens shape
+    if "observation.language_tokens" in lerobot_observation:
+        lang_tokens = lerobot_observation["observation.language_tokens"]
+        print(f"\nLanguage tokens shape: {lang_tokens.shape}")
+        # Should have batch dimension and max_length from tokenizer
+        assert lang_tokens.dim() == 2, f"Expected 2D tensor, got {lang_tokens.dim()}D"
+
+    print("\nPreprocessor outputs verified!")
+
+
+@require_cuda
+@require_hf_token
+def test_pi0_fast_action_generation(policy, preprocessor):
+    """Test PI0Fast LeRobot implementation generates expected actions."""
+    print("\n" + "=" * 80)
+    print("Test: PI0Fast Action Generation Against Expected Values")
+    print("=" * 80)
+
+    set_seed_all(42)
+
+    print("\nCreating dummy data...")
+    batch = create_dummy_data()
+
+    print("\n[LeRobot] Running inference...")
+    lerobot_observation = preprocessor(deepcopy(batch))
+
+    # Reset seed for inference
+    torch.manual_seed(42)
+    with torch.no_grad():
+        lerobot_actions = policy.predict_action_chunk(lerobot_observation)
+        lerobot_actions = lerobot_actions.float().cpu()
+
+    print(f"LeRobot actions shape: {lerobot_actions.shape}")
+    print(f"LeRobot actions mean: {lerobot_actions.mean().item():.6f}")
+    print(f"LeRobot actions std: {lerobot_actions.std().item():.6f}")
+    print(f"LeRobot actions first 5: {lerobot_actions[0, 0, :5]}")
+
+    print("\nExpected values (from original PI0Fast):")
+    print(f"Expected actions shape: {EXPECTED_ACTIONS_SHAPE}")
+    print(f"Expected actions mean: {EXPECTED_ACTIONS_MEAN:.6f}")
+    print(f"Expected actions std: {EXPECTED_ACTIONS_STD:.6f}")
+    print(f"Expected actions first 5: {EXPECTED_ACTIONS_FIRST_5}")
+
+    print("\nAction Comparison:")
+    print("-" * 80)
+
+    # Compare shapes
+    actual_shape = tuple(lerobot_actions.shape)
+    print(f"Actual shape: {actual_shape}")
+
+    assert actual_shape == EXPECTED_ACTIONS_SHAPE, (
+        f"Shape mismatch: {actual_shape} vs {EXPECTED_ACTIONS_SHAPE}"
+    )
+    print(f"Shape matches: {actual_shape}")
+
+    # Compare statistics
+    actual_mean = lerobot_actions.mean().item()
+    actual_std = lerobot_actions.std().item()
+
+    print(f"\nMean: {actual_mean:.6f} (expected: {EXPECTED_ACTIONS_MEAN:.6f})")
+    print(f"Std: {actual_std:.6f} (expected: {EXPECTED_ACTIONS_STD:.6f})")
+
+    # Compare first 5 actions
+    actual_first_5 = lerobot_actions[0, 0, :5]
+    print("\nFirst 5 actions comparison:")
+    print(f"  Actual:   {actual_first_5}")
+    print(f"  Expected: {EXPECTED_ACTIONS_FIRST_5}")
+
+    first_5_diff = torch.abs(actual_first_5 - EXPECTED_ACTIONS_FIRST_5)
+    print(f"  Max diff: {first_5_diff.max().item():.6e}")
+    print(f"  Mean diff: {first_5_diff.mean().item():.6e}")
+
+    # Check with different tolerances
+    tolerances = [1e-5, 1e-4, 1e-3, 1e-2]
+    for tol in tolerances:
+        is_close = torch.allclose(actual_first_5, EXPECTED_ACTIONS_FIRST_5, atol=tol)
+        status = "Success" if is_close else "Failure"
+        print(f"{status}: First 5 actions close (atol={tol}): {is_close}")
+
+    # Assert with reasonable tolerance
+    tolerance = 1e-3
+    assert torch.allclose(actual_first_5, EXPECTED_ACTIONS_FIRST_5, atol=tolerance), (
+        f"First 5 actions differ by more than tolerance ({tolerance})"
+    )
+    print(f"\nSuccess: Actions match expected values within tolerance ({tolerance})!")
+
+    print("\nAction generation test completed (values printed for reference)!")
+
+
+@require_cuda
+@require_hf_token
+def test_pi0_fast_inference_reproducibility(policy, preprocessor):
+    """Test that PI0Fast inference is reproducible with the same seed."""
+    print("\n" + "=" * 80)
+    print("Test: PI0Fast Inference Reproducibility")
+    print("=" * 80)
+
+    print("\nCreating dummy data...")
+    batch = create_dummy_data()
+
+    # First inference
+    print("\n[Run 1] Running inference...")
+    set_seed_all(42)
+    lerobot_observation = preprocessor(deepcopy(batch))
+    with torch.no_grad():
+        actions_1 = policy.predict_action_chunk(lerobot_observation)
+        actions_1 = actions_1.float().cpu()
+
+    # Second inference with same seed
+    print("\n[Run 2] Running inference with same seed...")
+    set_seed_all(42)
+    lerobot_observation = preprocessor(deepcopy(batch))
+    with torch.no_grad():
+        actions_2 = policy.predict_action_chunk(lerobot_observation)
+        actions_2 = actions_2.float().cpu()
+
+    print("\nComparing two runs:")
+    print("-" * 80)
+    if torch.allclose(actions_1, actions_2, atol=1e-8):
+        print("Inference is perfectly reproducible!")
+    else:
+        diff = torch.abs(actions_1 - actions_2)
+        print("Small differences detected:")
+        print(f"  Max diff: {diff.max().item():.6e}")
+        print(f"  Mean diff: {diff.mean().item():.6e}")
+
+    assert torch.allclose(actions_1, actions_2, atol=1e-6), "Inference should be reproducible!"
+
+    print("\nInference is reproducible!")
+
+
+@require_cuda
+@require_hf_token
+def test_pi0_fast_forward_pass_logits(policy, preprocessor):
+    """Test PI0Fast forward pass and compare logits against expected values."""
+    print("\n" + "=" * 80)
+    print("Test: PI0Fast Forward Pass Logits")
+    print("=" * 80)
+
+    set_seed_all(42)
+
+    print("\nCreating dummy data with action tokens...")
+    batch = create_dummy_data()
+
+    # Preprocess the batch
+    lerobot_observation = preprocessor(deepcopy(batch))
+
+    # For forward pass, we need action tokens
+    # Create dummy action tokens for testing
+    batch_size = 1
+    max_action_tokens = policy.config.max_action_tokens
+
+    # Create dummy action tokens (in practice, these come from the FAST tokenizer)
+    dummy_action_tokens = torch.randint(
+        0, 1000, (batch_size, max_action_tokens), dtype=torch.long, device=DEVICE
+    )
+    dummy_action_masks = torch.ones(batch_size, max_action_tokens, dtype=torch.bool, device=DEVICE)
+
+    # Add action tokens to the observation
+    lerobot_observation[ACTION_TOKENS] = dummy_action_tokens
+    lerobot_observation[ACTION_TOKEN_MASK] = dummy_action_masks
+
+    print("\n[LeRobot] Running forward pass...")
+    policy.train()
+    with torch.no_grad():
+        loss, loss_dict = policy.forward(lerobot_observation)
+
+    print(f"Loss: {loss.item():.6f}")
+    print(f"FAST Loss: {loss_dict['ce_loss']:.6f}")
+
+    print("\nForward pass completed successfully!")
+    print(f"Loss value: {loss.item():.6f}")
+
+    # The loss should be a positive value
+    assert loss.item() > 0, "Loss should be positive"
+    assert not torch.isnan(loss), "Loss should not be NaN"
+    assert not torch.isinf(loss), "Loss should not be infinite"
+
+    print("\nForward pass test passed!")
+
+
+@require_cuda
+@require_hf_token
+def test_pi0_fast_action_token_sampling(policy, preprocessor):
+    """Test PI0Fast action token sampling (autoregressive decoding)."""
+    print("\n" + "=" * 80)
+    print("Test: PI0Fast Action Token Sampling")
+    print("=" * 80)
+
+    set_seed_all(42)
+
+    print("\nCreating dummy data...")
+    batch = create_dummy_data()
+
+    print("\n[LeRobot] Preprocessing...")
+    lerobot_observation = preprocessor(deepcopy(batch))
+
+    # Prepare inputs for model
+    images, img_masks = policy._preprocess_images(lerobot_observation)
+    tokens = lerobot_observation[OBS_LANGUAGE_TOKENS]
+    masks = lerobot_observation[OBS_LANGUAGE_ATTENTION_MASK]
+
+    print("\n[LeRobot] Sampling action tokens...")
+    torch.manual_seed(42)
+    with torch.no_grad():
+        action_tokens = policy.model.sample_actions_fast(
+            images,
+            img_masks,
+            tokens,
+            masks,
+            max_decoding_steps=2,
+            temperature=0.0,  # Greedy decoding for reproducibility
+        )
+
+    print(f"Action tokens shape: {action_tokens.shape}")
+    print(f"Action tokens first 10: {action_tokens[0, :10].tolist()}")
+
+    print("\nExpected values (from original PI0Fast):")
+    print(f"Expected shape: {EXPECTED_ACTION_TOKENS_SHAPE}")
+    print(f"Expected first 5: {EXPECTED_ACTION_TOKENS_FIRST_5.tolist()}")
+
+    # Verify shape
+    actual_shape = tuple(action_tokens.shape)
+    print(f"\nActual shape: {actual_shape}")
+
+    assert actual_shape == EXPECTED_ACTION_TOKENS_SHAPE, (
+        f"Shape mismatch: {actual_shape} vs {EXPECTED_ACTION_TOKENS_SHAPE}"
+    )
+
+    # Compare first 5 tokens
+    actual_first_5 = action_tokens[0, :5].cpu()
+    assert torch.equal(actual_first_5, EXPECTED_ACTION_TOKENS_FIRST_5), (
+        f"First 5 tokens mismatch: {actual_first_5} vs {EXPECTED_ACTION_TOKENS_FIRST_5}"
+    )
+
+    print("\nAction token sampling test completed!")
+
+
+@require_cuda
+@require_hf_token
+def test_pi0_fast_detokenization(policy, preprocessor):
+    """Test PI0Fast action detokenization (FAST decoding)."""
+    print("\n" + "=" * 80)
+    print("Test: PI0Fast Action Detokenization")
+    print("=" * 80)
+
+    set_seed_all(42)
+
+    print("\nCreating dummy data...")
+    batch = create_dummy_data()
+
+    print("\n[LeRobot] Preprocessing...")
+    lerobot_observation = preprocessor(deepcopy(batch))
+
+    # Prepare inputs for model
+    images, img_masks = policy._preprocess_images(lerobot_observation)
+    tokens = lerobot_observation[OBS_LANGUAGE_TOKENS]
+    masks = lerobot_observation[OBS_LANGUAGE_ATTENTION_MASK]
+
+    print("\n[LeRobot] Sampling action tokens...")
+    torch.manual_seed(42)
+    with torch.no_grad():
+        action_tokens = policy.model.sample_actions_fast(
+            images,
+            img_masks,
+            tokens,
+            masks,
+            max_decoding_steps=2,
+            temperature=0.0,
+        )
+
+    print(f"Action tokens shape: {action_tokens.shape}")
+
+    # Detokenize
+    print("\n[LeRobot] Detokenizing action tokens...")
+    action_horizon = policy.config.n_action_steps
+    action_dim = policy.config.output_features["action"].shape[0]
+
+    try:
+        continuous_actions = policy.detokenize_actions(
+            action_tokens, action_horizon=action_horizon, action_dim=action_dim
+        )
+        print(f"Continuous actions shape: {continuous_actions.shape}")
+        print(f"Continuous actions mean: {continuous_actions.mean().item():.6f}")
+        print(f"Continuous actions std: {continuous_actions.std().item():.6f}")
+        print(f"Continuous actions first 5: {continuous_actions[0, 0, :5]}")
+        print("\nDetokenization successful!")
+    except Exception as e:
+        print(f"\nDetokenization failed with error: {e}")
+        print("This may be expected if the action tokens are not valid FAST tokens.")
+        print("The test will pass as long as the sampling works correctly.")
diff --git a/lerobot/tests/policies/pi0_pi05/test_pi0.py b/lerobot/tests/policies/pi0_pi05/test_pi0.py
new file mode 100644
index 0000000000000000000000000000000000000000..5a985e03c91597ab84bdaa53e3f5bf6707394807
--- /dev/null
+++ b/lerobot/tests/policies/pi0_pi05/test_pi0.py
@@ -0,0 +1,127 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test script to verify PI0 policy integration with LeRobot"""
+
+import pytest
+import torch
+
+pytest.importorskip("transformers")
+
+from lerobot.policies.factory import make_policy_config  # noqa: E402
+from lerobot.policies.pi0 import (  # noqa: E402
+    PI0Config,
+    PI0Policy,
+    make_pi0_pre_post_processors,  # noqa: E402
+)
+from lerobot.utils.random_utils import set_seed  # noqa: E402
+from tests.utils import require_cuda, require_hf_token  # noqa: E402
+
+
+@require_cuda
+@require_hf_token
+def test_policy_instantiation():
+    # Create config
+    set_seed(42)
+    config = PI0Config(max_action_dim=7, max_state_dim=14, dtype="float32")
+
+    # Set up input_features and output_features in the config
+    from lerobot.configs.types import FeatureType, PolicyFeature
+
+    config.input_features = {
+        "observation.state": PolicyFeature(
+            type=FeatureType.STATE,
+            shape=(14,),
+        ),
+        "observation.images.base_0_rgb": PolicyFeature(
+            type=FeatureType.VISUAL,
+            shape=(3, 224, 224),
+        ),
+    }
+
+    config.output_features = {
+        "action": PolicyFeature(
+            type=FeatureType.ACTION,
+            shape=(7,),
+        ),
+    }
+
+    # Create dummy dataset stats
+    dataset_stats = {
+        "observation.state": {
+            "mean": torch.zeros(14),
+            "std": torch.ones(14),
+        },
+        "action": {
+            "mean": torch.zeros(7),
+            "std": torch.ones(7),
+        },
+        "observation.images.base_0_rgb": {
+            "mean": torch.zeros(3, 224, 224),
+            "std": torch.ones(3, 224, 224),
+        },
+    }
+
+    # Instantiate policy
+    policy = PI0Policy(config)
+    preprocessor, postprocessor = make_pi0_pre_post_processors(config=config, dataset_stats=dataset_stats)
+    # Test forward pass with dummy data
+    batch_size = 1
+    device = config.device
+    batch = {
+        "observation.state": torch.randn(batch_size, 14, dtype=torch.float32, device=device),
+        "action": torch.randn(batch_size, config.chunk_size, 7, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),  # Use rand for [0,1] range
+        "task": ["Pick up the object"] * batch_size,
+    }
+    batch = preprocessor(batch)
+    try:
+        loss, loss_dict = policy.forward(batch)
+        print(f"Forward pass successful. Loss: {loss_dict['loss']:.4f}")
+    except Exception as e:
+        print(f"Forward pass failed: {e}")
+        raise
+
+    try:
+        with torch.no_grad():
+            action = policy.select_action(batch)
+            action = postprocessor(action)
+            print(f"Action: {action}")
+        print(f"Action prediction successful. Action shape: {action.shape}")
+    except Exception as e:
+        print(f"Action prediction failed: {e}")
+        raise
+
+
+@require_cuda
+@require_hf_token
+def test_config_creation():
+    """Test policy config creation through factory."""
+    try:
+        config = make_policy_config(
+            policy_type="pi0",
+            max_action_dim=7,
+            max_state_dim=14,
+        )
+        print("Config created successfully through factory")
+        print(f"  Config type: {type(config).__name__}")
+        print(f"  PaliGemma variant: {config.paligemma_variant}")
+        print(f"  Action expert variant: {config.action_expert_variant}")
+    except Exception as e:
+        print(f"Config creation failed: {e}")
+        raise
diff --git a/lerobot/tests/policies/pi0_pi05/test_pi05.py b/lerobot/tests/policies/pi0_pi05/test_pi05.py
new file mode 100644
index 0000000000000000000000000000000000000000..f0da2971bac6176311b7ee7169403f2c97a8a735
--- /dev/null
+++ b/lerobot/tests/policies/pi0_pi05/test_pi05.py
@@ -0,0 +1,163 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test script to verify PI0.5 (pi05) support in PI0 policy"""
+
+import pytest
+import torch
+
+pytest.importorskip("transformers")
+
+from lerobot.policies.factory import make_policy_config  # noqa: E402
+from lerobot.policies.pi05 import (  # noqa: E402
+    PI05Config,
+    PI05Policy,
+    make_pi05_pre_post_processors,  # noqa: E402
+)
+from lerobot.utils.random_utils import set_seed
+from tests.utils import require_cuda, require_hf_token  # noqa: E402
+
+
+@require_cuda
+@require_hf_token
+def test_policy_instantiation():
+    # Create config
+    set_seed(42)
+    config = PI05Config(max_action_dim=7, max_state_dim=14, dtype="float32")
+
+    # Set up input_features and output_features in the config
+    from lerobot.configs.types import FeatureType, PolicyFeature
+
+    config.input_features = {
+        "observation.state": PolicyFeature(
+            type=FeatureType.STATE,
+            shape=(14,),
+        ),
+        "observation.images.base_0_rgb": PolicyFeature(
+            type=FeatureType.VISUAL,
+            shape=(3, 224, 224),
+        ),
+    }
+
+    config.output_features = {
+        "action": PolicyFeature(
+            type=FeatureType.ACTION,
+            shape=(7,),
+        ),
+    }
+
+    assert config.tokenizer_max_length == 200, (
+        f"Expected tokenizer_max_length=200 for pi05, got {config.tokenizer_max_length}"
+    )
+
+    # Create dummy dataset stats
+    dataset_stats = {
+        "observation.state": {
+            "mean": torch.zeros(14),
+            "std": torch.ones(14),
+            "min": torch.zeros(14),
+            "max": torch.ones(14),
+            "q01": torch.zeros(14),
+            "q99": torch.ones(14),
+        },
+        "action": {
+            "mean": torch.zeros(7),
+            "std": torch.ones(7),
+            "min": torch.zeros(7),
+            "max": torch.ones(7),
+            "q01": torch.zeros(7),
+            "q99": torch.ones(7),
+        },
+        "observation.images.base_0_rgb": {
+            "mean": torch.zeros(3, 224, 224),
+            "std": torch.ones(3, 224, 224),
+            "q01": torch.zeros(3, 224, 224),
+            "q99": torch.ones(3, 224, 224),
+        },
+    }
+
+    # Instantiate policy
+    policy = PI05Policy(config)
+    # Test forward pass with dummy data
+    batch_size = 1
+    preprocessor, postprocessor = make_pi05_pre_post_processors(config=config, dataset_stats=dataset_stats)
+    device = config.device
+    batch = {
+        "observation.state": torch.randn(batch_size, 14, dtype=torch.float32, device=device),
+        "action": torch.randn(batch_size, config.chunk_size, 7, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),  # Use rand for [0,1] range
+        "task": ["Pick up the object"] * batch_size,
+    }
+    batch = preprocessor(batch)
+    try:
+        loss, loss_dict = policy.forward(batch)
+        print(f"Forward pass successful. Loss: {loss_dict['loss']:.4f}")
+    except Exception as e:
+        print(f"Forward pass failed: {e}")
+        raise
+    try:
+        with torch.no_grad():
+            action = policy.select_action(batch)
+            action = postprocessor(action)
+            print(f"Action: {action}")
+        print(f"Action prediction successful. Action shape: {action.shape}")
+    except Exception as e:
+        print(f"Action prediction failed: {e}")
+        raise
+
+    # Verify pi05 model components exist
+    # Check that time_mlp layers exist (for AdaRMS conditioning)
+    assert hasattr(policy.model, "time_mlp_in"), "Missing time_mlp_in layer for pi05"
+    assert hasattr(policy.model, "time_mlp_out"), "Missing time_mlp_out layer for pi05"
+
+    # Check that action_time_mlp layers don't exist (pi0 only)
+    assert not hasattr(policy.model, "action_time_mlp_in"), "action_time_mlp_in should not exist in pi05 mode"
+    assert not hasattr(policy.model, "action_time_mlp_out"), (
+        "action_time_mlp_out should not exist in pi05 mode"
+    )
+
+    # Check that state_proj doesn't exist in pi05 mode
+    assert not hasattr(policy.model, "state_proj"), "state_proj should not exist in pi05 mode"
+
+    # Check AdaRMS configuration in the underlying model
+    adarms_config = policy.model.paligemma_with_expert.paligemma.config.text_config.use_adarms
+    assert adarms_config == False, f"PaliGemma should not use AdaRMS, got {adarms_config}"  # noqa: E712
+
+    adarms_expert_config = policy.model.paligemma_with_expert.gemma_expert.config.use_adarms
+    assert adarms_expert_config == True, (  # noqa: E712
+        f"Action expert should use AdaRMS in pi05, got {adarms_expert_config}"
+    )
+
+
+@require_cuda
+@require_hf_token
+def test_config_creation():
+    """Test policy config creation through factory."""
+    try:
+        config = make_policy_config(
+            policy_type="pi0",
+            max_action_dim=7,
+            max_state_dim=14,
+        )
+        print("Config created successfully through factory")
+        print(f"  Config type: {type(config).__name__}")
+        print(f"  PaliGemma variant: {config.paligemma_variant}")
+        print(f"  Action expert variant: {config.action_expert_variant}")
+    except Exception as e:
+        print(f"Config creation failed: {e}")
+        raise
diff --git a/lerobot/tests/policies/pi0_pi05/test_pi05_original_vs_lerobot.py b/lerobot/tests/policies/pi0_pi05/test_pi05_original_vs_lerobot.py
new file mode 100644
index 0000000000000000000000000000000000000000..a965132b022999a8a4126ad87874f1787be91a7a
--- /dev/null
+++ b/lerobot/tests/policies/pi0_pi05/test_pi05_original_vs_lerobot.py
@@ -0,0 +1,436 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test script to verify PI0OpenPI policy integration with LeRobot vs the original implementation"""
+
+import os
+from copy import deepcopy
+from typing import Any
+
+import numpy as np
+import pytest
+import torch
+
+# Skip if openpi or transformers is not available
+pytest.importorskip("openpi")
+pytest.importorskip("transformers")
+
+# Skip this entire module in CI
+pytestmark = pytest.mark.skipif(
+    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
+    reason="This test requires local OpenPI installation and is not meant for CI",
+)
+
+from openpi.models_pytorch import preprocessing_pytorch as openpi_preprocessing  # noqa: E402
+
+# NOTE: Assumes PYTHONPATH is set to include OpenPI src as per instructions.
+from openpi.models_pytorch.pi0_pytorch import PI0Pytorch  # noqa: E402
+from transformers import AutoTokenizer  # noqa: E402
+
+from lerobot.policies.pi05 import PI05Config, PI05Policy  # noqa: E402
+from lerobot.policies.pi05.processor_pi05 import make_pi05_pre_post_processors  # noqa: E402
+from lerobot.processor import PolicyProcessorPipeline  # noqa: E402
+from lerobot.types import PolicyAction  # noqa: E402
+
+# TODO: ADDING DEFAULT IMAGES_FEATURES TO CONFIG
+DUMMY_ACTION_DIM = 32
+DUMMY_STATE_DIM = 32
+DUMMY_ACTION_HORIZON = 50
+DUMMY_MAX_TOKEN_LEN = 200
+DEVICE = "cpu"  # Use CPU to avoid memory issues for testing
+
+DUMMY_DATASET_STATS = {
+    "observation.state": {
+        "mean": torch.zeros(DUMMY_STATE_DIM),
+        "std": torch.ones(DUMMY_STATE_DIM),
+        "q01": torch.zeros(DUMMY_STATE_DIM),
+        "q99": torch.ones(DUMMY_STATE_DIM),
+    },
+    "action": {
+        "mean": torch.zeros(DUMMY_ACTION_DIM),
+        "std": torch.ones(DUMMY_ACTION_DIM),
+        "q01": torch.zeros(DUMMY_ACTION_DIM),
+        "q99": torch.ones(DUMMY_ACTION_DIM),
+    },
+    "images": {
+        "base_0_rgb": {
+            "mean": torch.zeros(3, 224, 224),
+            "std": torch.ones(3, 224, 224),
+            "q01": torch.zeros(3, 224, 224),
+            "q99": torch.ones(3, 224, 224),
+        },
+        "left_wrist_0_rgb": {
+            "mean": torch.zeros(3, 224, 224),
+            "std": torch.ones(3, 224, 224),
+            "q01": torch.zeros(3, 224, 224),
+            "q99": torch.ones(3, 224, 224),
+        },
+        "right_wrist_0_rgb": {
+            "mean": torch.zeros(3, 224, 224),
+            "std": torch.ones(3, 224, 224),
+            "q01": torch.zeros(3, 224, 224),
+            "q99": torch.ones(3, 224, 224),
+        },
+    },
+}
+
+
+class PI05BaseOriginalConfig:
+    action_dim: int = DUMMY_ACTION_DIM
+    action_horizon: int = DUMMY_ACTION_HORIZON
+    paligemma_variant: str = "gemma_2b"
+    action_expert_variant: str = "gemma_300m"
+    precision: str = "float32"
+    pi05: bool = True
+    dtype: str = "float32"
+
+
+def instantiate_lerobot_pi05(
+    from_pretrained: bool = False,
+) -> tuple[
+    PI05Policy,
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    if from_pretrained:
+        # Load the policy first
+        policy = PI05Policy.from_pretrained(pretrained_name_or_path="lerobot/pi05_base", strict=True)
+    else:
+        config = PI05Config(max_action_dim=DUMMY_ACTION_DIM, max_state_dim=DUMMY_STATE_DIM, dtype="float32")
+        policy = PI05Policy(config)
+
+    policy.to(DEVICE)
+    policy.config.device = DEVICE
+    preprocessor, postprocessor = make_pi05_pre_post_processors(
+        config=policy.config, dataset_stats=DUMMY_DATASET_STATS
+    )
+    return (policy, preprocessor, postprocessor)
+
+
+def instantiate_original_pi05(from_pretrained: bool = False, model_path: str | None = None):
+    config = PI05BaseOriginalConfig()
+    policy = PI0Pytorch(config)
+
+    if from_pretrained:
+        try:
+            print("Loading converted PyTorch weights from HuggingFace Hub (lerobot/pi05_base)...")
+
+            # Download the model from HuggingFace Hub
+            import safetensors.torch
+            from huggingface_hub import snapshot_download
+
+            # Download the entire repository
+            if model_path and os.path.exists(model_path):
+                cache_dir = model_path
+                print(f"Using cached model from: {cache_dir}")
+            else:
+                cache_dir = snapshot_download(repo_id="lerobot/pi05_base", repo_type="model")
+                print(f"Downloaded model to: {cache_dir}")
+
+            # Try to load safetensors format first
+            model_file = os.path.join(cache_dir, "model.safetensors")
+            if os.path.exists(model_file):
+                state_dict = safetensors.torch.load_file(model_file)
+                print(f"Loaded {len(state_dict)} parameters from safetensors")
+            else:
+                raise FileNotFoundError(f"No safetensors file found in {cache_dir}")
+
+            # Load the state dict into the model
+            missing_keys, unexpected_keys = policy.load_state_dict(state_dict, strict=False)
+
+            if missing_keys:
+                print(f"Missing keys: {len(missing_keys)}")
+                if len(missing_keys) <= 5:
+                    for key in missing_keys:
+                        print(f"    - {key}")
+                else:
+                    for key in missing_keys[:5]:
+                        print(f"    - {key}")
+                    print(f"    ... and {len(missing_keys) - 5} more")
+
+            if unexpected_keys:
+                print(f"Unexpected keys: {len(unexpected_keys)}")
+                if len(unexpected_keys) <= 5:
+                    for key in unexpected_keys:
+                        print(f"    - {key}")
+                else:
+                    for key in unexpected_keys[:5]:
+                        print(f"    - {key}")
+                    print(f"    ... and {len(unexpected_keys) - 5} more")
+
+            if not missing_keys and not unexpected_keys:
+                print("All pretrained weights loaded successfully!")
+            else:
+                print("Pretrained weights loaded with some missing/unexpected keys (this may be normal)")
+
+        except Exception as e:
+            print(f"Failed to load pretrained weights: {e}")
+            print("   Using randomly initialized weights...")
+            import traceback
+
+            traceback.print_exc()
+
+    policy.to(DEVICE)
+    return policy
+
+
+def create_dummy_data():
+    batch_size = 2  # Reduce batch size for testing
+    device = DEVICE
+
+    # Use the exact same prompt for both implementations
+    prompt = "Pick up the red block and place it in the bin"
+
+    batch = {
+        "observation.state": torch.randn(batch_size, DUMMY_STATE_DIM, dtype=torch.float32, device=device),
+        "action": torch.randn(
+            batch_size, DUMMY_ACTION_HORIZON, DUMMY_ACTION_DIM, dtype=torch.float32, device=device
+        ),
+        # Create images in [0, 1] range as expected by LeRobot (will be converted to [-1, 1] internally)
+        "observation.images.base_0_rgb": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),
+        "observation.images.left_wrist_0_rgb": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),
+        "observation.images.right_wrist_0_rgb": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),
+        # Add the task prompt for LeRobot - provide as list with single element to trigger expansion
+        "task": [prompt for _ in range(batch_size)],
+    }
+    return batch
+
+
+def extract_lerobot_processed_inputs(lerobot_pi0, batch):
+    """Extract the exact same processed inputs that LeRobot uses internally."""
+    # Get the tokenized language from LeRobot's internal method
+    lang_tokens, lang_masks = lerobot_pi0._tokenize_language(batch)
+
+    # Get the preprocessed images from LeRobot's internal method
+    images, img_masks = lerobot_pi0._preprocess_images(batch, train=False)
+
+    # Create dummy token_ar_mask and token_loss_mask for original implementation
+    token_ar_mask = torch.zeros_like(lang_tokens, dtype=torch.int32)
+    token_loss_mask = torch.ones_like(lang_masks, dtype=torch.bool)
+
+    return images, img_masks, lang_tokens, lang_masks, token_ar_mask, token_loss_mask
+
+
+class PI05Observation:
+    """Observation class that matches the original OpenPI format."""
+
+    def __init__(
+        self,
+        state,
+        images,
+        image_masks,
+        tokenized_prompt,
+        tokenized_prompt_mask,
+        token_ar_mask,
+        token_loss_mask,
+    ):
+        self.state = state
+        self.images = images
+        self.image_masks = image_masks
+        self.tokenized_prompt = tokenized_prompt
+        self.tokenized_prompt_mask = tokenized_prompt_mask
+        self.token_ar_mask = token_ar_mask
+        self.token_loss_mask = token_loss_mask
+
+
+def create_original_observation_with_openpi_preprocessing(batch):
+    """Create observation object for OpenPI using OpenPI's own preprocessing with pi05 state tokenizer."""
+    batch_size = batch["observation.state"].shape[0]
+    device = batch["observation.state"].device
+
+    # Create tokenizer for OpenPI (same as LeRobot uses)
+    tokenizer = AutoTokenizer.from_pretrained("google/paligemma-3b-pt-224")
+
+    # Get task description (pi05 processor handles all text formatting)
+    tasks = batch.get("task", ["Pick up the object"] * batch_size)
+    if isinstance(tasks, str):
+        tasks = [tasks] * batch_size
+    elif len(tasks) == 1:
+        tasks = tasks * batch_size
+
+    # Use pi05 state and input tokenizer logic (same as Pi05PrepareStateTokenizerProcessorStep)
+    state = batch["observation.state"]
+    state = deepcopy(state)
+
+    # Prepare state (pad to max_state_dim)
+    from lerobot.policies.pi05.modeling_pi05 import pad_vector
+
+    state = pad_vector(state, DUMMY_STATE_DIM)
+
+    # Normalize state to [-1, 1] range if needed (assuming it's already normalized from normalize_inputs)
+    # Discretize into 256 bins (see openpi `PaligemmaTokenizer.tokenize()`)
+    state_np = state.cpu().numpy()
+    discretized_states = np.digitize(state_np, bins=np.linspace(-1, 1, 256 + 1)[:-1]) - 1
+
+    # Create pi05-formatted prompts that include state information
+    full_prompts = []
+    for i, task in enumerate(tasks):
+        cleaned_text = task.strip().replace("_", " ").replace("\n", " ")
+        state_str = " ".join(map(str, discretized_states[i]))
+        full_prompt = f"Task: {cleaned_text}, State: {state_str};\nAction: "
+        full_prompts.append(full_prompt)
+
+    # Tokenize with max_length padding to match OpenPI's expected format
+    tokenized = tokenizer(
+        full_prompts,
+        padding="max_length",
+        padding_side="right",
+        truncation=True,
+        max_length=DUMMY_MAX_TOKEN_LEN,
+        return_tensors="pt",
+    )
+
+    lang_tokens = tokenized["input_ids"].to(device)
+    lang_masks = tokenized["attention_mask"].to(device, dtype=torch.bool)
+
+    # Create dummy token_ar_mask and token_loss_mask for OpenPI
+    token_ar_mask = torch.zeros_like(lang_tokens, dtype=torch.int32)
+    token_loss_mask = torch.ones_like(lang_masks, dtype=torch.bool)
+
+    # Convert LeRobot images format to OpenPI format (convert [0,1] to [-1,1] range)
+    image_dict = {
+        "base_0_rgb": batch["observation.images.base_0_rgb"] * 2.0 - 1.0,
+        "left_wrist_0_rgb": batch["observation.images.left_wrist_0_rgb"] * 2.0 - 1.0,
+        "right_wrist_0_rgb": batch["observation.images.right_wrist_0_rgb"] * 2.0 - 1.0,
+    }
+
+    # Create image masks (all ones for real images)
+    image_masks_dict = {}
+    for key in image_dict:
+        image_masks_dict[key] = torch.ones(batch_size, dtype=torch.bool, device=device)
+
+    # Create raw observation object (before preprocessing)
+    raw_observation = PI05Observation(
+        state=batch["observation.state"],
+        images=image_dict,
+        image_masks=image_masks_dict,
+        tokenized_prompt=lang_tokens,
+        tokenized_prompt_mask=lang_masks,
+        token_ar_mask=token_ar_mask,
+        token_loss_mask=token_loss_mask,
+    )
+
+    # Now use OpenPI's preprocessing
+    processed_obs = openpi_preprocessing.preprocess_observation_pytorch(raw_observation, train=False)
+
+    return processed_obs
+
+
+def create_original_observation_from_lerobot(lerobot_pi0, batch):
+    """Create observation object compatible with original OpenPI using the exact same inputs as LeRobot."""
+    _batch_size = batch["observation.state"].shape[0]
+    _device = batch["observation.state"].device
+
+    # Extract the exact same processed inputs that LeRobot uses
+    images, img_masks, lang_tokens, lang_masks, token_ar_mask, token_loss_mask = (
+        extract_lerobot_processed_inputs(lerobot_pi0, batch)
+    )
+
+    # Convert images list to dict with original OpenPI keys
+    image_dict = {
+        "base_0_rgb": images[0],
+        "left_wrist_0_rgb": images[1],
+        "right_wrist_0_rgb": images[2],
+    }
+
+    # Convert image masks list to dict with original OpenPI keys
+    image_masks_dict = {
+        "base_0_rgb": img_masks[0],
+        "left_wrist_0_rgb": img_masks[1],
+        "right_wrist_0_rgb": img_masks[2],
+    }
+
+    return PI05Observation(
+        state=batch["observation.state"],
+        images=image_dict,
+        image_masks=image_masks_dict,
+        tokenized_prompt=lang_tokens,
+        tokenized_prompt_mask=lang_masks,
+        token_ar_mask=token_ar_mask,
+        token_loss_mask=token_loss_mask,
+    )
+
+
+def test_pi05_original_vs_lerobot():
+    """Test PI05 original implementation vs LeRobot implementation."""
+    print("Initializing models...")
+    lerobot_pi05, lerobot_preprocessor, lerobot_postprocessor = instantiate_lerobot_pi05(
+        from_pretrained=True
+    )  # Load pretrained LeRobot model
+    original_pi0 = instantiate_original_pi05(
+        from_pretrained=True
+    )  # Load pretrained OpenPI model from HuggingFace Hub
+
+    print("Creating dummy data...")
+    batch = create_dummy_data()
+    batch_lerobot = deepcopy(batch)
+
+    # Test each model with its own preprocessing (more realistic end-to-end test)
+    print("\nTest each model with its own preprocessing")
+    print("Creating observation for OpenPI using OpenPI's own preprocessing...")
+    pi0_obs_openpi = create_original_observation_with_openpi_preprocessing(batch)
+
+    print(f"Task prompt: '{batch['task'][0]}'")
+    print(f"OpenPI tokenized prompt shape: {pi0_obs_openpi.tokenized_prompt.shape}")
+    print(f"OpenPI image shapes: {[img.shape for img in pi0_obs_openpi.images.values()]}")
+    print(f"OpenPI state shape: {pi0_obs_openpi.state.shape}")
+
+    print("Testing OpenPI with own preprocessing...")
+    original_pi0.eval()
+    torch.manual_seed(42)  # Set seed for reproducibility
+    batch_size = batch["observation.state"].shape[0]
+    noise_shape = (batch_size, DUMMY_ACTION_HORIZON, DUMMY_ACTION_DIM)
+    fixed_noise = torch.randn(noise_shape, dtype=torch.float32, device=DEVICE)
+
+    with torch.no_grad():
+        openpi_actions = original_pi0.sample_actions(
+            device=DEVICE, observation=pi0_obs_openpi, noise=fixed_noise, num_steps=10
+        )
+        openpi_actions_unit = openpi_actions[:, 0, :]
+    print(f"OpenPI (own preprocessing) Actions shape: {openpi_actions.shape}")
+    print(f"OpenPI (own preprocessing) Actions unit shape: {openpi_actions_unit.shape}")
+    print(f"OpenPI (own preprocessing) Actions mean: {openpi_actions.mean().item():.6f}")
+    print(f"OpenPI (own preprocessing) Actions std: {openpi_actions.std().item():.6f}")
+
+    print("Testing LeRobot with own preprocessing...")
+    lerobot_pi05.eval()
+    torch.manual_seed(42)  # Set the same seed
+
+    batch_lerobot_processed = lerobot_preprocessor(batch_lerobot)
+    with torch.no_grad():
+        lerobot_actions_own = lerobot_pi05.predict_action_chunk(
+            batch_lerobot_processed
+        )  # batch_size, n_action_steps, action_dim
+        lerobot_actions_unit = lerobot_actions_own[:, 0, :]
+    print(f"LeRobot (own preprocessing) Actions shape: {lerobot_actions_own.shape}")
+    print(f"LeRobot (own preprocessing) Actions unit shape: {lerobot_actions_unit.shape}")
+    print(f"LeRobot (own preprocessing) Actions mean: {lerobot_actions_own.mean().item():.6f}")
+    print(f"LeRobot (own preprocessing) Actions std: {lerobot_actions_own.std().item():.6f}")
+
+    print("\nComparing end-to-end implementations:")
+    print(f"Actions close (atol=1e-4): {torch.allclose(lerobot_actions_own, openpi_actions, atol=1e-4)}")
+    print(f"Actions close (atol=1e-2): {torch.allclose(lerobot_actions_own, openpi_actions, atol=1e-2)}")
+    print(f"Max absolute difference: {torch.abs(lerobot_actions_own - openpi_actions).max().item():.6f}")
+
+    assert torch.allclose(lerobot_actions_own, openpi_actions, atol=1e-4)
+    assert torch.allclose(lerobot_actions_own, openpi_actions, atol=1e-2)
+    assert torch.abs(lerobot_actions_own - openpi_actions).max().item() < 1e-4
diff --git a/lerobot/tests/policies/pi0_pi05/test_pi05_rtc.py b/lerobot/tests/policies/pi0_pi05/test_pi05_rtc.py
new file mode 100644
index 0000000000000000000000000000000000000000..0dc240638bf788984845bee9a84845f5c9207989
--- /dev/null
+++ b/lerobot/tests/policies/pi0_pi05/test_pi05_rtc.py
@@ -0,0 +1,337 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test PI0.5 policy with Real-Time Chunking (RTC) enabled during inference."""
+
+import os
+
+import pytest
+import torch
+
+# Skip this entire module in CI
+pytestmark = pytest.mark.skipif(
+    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
+    reason="TODO: This test seems to hang the CI",
+)
+
+
+from lerobot.configs.types import FeatureType, PolicyFeature, RTCAttentionSchedule  # noqa: E402
+from lerobot.policies.pi05 import PI05Config, PI05Policy, make_pi05_pre_post_processors  # noqa: E402
+from lerobot.policies.rtc.configuration_rtc import RTCConfig  # noqa: E402
+from lerobot.utils.random_utils import set_seed  # noqa: E402
+from tests.utils import require_cuda  # noqa: E402
+
+
+@require_cuda
+def test_pi05_rtc_initialization():
+    """Test PI0.5 policy can initialize RTC processor."""
+    set_seed(42)
+
+    config = PI05Config(max_action_dim=7, max_state_dim=14, dtype="float32")
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Instantiate policy
+    policy = PI05Policy(config)
+
+    # Verify RTC processor is initialized
+    assert hasattr(policy, "rtc_processor")
+    assert policy.rtc_processor is not None
+    assert policy.rtc_processor.rtc_config.enabled is True
+
+    print("✓ PI0.5 RTC initialization: Test passed")
+
+
+@require_cuda
+def test_pi05_rtc_initialization_without_rtc_config():
+    """Test PI0.5 policy can initialize without RTC config."""
+    set_seed(42)
+
+    config = PI05Config(max_action_dim=7, max_state_dim=14, dtype="float32")
+
+    # Instantiate policy
+    policy = PI05Policy(config)
+
+    # Verify RTC processor is not initialized
+    assert hasattr(policy, "rtc_processor")
+    assert policy.rtc_processor is None
+    assert policy.model.rtc_processor is None
+    assert policy._rtc_enabled() is False
+
+    print("✓ PI0.5 RTC initialization without RTC config: Test passed")
+
+
+@require_cuda
+def test_pi05_rtc_inference_with_prev_chunk():
+    """Test PI0.5 policy inference with RTC and previous chunk."""
+    set_seed(42)
+
+    config = PI05Config(max_action_dim=7, max_state_dim=14, chunk_size=50, dtype="float32")
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats (PI0.5 uses QUANTILES normalization)
+    dataset_stats = {
+        "observation.state": {
+            "mean": torch.zeros(14),
+            "std": torch.ones(14),
+            "q01": -torch.ones(14),
+            "q99": torch.ones(14),
+        },
+        "action": {
+            "mean": torch.zeros(7),
+            "std": torch.ones(7),
+            "q01": -torch.ones(7),
+            "q99": torch.ones(7),
+        },
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    # Instantiate policy and preprocessor
+    policy = PI05Policy(config)
+    policy.eval()
+    preprocessor, _ = make_pi05_pre_post_processors(config=config, dataset_stats=dataset_stats)
+
+    device = config.device
+
+    # Create dummy batch
+    batch = {
+        "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+        "task": ["Pick up the object"],
+    }
+    batch = preprocessor(batch)
+
+    # Create previous chunk
+    prev_chunk = torch.randn(1, 25, 7, dtype=torch.float32, device=device)
+
+    with torch.no_grad():
+        # Use same noise for fair comparison
+        noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+
+        # Test with RTC and previous chunk
+        actions_with_rtc = policy.predict_action_chunk(
+            batch,
+            noise=noise.clone(),
+            prev_chunk_left_over=prev_chunk,
+            inference_delay=4,
+            execution_horizon=10,
+        )
+
+        # Test without RTC for comparison
+        policy.config.rtc_config.enabled = False
+        actions_without_rtc = policy.predict_action_chunk(batch, noise=noise.clone())
+        policy.config.rtc_config.enabled = True
+
+    # Verify shapes
+    assert actions_with_rtc.shape == (1, config.chunk_size, 7)
+    assert actions_without_rtc.shape == (1, config.chunk_size, 7)
+
+    # With previous chunk, actions should be different (RTC guidance applied)
+    assert not torch.allclose(actions_with_rtc, actions_without_rtc, rtol=1e-3)
+
+    print("✓ PI0.5 RTC inference with prev_chunk: Test passed")
+
+
+@require_cuda
+def test_pi05_rtc_inference_without_prev_chunk():
+    """Test PI0.5 policy inference with RTC but no previous chunk (RTC should have no effect)."""
+    set_seed(42)
+
+    config = PI05Config(max_action_dim=7, max_state_dim=14, chunk_size=50, dtype="float32")
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats (PI0.5 uses QUANTILES normalization)
+    dataset_stats = {
+        "observation.state": {
+            "mean": torch.zeros(14),
+            "std": torch.ones(14),
+            "q01": -torch.ones(14),
+            "q99": torch.ones(14),
+        },
+        "action": {
+            "mean": torch.zeros(7),
+            "std": torch.ones(7),
+            "q01": -torch.ones(7),
+            "q99": torch.ones(7),
+        },
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    # Instantiate policy and preprocessor
+    policy = PI05Policy(config)
+    policy.eval()
+    preprocessor, _ = make_pi05_pre_post_processors(config=config, dataset_stats=dataset_stats)
+
+    device = config.device
+
+    # Create dummy batch
+    batch = {
+        "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+        "task": ["Pick up the object"],
+    }
+    batch = preprocessor(batch)
+
+    with torch.no_grad():
+        # Use same noise for fair comparison
+        noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+
+        # Test with RTC enabled but no previous chunk
+        actions_with_rtc_no_prev = policy.predict_action_chunk(
+            batch,
+            noise=noise.clone(),
+            prev_chunk_left_over=None,
+        )
+
+        # Test without RTC
+        policy.config.rtc_config.enabled = False
+        actions_without_rtc = policy.predict_action_chunk(batch, noise=noise.clone())
+        policy.config.rtc_config.enabled = True
+
+    # Without previous chunk, RTC should have no effect
+    assert torch.allclose(actions_with_rtc_no_prev, actions_without_rtc, rtol=1e-5)
+
+    print("✓ PI0.5 RTC inference without prev_chunk: Test passed")
+
+
+@require_cuda
+def test_pi05_rtc_validation_rules():
+    """Test PI0.5 policy with RTC follows all three validation rules."""
+    set_seed(42)
+
+    config = PI05Config(max_action_dim=7, max_state_dim=14, chunk_size=50, dtype="float32")
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats (PI0.5 uses QUANTILES normalization)
+    dataset_stats = {
+        "observation.state": {
+            "mean": torch.zeros(14),
+            "std": torch.ones(14),
+            "q01": -torch.ones(14),
+            "q99": torch.ones(14),
+        },
+        "action": {
+            "mean": torch.zeros(7),
+            "std": torch.ones(7),
+            "q01": -torch.ones(7),
+            "q99": torch.ones(7),
+        },
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    # Instantiate policy and preprocessor
+    policy = PI05Policy(config)
+    policy.eval()
+    preprocessor, _ = make_pi05_pre_post_processors(config=config, dataset_stats=dataset_stats)
+
+    device = config.device
+
+    # Create dummy batch
+    batch = {
+        "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+        "task": ["Pick up the object"],
+    }
+    batch = preprocessor(batch)
+
+    # Create previous chunk
+    prev_chunk = torch.randn(1, 25, 7, dtype=torch.float32, device=device)
+
+    inference_delay = 4
+    execution_horizon = 10
+
+    with torch.no_grad():
+        # Use same noise for fair comparison
+        noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+
+        # Test with RTC
+        actions_with_rtc = policy.predict_action_chunk(
+            batch,
+            noise=noise.clone(),
+            prev_chunk_left_over=prev_chunk,
+            inference_delay=inference_delay,
+            execution_horizon=execution_horizon,
+        )
+
+        # Test without RTC
+        policy.config.rtc_config.enabled = False
+        actions_without_rtc = policy.predict_action_chunk(batch, noise=noise.clone())
+        policy.config.rtc_config.enabled = True
+
+    assert not torch.allclose(actions_with_rtc, actions_without_rtc, rtol=1e-3)
diff --git a/lerobot/tests/policies/pi0_pi05/test_pi0_original_vs_lerobot.py b/lerobot/tests/policies/pi0_pi05/test_pi0_original_vs_lerobot.py
new file mode 100644
index 0000000000000000000000000000000000000000..62e34b70d64a08f0c2cb6d7bba191cef8a1f1626
--- /dev/null
+++ b/lerobot/tests/policies/pi0_pi05/test_pi0_original_vs_lerobot.py
@@ -0,0 +1,427 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test script to verify PI0 policy integration with LeRobot vs the original implementation"""
+
+import os
+from copy import deepcopy
+from typing import Any
+
+import pytest
+import torch
+
+# Skip if openpi or transformers is not available
+pytest.importorskip("openpi")
+pytest.importorskip("transformers")
+
+# Skip this entire module in CI
+pytestmark = pytest.mark.skipif(
+    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
+    reason="This test requires local OpenPI installation and is not meant for CI",
+)
+
+from openpi.models_pytorch import preprocessing_pytorch as openpi_preprocessing  # noqa: E402
+
+# NOTE: Assumes PYTHONPATH is set to include OpenPI src as per instructions.
+from openpi.models_pytorch.pi0_pytorch import PI0Pytorch  # noqa: E402
+from transformers import AutoTokenizer  # noqa: E402
+
+from lerobot.policies.pi0 import PI0Config, PI0Policy  # noqa: E402
+from lerobot.policies.pi0.processor_pi0 import make_pi0_pre_post_processors  # noqa: E402
+from lerobot.processor import PolicyProcessorPipeline  # noqa: E402
+from lerobot.types import PolicyAction  # noqa: E402
+
+# TODO: ADDING DEFAULT IMAGES_FEATURES TO CONFIG
+DUMMY_ACTION_DIM = 32
+DUMMY_STATE_DIM = 32
+DUMMY_ACTION_HORIZON = 50
+DUMMY_MAX_TOKEN_LEN = 48  # Default for PI0 (non-pi05)
+DEVICE = "cpu"  # Use CPU to avoid memory issues for testing
+
+DUMMY_DATASET_STATS = {
+    "observation.state": {
+        "mean": torch.zeros(DUMMY_STATE_DIM),
+        "std": torch.ones(DUMMY_STATE_DIM),
+        "q01": torch.zeros(DUMMY_STATE_DIM),
+        "q99": torch.ones(DUMMY_STATE_DIM),
+    },
+    "action": {
+        "mean": torch.zeros(DUMMY_ACTION_DIM),
+        "std": torch.ones(DUMMY_ACTION_DIM),
+        "q01": torch.zeros(DUMMY_ACTION_DIM),
+        "q99": torch.ones(DUMMY_ACTION_DIM),
+    },
+    "images": {
+        "base_0_rgb": {
+            "mean": torch.zeros(3, 224, 224),
+            "std": torch.ones(3, 224, 224),
+            "q01": torch.zeros(3, 224, 224),
+            "q99": torch.ones(3, 224, 224),
+        },
+        "left_wrist_0_rgb": {
+            "mean": torch.zeros(3, 224, 224),
+            "std": torch.ones(3, 224, 224),
+            "q01": torch.zeros(3, 224, 224),
+            "q99": torch.ones(3, 224, 224),
+        },
+        "right_wrist_0_rgb": {
+            "mean": torch.zeros(3, 224, 224),
+            "std": torch.ones(3, 224, 224),
+            "q01": torch.zeros(3, 224, 224),
+            "q99": torch.ones(3, 224, 224),
+        },
+    },
+}
+
+
+class PI0BaseOriginalConfig:
+    action_dim: int = DUMMY_ACTION_DIM
+    action_horizon: int = DUMMY_ACTION_HORIZON
+    paligemma_variant: str = "gemma_2b"
+    action_expert_variant: str = "gemma_300m"
+    precision: str = "float32"
+    pi05: bool = False
+    dtype: str = "float32"
+
+
+def instantiate_lerobot_pi0(
+    from_pretrained: bool = False,
+) -> tuple[
+    PI0Policy,
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    if from_pretrained:
+        # Load the policy first
+        policy = PI0Policy.from_pretrained(pretrained_name_or_path="lerobot/pi0_base", strict=True)
+    else:
+        config = PI0Config(max_action_dim=DUMMY_ACTION_DIM, max_state_dim=DUMMY_STATE_DIM, dtype="float32")
+        policy = PI0Policy(config)
+
+    policy.to(DEVICE)
+    policy.config.device = DEVICE
+    preprocessor, postprocessor = make_pi0_pre_post_processors(
+        config=policy.config, dataset_stats=DUMMY_DATASET_STATS
+    )
+    return (policy, preprocessor, postprocessor)
+
+
+def instantiate_original_pi0(from_pretrained: bool = False, model_path: str = None):
+    config = PI0BaseOriginalConfig()
+    policy = PI0Pytorch(config)
+
+    if from_pretrained:
+        try:
+            print("Loading converted PyTorch weights from HuggingFace Hub (lerobot/pi0_base)...")
+
+            # Download the model from HuggingFace Hub
+            import safetensors.torch
+            from huggingface_hub import snapshot_download
+
+            # Download the entire repository
+            if model_path and os.path.exists(model_path):
+                cache_dir = model_path
+                print(f"Using cached model from: {cache_dir}")
+            else:
+                cache_dir = snapshot_download(repo_id="lerobot/pi0_base", repo_type="model")
+                print(f"Downloaded model to: {cache_dir}")
+
+            # Try to load safetensors format first
+            model_file = os.path.join(cache_dir, "model.safetensors")
+            if os.path.exists(model_file):
+                state_dict = safetensors.torch.load_file(model_file)
+                print(f"Loaded {len(state_dict)} parameters from safetensors")
+            else:
+                raise FileNotFoundError(f"No safetensors file found in {cache_dir}")
+
+            # Load the state dict into the model
+            missing_keys, unexpected_keys = policy.load_state_dict(state_dict, strict=False)
+
+            if missing_keys:
+                print(f"Missing keys: {len(missing_keys)}")
+                if len(missing_keys) <= 5:
+                    for key in missing_keys:
+                        print(f"    - {key}")
+                else:
+                    for key in missing_keys[:5]:
+                        print(f"    - {key}")
+                    print(f"    ... and {len(missing_keys) - 5} more")
+
+            if unexpected_keys:
+                print(f"Unexpected keys: {len(unexpected_keys)}")
+                if len(unexpected_keys) <= 5:
+                    for key in unexpected_keys:
+                        print(f"    - {key}")
+                else:
+                    for key in unexpected_keys[:5]:
+                        print(f"    - {key}")
+                    print(f"    ... and {len(unexpected_keys) - 5} more")
+
+            if not missing_keys and not unexpected_keys:
+                print("All pretrained weights loaded successfully!")
+            else:
+                print("Pretrained weights loaded with some missing/unexpected keys (this may be normal)")
+
+        except Exception as e:
+            print(f"Failed to load pretrained weights: {e}")
+            print("   Using randomly initialized weights...")
+            import traceback
+
+            traceback.print_exc()
+
+    policy.to(DEVICE)
+    return policy
+
+
+def create_dummy_data():
+    batch_size = 2  # Reduce batch size for testing
+    device = DEVICE
+
+    # Use the exact same prompt for both implementations
+    prompt = "Pick up the red block and place it in the bin"
+
+    batch = {
+        "observation.state": torch.randn(batch_size, DUMMY_STATE_DIM, dtype=torch.float32, device=device),
+        "action": torch.randn(
+            batch_size, DUMMY_ACTION_HORIZON, DUMMY_ACTION_DIM, dtype=torch.float32, device=device
+        ),
+        # Create images in [0, 1] range as expected by LeRobot (will be converted to [-1, 1] internally)
+        "observation.images.base_0_rgb": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),
+        "observation.images.left_wrist_0_rgb": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),
+        "observation.images.right_wrist_0_rgb": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),
+        # Add the task prompt for LeRobot - provide as list with single element to trigger expansion
+        "task": [prompt for _ in range(batch_size)],
+    }
+    return batch
+
+
+def extract_lerobot_processed_inputs(lerobot_pi0, batch):
+    """Extract the exact same processed inputs that LeRobot uses internally."""
+    # Get the tokenized language from LeRobot's internal method
+    lang_tokens, lang_masks = lerobot_pi0._tokenize_language(batch)
+
+    # Get the preprocessed images from LeRobot's internal method
+    images, img_masks = lerobot_pi0._preprocess_images(batch, train=False)
+
+    # Create dummy token_ar_mask and token_loss_mask for original implementation
+    token_ar_mask = torch.zeros_like(lang_tokens, dtype=torch.int32)
+    token_loss_mask = torch.ones_like(lang_masks, dtype=torch.bool)
+
+    return images, img_masks, lang_tokens, lang_masks, token_ar_mask, token_loss_mask
+
+
+class PI0Observation:
+    """Observation class that matches the original OpenPI format."""
+
+    def __init__(
+        self,
+        state,
+        images,
+        image_masks,
+        tokenized_prompt,
+        tokenized_prompt_mask,
+        token_ar_mask,
+        token_loss_mask,
+    ):
+        self.state = state
+        self.images = images
+        self.image_masks = image_masks
+        self.tokenized_prompt = tokenized_prompt
+        self.tokenized_prompt_mask = tokenized_prompt_mask
+        self.token_ar_mask = token_ar_mask
+        self.token_loss_mask = token_loss_mask
+
+
+def create_original_observation_with_openpi_preprocessing(batch):
+    """Create observation object for OpenPI using OpenPI's own preprocessing."""
+    batch_size = batch["observation.state"].shape[0]
+    device = batch["observation.state"].device
+
+    # Create tokenizer for OpenPI (same as LeRobot uses)
+    tokenizer = AutoTokenizer.from_pretrained("google/paligemma-3b-pt-224")
+
+    # Get task description
+    if "task" in batch:
+        tasks = batch["task"]
+        if isinstance(tasks, str):
+            # Single string: add newline if not present, then convert to list
+            if not tasks.endswith("\n"):
+                tasks = f"{tasks}\n"
+            tasks = [tasks]
+        elif isinstance(tasks, list) and all(isinstance(t, str) for t in tasks):
+            # List of strings: add newline to each if not present
+            tasks = [t if t.endswith("\n") else f"{t}\n" for t in tasks]
+            if len(tasks) == 1:
+                # Expand to batch size
+                tasks = tasks * batch_size
+                if len(tasks) != batch_size:
+                    raise ValueError(f"Expected batch size {batch_size}, got {len(tasks)}")
+        # If task is neither string nor list of strings, leave unchanged
+    else:
+        # Default task if not provided
+        tasks = ["Pick up the object\n"] * batch_size
+
+    # Tokenize with max_length padding to match OpenPI's expected format
+    tokenized = tokenizer(
+        tasks,
+        padding="max_length",
+        padding_side="right",
+        truncation=True,
+        max_length=DUMMY_MAX_TOKEN_LEN,
+        return_tensors="pt",
+    )
+
+    lang_tokens = tokenized["input_ids"].to(device)
+    lang_masks = tokenized["attention_mask"].to(device, dtype=torch.bool)
+
+    # Create dummy token_ar_mask and token_loss_mask for OpenPI
+    token_ar_mask = torch.zeros_like(lang_tokens, dtype=torch.int32)
+    token_loss_mask = torch.ones_like(lang_masks, dtype=torch.bool)
+
+    # Convert LeRobot images format to OpenPI format (convert [0,1] to [-1,1] range)
+    image_dict = {
+        "base_0_rgb": batch["observation.images.base_0_rgb"] * 2.0 - 1.0,
+        "left_wrist_0_rgb": batch["observation.images.left_wrist_0_rgb"] * 2.0 - 1.0,
+        "right_wrist_0_rgb": batch["observation.images.right_wrist_0_rgb"] * 2.0 - 1.0,
+    }
+
+    # Create image masks (all ones for real images)
+    image_masks_dict = {}
+    for key in image_dict:
+        image_masks_dict[key] = torch.ones(batch_size, dtype=torch.bool, device=device)
+
+    # Create raw observation object (before preprocessing)
+    raw_observation = PI0Observation(
+        state=batch["observation.state"],
+        images=image_dict,
+        image_masks=image_masks_dict,
+        tokenized_prompt=lang_tokens,
+        tokenized_prompt_mask=lang_masks,
+        token_ar_mask=token_ar_mask,
+        token_loss_mask=token_loss_mask,
+    )
+
+    # Now use OpenPI's preprocessing
+    processed_obs = openpi_preprocessing.preprocess_observation_pytorch(raw_observation, train=False)
+
+    return processed_obs
+
+
+def create_original_observation_from_lerobot(lerobot_pi0, batch):
+    """Create observation object compatible with original OpenPI using the exact same inputs as LeRobot."""
+    _batch_size = batch["observation.state"].shape[0]
+    _device = batch["observation.state"].device
+
+    # Extract the exact same processed inputs that LeRobot uses
+    images, img_masks, lang_tokens, lang_masks, token_ar_mask, token_loss_mask = (
+        extract_lerobot_processed_inputs(lerobot_pi0, batch)
+    )
+
+    # Convert images list to dict with original OpenPI keys
+    image_dict = {
+        "base_0_rgb": images[0],
+        "left_wrist_0_rgb": images[1],
+        "right_wrist_0_rgb": images[2],
+    }
+
+    # Convert image masks list to dict with original OpenPI keys
+    image_masks_dict = {
+        "base_0_rgb": img_masks[0],
+        "left_wrist_0_rgb": img_masks[1],
+        "right_wrist_0_rgb": img_masks[2],
+    }
+
+    return PI0Observation(
+        state=batch["observation.state"],
+        images=image_dict,
+        image_masks=image_masks_dict,
+        tokenized_prompt=lang_tokens,
+        tokenized_prompt_mask=lang_masks,
+        token_ar_mask=token_ar_mask,
+        token_loss_mask=token_loss_mask,
+    )
+
+
+def test_pi0_original_vs_lerobot():
+    """Test PI0 original implementation vs LeRobot implementation."""
+    print("Initializing models...")
+    lerobot_pi0, lerobot_preprocessor, lerobot_postprocessor = instantiate_lerobot_pi0(
+        from_pretrained=True
+    )  # Load pretrained LeRobot model
+    original_pi0 = instantiate_original_pi0(
+        from_pretrained=True
+    )  # Load pretrained OpenPI model from HuggingFace Hub
+
+    print("Creating dummy data...")
+    batch = create_dummy_data()
+    batch_lerobot = deepcopy(batch)
+
+    # Test each model with its own preprocessing (more realistic end-to-end test)
+    print("\nTest each model with its own preprocessing")
+    print("Creating observation for OpenPI using OpenPI's own preprocessing...")
+    pi0_obs_openpi = create_original_observation_with_openpi_preprocessing(batch)
+
+    print(f"Task prompt: '{batch['task'][0]}'")
+    print(f"OpenPI tokenized prompt shape: {pi0_obs_openpi.tokenized_prompt.shape}")
+    print(f"OpenPI image shapes: {[img.shape for img in pi0_obs_openpi.images.values()]}")
+    print(f"OpenPI state shape: {pi0_obs_openpi.state.shape}")
+
+    print("Testing OpenPI with own preprocessing...")
+    original_pi0.eval()
+    torch.manual_seed(42)  # Set seed for reproducibility
+    batch_size = batch["observation.state"].shape[0]
+    noise_shape = (batch_size, DUMMY_ACTION_HORIZON, DUMMY_ACTION_DIM)
+    fixed_noise = torch.randn(noise_shape, dtype=torch.float32, device=DEVICE)
+
+    with torch.no_grad():
+        openpi_actions = original_pi0.sample_actions(
+            device=DEVICE, observation=pi0_obs_openpi, noise=fixed_noise, num_steps=10
+        )
+        openpi_actions_unit = openpi_actions[:, 0, :]
+    print(f"OpenPI (own preprocessing) Actions shape: {openpi_actions.shape}")
+    print(f"OpenPI (own preprocessing) Actions unit shape: {openpi_actions_unit.shape}")
+    print(f"OpenPI (own preprocessing) Actions mean: {openpi_actions.mean().item():.6f}")
+    print(f"OpenPI (own preprocessing) Actions std: {openpi_actions.std().item():.6f}")
+
+    print("Testing LeRobot with own preprocessing...")
+    lerobot_pi0.eval()
+    torch.manual_seed(42)  # Set the same seed
+
+    batch_lerobot_processed = lerobot_preprocessor(batch_lerobot)
+    with torch.no_grad():
+        lerobot_actions_own = lerobot_pi0.predict_action_chunk(
+            batch_lerobot_processed
+        )  # batch_size, n_action_steps, action_dim
+        lerobot_actions_unit = lerobot_actions_own[:, 0, :]
+    print(f"LeRobot (own preprocessing) Actions shape: {lerobot_actions_own.shape}")
+    print(f"LeRobot (own preprocessing) Actions unit shape: {lerobot_actions_unit.shape}")
+    print(f"LeRobot (own preprocessing) Actions mean: {lerobot_actions_own.mean().item():.6f}")
+    print(f"LeRobot (own preprocessing) Actions std: {lerobot_actions_own.std().item():.6f}")
+
+    print("\nComparing end-to-end implementations:")
+    print(f"Actions close (atol=1e-4): {torch.allclose(lerobot_actions_own, openpi_actions, atol=1e-4)}")
+    print(f"Actions close (atol=1e-2): {torch.allclose(lerobot_actions_own, openpi_actions, atol=1e-2)}")
+    print(f"Max absolute difference: {torch.abs(lerobot_actions_own - openpi_actions).max().item():.6f}")
+
+    assert torch.allclose(lerobot_actions_own, openpi_actions, atol=1e-4)
+    assert torch.allclose(lerobot_actions_own, openpi_actions, atol=1e-2)
+    assert torch.abs(lerobot_actions_own - openpi_actions).max().item() < 1e-4
diff --git a/lerobot/tests/policies/pi0_pi05/test_pi0_rtc.py b/lerobot/tests/policies/pi0_pi05/test_pi0_rtc.py
new file mode 100644
index 0000000000000000000000000000000000000000..4105e2068a96e31c23c6bba43a9f57fe6f3cff54
--- /dev/null
+++ b/lerobot/tests/policies/pi0_pi05/test_pi0_rtc.py
@@ -0,0 +1,380 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test PI0 policy with Real-Time Chunking (RTC) enabled during inference."""
+
+import os
+
+import pytest
+import torch
+
+# Skip this entire module in CI
+pytestmark = pytest.mark.skipif(
+    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
+    reason="TODO: This test seems to hang the CI",
+)
+
+
+from lerobot.configs.types import FeatureType, PolicyFeature, RTCAttentionSchedule  # noqa: E402
+from lerobot.policies.pi0 import PI0Config, PI0Policy, make_pi0_pre_post_processors  # noqa: E402
+from lerobot.policies.rtc.configuration_rtc import RTCConfig  # noqa: E402
+from lerobot.utils.random_utils import set_seed  # noqa: E402
+from tests.utils import require_cuda  # noqa: E402
+
+
+@require_cuda
+def test_pi0_rtc_initialization():
+    """Test PI0 policy can initialize RTC processor."""
+    set_seed(42)
+
+    config = PI0Config(max_action_dim=7, max_state_dim=14, dtype="float32")
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Instantiate policy
+    policy = PI0Policy(config)
+
+    # Verify RTC processor is initialized
+    assert hasattr(policy, "rtc_processor")
+    assert policy.rtc_processor is not None
+    assert policy.rtc_processor.rtc_config.enabled is True
+
+    print("✓ PI0 RTC initialization: Test passed")
+
+
+@require_cuda
+def test_pi0_rtc_initialization_without_rtc_config():
+    """Test PI0 policy can initialize without RTC config."""
+    set_seed(42)
+
+    config = PI0Config(max_action_dim=7, max_state_dim=14, dtype="float32")
+
+    # Instantiate policy
+    policy = PI0Policy(config)
+
+    # Verify RTC processor is not initialized
+    assert hasattr(policy, "rtc_processor")
+    assert policy.rtc_processor is None
+    assert policy.model.rtc_processor is None
+    assert policy._rtc_enabled() is False
+
+    print("✓ PI0 RTC initialization without RTC config: Test passed")
+
+
+@require_cuda
+def test_pi0_rtc_inference_with_prev_chunk():
+    """Test PI0 policy inference with RTC and previous chunk."""
+    set_seed(42)
+
+    config = PI0Config(max_action_dim=7, max_state_dim=14, chunk_size=50, dtype="float32")
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats
+    dataset_stats = {
+        "observation.state": {"mean": torch.zeros(14), "std": torch.ones(14)},
+        "action": {"mean": torch.zeros(7), "std": torch.ones(7)},
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    # Instantiate policy and preprocessor
+    policy = PI0Policy(config)
+    policy.eval()
+    preprocessor, _ = make_pi0_pre_post_processors(config=config, dataset_stats=dataset_stats)
+
+    device = config.device
+
+    # Create dummy batch
+    batch = {
+        "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+        "task": ["Pick up the object"],
+    }
+    batch = preprocessor(batch)
+
+    # Create previous chunk
+    prev_chunk = torch.randn(1, 25, 7, dtype=torch.float32, device=device)
+
+    with torch.no_grad():
+        # Use same noise for fair comparison
+        noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+
+        # Test with RTC and previous chunk
+        actions_with_rtc = policy.predict_action_chunk(
+            batch,
+            noise=noise.clone(),
+            prev_chunk_left_over=prev_chunk,
+            inference_delay=4,
+            execution_horizon=10,
+        )
+
+        # Test without RTC for comparison
+        policy.config.rtc_config.enabled = False
+        actions_without_rtc = policy.predict_action_chunk(batch, noise=noise.clone())
+        policy.config.rtc_config.enabled = True
+
+    # Verify shapes
+    assert actions_with_rtc.shape == (1, config.chunk_size, 7)
+    assert actions_without_rtc.shape == (1, config.chunk_size, 7)
+
+    # With previous chunk, actions should be different (RTC guidance applied)
+    assert not torch.allclose(actions_with_rtc, actions_without_rtc, rtol=1e-3)
+
+    print("✓ PI0 RTC inference with prev_chunk: Test passed")
+
+
+@require_cuda
+def test_pi0_rtc_inference_without_prev_chunk():
+    """Test PI0 policy inference with RTC but no previous chunk (RTC should have no effect)."""
+    set_seed(42)
+
+    config = PI0Config(max_action_dim=7, max_state_dim=14, chunk_size=50, dtype="float32")
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats
+    dataset_stats = {
+        "observation.state": {"mean": torch.zeros(14), "std": torch.ones(14)},
+        "action": {"mean": torch.zeros(7), "std": torch.ones(7)},
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    # Instantiate policy and preprocessor
+    policy = PI0Policy(config)
+    policy.eval()
+    preprocessor, _ = make_pi0_pre_post_processors(config=config, dataset_stats=dataset_stats)
+
+    device = config.device
+
+    # Create dummy batch
+    batch = {
+        "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+        "task": ["Pick up the object"],
+    }
+    batch = preprocessor(batch)
+
+    with torch.no_grad():
+        # Use same noise for fair comparison
+        noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+
+        # Test with RTC enabled but no previous chunk
+        actions_with_rtc_no_prev = policy.predict_action_chunk(
+            batch,
+            noise=noise.clone(),
+            prev_chunk_left_over=None,
+        )
+
+        # Test without RTC
+        policy.config.rtc_config.enabled = False
+        actions_without_rtc = policy.predict_action_chunk(batch, noise=noise.clone())
+        policy.config.rtc_config.enabled = True
+
+    # Without previous chunk, RTC should have no effect
+    assert torch.allclose(actions_with_rtc_no_prev, actions_without_rtc, rtol=1e-5)
+
+    print("✓ PI0 RTC inference without prev_chunk: Test passed")
+
+
+@require_cuda
+def test_pi0_rtc_validation_rules():
+    """Test PI0 policy with RTC follows all three validation rules."""
+    set_seed(42)
+
+    config = PI0Config(max_action_dim=7, max_state_dim=14, chunk_size=50, dtype="float32")
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats
+    dataset_stats = {
+        "observation.state": {"mean": torch.zeros(14), "std": torch.ones(14)},
+        "action": {"mean": torch.zeros(7), "std": torch.ones(7)},
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    # Instantiate policy and preprocessor
+    policy = PI0Policy(config)
+    policy.eval()
+    preprocessor, _ = make_pi0_pre_post_processors(config=config, dataset_stats=dataset_stats)
+
+    device = config.device
+
+    # Create dummy batch
+    batch = {
+        "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+        "task": ["Pick up the object"],
+    }
+    batch = preprocessor(batch)
+
+    # Create previous chunk
+    prev_chunk = torch.randn(1, 25, 7, dtype=torch.float32, device=device)
+
+    inference_delay = 4
+    execution_horizon = 10
+
+    with torch.no_grad():
+        # Use same noise for fair comparison
+        noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+
+        # Test with RTC
+        actions_with_rtc = policy.predict_action_chunk(
+            batch,
+            noise=noise.clone(),
+            prev_chunk_left_over=prev_chunk,
+            inference_delay=inference_delay,
+            execution_horizon=execution_horizon,
+        )
+
+        # Test without RTC
+        policy.config.rtc_config.enabled = False
+        actions_without_rtc = policy.predict_action_chunk(batch, noise=noise.clone())
+        policy.config.rtc_config.enabled = True
+
+    assert not torch.allclose(actions_with_rtc, actions_without_rtc, rtol=1e-3)
+
+    """Test PI0 with different RTC attention schedules."""
+    set_seed(42)
+
+    schedules = [
+        RTCAttentionSchedule.ZEROS,
+        RTCAttentionSchedule.ONES,
+        RTCAttentionSchedule.LINEAR,
+        RTCAttentionSchedule.EXP,
+    ]
+
+    config = PI0Config(max_action_dim=7, max_state_dim=14, chunk_size=50, dtype="float32")
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats
+    dataset_stats = {
+        "observation.state": {"mean": torch.zeros(14), "std": torch.ones(14)},
+        "action": {"mean": torch.zeros(7), "std": torch.ones(7)},
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    device = config.device
+
+    for schedule in schedules:
+        print(f"Testing schedule: {schedule}")
+
+        # Add RTC config with specific schedule
+        config.rtc_config = RTCConfig(
+            enabled=True,
+            execution_horizon=10,
+            max_guidance_weight=5.0,
+            prefix_attention_schedule=schedule,
+            debug=False,
+        )
+
+        # Instantiate policy
+        policy = PI0Policy(config)
+        policy.eval()
+        preprocessor, _ = make_pi0_pre_post_processors(config=config, dataset_stats=dataset_stats)
+
+        # Create dummy batch
+        batch = {
+            "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+            "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+            "task": ["Pick up the object"],
+        }
+        batch = preprocessor(batch)
+
+        # Create previous chunk
+        prev_chunk = torch.randn(1, 25, 7, dtype=torch.float32, device=device)
+
+        with torch.no_grad():
+            noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+            actions = policy.predict_action_chunk(
+                batch,
+                noise=noise,
+                prev_chunk_left_over=prev_chunk,
+                inference_delay=4,
+                execution_horizon=10,
+            )
+
+        # Verify shape
+        assert actions.shape == (1, config.chunk_size, 7)
+        print(f"  ✓ Schedule {schedule}: Test passed")
+
+    print("✓ PI0 RTC different schedules: All schedules tested")
diff --git a/lerobot/tests/policies/rtc/test_action_queue.py b/lerobot/tests/policies/rtc/test_action_queue.py
new file mode 100644
index 0000000000000000000000000000000000000000..2f9b843842fd533f5c2f5c900a16b3de1f69d8ae
--- /dev/null
+++ b/lerobot/tests/policies/rtc/test_action_queue.py
@@ -0,0 +1,825 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Tests for RTC ActionQueue module."""
+
+import threading
+import time
+
+import pytest
+import torch
+
+from lerobot.policies.rtc.action_queue import ActionQueue
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+
+# ====================== Fixtures ======================
+
+
+@pytest.fixture
+def rtc_config_enabled():
+    """Create an RTC config with RTC enabled."""
+    return RTCConfig(enabled=True, execution_horizon=10, max_guidance_weight=1.0)
+
+
+@pytest.fixture
+def rtc_config_disabled():
+    """Create an RTC config with RTC disabled."""
+    return RTCConfig(enabled=False, execution_horizon=10, max_guidance_weight=1.0)
+
+
+@pytest.fixture
+def sample_actions():
+    """Create sample action tensors for testing."""
+    return {
+        "original": torch.randn(50, 6),  # (time_steps, action_dim)
+        "processed": torch.randn(50, 6),
+        "short": torch.randn(10, 6),
+        "longer": torch.randn(100, 6),
+    }
+
+
+@pytest.fixture
+def action_queue_rtc_enabled(rtc_config_enabled):
+    """Create an ActionQueue with RTC enabled."""
+    return ActionQueue(rtc_config_enabled)
+
+
+@pytest.fixture
+def action_queue_rtc_disabled(rtc_config_disabled):
+    """Create an ActionQueue with RTC disabled."""
+    return ActionQueue(rtc_config_disabled)
+
+
+# ====================== Initialization Tests ======================
+
+
+def test_action_queue_initialization_rtc_enabled(rtc_config_enabled):
+    """Test ActionQueue initializes correctly with RTC enabled."""
+    queue = ActionQueue(rtc_config_enabled)
+    assert queue.queue is None
+    assert queue.original_queue is None
+    assert queue.last_index == 0
+    assert queue.cfg.enabled is True
+
+
+def test_action_queue_initialization_rtc_disabled(rtc_config_disabled):
+    """Test ActionQueue initializes correctly with RTC disabled."""
+    queue = ActionQueue(rtc_config_disabled)
+    assert queue.queue is None
+    assert queue.original_queue is None
+    assert queue.last_index == 0
+    assert queue.cfg.enabled is False
+
+
+# ====================== get() Tests ======================
+
+
+def test_get_returns_none_when_empty(action_queue_rtc_enabled):
+    """Test get() returns None when queue is empty."""
+    action = action_queue_rtc_enabled.get()
+    assert action is None
+
+
+def test_get_returns_actions_sequentially(action_queue_rtc_enabled, sample_actions):
+    """Test get() returns actions in sequence."""
+    # Initialize queue with actions
+    action_queue_rtc_enabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=0)
+
+    # Get first action
+    action1 = action_queue_rtc_enabled.get()
+    assert action1 is not None
+    assert action1.shape == (6,)
+    assert torch.equal(action1, sample_actions["processed"][0])
+
+    # Get second action
+    action2 = action_queue_rtc_enabled.get()
+    assert action2 is not None
+    assert torch.equal(action2, sample_actions["processed"][1])
+
+
+def test_get_returns_none_after_exhaustion(action_queue_rtc_enabled, sample_actions):
+    """Test get() returns None after all actions are consumed."""
+    # Use short action sequence
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # Consume all actions
+    for _ in range(10):
+        action = action_queue_rtc_enabled.get()
+        assert action is not None
+
+    # Next get should return None
+    action = action_queue_rtc_enabled.get()
+    assert action is None
+
+
+def test_get_increments_last_index(action_queue_rtc_enabled, sample_actions):
+    """Test get() increments last_index correctly."""
+    action_queue_rtc_enabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=0)
+
+    assert action_queue_rtc_enabled.last_index == 0
+    action_queue_rtc_enabled.get()
+    assert action_queue_rtc_enabled.last_index == 1
+    action_queue_rtc_enabled.get()
+    assert action_queue_rtc_enabled.last_index == 2
+
+
+# ====================== qsize() Tests ======================
+
+
+def test_qsize_returns_zero_when_empty(action_queue_rtc_enabled):
+    """Test qsize() returns 0 when queue is empty."""
+    assert action_queue_rtc_enabled.qsize() == 0
+
+
+def test_qsize_returns_correct_size(action_queue_rtc_enabled, sample_actions):
+    """Test qsize() returns correct number of remaining actions."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+    assert action_queue_rtc_enabled.qsize() == 10
+
+    action_queue_rtc_enabled.get()
+    assert action_queue_rtc_enabled.qsize() == 9
+
+    action_queue_rtc_enabled.get()
+    assert action_queue_rtc_enabled.qsize() == 8
+
+
+def test_qsize_after_exhaustion(action_queue_rtc_enabled, sample_actions):
+    """Test qsize() returns 0 after queue is exhausted."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # Consume all actions
+    for _ in range(10):
+        action_queue_rtc_enabled.get()
+
+    assert action_queue_rtc_enabled.qsize() == 0
+
+
+# ====================== empty() Tests ======================
+
+
+def test_empty_returns_true_when_empty(action_queue_rtc_enabled):
+    """Test empty() returns True when queue is empty."""
+    assert action_queue_rtc_enabled.empty() is True
+
+
+def test_empty_returns_false_when_not_empty(action_queue_rtc_enabled, sample_actions):
+    """Test empty() returns False when queue has actions."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+    assert action_queue_rtc_enabled.empty() is False
+
+
+def test_empty_after_partial_consumption(action_queue_rtc_enabled, sample_actions):
+    """Test empty() returns False after partial consumption."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    action_queue_rtc_enabled.get()
+    action_queue_rtc_enabled.get()
+
+    assert action_queue_rtc_enabled.empty() is False
+
+
+def test_empty_after_full_consumption(action_queue_rtc_enabled, sample_actions):
+    """Test empty() returns True after all actions consumed."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # Consume all
+    for _ in range(10):
+        action_queue_rtc_enabled.get()
+
+    assert action_queue_rtc_enabled.empty() is True
+
+
+# ====================== get_action_index() Tests ======================
+
+
+def test_get_action_index_initial_value(action_queue_rtc_enabled):
+    """Test get_action_index() returns 0 initially."""
+    assert action_queue_rtc_enabled.get_action_index() == 0
+
+
+def test_get_action_index_after_consumption(action_queue_rtc_enabled, sample_actions):
+    """Test get_action_index() tracks consumption correctly."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    assert action_queue_rtc_enabled.get_action_index() == 0
+    action_queue_rtc_enabled.get()
+    assert action_queue_rtc_enabled.get_action_index() == 1
+    action_queue_rtc_enabled.get()
+    action_queue_rtc_enabled.get()
+    assert action_queue_rtc_enabled.get_action_index() == 3
+
+
+# ====================== get_left_over() Tests ======================
+
+
+def test_get_left_over_returns_none_when_empty(action_queue_rtc_enabled):
+    """Test get_left_over() returns None when queue is empty."""
+    leftover = action_queue_rtc_enabled.get_left_over()
+    assert leftover is None
+
+
+def test_get_left_over_returns_all_when_unconsumed(action_queue_rtc_enabled, sample_actions):
+    """Test get_left_over() returns all original actions when none consumed."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    leftover = action_queue_rtc_enabled.get_left_over()
+    assert leftover is not None
+    assert leftover.shape == (10, 6)
+    assert torch.equal(leftover, sample_actions["short"])
+
+
+def test_get_left_over_returns_remaining_after_consumption(action_queue_rtc_enabled, sample_actions):
+    """Test get_left_over() returns only remaining original actions."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # Consume 3 actions
+    action_queue_rtc_enabled.get()
+    action_queue_rtc_enabled.get()
+    action_queue_rtc_enabled.get()
+
+    leftover = action_queue_rtc_enabled.get_left_over()
+    assert leftover is not None
+    assert leftover.shape == (7, 6)
+    assert torch.equal(leftover, sample_actions["short"][3:])
+
+
+def test_get_left_over_returns_empty_after_exhaustion(action_queue_rtc_enabled, sample_actions):
+    """Test get_left_over() returns empty tensor after all consumed."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # Consume all
+    for _ in range(10):
+        action_queue_rtc_enabled.get()
+
+    leftover = action_queue_rtc_enabled.get_left_over()
+    assert leftover is not None
+    assert leftover.shape == (0, 6)
+
+
+# ====================== merge() with RTC Enabled Tests ======================
+
+
+def test_merge_replaces_queue_when_rtc_enabled(action_queue_rtc_enabled, sample_actions):
+    """Test merge() replaces queue when RTC is enabled."""
+    # Add initial actions
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+    assert action_queue_rtc_enabled.qsize() == 10
+
+    # Consume some actions
+    action_queue_rtc_enabled.get()
+    action_queue_rtc_enabled.get()
+    assert action_queue_rtc_enabled.qsize() == 8
+
+    # Merge new actions - should replace, not append
+    action_queue_rtc_enabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=5)
+
+    # Queue should be replaced with new actions minus delay
+    # Original has 50 actions, delay is 5, so remaining is 45
+    assert action_queue_rtc_enabled.qsize() == 45
+    assert action_queue_rtc_enabled.get_action_index() == 0
+
+
+def test_merge_respects_real_delay(action_queue_rtc_enabled, sample_actions):
+    """Test merge() correctly applies real_delay when RTC is enabled."""
+    delay = 10
+    action_queue_rtc_enabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=delay)
+
+    # Queue should have original length minus delay
+    expected_size = len(sample_actions["original"]) - delay
+    assert action_queue_rtc_enabled.qsize() == expected_size
+
+    # First action should be the one at index [delay]
+    first_action = action_queue_rtc_enabled.get()
+    assert torch.equal(first_action, sample_actions["processed"][delay])
+
+
+def test_merge_resets_last_index_when_rtc_enabled(action_queue_rtc_enabled, sample_actions):
+    """Test merge() resets last_index to 0 when RTC is enabled."""
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+    action_queue_rtc_enabled.get()
+    action_queue_rtc_enabled.get()
+    assert action_queue_rtc_enabled.last_index == 2
+
+    # Merge new actions
+    action_queue_rtc_enabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=5)
+
+    assert action_queue_rtc_enabled.last_index == 0
+
+
+def test_merge_with_zero_delay(action_queue_rtc_enabled, sample_actions):
+    """Test merge() with zero delay keeps all actions."""
+    action_queue_rtc_enabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=0)
+
+    assert action_queue_rtc_enabled.qsize() == len(sample_actions["original"])
+
+
+def test_merge_with_large_delay(action_queue_rtc_enabled, sample_actions):
+    """Test merge() with delay larger than action sequence."""
+    # Delay is larger than sequence length
+    delay = 100
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=delay)
+
+    # Queue should be empty (delay >= length)
+    assert action_queue_rtc_enabled.qsize() == 0
+
+
+# ====================== merge() with RTC Disabled Tests ======================
+
+
+def test_merge_appends_when_rtc_disabled(action_queue_rtc_disabled, sample_actions):
+    """Test merge() appends actions when RTC is disabled."""
+    # Add initial actions
+    action_queue_rtc_disabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+    initial_size = action_queue_rtc_disabled.qsize()
+    assert initial_size == 10
+
+    # Merge more actions
+    action_queue_rtc_disabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # Should have appended
+    assert action_queue_rtc_disabled.qsize() == initial_size + 10
+
+
+def test_merge_removes_consumed_actions_when_appending(action_queue_rtc_disabled, sample_actions):
+    """Test merge() removes consumed actions before appending when RTC is disabled."""
+    # Add initial actions
+    action_queue_rtc_disabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+    assert action_queue_rtc_disabled.qsize() == 10
+
+    # Consume 3 actions
+    action_queue_rtc_disabled.get()
+    action_queue_rtc_disabled.get()
+    action_queue_rtc_disabled.get()
+    assert action_queue_rtc_disabled.qsize() == 7
+
+    # Merge more actions
+    action_queue_rtc_disabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # Should have 7 remaining + 10 new = 17
+    assert action_queue_rtc_disabled.qsize() == 17
+
+
+def test_merge_resets_last_index_after_append(action_queue_rtc_disabled, sample_actions):
+    """Test merge() resets last_index after appending when RTC is disabled."""
+    action_queue_rtc_disabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+    action_queue_rtc_disabled.get()
+    action_queue_rtc_disabled.get()
+    assert action_queue_rtc_disabled.last_index == 2
+
+    # Merge more actions
+    action_queue_rtc_disabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # last_index should be reset to 0
+    assert action_queue_rtc_disabled.last_index == 0
+
+
+def test_merge_ignores_delay_when_rtc_disabled(action_queue_rtc_disabled, sample_actions):
+    """Test merge() ignores real_delay parameter when RTC is disabled."""
+    action_queue_rtc_disabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=10)
+
+    # All actions should be in queue (delay ignored)
+    assert action_queue_rtc_disabled.qsize() == len(sample_actions["original"])
+
+
+def test_merge_first_call_with_rtc_disabled(action_queue_rtc_disabled, sample_actions):
+    """Test merge() on first call with RTC disabled."""
+    action_queue_rtc_disabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=0)
+
+    assert action_queue_rtc_disabled.qsize() == len(sample_actions["original"])
+    assert action_queue_rtc_disabled.last_index == 0
+
+
+# ====================== merge() with Different Action Shapes Tests ======================
+
+
+def test_merge_with_different_action_dims():
+    """Test merge() handles actions with different dimensions."""
+    cfg = RTCConfig(enabled=True, execution_horizon=10)
+    queue = ActionQueue(cfg)
+
+    # Actions with 4 dimensions instead of 6
+    actions_4d = torch.randn(20, 4)
+    queue.merge(actions_4d, actions_4d, real_delay=5)
+
+    action = queue.get()
+    assert action.shape == (4,)
+
+
+def test_merge_with_different_lengths():
+    """Test merge() handles action sequences of varying lengths."""
+    cfg = RTCConfig(enabled=False, execution_horizon=10)
+    queue = ActionQueue(cfg)
+
+    # Add sequences of different lengths
+    queue.merge(torch.randn(10, 6), torch.randn(10, 6), real_delay=0)
+    assert queue.qsize() == 10
+
+    queue.merge(torch.randn(25, 6), torch.randn(25, 6), real_delay=0)
+    assert queue.qsize() == 35
+
+
+# ====================== merge() Delay Validation Tests ======================
+
+
+def test_merge_validates_delay_consistency(action_queue_rtc_enabled, sample_actions, caplog):
+    """Test merge() validates that real_delay matches action index difference."""
+    import logging
+
+    caplog.set_level(logging.WARNING)
+
+    # Initialize queue
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # Consume 5 actions
+    for _ in range(5):
+        action_queue_rtc_enabled.get()
+
+    # Merge with mismatched delay (should log warning)
+    # We consumed 5 actions, so index is 5. If we pass action_index_before_inference=0,
+    # then indexes_diff=5, but if real_delay=3, it will warn
+    action_queue_rtc_enabled.merge(
+        sample_actions["original"],
+        sample_actions["processed"],
+        real_delay=3,
+        action_index_before_inference=0,
+    )
+
+    # Check warning was logged
+    assert "Indexes diff is not equal to real delay" in caplog.text
+
+
+def test_merge_no_warning_when_delays_match(action_queue_rtc_enabled, sample_actions, caplog):
+    """Test merge() doesn't warn when delays are consistent."""
+    import logging
+
+    caplog.set_level(logging.WARNING)
+
+    # Initialize queue
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    # Consume 5 actions
+    for _ in range(5):
+        action_queue_rtc_enabled.get()
+
+    # Merge with matching delay
+    action_queue_rtc_enabled.merge(
+        sample_actions["original"],
+        sample_actions["processed"],
+        real_delay=5,
+        action_index_before_inference=0,
+    )
+
+    # Should not have warning
+    assert "Indexes diff is not equal to real delay" not in caplog.text
+
+
+def test_merge_skips_validation_when_action_index_none(action_queue_rtc_enabled, sample_actions, caplog):
+    """Test merge() skips delay validation when action_index_before_inference is None."""
+    import logging
+
+    caplog.set_level(logging.WARNING)
+
+    action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+    for _ in range(5):
+        action_queue_rtc_enabled.get()
+
+    # Pass None for action_index_before_inference
+    action_queue_rtc_enabled.merge(
+        sample_actions["original"],
+        sample_actions["processed"],
+        real_delay=999,  # Doesn't matter
+        action_index_before_inference=None,
+    )
+
+    # Should not warn (validation skipped)
+    assert "Indexes diff is not equal to real delay" not in caplog.text
+
+
+# ====================== Thread Safety Tests ======================
+
+
+def test_get_is_thread_safe(action_queue_rtc_enabled, sample_actions):
+    """Test get() is thread-safe with multiple consumers."""
+    action_queue_rtc_enabled.merge(sample_actions["longer"], sample_actions["longer"], real_delay=0)
+
+    results = []
+    errors = []
+
+    def consumer():
+        try:
+            for _ in range(25):
+                action = action_queue_rtc_enabled.get()
+                if action is not None:
+                    results.append(action)
+                time.sleep(0.001)
+        except Exception as e:
+            errors.append(e)
+
+    threads = [threading.Thread(target=consumer) for _ in range(4)]
+
+    for t in threads:
+        t.start()
+
+    for t in threads:
+        t.join()
+
+    # Should not have errors
+    assert len(errors) == 0
+
+    # Should have consumed all actions (100 total, 4 threads * 25 each)
+    assert len(results) == 100
+
+    # All results should be unique (no duplicate consumption)
+    # We can verify by checking that indices are not duplicated
+    # Since we don't track indices in results, we check total count is correct
+    assert action_queue_rtc_enabled.qsize() == 0
+
+
+def test_merge_is_thread_safe(action_queue_rtc_disabled, sample_actions):
+    """Test merge() is thread-safe with multiple producers."""
+    errors = []
+
+    def producer():
+        try:
+            for _ in range(5):
+                action_queue_rtc_disabled.merge(
+                    sample_actions["short"], sample_actions["short"], real_delay=0
+                )
+                time.sleep(0.001)
+        except Exception as e:
+            errors.append(e)
+
+    threads = [threading.Thread(target=producer) for _ in range(3)]
+
+    for t in threads:
+        t.start()
+
+    for t in threads:
+        t.join()
+
+    # Should not have errors
+    assert len(errors) == 0
+
+    # Should have accumulated all actions (3 threads * 5 merges * 10 actions = 150)
+    assert action_queue_rtc_disabled.qsize() == 150
+
+
+def test_concurrent_get_and_merge(action_queue_rtc_disabled, sample_actions):
+    """Test concurrent get() and merge() operations."""
+    errors = []
+    consumed_count = [0]
+
+    def consumer():
+        try:
+            for _ in range(50):
+                action = action_queue_rtc_disabled.get()
+                if action is not None:
+                    consumed_count[0] += 1
+                time.sleep(0.001)
+        except Exception as e:
+            errors.append(e)
+
+    def producer():
+        try:
+            for _ in range(10):
+                action_queue_rtc_disabled.merge(
+                    sample_actions["short"], sample_actions["short"], real_delay=0
+                )
+                time.sleep(0.005)
+        except Exception as e:
+            errors.append(e)
+
+    consumer_threads = [threading.Thread(target=consumer) for _ in range(2)]
+    producer_threads = [threading.Thread(target=producer) for _ in range(2)]
+
+    for t in consumer_threads + producer_threads:
+        t.start()
+
+    for t in consumer_threads + producer_threads:
+        t.join()
+
+    # Should not have errors
+    assert len(errors) == 0
+
+    # Should have consumed some or all actions (non-deterministic due to timing)
+    # Total produced: 2 producers * 10 merges * 10 actions = 200
+    # Total consumed attempts: 2 consumers * 50 = 100
+    assert consumed_count[0] <= 200
+
+
+# ====================== get_left_over() Thread Safety Tests ======================
+
+
+def test_get_left_over_is_thread_safe(action_queue_rtc_enabled, sample_actions):
+    """Test get_left_over() is thread-safe with concurrent access."""
+    action_queue_rtc_enabled.merge(sample_actions["longer"], sample_actions["longer"], real_delay=0)
+
+    errors = []
+    leftovers = []
+
+    def reader():
+        try:
+            for _ in range(20):
+                leftover = action_queue_rtc_enabled.get_left_over()
+                if leftover is not None:
+                    leftovers.append(leftover.shape[0])
+                time.sleep(0.001)
+        except Exception as e:
+            errors.append(e)
+
+    threads = [threading.Thread(target=reader) for _ in range(3)]
+
+    # Also consume some actions concurrently
+    def consumer():
+        try:
+            for _ in range(10):
+                action_queue_rtc_enabled.get()
+                time.sleep(0.002)
+        except Exception as e:
+            errors.append(e)
+
+    consumer_thread = threading.Thread(target=consumer)
+
+    all_threads = threads + [consumer_thread]
+
+    for t in all_threads:
+        t.start()
+
+    for t in all_threads:
+        t.join()
+
+    # Should not have errors
+    assert len(errors) == 0
+
+    # Leftovers should be monotonically decreasing or stable
+    # (as actions are consumed, leftover size decreases)
+    assert len(leftovers) > 0
+
+
+# ====================== Edge Cases Tests ======================
+
+
+def test_queue_with_single_action(action_queue_rtc_enabled):
+    """Test queue behavior with a single action."""
+    single_action_original = torch.randn(1, 6)
+    single_action_processed = torch.randn(1, 6)
+
+    action_queue_rtc_enabled.merge(single_action_original, single_action_processed, real_delay=0)
+
+    assert action_queue_rtc_enabled.qsize() == 1
+    action = action_queue_rtc_enabled.get()
+    assert action is not None
+    assert action.shape == (6,)
+    assert action_queue_rtc_enabled.qsize() == 0
+
+
+def test_queue_behavior_after_multiple_merge_cycles(action_queue_rtc_enabled, sample_actions):
+    """Test queue maintains correct state through multiple merge cycles."""
+    for _ in range(5):
+        action_queue_rtc_enabled.merge(sample_actions["short"], sample_actions["short"], real_delay=0)
+
+        # Consume half
+        for _ in range(5):
+            action_queue_rtc_enabled.get()
+
+        # Merge again
+        action_queue_rtc_enabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=3)
+
+        assert action_queue_rtc_enabled.qsize() > 0
+
+
+def test_queue_with_all_zeros_actions(action_queue_rtc_enabled):
+    """Test queue handles all-zero action tensors."""
+    zeros_actions = torch.zeros(20, 6)
+    action_queue_rtc_enabled.merge(zeros_actions, zeros_actions, real_delay=0)
+
+    action = action_queue_rtc_enabled.get()
+    assert torch.all(action == 0)
+
+
+def test_queue_clones_input_tensors(action_queue_rtc_enabled, sample_actions):
+    """Test that merge() clones input tensors, not storing references."""
+    original_copy = sample_actions["original"].clone()
+    processed_copy = sample_actions["processed"].clone()
+
+    action_queue_rtc_enabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=0)
+
+    # Modify original tensors
+    sample_actions["original"].fill_(999.0)
+    sample_actions["processed"].fill_(-999.0)
+
+    # Queue should have cloned values
+    action = action_queue_rtc_enabled.get()
+    assert not torch.equal(action, sample_actions["processed"][0])
+    assert torch.equal(action, processed_copy[0])
+
+    leftover = action_queue_rtc_enabled.get_left_over()
+    assert not torch.equal(leftover, sample_actions["original"][1:])
+    assert torch.equal(leftover, original_copy[1:])
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_queue_handles_gpu_tensors():
+    """Test queue correctly handles GPU tensors."""
+    cfg = RTCConfig(enabled=True, execution_horizon=10)
+    queue = ActionQueue(cfg)
+
+    actions_gpu = torch.randn(20, 6, device="cuda")
+    queue.merge(actions_gpu, actions_gpu, real_delay=0)
+
+    action = queue.get()
+    assert action.device.type == "cuda"
+
+    leftover = queue.get_left_over()
+    assert leftover.device.type == "cuda"
+
+
+def test_queue_handles_different_dtypes():
+    """Test queue handles actions with different dtypes."""
+    cfg = RTCConfig(enabled=True, execution_horizon=10)
+    queue = ActionQueue(cfg)
+
+    # Use float64 instead of default float32
+    actions_f64 = torch.randn(20, 6, dtype=torch.float64)
+    queue.merge(actions_f64, actions_f64, real_delay=0)
+
+    action = queue.get()
+    assert action.dtype == torch.float64
+
+
+def test_empty_with_none_queue(action_queue_rtc_enabled):
+    """Test empty() correctly handles None queue."""
+    assert action_queue_rtc_enabled.queue is None
+    assert action_queue_rtc_enabled.empty() is True
+
+
+def test_qsize_with_none_queue(action_queue_rtc_enabled):
+    """Test qsize() correctly handles None queue."""
+    assert action_queue_rtc_enabled.queue is None
+    assert action_queue_rtc_enabled.qsize() == 0
+
+
+# ====================== Integration Tests ======================
+
+
+def test_typical_rtc_workflow(action_queue_rtc_enabled, sample_actions):
+    """Test a typical RTC workflow: merge, consume, merge with delay."""
+    # First inference
+    action_queue_rtc_enabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=0)
+    initial_size = action_queue_rtc_enabled.qsize()
+    assert initial_size == 50
+
+    # Consume 10 actions (execution_horizon)
+    for _ in range(10):
+        action = action_queue_rtc_enabled.get()
+        assert action is not None
+
+    assert action_queue_rtc_enabled.qsize() == 40
+
+    # Second inference with delay
+    action_index_before = action_queue_rtc_enabled.get_action_index()
+
+    action_queue_rtc_enabled.merge(
+        sample_actions["original"],
+        sample_actions["processed"],
+        real_delay=5,
+        action_index_before_inference=action_index_before,
+    )
+
+    # Queue should be replaced, minus delay
+    assert action_queue_rtc_enabled.qsize() == 45
+    assert action_queue_rtc_enabled.get_action_index() == 0
+
+
+def test_typical_non_rtc_workflow(action_queue_rtc_disabled, sample_actions):
+    """Test a typical non-RTC workflow: merge, consume, merge again."""
+    # First inference
+    action_queue_rtc_disabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=0)
+    assert action_queue_rtc_disabled.qsize() == 50
+
+    # Consume 40 actions
+    for _ in range(40):
+        action = action_queue_rtc_disabled.get()
+        assert action is not None
+
+    assert action_queue_rtc_disabled.qsize() == 10
+
+    # Second inference (should append)
+    action_queue_rtc_disabled.merge(sample_actions["original"], sample_actions["processed"], real_delay=0)
+
+    # Should have 10 remaining + 50 new = 60
+    assert action_queue_rtc_disabled.qsize() == 60
diff --git a/lerobot/tests/policies/rtc/test_configuration_rtc.py b/lerobot/tests/policies/rtc/test_configuration_rtc.py
new file mode 100644
index 0000000000000000000000000000000000000000..bb4550eaa637ef958b62dd9b0e4cafd418187bfe
--- /dev/null
+++ b/lerobot/tests/policies/rtc/test_configuration_rtc.py
@@ -0,0 +1,65 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Tests for RTC configuration module."""
+
+from lerobot.configs.types import RTCAttentionSchedule
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+
+# ====================== Initialization Tests ======================
+
+
+def test_rtc_config_default_initialization():
+    """Test RTCConfig initializes with default values."""
+    config = RTCConfig()
+
+    assert config.enabled is False
+    assert config.prefix_attention_schedule == RTCAttentionSchedule.LINEAR
+    assert config.max_guidance_weight == 10.0
+    assert config.execution_horizon == 10
+    assert config.debug is False
+    assert config.debug_maxlen == 100
+
+
+def test_rtc_config_custom_initialization():
+    """Test RTCConfig initializes with custom values."""
+    config = RTCConfig(
+        enabled=True,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        max_guidance_weight=5.0,
+        execution_horizon=20,
+        debug=True,
+        debug_maxlen=200,
+    )
+
+    assert config.enabled is True
+    assert config.prefix_attention_schedule == RTCAttentionSchedule.EXP
+    assert config.max_guidance_weight == 5.0
+    assert config.execution_horizon == 20
+    assert config.debug is True
+    assert config.debug_maxlen == 200
+
+
+def test_rtc_config_partial_initialization():
+    """Test RTCConfig with partial custom values."""
+    config = RTCConfig(enabled=True, max_guidance_weight=15.0)
+
+    assert config.enabled is True
+    assert config.max_guidance_weight == 15.0
+    # Other values should be defaults
+    assert config.prefix_attention_schedule == RTCAttentionSchedule.LINEAR
+    assert config.execution_horizon == 10
+    assert config.debug is False
diff --git a/lerobot/tests/policies/rtc/test_debug_tracker.py b/lerobot/tests/policies/rtc/test_debug_tracker.py
new file mode 100644
index 0000000000000000000000000000000000000000..f9bea558b0870d19a844bf3efe4ab473910ed35e
--- /dev/null
+++ b/lerobot/tests/policies/rtc/test_debug_tracker.py
@@ -0,0 +1,488 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Tests for RTC debug tracker module."""
+
+import pytest
+import torch
+
+from lerobot.policies.rtc.debug_tracker import DebugStep, Tracker
+
+# ====================== Fixtures ======================
+
+
+@pytest.fixture
+def sample_tensors():
+    """Create sample tensors for testing."""
+    return {
+        "x_t": torch.randn(1, 50, 6),
+        "v_t": torch.randn(1, 50, 6),
+        "x1_t": torch.randn(1, 50, 6),
+        "correction": torch.randn(1, 50, 6),
+        "err": torch.randn(1, 50, 6),
+        "weights": torch.randn(1, 50, 1),
+    }
+
+
+@pytest.fixture
+def enabled_tracker():
+    """Create an enabled tracker with default settings."""
+    return Tracker(enabled=True, maxlen=100)
+
+
+@pytest.fixture
+def disabled_tracker():
+    """Create a disabled tracker."""
+    return Tracker(enabled=False)
+
+
+# ====================== DebugStep Tests ======================
+
+
+def test_debug_step_initialization():
+    """Test that DebugStep can be initialized with default values."""
+    step = DebugStep()
+    assert step.step_idx == 0
+    assert step.x_t is None
+    assert step.v_t is None
+    assert step.x1_t is None
+    assert step.correction is None
+    assert step.err is None
+    assert step.weights is None
+    assert step.guidance_weight is None
+    assert step.time is None
+    assert step.inference_delay is None
+    assert step.execution_horizon is None
+    assert step.metadata == {}
+
+
+def test_debug_step_with_values(sample_tensors):
+    """Test DebugStep initialization with actual values."""
+    step = DebugStep(
+        step_idx=5,
+        x_t=sample_tensors["x_t"],
+        v_t=sample_tensors["v_t"],
+        x1_t=sample_tensors["x1_t"],
+        correction=sample_tensors["correction"],
+        err=sample_tensors["err"],
+        weights=sample_tensors["weights"],
+        guidance_weight=2.5,
+        time=0.8,
+        inference_delay=4,
+        execution_horizon=8,
+        metadata={"custom_key": "custom_value"},
+    )
+
+    assert step.step_idx == 5
+    assert torch.equal(step.x_t, sample_tensors["x_t"])
+    assert torch.equal(step.v_t, sample_tensors["v_t"])
+    assert torch.equal(step.x1_t, sample_tensors["x1_t"])
+    assert torch.equal(step.correction, sample_tensors["correction"])
+    assert torch.equal(step.err, sample_tensors["err"])
+    assert torch.equal(step.weights, sample_tensors["weights"])
+    assert step.guidance_weight == 2.5
+    assert step.time == 0.8
+    assert step.inference_delay == 4
+    assert step.execution_horizon == 8
+    assert step.metadata == {"custom_key": "custom_value"}
+
+
+def test_debug_step_to_dict_without_tensors(sample_tensors):
+    """Test converting DebugStep to dictionary without tensor values."""
+    step = DebugStep(
+        step_idx=3,
+        x_t=sample_tensors["x_t"],
+        v_t=sample_tensors["v_t"],
+        guidance_weight=torch.tensor(3.0),
+        time=torch.tensor(0.5),
+        inference_delay=2,
+        execution_horizon=10,
+    )
+
+    result = step.to_dict(include_tensors=False)
+
+    assert result["step_idx"] == 3
+    assert result["guidance_weight"] == 3.0
+    assert result["time"] == 0.5
+    assert result["inference_delay"] == 2
+    assert result["execution_horizon"] == 10
+
+    # Check tensor statistics are included
+    assert "x_t_stats" in result
+    assert "v_t_stats" in result
+    assert "x1_t_stats" not in result  # x1_t was None
+
+    # Verify statistics structure
+    assert "shape" in result["x_t_stats"]
+    assert "mean" in result["x_t_stats"]
+    assert "std" in result["x_t_stats"]
+    assert "min" in result["x_t_stats"]
+    assert "max" in result["x_t_stats"]
+
+    # Verify shape matches original tensor
+    assert result["x_t_stats"]["shape"] == tuple(sample_tensors["x_t"].shape)
+
+
+def test_debug_step_to_dict_with_tensors(sample_tensors):
+    """Test converting DebugStep to dictionary with tensor values."""
+    step = DebugStep(
+        step_idx=1,
+        x_t=sample_tensors["x_t"],
+        v_t=sample_tensors["v_t"],
+        guidance_weight=1.5,
+        time=0.9,
+    )
+
+    result = step.to_dict(include_tensors=True)
+
+    assert result["step_idx"] == 1
+    assert result["guidance_weight"] == 1.5
+    assert result["time"] == 0.9
+
+    # Check tensors are included (as CPU tensors)
+    assert "x_t" in result
+    assert "v_t" in result
+    assert isinstance(result["x_t"], torch.Tensor)
+    assert isinstance(result["v_t"], torch.Tensor)
+    assert result["x_t"].device.type == "cpu"
+    assert result["v_t"].device.type == "cpu"
+
+
+def test_debug_step_to_dict_with_none_guidance_weight():
+    """Test to_dict handles None guidance_weight correctly."""
+    step = DebugStep(step_idx=0, time=1.0, guidance_weight=None)
+    result = step.to_dict(include_tensors=False)
+    assert result["guidance_weight"] is None
+
+
+def test_tracker_initialization_enabled():
+    """Test tracker initialization when enabled."""
+    tracker = Tracker(enabled=True, maxlen=50)
+    assert tracker.enabled is True
+    assert tracker._steps == {}
+    assert tracker._maxlen == 50
+    assert tracker._step_counter == 0
+    assert len(tracker) == 0
+
+
+def test_tracker_reset_when_enabled(enabled_tracker, sample_tensors):
+    """Test reset clears all steps when tracker is enabled."""
+    # Add some steps
+    enabled_tracker.track(time=1.0, x_t=sample_tensors["x_t"])
+    enabled_tracker.track(time=0.9, x_t=sample_tensors["x_t"])
+    assert len(enabled_tracker) == 2
+
+    # Reset
+    enabled_tracker.reset()
+    assert len(enabled_tracker) == 0
+    assert enabled_tracker._step_counter == 0
+    assert enabled_tracker._steps == {}
+
+
+def test_tracker_reset_when_disabled(disabled_tracker):
+    """Test reset on disabled tracker doesn't cause errors."""
+    disabled_tracker.reset()
+    assert len(disabled_tracker) == 0
+
+
+# ====================== Tracker.track() Tests ======================
+
+
+def test_track_creates_new_step(enabled_tracker, sample_tensors):
+    """Test that track creates a new step when time doesn't exist."""
+    enabled_tracker.track(
+        time=1.0,
+        x_t=sample_tensors["x_t"],
+        v_t=sample_tensors["v_t"],
+        guidance_weight=5.0,
+        inference_delay=4,
+        execution_horizon=8,
+    )
+
+    assert len(enabled_tracker) == 1
+    steps = enabled_tracker.get_all_steps()
+    assert len(steps) == 1
+    assert steps[0].step_idx == 0
+    assert steps[0].time == 1.0
+    assert torch.equal(steps[0].x_t, sample_tensors["x_t"])
+    assert torch.equal(steps[0].v_t, sample_tensors["v_t"])
+    assert steps[0].guidance_weight == 5.0
+    assert steps[0].inference_delay == 4
+    assert steps[0].execution_horizon == 8
+
+
+def test_track_updates_existing_step(enabled_tracker, sample_tensors):
+    """Test that track updates an existing step at the same time."""
+    # Create initial step
+    enabled_tracker.track(time=0.9, x_t=sample_tensors["x_t"])
+    assert len(enabled_tracker) == 1
+    steps = enabled_tracker.get_all_steps()
+    assert steps[0].v_t is None
+
+    # Update the same timestep with v_t
+    enabled_tracker.track(time=0.9, v_t=sample_tensors["v_t"])
+    assert len(enabled_tracker) == 1  # Still only one step
+    steps = enabled_tracker.get_all_steps()
+    assert torch.equal(steps[0].x_t, sample_tensors["x_t"])  # Original x_t preserved
+    assert torch.equal(steps[0].v_t, sample_tensors["v_t"])  # New v_t added
+
+
+def test_track_with_tensor_time(enabled_tracker, sample_tensors):
+    """Test track handles tensor time values correctly."""
+    time_tensor = torch.tensor(0.8)
+    enabled_tracker.track(time=time_tensor, x_t=sample_tensors["x_t"])
+
+    steps = enabled_tracker.get_all_steps()
+    assert len(steps) == 1
+    assert abs(steps[0].time - 0.8) < 1e-6  # Use approximate comparison for floating point
+
+
+def test_track_time_rounding(enabled_tracker, sample_tensors):
+    """Test that track rounds time to avoid floating point precision issues."""
+    # These times should be treated as the same after rounding to 6 decimals
+    enabled_tracker.track(time=0.9000001, x_t=sample_tensors["x_t"])
+    enabled_tracker.track(time=0.9000002, v_t=sample_tensors["v_t"])
+
+    # Should still be one step (times rounded to same value)
+    assert len(enabled_tracker) == 1
+    steps = enabled_tracker.get_all_steps()
+    assert torch.equal(steps[0].x_t, sample_tensors["x_t"])
+    assert torch.equal(steps[0].v_t, sample_tensors["v_t"])
+
+
+def test_track_does_nothing_when_disabled(disabled_tracker, sample_tensors):
+    """Test that track does nothing when tracker is disabled."""
+    disabled_tracker.track(time=1.0, x_t=sample_tensors["x_t"])
+    assert len(disabled_tracker) == 0
+
+
+def test_track_with_metadata(enabled_tracker, sample_tensors):
+    """Test track stores custom metadata."""
+    enabled_tracker.track(time=0.7, x_t=sample_tensors["x_t"], custom_field="custom_value", count=42)
+
+    steps = enabled_tracker.get_all_steps()
+    assert steps[0].metadata["custom_field"] == "custom_value"
+    assert steps[0].metadata["count"] == 42
+
+
+def test_track_updates_metadata(enabled_tracker):
+    """Test that track updates metadata for existing steps."""
+    enabled_tracker.track(time=0.6, meta1="value1")
+    enabled_tracker.track(time=0.6, meta2="value2")
+
+    steps = enabled_tracker.get_all_steps()
+    assert steps[0].metadata["meta1"] == "value1"
+    assert steps[0].metadata["meta2"] == "value2"
+
+
+def test_track_clones_tensors(enabled_tracker, sample_tensors):
+    """Test that track clones tensors instead of storing references."""
+    x_t_original = sample_tensors["x_t"].clone()
+    enabled_tracker.track(time=0.5, x_t=sample_tensors["x_t"])
+
+    # Modify original tensor
+    sample_tensors["x_t"].fill_(999.0)
+
+    # Tracked tensor should not be affected
+    steps = enabled_tracker.get_all_steps()
+    assert not torch.equal(steps[0].x_t, sample_tensors["x_t"])
+    assert torch.equal(steps[0].x_t, x_t_original)
+
+
+def test_track_with_none_values(enabled_tracker):
+    """Test track handles None values correctly."""
+    enabled_tracker.track(
+        time=0.4,
+        x_t=None,
+        v_t=None,
+        guidance_weight=None,
+        inference_delay=None,
+    )
+
+    steps = enabled_tracker.get_all_steps()
+    assert len(steps) == 1
+    assert steps[0].x_t is None
+    assert steps[0].v_t is None
+    assert steps[0].guidance_weight is None
+    assert steps[0].inference_delay is None
+
+
+def test_track_updates_only_non_none_fields(enabled_tracker, sample_tensors):
+    """Test that update preserves existing values when None is passed."""
+    # Create step with x_t
+    enabled_tracker.track(time=0.3, x_t=sample_tensors["x_t"], guidance_weight=2.0)
+
+    # Update with v_t only (pass None for other fields)
+    enabled_tracker.track(time=0.3, v_t=sample_tensors["v_t"], x_t=None, guidance_weight=None)
+
+    # Original values should be preserved
+    steps = enabled_tracker.get_all_steps()
+    assert torch.equal(steps[0].x_t, sample_tensors["x_t"])  # Still has x_t
+    assert torch.equal(steps[0].v_t, sample_tensors["v_t"])  # Now has v_t
+    assert steps[0].guidance_weight == 2.0  # Still has guidance_weight
+
+
+# ====================== Tracker.maxlen Tests ======================
+
+
+def test_tracker_enforces_maxlen():
+    """Test that tracker enforces maxlen limit."""
+    tracker = Tracker(enabled=True, maxlen=3)
+
+    # Add 5 steps
+    for i in range(5):
+        time = 1.0 - i * 0.1  # 1.0, 0.9, 0.8, 0.7, 0.6
+        tracker.track(time=time, x_t=torch.randn(1, 10, 6))
+
+    # Should only keep the last 3
+    assert len(tracker) == 3
+
+    # Verify oldest steps were removed (should have 0.6, 0.7, 0.8)
+    steps = tracker.get_all_steps()
+    times = sorted([step.time for step in steps])
+    assert times == [0.6, 0.7, 0.8]
+
+
+def test_tracker_step_idx_increments_despite_maxlen():
+    """Test that step_idx continues incrementing even when maxlen is enforced."""
+    tracker = Tracker(enabled=True, maxlen=2)
+
+    # Add 4 steps
+    for i in range(4):
+        time = 1.0 - i * 0.1
+        tracker.track(time=time, x_t=torch.randn(1, 10, 6))
+
+    # Should have 2 steps with step_idx 2 and 3 (oldest removed)
+    steps = sorted(tracker.get_all_steps(), key=lambda s: s.step_idx)
+    assert len(steps) == 2
+    assert steps[0].step_idx == 2
+    assert steps[1].step_idx == 3
+
+
+def test_tracker_without_maxlen_keeps_all():
+    """Test that tracker without maxlen keeps all steps."""
+    tracker = Tracker(enabled=True, maxlen=None)
+
+    # Add 100 steps
+    for i in range(100):
+        time = 1.0 - i * 0.01
+        tracker.track(time=time, x_t=torch.randn(1, 10, 6))
+
+    assert len(tracker) == 100
+
+
+def test_get_all_steps_returns_empty_when_disabled(disabled_tracker):
+    """Test get_all_steps returns empty list when disabled."""
+    steps = disabled_tracker.get_all_steps()
+    assert steps == []
+    assert isinstance(steps, list)
+
+
+def test_get_all_steps_returns_empty_when_no_steps(enabled_tracker):
+    """Test get_all_steps returns empty list when no steps tracked."""
+    steps = enabled_tracker.get_all_steps()
+    assert steps == []
+
+
+def test_get_all_steps_returns_all_tracked_steps(enabled_tracker, sample_tensors):
+    """Test get_all_steps returns all tracked steps."""
+    # Track 5 steps
+    for i in range(5):
+        time = 1.0 - i * 0.1
+        enabled_tracker.track(time=time, x_t=sample_tensors["x_t"])
+
+    steps = enabled_tracker.get_all_steps()
+    assert len(steps) == 5
+
+    # Verify all are DebugStep instances
+    for step in steps:
+        assert isinstance(step, DebugStep)
+
+
+def test_get_all_steps_preserves_insertion_order(enabled_tracker):
+    """Test that get_all_steps preserves insertion order (Python 3.7+)."""
+    times = [0.9, 0.8, 0.7, 0.6, 0.5]
+    for time in times:
+        enabled_tracker.track(time=time, x_t=torch.randn(1, 10, 6))
+
+    steps = enabled_tracker.get_all_steps()
+    retrieved_times = [step.time for step in steps]
+
+    # Should be in insertion order
+    assert retrieved_times == times
+
+
+# ====================== Tracker.__len__() Tests ======================
+
+
+def test_len_returns_zero_when_disabled(disabled_tracker):
+    """Test __len__ returns 0 when tracker is disabled."""
+    assert len(disabled_tracker) == 0
+
+
+def test_len_returns_zero_when_empty(enabled_tracker):
+    """Test __len__ returns 0 when no steps are tracked."""
+    assert len(enabled_tracker) == 0
+
+
+def test_len_returns_correct_count(enabled_tracker, sample_tensors):
+    """Test __len__ returns correct number of tracked steps."""
+    assert len(enabled_tracker) == 0
+
+    enabled_tracker.track(time=1.0, x_t=sample_tensors["x_t"])
+    assert len(enabled_tracker) == 1
+
+    enabled_tracker.track(time=0.9, x_t=sample_tensors["x_t"])
+    assert len(enabled_tracker) == 2
+
+    enabled_tracker.track(time=0.8, x_t=sample_tensors["x_t"])
+    assert len(enabled_tracker) == 3
+
+
+def test_len_after_reset(enabled_tracker, sample_tensors):
+    """Test __len__ returns 0 after reset."""
+    enabled_tracker.track(time=1.0, x_t=sample_tensors["x_t"])
+    enabled_tracker.track(time=0.9, x_t=sample_tensors["x_t"])
+    assert len(enabled_tracker) == 2
+
+    enabled_tracker.reset()
+    assert len(enabled_tracker) == 0
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_tracker_handles_gpu_tensors():
+    """Test tracker correctly handles GPU tensors."""
+    tracker = Tracker(enabled=True, maxlen=10)
+    x_t_gpu = torch.randn(1, 50, 6, device="cuda")
+
+    tracker.track(time=1.0, x_t=x_t_gpu)
+
+    steps = tracker.get_all_steps()
+    # Tracker should clone and detach tensors
+    assert steps[0].x_t.device.type == "cuda"
+
+
+def test_tracker_with_varying_tensor_shapes(enabled_tracker):
+    """Test tracker handles varying tensor shapes across steps."""
+    enabled_tracker.track(time=1.0, x_t=torch.randn(1, 50, 6))
+    enabled_tracker.track(time=0.9, x_t=torch.randn(1, 25, 6))
+    enabled_tracker.track(time=0.8, x_t=torch.randn(2, 50, 8))
+
+    steps = enabled_tracker.get_all_steps()
+    assert len(steps) == 3
+    assert steps[0].x_t.shape == (1, 50, 6)
+    assert steps[1].x_t.shape == (1, 25, 6)
+    assert steps[2].x_t.shape == (2, 50, 8)
diff --git a/lerobot/tests/policies/rtc/test_latency_tracker.py b/lerobot/tests/policies/rtc/test_latency_tracker.py
new file mode 100644
index 0000000000000000000000000000000000000000..ee8ca9e1152252e485a217eae21e3e1df6cdecbd
--- /dev/null
+++ b/lerobot/tests/policies/rtc/test_latency_tracker.py
@@ -0,0 +1,322 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Tests for RTC LatencyTracker module."""
+
+import pytest
+
+from lerobot.policies.rtc.latency_tracker import LatencyTracker
+
+# ====================== Fixtures ======================
+
+
+@pytest.fixture
+def tracker():
+    """Create a LatencyTracker with default maxlen."""
+    return LatencyTracker(maxlen=100)
+
+
+@pytest.fixture
+def small_tracker():
+    """Create a LatencyTracker with small maxlen for overflow testing."""
+    return LatencyTracker(maxlen=5)
+
+
+# ====================== Initialization Tests ======================
+
+
+def test_latency_tracker_initialization():
+    """Test LatencyTracker initializes correctly."""
+    tracker = LatencyTracker(maxlen=50)
+    assert len(tracker) == 0
+    assert tracker.max_latency == 0.0
+    assert tracker.max() == 0.0
+
+
+def test_latency_tracker_default_maxlen():
+    """Test LatencyTracker uses default maxlen."""
+    tracker = LatencyTracker()
+    # Should accept default maxlen=100
+    assert len(tracker) == 0
+
+
+# ====================== add() Tests ======================
+
+
+def test_add_single_latency(tracker):
+    """Test adding a single latency value."""
+    tracker.add(0.5)
+    assert len(tracker) == 1
+    assert tracker.max() == 0.5
+
+
+def test_add_multiple_latencies(tracker):
+    """Test adding multiple latency values."""
+    latencies = [0.1, 0.5, 0.3, 0.8, 0.2]
+    for lat in latencies:
+        tracker.add(lat)
+
+    assert len(tracker) == 5
+    assert tracker.max() == 0.8
+
+
+def test_add_negative_latency_ignored(tracker):
+    """Test that negative latencies are ignored."""
+    tracker.add(0.5)
+    tracker.add(-0.1)
+    tracker.add(0.3)
+
+    # Should only have 2 valid latencies
+    assert len(tracker) == 2
+    assert tracker.max() == 0.5
+
+
+def test_add_zero_latency(tracker):
+    """Test adding zero latency."""
+    tracker.add(0.0)
+    assert len(tracker) == 1
+    assert tracker.max() == 0.0
+
+
+def test_add_converts_to_float(tracker):
+    """Test add() converts input to float."""
+    tracker.add(5)  # Integer
+    tracker.add("3.5")  # String
+
+    assert len(tracker) == 2
+    assert tracker.max() == 5.0
+
+
+def test_add_updates_max_latency(tracker):
+    """Test that max_latency is updated correctly."""
+    tracker.add(0.5)
+    assert tracker.max_latency == 0.5
+
+    tracker.add(0.3)
+    assert tracker.max_latency == 0.5  # Should not decrease
+
+    tracker.add(0.9)
+    assert tracker.max_latency == 0.9  # Should increase
+
+
+# ====================== reset() Tests ======================
+
+
+def test_reset_clears_values(tracker):
+    """Test reset() clears all values."""
+    tracker.add(0.5)
+    tracker.add(0.8)
+    tracker.add(0.3)
+    assert len(tracker) == 3
+
+    tracker.reset()
+    assert len(tracker) == 0
+    assert tracker.max_latency == 0.0
+
+
+def test_reset_clears_max_latency(tracker):
+    """Test reset() resets max_latency."""
+    tracker.add(1.5)
+    assert tracker.max_latency == 1.5
+
+    tracker.reset()
+    assert tracker.max_latency == 0.0
+
+
+def test_reset_allows_new_values(tracker):
+    """Test that tracker works correctly after reset."""
+    tracker.add(0.5)
+    tracker.reset()
+
+    tracker.add(0.3)
+    assert len(tracker) == 1
+    assert tracker.max() == 0.3
+
+
+# ====================== max() Tests ======================
+
+
+def test_max_returns_zero_when_empty(tracker):
+    """Test max() returns 0.0 when tracker is empty."""
+    assert tracker.max() == 0.0
+
+
+def test_max_returns_maximum_value(tracker):
+    """Test max() returns the maximum latency."""
+    latencies = [0.2, 0.8, 0.3, 0.5, 0.1]
+    for lat in latencies:
+        tracker.add(lat)
+
+    assert tracker.max() == 0.8
+
+
+def test_max_persists_after_sliding_window(small_tracker):
+    """Test max() persists even after values slide out of window."""
+    # Add values that will exceed maxlen=5
+    small_tracker.add(0.1)
+    small_tracker.add(0.9)  # This is max
+    small_tracker.add(0.2)
+    small_tracker.add(0.3)
+    small_tracker.add(0.4)
+    small_tracker.add(0.5)  # This pushes out 0.1
+
+    # Max should still be 0.9 even though only last 5 values kept
+    assert small_tracker.max() == 0.9
+
+
+def test_max_after_reset(tracker):
+    """Test max() returns 0.0 after reset."""
+    tracker.add(1.5)
+    tracker.reset()
+    assert tracker.max() == 0.0
+
+
+# ====================== p95() Tests ======================
+
+
+def test_p95_returns_zero_when_empty(tracker):
+    """Test p95() returns 0.0 when tracker is empty."""
+    assert tracker.p95() == 0.0
+
+
+def test_p95_returns_95th_percentile(tracker):
+    """Test p95() returns the 95th percentile."""
+    # Add 100 values
+    for i in range(100):
+        tracker.add(i / 100.0)
+
+    p95 = tracker.p95()
+    assert 0.93 <= p95 <= 0.96
+
+
+def test_p95_equals_percentile_95(tracker):
+    """Test p95() equals percentile(0.95)."""
+    for i in range(50):
+        tracker.add(i / 50.0)
+
+    assert tracker.p95() == tracker.percentile(0.95)
+
+
+# ====================== Edge Cases Tests ======================
+
+
+def test_single_value(tracker):
+    """Test tracker behavior with single value."""
+    tracker.add(0.75)
+
+    assert len(tracker) == 1
+    assert tracker.max() == 0.75
+    assert tracker.percentile(0.0) == 0.75
+    assert tracker.percentile(0.5) == 0.75
+    assert tracker.percentile(1.0) == 0.75
+
+
+def test_all_same_values(tracker):
+    """Test tracker with all identical values."""
+    for _ in range(10):
+        tracker.add(0.5)
+
+    assert len(tracker) == 10
+    assert tracker.max() == 0.5
+    assert tracker.percentile(0.0) == 0.5
+    assert tracker.percentile(0.5) == 0.5
+    assert tracker.percentile(1.0) == 0.5
+
+
+def test_very_small_values(tracker):
+    """Test tracker with very small float values."""
+    tracker.add(1e-10)
+    tracker.add(2e-10)
+    tracker.add(3e-10)
+
+    assert len(tracker) == 3
+    assert tracker.max() == pytest.approx(3e-10)
+
+
+def test_very_large_values(tracker):
+    """Test tracker with very large float values."""
+    tracker.add(1e10)
+    tracker.add(2e10)
+    tracker.add(3e10)
+
+    assert len(tracker) == 3
+    assert tracker.max() == pytest.approx(3e10)
+
+
+# ====================== Integration Tests ======================
+
+
+def test_typical_usage_pattern(tracker):
+    """Test a typical usage pattern of the tracker."""
+    # Simulate adding latencies over time
+    latencies = [0.05, 0.08, 0.12, 0.07, 0.15, 0.09, 0.11, 0.06, 0.14, 0.10]
+
+    for lat in latencies:
+        tracker.add(lat)
+
+    # Check statistics
+    assert len(tracker) == 10
+    assert tracker.max() == 0.15
+
+    # p95 should be close to max since we have only 10 values
+    p95 = tracker.p95()
+    assert p95 >= tracker.percentile(0.5)  # p95 should be >= median
+    assert p95 <= tracker.max()  # p95 should be <= max
+
+
+def test_reset_and_reuse(tracker):
+    """Test resetting and reusing tracker."""
+    # First batch
+    tracker.add(1.0)
+    tracker.add(2.0)
+    assert tracker.max() == 2.0
+
+    # Reset
+    tracker.reset()
+
+    # Second batch
+    tracker.add(0.5)
+    tracker.add(0.8)
+    assert len(tracker) == 2
+    assert tracker.max() == 0.8
+    assert tracker.percentile(0.5) <= 0.8
+
+
+# ====================== Type Conversion Tests ======================
+
+
+def test_add_with_integer(tracker):
+    """Test adding integer values."""
+    tracker.add(5)
+    assert len(tracker) == 1
+    assert tracker.max() == 5.0
+
+
+def test_add_with_string_number(tracker):
+    """Test adding string representation of number."""
+    tracker.add("3.14")
+    assert len(tracker) == 1
+    assert tracker.max() == pytest.approx(3.14)
+
+
+def test_percentile_converts_q_to_float(tracker):
+    """Test percentile converts q parameter to float."""
+    tracker.add(0.5)
+    tracker.add(0.8)
+
+    # Pass integer q
+    result = tracker.percentile(1)
+    assert result == 0.8
diff --git a/lerobot/tests/policies/rtc/test_modeling_rtc.py b/lerobot/tests/policies/rtc/test_modeling_rtc.py
new file mode 100644
index 0000000000000000000000000000000000000000..e7fdc09c65720488ac83f93f9ddc17179d333b67
--- /dev/null
+++ b/lerobot/tests/policies/rtc/test_modeling_rtc.py
@@ -0,0 +1,773 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Tests for RTC modeling module (RTCProcessor)."""
+
+import pytest
+import torch
+
+from lerobot.configs.types import RTCAttentionSchedule
+from lerobot.policies.rtc.configuration_rtc import RTCConfig
+from lerobot.policies.rtc.modeling_rtc import RTCProcessor
+
+# ====================== Fixtures ======================
+
+
+@pytest.fixture
+def rtc_config_debug_enabled():
+    """Create RTC config with debug enabled."""
+    return RTCConfig(
+        enabled=True,
+        prefix_attention_schedule=RTCAttentionSchedule.LINEAR,
+        max_guidance_weight=10.0,
+        execution_horizon=10,
+        debug=True,
+        debug_maxlen=100,
+    )
+
+
+@pytest.fixture
+def rtc_config_debug_disabled():
+    """Create RTC config with debug disabled."""
+    return RTCConfig(
+        enabled=True,
+        prefix_attention_schedule=RTCAttentionSchedule.LINEAR,
+        max_guidance_weight=10.0,
+        execution_horizon=10,
+        debug=False,
+    )
+
+
+@pytest.fixture
+def rtc_processor_debug_enabled(rtc_config_debug_enabled):
+    """Create RTCProcessor with debug enabled."""
+    return RTCProcessor(rtc_config_debug_enabled)
+
+
+@pytest.fixture
+def rtc_processor_debug_disabled(rtc_config_debug_disabled):
+    """Create RTCProcessor with debug disabled."""
+    return RTCProcessor(rtc_config_debug_disabled)
+
+
+@pytest.fixture
+def sample_x_t():
+    """Create sample x_t tensor (batch, time, action_dim)."""
+    return torch.randn(1, 50, 6)
+
+
+@pytest.fixture
+def sample_prev_chunk():
+    """Create sample previous chunk tensor."""
+    return torch.randn(1, 50, 6)
+
+
+# ====================== Initialization Tests ======================
+
+
+def test_rtc_processor_initialization_with_debug(rtc_config_debug_enabled):
+    """Test RTCProcessor initializes with debug tracker."""
+    processor = RTCProcessor(rtc_config_debug_enabled)
+    assert processor.rtc_config == rtc_config_debug_enabled
+    assert processor.tracker is not None
+    assert processor.tracker.enabled is True
+
+
+def test_rtc_processor_initialization_without_debug(rtc_config_debug_disabled):
+    """Test RTCProcessor initializes without debug tracker."""
+    processor = RTCProcessor(rtc_config_debug_disabled)
+    assert processor.rtc_config == rtc_config_debug_disabled
+    assert processor.tracker is None
+
+
+# ====================== Tracker Proxy Methods Tests ======================
+
+
+def test_track_when_tracker_enabled(rtc_processor_debug_enabled, sample_x_t):
+    """Test track() forwards to tracker when enabled."""
+    rtc_processor_debug_enabled.track(
+        time=torch.tensor(0.5),
+        x_t=sample_x_t,
+        v_t=sample_x_t,
+        guidance_weight=2.0,
+    )
+
+    # Should have tracked one step
+    steps = rtc_processor_debug_enabled.get_all_debug_steps()
+    assert len(steps) == 1
+    assert steps[0].time == 0.5
+
+
+def test_track_when_tracker_disabled(rtc_processor_debug_disabled, sample_x_t):
+    """Test track() does nothing when tracker disabled."""
+    # Should not raise error
+    rtc_processor_debug_disabled.track(
+        time=torch.tensor(0.5),
+        x_t=sample_x_t,
+        v_t=sample_x_t,
+    )
+
+    # Should return empty list
+    steps = rtc_processor_debug_disabled.get_all_debug_steps()
+    assert len(steps) == 0
+
+
+def test_get_all_debug_steps_when_enabled(rtc_processor_debug_enabled, sample_x_t):
+    """Test get_all_debug_steps() returns tracked steps."""
+    rtc_processor_debug_enabled.track(time=torch.tensor(0.5), x_t=sample_x_t)
+    rtc_processor_debug_enabled.track(time=torch.tensor(0.4), x_t=sample_x_t)
+
+    steps = rtc_processor_debug_enabled.get_all_debug_steps()
+    assert len(steps) == 2
+
+
+def test_get_all_debug_steps_when_disabled(rtc_processor_debug_disabled):
+    """Test get_all_debug_steps() returns empty list when disabled."""
+    steps = rtc_processor_debug_disabled.get_all_debug_steps()
+    assert steps == []
+    assert isinstance(steps, list)
+
+
+def test_is_debug_enabled_when_tracker_exists(rtc_processor_debug_enabled):
+    """Test is_debug_enabled() returns True when tracker enabled."""
+    assert rtc_processor_debug_enabled.is_debug_enabled() is True
+
+
+def test_is_debug_enabled_when_tracker_disabled(rtc_processor_debug_disabled):
+    """Test is_debug_enabled() returns False when tracker disabled."""
+    assert rtc_processor_debug_disabled.is_debug_enabled() is False
+
+
+def test_reset_tracker_when_enabled(rtc_processor_debug_enabled, sample_x_t):
+    """Test reset_tracker() clears tracked steps."""
+    rtc_processor_debug_enabled.track(time=torch.tensor(0.5), x_t=sample_x_t)
+    rtc_processor_debug_enabled.track(time=torch.tensor(0.4), x_t=sample_x_t)
+    assert len(rtc_processor_debug_enabled.get_all_debug_steps()) == 2
+
+    rtc_processor_debug_enabled.reset_tracker()
+    assert len(rtc_processor_debug_enabled.get_all_debug_steps()) == 0
+
+
+def test_reset_tracker_when_disabled(rtc_processor_debug_disabled):
+    """Test reset_tracker() doesn't error when tracker disabled."""
+    rtc_processor_debug_disabled.reset_tracker()  # Should not raise
+
+
+# ====================== get_prefix_weights Tests ======================
+
+
+def test_get_prefix_weights_zeros_schedule():
+    """Test get_prefix_weights with ZEROS schedule."""
+    config = RTCConfig(prefix_attention_schedule=RTCAttentionSchedule.ZEROS)
+    processor = RTCProcessor(config)
+
+    weights = processor.get_prefix_weights(start=5, end=10, total=20)
+
+    # First 5 should be 1.0, rest should be 0.0
+    assert weights.shape == (20,)
+    assert torch.all(weights[:5] == 1.0)
+    assert torch.all(weights[5:] == 0.0)
+
+
+def test_get_prefix_weights_ones_schedule():
+    """Test get_prefix_weights with ONES schedule."""
+    config = RTCConfig(prefix_attention_schedule=RTCAttentionSchedule.ONES)
+    processor = RTCProcessor(config)
+
+    weights = processor.get_prefix_weights(start=5, end=15, total=20)
+
+    # First 15 should be 1.0, rest should be 0.0
+    assert weights.shape == (20,)
+    assert torch.all(weights[:15] == 1.0)
+    assert torch.all(weights[15:] == 0.0)
+
+
+def test_get_prefix_weights_linear_schedule():
+    """Test get_prefix_weights with LINEAR schedule."""
+    config = RTCConfig(prefix_attention_schedule=RTCAttentionSchedule.LINEAR)
+    processor = RTCProcessor(config)
+
+    weights = processor.get_prefix_weights(start=5, end=14, total=25)
+
+    # Should have shape (20,)
+    assert weights.shape == (25,)
+
+    # First 5 should be 1.0 (leading ones)
+    assert torch.all(weights[:5] == 1.0)
+
+    # Middle section (5:15) should be linearly decreasing from 1 to 0
+    middle_weights = torch.tensor([0.9, 0.8, 0.7, 0.6, 0.5, 0.4, 0.3, 0.2, 0.1])
+    assert torch.allclose(weights[5:14], middle_weights)
+
+    # Last 5 should be 0.0 (trailing zeros)
+    assert torch.all(weights[14:] == 0.0)
+
+
+def test_get_prefix_weights_exp_schedule():
+    """Test get_prefix_weights with EXP schedule."""
+    config = RTCConfig(prefix_attention_schedule=RTCAttentionSchedule.EXP)
+    processor = RTCProcessor(config)
+
+    weights = processor.get_prefix_weights(start=5, end=14, total=25)
+
+    # Should have shape (20,)
+    assert weights.shape == (25,)
+
+    # First 5 should be 1.0 (leading ones)
+    assert torch.all(weights[:5] == 1.0)
+
+    # Middle section should be exponentially weighted
+    middle_weights = torch.tensor([0.7645, 0.5706, 0.4130, 0.2871, 0.1888, 0.1145, 0.0611, 0.0258, 0.0061])
+    assert torch.allclose(weights[5:14], middle_weights, atol=1e-4)
+
+    # Last 5 should be 0.0 (trailing zeros)
+    assert torch.all(weights[14:] == 0.0)
+
+
+def test_get_prefix_weights_with_start_equals_end():
+    """Test get_prefix_weights when start equals end."""
+    config = RTCConfig(prefix_attention_schedule=RTCAttentionSchedule.LINEAR)
+    processor = RTCProcessor(config)
+
+    weights = processor.get_prefix_weights(start=10, end=10, total=20)
+
+    # Should have ones up to start, then zeros
+    assert torch.all(weights[:10] == 1.0)
+    assert torch.all(weights[10:] == 0.0)
+
+
+def test_get_prefix_weights_with_start_greater_than_end():
+    """Test get_prefix_weights when start > end (gets clamped)."""
+    config = RTCConfig(prefix_attention_schedule=RTCAttentionSchedule.LINEAR)
+    processor = RTCProcessor(config)
+
+    # start > end should use min(start, end) = end
+    weights = processor.get_prefix_weights(start=15, end=10, total=20)
+
+    # Should have ones up to end (10), then zeros
+    assert torch.all(weights[:10] == 1.0)
+    assert torch.all(weights[10:] == 0.0)
+
+
+# ====================== Helper Method Tests ======================
+
+
+def test_linweights_with_end_equals_start():
+    """Test _linweights when end equals start."""
+    config = RTCConfig()
+    processor = RTCProcessor(config)
+
+    weights = processor._linweights(start=10, end=10, total=20)
+
+    # Should return empty tensor
+    assert len(weights) == 0
+
+
+def test_linweights_with_end_less_than_start():
+    """Test _linweights when end < start."""
+    config = RTCConfig()
+    processor = RTCProcessor(config)
+
+    weights = processor._linweights(start=15, end=10, total=20)
+
+    # Should return empty tensor
+    assert len(weights) == 0
+
+
+def test_add_trailing_zeros_normal():
+    """Test _add_trailing_zeros adds zeros correctly."""
+    config = RTCConfig()
+    processor = RTCProcessor(config)
+
+    weights = torch.tensor([1.0, 0.8, 0.6, 0.4, 0.2])
+    result = processor._add_trailing_zeros(weights, total=10, end=5)
+
+    # Should add 5 zeros (total - end = 10 - 5 = 5)
+    assert len(result) == 10
+    assert torch.all(result[:5] == weights)
+    assert torch.all(result[5:] == 0.0)
+
+
+def test_add_trailing_zeros_no_zeros_needed():
+    """Test _add_trailing_zeros when no zeros needed."""
+    config = RTCConfig()
+    processor = RTCProcessor(config)
+
+    weights = torch.tensor([1.0, 0.8, 0.6])
+    result = processor._add_trailing_zeros(weights, total=3, end=5)
+
+    # zeros_len = 3 - 5 = -2 <= 0, so no zeros added
+    assert torch.equal(result, weights)
+
+
+def test_add_leading_ones_normal():
+    """Test _add_leading_ones adds ones correctly."""
+    config = RTCConfig()
+    processor = RTCProcessor(config)
+
+    weights = torch.tensor([0.8, 0.6, 0.4, 0.2, 0.0])
+    result = processor._add_leading_ones(weights, start=3, total=10)
+
+    # Should add 3 ones at the start
+    assert len(result) == 8
+    assert torch.all(result[:3] == 1.0)
+    assert torch.all(result[3:] == weights)
+
+
+def test_add_leading_ones_no_ones_needed():
+    """Test _add_leading_ones when no ones needed."""
+    config = RTCConfig()
+    processor = RTCProcessor(config)
+
+    weights = torch.tensor([0.8, 0.6, 0.4])
+    result = processor._add_leading_ones(weights, start=0, total=10)
+
+    # ones_len = 0, so no ones added
+    assert torch.equal(result, weights)
+
+
+def test_get_prefix_weights_with_start_equals_total():
+    """Test get_prefix_weights when start equals total."""
+    config = RTCConfig(prefix_attention_schedule=RTCAttentionSchedule.LINEAR)
+    processor = RTCProcessor(config)
+
+    weights = processor.get_prefix_weights(start=10, end=10, total=20)
+
+    # Should have ones up to start, then zeros
+    assert len(weights) == 20
+    assert torch.all(weights[:10] == 1.0)
+    assert torch.all(weights[10:] == 0.0)
+
+
+def test_get_prefix_weights_with_total_less_than_start():
+    """Test get_prefix_weights when total less than start."""
+    config = RTCConfig(prefix_attention_schedule=RTCAttentionSchedule.LINEAR)
+    processor = RTCProcessor(config)
+
+    weights = processor.get_prefix_weights(start=10, end=10, total=5)
+
+    # Should have ones up to start, then zeros
+    assert len(weights) == 5
+    assert torch.all(weights == 1.0)
+
+
+# ====================== denoise_step Tests ======================
+
+
+def test_denoise_step_without_prev_chunk(rtc_processor_debug_disabled):
+    """Test denoise_step without previous chunk (no guidance)."""
+    x_t = torch.randn(1, 50, 6)
+
+    # Mock denoiser that returns fixed velocity
+    def mock_denoiser(x):
+        return torch.ones_like(x) * 0.5
+
+    result = rtc_processor_debug_disabled.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=None,
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    # Should return v_t unchanged (no guidance)
+    expected = mock_denoiser(x_t)
+    assert torch.allclose(result, expected)
+
+
+def test_denoise_step_with_prev_chunk(rtc_processor_debug_disabled):
+    """Test denoise_step with previous chunk applies guidance."""
+    x_t = torch.ones(1, 20, 1)
+    prev_chunk = torch.full((1, 20, 1), 0.1)
+
+    def mock_denoiser(x):
+        return x * 0.5
+
+    result = rtc_processor_debug_disabled.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=prev_chunk,
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    expected_result = torch.tensor(
+        [
+            [
+                [1.8000],
+                [1.8000],
+                [1.8000],
+                [1.8000],
+                [1.8000],
+                [1.5833],
+                [1.3667],
+                [1.1500],
+                [0.9333],
+                [0.7167],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+            ]
+        ]
+    )
+
+    assert torch.allclose(result, expected_result, atol=1e-4)
+
+
+def test_denoise_step_adds_batch_dimension():
+    """Test denoise_step handles 2D input by adding batch dimension."""
+    config = RTCConfig(execution_horizon=10, max_guidance_weight=5.0)
+    processor = RTCProcessor(config)
+
+    # 2D input (no batch dimension)
+    x_t = torch.randn(10, 6)
+    prev_chunk = torch.randn(5, 6)
+
+    def mock_denoiser(x):
+        return x * 0.5
+
+    result = processor.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=prev_chunk,
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    # Output should be 2D (batch dimension removed)
+    assert result.ndim == 2
+    assert result.shape == (10, 6)
+
+
+def test_denoise_step_uses_custom_execution_horizon():
+    """Test denoise_step uses custom execution_horizon parameter."""
+    config = RTCConfig(execution_horizon=10)
+    processor = RTCProcessor(config)
+
+    x_t = torch.ones(1, 20, 1)
+    prev_chunk = torch.full((1, 15, 1), 0.1)
+
+    def mock_denoiser(x):
+        return x * 0.5
+
+    result = processor.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=prev_chunk,
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=mock_denoiser,
+        execution_horizon=15,
+    )
+
+    expected_result = torch.tensor(
+        [
+            [
+                [1.8000],
+                [1.8000],
+                [1.8000],
+                [1.8000],
+                [1.8000],
+                [1.6818],
+                [1.5636],
+                [1.4455],
+                [1.3273],
+                [1.2091],
+                [1.0909],
+                [0.9727],
+                [0.8545],
+                [0.7364],
+                [0.6182],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+            ]
+        ]
+    )
+
+    assert torch.allclose(result, expected_result, atol=1e-4)
+
+
+def test_denoise_step_guidance_weight_at_time_zero():
+    """Test denoise_step handles time=0 (tau=1) without NaN/Inf."""
+    config = RTCConfig(max_guidance_weight=10.0)
+    processor = RTCProcessor(config)
+
+    x_t = torch.ones(1, 20, 1)
+    prev_chunk = torch.full((1, 20, 1), 0.1)
+
+    def mock_denoiser(x):
+        return x * 0.5
+
+    result = processor.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=prev_chunk,
+        inference_delay=5,
+        time=torch.tensor(0.0),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    expected_result = torch.tensor(
+        [
+            [
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+                [0.5000],
+            ]
+        ]
+    )
+
+    assert torch.allclose(result, expected_result, atol=1e-4)
+
+
+def test_denoise_step_with_real_denoise_step_partial():
+    """Test denoise_step with a real denoiser."""
+    config = RTCConfig(max_guidance_weight=10.0)
+    processor = RTCProcessor(config)
+
+    batch_size = 10
+    action_dim = 6
+    chunk_size = 20
+
+    x_t = torch.ones(batch_size, chunk_size, action_dim)
+    prev_chunk = torch.full((batch_size, chunk_size, action_dim), 0.1)
+
+    velocity_function = torch.nn.Sequential(
+        torch.nn.Linear(action_dim, 1000),
+        torch.nn.ReLU(),
+        torch.nn.Linear(1000, 256),
+        torch.nn.ReLU(),
+        torch.nn.Linear(256, action_dim),
+    )
+
+    def mock_denoiser(x):
+        return velocity_function(x)
+
+    result = processor.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=prev_chunk,
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    assert result.shape == (batch_size, chunk_size, action_dim)
+
+
+def test_denoise_step_guidance_weight_at_time_one():
+    """Test denoise_step handles time=1 (tau=0) with max_guidance_weight clamping."""
+    config = RTCConfig(max_guidance_weight=10.0)
+    processor = RTCProcessor(config)
+
+    x_t = torch.randn(1, 50, 6)
+    prev_chunk = torch.randn(1, 50, 6)
+
+    def mock_denoiser(x):
+        return torch.ones_like(x) * 0.5
+
+    # Time = 1 => tau = 0, c = (1-tau)/tau = 1/0 = inf (clamped to max_guidance_weight)
+    result = processor.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=prev_chunk,
+        inference_delay=5,
+        time=torch.tensor(1.0),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    # Should clamp to max_guidance_weight (no Inf)
+    assert not torch.any(torch.isinf(result))
+
+
+def test_denoise_step_tracks_debug_info(rtc_processor_debug_enabled):
+    """Test denoise_step tracks debug information when enabled."""
+    x_t = torch.randn(1, 50, 6)
+    prev_chunk = torch.randn(1, 50, 6)
+
+    def mock_denoiser(x):
+        return torch.ones_like(x) * 0.5
+
+    rtc_processor_debug_enabled.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=prev_chunk,
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    # Should have tracked one step
+    steps = rtc_processor_debug_enabled.get_all_debug_steps()
+    assert len(steps) == 1
+
+    # Check tracked values
+    step = steps[0]
+    assert step.time == 0.5
+    assert step.x1_t is not None
+    assert step.correction is not None
+    assert step.err is not None
+    assert step.weights is not None
+    assert step.guidance_weight is not None
+    assert step.inference_delay == 5
+
+
+def test_denoise_step_doesnt_track_without_debug(rtc_processor_debug_disabled):
+    """Test denoise_step doesn't track when debug disabled."""
+    x_t = torch.randn(1, 50, 6)
+    prev_chunk = torch.randn(1, 50, 6)
+
+    def mock_denoiser(x):
+        return torch.ones_like(x) * 0.5
+
+    rtc_processor_debug_disabled.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=prev_chunk,
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    # Should not track
+    steps = rtc_processor_debug_disabled.get_all_debug_steps()
+    assert len(steps) == 0
+
+
+# ====================== Integration Tests ======================
+
+
+def test_denoise_step_full_workflow():
+    """Test complete denoise_step workflow."""
+    config = RTCConfig(
+        enabled=True,
+        prefix_attention_schedule=RTCAttentionSchedule.LINEAR,
+        max_guidance_weight=5.0,
+        execution_horizon=10,
+        debug=True,
+    )
+    processor = RTCProcessor(config)
+
+    # Simulate two denoising steps
+    x_t1 = torch.randn(1, 50, 6)
+    x_t2 = torch.randn(1, 50, 6)
+
+    def mock_denoiser(x):
+        return torch.randn_like(x) * 0.1
+
+    # First step - no guidance
+    result1 = processor.denoise_step(
+        x_t=x_t1,
+        prev_chunk_left_over=None,
+        inference_delay=5,
+        time=torch.tensor(0.8),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    # Second step - with guidance
+    result2 = processor.denoise_step(
+        x_t=x_t2,
+        prev_chunk_left_over=result1,
+        inference_delay=5,
+        time=torch.tensor(0.6),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    # Both should complete successfully
+    assert result1.shape == (1, 50, 6)
+    assert result2.shape == (1, 50, 6)
+
+    # Should have tracked one step (second one, first had no prev_chunk)
+    steps = processor.get_all_debug_steps()
+    assert len(steps) == 1
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_denoise_step_with_cuda_tensors():
+    """Test denoise_step works with CUDA tensors."""
+    config = RTCConfig(execution_horizon=10, max_guidance_weight=5.0)
+    processor = RTCProcessor(config)
+
+    x_t = torch.randn(1, 50, 6, device="cuda")
+    prev_chunk = torch.randn(1, 50, 6, device="cuda")
+
+    def mock_denoiser(x):
+        return torch.ones_like(x) * 0.5
+
+    result = processor.denoise_step(
+        x_t=x_t,
+        prev_chunk_left_over=prev_chunk,
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=mock_denoiser,
+    )
+
+    # Result should be on CUDA
+    assert result.device.type == "cuda"
+    assert result.shape == x_t.shape
+
+
+def test_denoise_step_deterministic_with_same_inputs():
+    """Test denoise_step produces same output with same inputs."""
+    config = RTCConfig(execution_horizon=10, max_guidance_weight=5.0)
+    processor = RTCProcessor(config)
+
+    torch.manual_seed(42)
+    x_t = torch.randn(1, 50, 6)
+    prev_chunk = torch.randn(1, 50, 6)
+
+    def deterministic_denoiser(x):
+        return torch.ones_like(x) * 0.5
+
+    result1 = processor.denoise_step(
+        x_t=x_t.clone(),
+        prev_chunk_left_over=prev_chunk.clone(),
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=deterministic_denoiser,
+    )
+
+    result2 = processor.denoise_step(
+        x_t=x_t.clone(),
+        prev_chunk_left_over=prev_chunk.clone(),
+        inference_delay=5,
+        time=torch.tensor(0.5),
+        original_denoise_step_partial=deterministic_denoiser,
+    )
+
+    # Should produce identical results
+    assert torch.allclose(result1, result2)
diff --git a/lerobot/tests/policies/smolvla/test_smolvla_rtc.py b/lerobot/tests/policies/smolvla/test_smolvla_rtc.py
new file mode 100644
index 0000000000000000000000000000000000000000..53e74d940b40cc9c47734b62970cdfb5807d9eed
--- /dev/null
+++ b/lerobot/tests/policies/smolvla/test_smolvla_rtc.py
@@ -0,0 +1,323 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test SmolVLA policy with Real-Time Chunking (RTC) enabled during inference."""
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, PolicyFeature, RTCAttentionSchedule  # noqa: E402
+from lerobot.policies.factory import make_pre_post_processors  # noqa: E402
+from lerobot.policies.rtc.configuration_rtc import RTCConfig  # noqa: E402
+from lerobot.policies.smolvla.configuration_smolvla import SmolVLAConfig  # noqa: F401
+from lerobot.utils.random_utils import set_seed  # noqa: E402
+from tests.utils import require_cuda, require_package  # noqa: E402
+
+
+@require_package("transformers")
+@require_cuda
+def test_smolvla_rtc_initialization():
+    from lerobot.policies.smolvla.modeling_smolvla import SmolVLAPolicy  # noqa: F401
+
+    """Test SmolVLA policy can initialize RTC processor."""
+    set_seed(42)
+
+    config = SmolVLAConfig(max_action_dim=7, chunk_size=50)
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Instantiate policy
+    policy = SmolVLAPolicy(config)
+
+    # Verify RTC processor is initialized
+    assert hasattr(policy, "rtc_processor")
+    assert policy.rtc_processor is not None
+    assert policy.rtc_processor.rtc_config.enabled is True
+
+    print("✓ SmolVLA RTC initialization: Test passed")
+
+
+@require_package("transformers")
+@require_cuda
+def test_smolvla_rtc_initialization_without_rtc_config():
+    from lerobot.policies.smolvla.modeling_smolvla import SmolVLAPolicy  # noqa: F401
+
+    """Test SmolVLA policy can initialize without RTC config."""
+    set_seed(42)
+
+    config = SmolVLAConfig(max_action_dim=7, chunk_size=50)
+
+    # Instantiate policy
+    policy = SmolVLAPolicy(config)
+
+    # Verify RTC processor is not initialized
+    assert hasattr(policy, "rtc_processor")
+    assert policy.rtc_processor is None
+    assert policy.model.rtc_processor is None
+    assert policy._rtc_enabled() is False
+
+    print("✓ SmolVLA RTC initialization without RTC config: Test passed")
+
+
+@require_package("transformers")
+@require_cuda
+@pytest.mark.skipif(True, reason="Requires pretrained SmolVLA model weights")
+def test_smolvla_rtc_inference_with_prev_chunk():
+    from lerobot.policies.smolvla.modeling_smolvla import SmolVLAPolicy  # noqa: F401
+
+    """Test SmolVLA policy inference with RTC and previous chunk."""
+    set_seed(42)
+
+    config = SmolVLAConfig(max_action_dim=7, chunk_size=50)
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats
+    dataset_stats = {
+        "observation.state": {"mean": torch.zeros(14), "std": torch.ones(14)},
+        "action": {"mean": torch.zeros(7), "std": torch.ones(7)},
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    # Instantiate policy and create preprocessor
+    policy = SmolVLAPolicy(config)
+    policy.eval()
+    preprocessor, _ = make_pre_post_processors(
+        policy_cfg=config, pretrained_path=None, dataset_stats=dataset_stats
+    )
+
+    device = config.device
+
+    # Create dummy batch
+    batch = {
+        "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+        "task": ["Pick up the object"],
+    }
+    batch = preprocessor(batch)
+
+    # Create previous chunk
+    prev_chunk = torch.randn(1, 25, 7, dtype=torch.float32, device=device)
+
+    with torch.no_grad():
+        # Use same noise for fair comparison
+        noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+
+        # Test with RTC and previous chunk
+        actions_with_rtc = policy.predict_action_chunk(
+            batch,
+            noise=noise.clone(),
+            prev_chunk_left_over=prev_chunk,
+            inference_delay=4,
+            execution_horizon=10,
+        )
+
+        # Test without RTC for comparison
+        policy.config.rtc_config.enabled = False
+        actions_without_rtc = policy.predict_action_chunk(batch, noise=noise.clone())
+        policy.config.rtc_config.enabled = True
+
+    # Verify shapes
+    assert actions_with_rtc.shape == (1, config.chunk_size, 7)
+    assert actions_without_rtc.shape == (1, config.chunk_size, 7)
+
+    # With previous chunk, actions should be different (RTC guidance applied)
+    assert not torch.allclose(actions_with_rtc, actions_without_rtc, rtol=1e-3)
+
+    print("✓ SmolVLA RTC inference with prev_chunk: Test passed")
+
+
+@require_package("transformers")
+@require_cuda
+@pytest.mark.skipif(True, reason="Requires pretrained SmolVLA model weights")
+def test_smolvla_rtc_inference_without_prev_chunk():
+    from lerobot.policies.smolvla.modeling_smolvla import SmolVLAPolicy  # noqa: F401
+
+    """Test SmolVLA policy inference with RTC but no previous chunk (RTC should have no effect)."""
+    set_seed(42)
+
+    config = SmolVLAConfig(max_action_dim=7, chunk_size=50)
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats
+    dataset_stats = {
+        "observation.state": {"mean": torch.zeros(14), "std": torch.ones(14)},
+        "action": {"mean": torch.zeros(7), "std": torch.ones(7)},
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    # Instantiate policy and create preprocessor
+    policy = SmolVLAPolicy(config)
+    policy.eval()
+    preprocessor, _ = make_pre_post_processors(
+        policy_cfg=config, pretrained_path=None, dataset_stats=dataset_stats
+    )
+
+    device = config.device
+
+    # Create dummy batch
+    batch = {
+        "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+        "task": ["Pick up the object"],
+    }
+    batch = preprocessor(batch)
+
+    with torch.no_grad():
+        # Use same noise for fair comparison
+        noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+
+        # Test with RTC enabled but no previous chunk
+        actions_with_rtc_no_prev = policy.predict_action_chunk(
+            batch,
+            noise=noise.clone(),
+            prev_chunk_left_over=None,
+        )
+
+        # Test without RTC
+        policy.config.rtc_config.enabled = False
+        actions_without_rtc = policy.predict_action_chunk(batch, noise=noise.clone())
+        policy.config.rtc_config.enabled = True
+
+    # Without previous chunk, RTC should have no effect
+    assert torch.allclose(actions_with_rtc_no_prev, actions_without_rtc, rtol=1e-5)
+
+    print("✓ SmolVLA RTC inference without prev_chunk: Test passed")
+
+
+@require_package("transformers")
+@require_cuda
+@pytest.mark.skipif(True, reason="Requires pretrained SmolVLA model weights")
+def test_smolvla_rtc_validation_rules():
+    from lerobot.policies.smolvla.modeling_smolvla import SmolVLAPolicy  # noqa: F401
+
+    """Test SmolVLA policy with RTC follows all three validation rules."""
+    set_seed(42)
+
+    config = SmolVLAConfig(max_action_dim=7, chunk_size=50)
+
+    # Add RTC config
+    config.rtc_config = RTCConfig(
+        enabled=True,
+        execution_horizon=10,
+        max_guidance_weight=5.0,
+        prefix_attention_schedule=RTCAttentionSchedule.EXP,
+        debug=False,
+    )
+
+    config.input_features = {
+        "observation.state": PolicyFeature(type=FeatureType.STATE, shape=(14,)),
+        "observation.images.base_0_rgb": PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "action": PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+
+    # Create dataset stats
+    dataset_stats = {
+        "observation.state": {"mean": torch.zeros(14), "std": torch.ones(14)},
+        "action": {"mean": torch.zeros(7), "std": torch.ones(7)},
+        "observation.images.base_0_rgb": {"mean": torch.zeros(3, 224, 224), "std": torch.ones(3, 224, 224)},
+    }
+
+    # Instantiate policy and create preprocessor
+    policy = SmolVLAPolicy(config)
+    policy.eval()
+    preprocessor, _ = make_pre_post_processors(
+        policy_cfg=config, pretrained_path=None, dataset_stats=dataset_stats
+    )
+
+    device = config.device
+
+    # Create dummy batch
+    batch = {
+        "observation.state": torch.randn(1, 14, dtype=torch.float32, device=device),
+        "observation.images.base_0_rgb": torch.rand(1, 3, 224, 224, dtype=torch.float32, device=device),
+        "task": ["Pick up the object"],
+    }
+    batch = preprocessor(batch)
+
+    # Create previous chunk
+    prev_chunk = torch.randn(1, 25, 7, dtype=torch.float32, device=device)
+
+    inference_delay = 4
+    execution_horizon = 10
+
+    with torch.no_grad():
+        # Use same noise for fair comparison
+        noise = policy.model.sample_noise((1, config.chunk_size, 7), device)
+
+        # Test with RTC
+        actions_with_rtc = policy.predict_action_chunk(
+            batch,
+            noise=noise.clone(),
+            prev_chunk_left_over=prev_chunk,
+            inference_delay=inference_delay,
+            execution_horizon=execution_horizon,
+        )
+
+        # Test without RTC
+        policy.config.rtc_config.enabled = False
+        actions_without_rtc = policy.predict_action_chunk(batch, noise=noise.clone())
+        policy.config.rtc_config.enabled = True
+
+    assert not torch.allclose(actions_with_rtc, actions_without_rtc, rtol=1e-3)
diff --git a/lerobot/tests/policies/test_policies.py b/lerobot/tests/policies/test_policies.py
new file mode 100644
index 0000000000000000000000000000000000000000..77a74d60e67a6d831eae9299498f95274fd8d219
--- /dev/null
+++ b/lerobot/tests/policies/test_policies.py
@@ -0,0 +1,506 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import inspect
+from copy import deepcopy
+from pathlib import Path
+
+import einops
+import pytest
+import torch
+from packaging import version
+from safetensors.torch import load_file
+
+from lerobot import available_policies
+from lerobot.configs.default import DatasetConfig
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.datasets.factory import make_dataset
+from lerobot.datasets.feature_utils import dataset_to_policy_features
+from lerobot.datasets.utils import cycle
+from lerobot.envs.factory import make_env, make_env_config
+from lerobot.envs.utils import preprocess_observation
+from lerobot.optim.factory import make_optimizer_and_scheduler
+from lerobot.policies.act.configuration_act import ACTConfig
+from lerobot.policies.act.modeling_act import ACTTemporalEnsembler
+from lerobot.policies.factory import (
+    get_policy_class,
+    make_policy,
+    make_policy_config,
+    make_pre_post_processors,
+)
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.vqbet.configuration_vqbet import VQBeTConfig
+from lerobot.policies.vqbet.modeling_vqbet import VQBeTHead
+from lerobot.utils.constants import ACTION, OBS_IMAGES, OBS_STATE
+from lerobot.utils.random_utils import seeded_context
+from tests.artifacts.policies.save_policy_to_safetensors import get_policy_stats
+from tests.utils import DEVICE, require_cpu, require_env, require_x86_64_kernel
+
+
+@pytest.fixture
+def dummy_dataset_metadata(lerobot_dataset_metadata_factory, info_factory, tmp_path):
+    # Create only one camera input which is squared to fit all current policy constraints
+    # e.g. vqbet and tdmpc works with one camera only, and tdmpc requires it to be squared
+    camera_features = {
+        f"{OBS_IMAGES}.laptop": {
+            "shape": (84, 84, 3),
+            "names": ["height", "width", "channels"],
+            "info": None,
+        },
+    }
+    motor_features = {
+        ACTION: {
+            "dtype": "float32",
+            "shape": (6,),
+            "names": ["shoulder_pan", "shoulder_lift", "elbow_flex", "wrist_flex", "wrist_roll", "gripper"],
+        },
+        OBS_STATE: {
+            "dtype": "float32",
+            "shape": (6,),
+            "names": ["shoulder_pan", "shoulder_lift", "elbow_flex", "wrist_flex", "wrist_roll", "gripper"],
+        },
+    }
+    info = info_factory(
+        total_episodes=1,
+        total_frames=1,
+        total_tasks=1,
+        camera_features=camera_features,
+        motor_features=motor_features,
+    )
+    ds_meta = lerobot_dataset_metadata_factory(root=tmp_path / "init", info=info)
+    return ds_meta
+
+
+@pytest.mark.parametrize("policy_name", available_policies)
+def test_get_policy_and_config_classes(policy_name: str):
+    """Check that the correct policy and config classes are returned."""
+    policy_cls = get_policy_class(policy_name)
+    policy_cfg = make_policy_config(policy_name)
+    assert policy_cls.name == policy_name
+    assert issubclass(
+        policy_cfg.__class__, inspect.signature(policy_cls.__init__).parameters["config"].annotation
+    )
+
+
+@pytest.mark.parametrize(
+    "ds_repo_id,env_name,env_kwargs,policy_name,policy_kwargs",
+    [
+        ("lerobot/pusht", "pusht", {}, "diffusion", {}),
+        ("lerobot/pusht", "pusht", {}, "vqbet", {}),
+        ("lerobot/pusht", "pusht", {}, "act", {}),
+        ("lerobot/aloha_sim_insertion_human", "aloha", {"task": "AlohaInsertion-v0"}, "act", {}),
+        (
+            "lerobot/aloha_sim_insertion_scripted",
+            "aloha",
+            {"task": "AlohaInsertion-v0"},
+            "act",
+            {},
+        ),
+        (
+            "lerobot/aloha_sim_insertion_human",
+            "aloha",
+            {"task": "AlohaInsertion-v0"},
+            "diffusion",
+            {},
+        ),
+        (
+            "lerobot/aloha_sim_transfer_cube_human",
+            "aloha",
+            {"task": "AlohaTransferCube-v0"},
+            "act",
+            {},
+        ),
+        (
+            "lerobot/aloha_sim_transfer_cube_scripted",
+            "aloha",
+            {"task": "AlohaTransferCube-v0"},
+            "act",
+            {},
+        ),
+    ],
+)
+@require_env
+def test_policy(ds_repo_id, env_name, env_kwargs, policy_name, policy_kwargs):
+    """
+    Tests:
+        - Making the policy object.
+        - Checking that the policy follows the correct protocol and subclasses nn.Module
+            and PyTorchModelHubMixin.
+        - Updating the policy.
+        - Using the policy to select actions at inference time.
+        - Test the action can be applied to the policy
+
+    Note: We test various combinations of policy and dataset. The combinations are by no means exhaustive,
+          and for now we add tests as we see fit.
+    """
+    if policy_name == "vqbet" and DEVICE == "mps":
+        pytest.skip("VQBet does not support MPS backend")
+    if policy_name == "act" and "aloha" in ds_repo_id and DEVICE == "mps":
+        pytest.skip("ACT with aloha has batch mutation issues on MPS")
+
+    train_cfg = TrainPipelineConfig(
+        # TODO(rcadene, aliberts): remove dataset download
+        dataset=DatasetConfig(repo_id=ds_repo_id, episodes=[0]),
+        policy=make_policy_config(policy_name, push_to_hub=False, **policy_kwargs),
+        env=make_env_config(env_name, **env_kwargs),
+    )
+    train_cfg.policy.device = DEVICE
+    train_cfg.validate()
+
+    # Check that we can make the policy object.
+    dataset = make_dataset(train_cfg)
+    preprocessor, _ = make_pre_post_processors(train_cfg.policy, None)
+    policy = make_policy(train_cfg.policy, ds_meta=dataset.meta)
+    assert isinstance(policy, PreTrainedPolicy)
+
+    # Check that we run select_actions and get the appropriate output.
+    envs = make_env(train_cfg.env, n_envs=2)
+
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        num_workers=0,
+        batch_size=2,
+        shuffle=True,
+        pin_memory=DEVICE != "cpu",
+        drop_last=True,
+    )
+    dl_iter = cycle(dataloader)
+
+    batch = next(dl_iter)
+
+    for key in batch:
+        if isinstance(batch[key], torch.Tensor):
+            batch[key] = batch[key].to(DEVICE, non_blocking=True)
+
+    # Test updating the policy (and test that it does not mutate the batch)
+    batch_ = deepcopy(batch)
+    policy.forward(batch)
+    assert set(batch) == set(batch_), "Batch keys are not the same after a forward pass."
+    assert all(
+        torch.equal(batch[k], batch_[k]) if isinstance(batch[k], torch.Tensor) else batch[k] == batch_[k]
+        for k in batch
+    ), "Batch values are not the same after a forward pass."
+
+    # reset the policy and environment
+    policy.reset()
+    # For testing purposes, we only need a single environment instance.
+    # So here we unwrap the first suite_name and first task_id to grab
+    # the actual env object (SyncVectorEnv) that exposes `.reset()`.
+    suite_name = next(iter(envs))
+    task_id = next(iter(envs[suite_name]))
+    env = envs[suite_name][task_id]
+    observation, _ = env.reset(seed=train_cfg.seed)
+
+    # apply transform to normalize the observations
+    observation = preprocess_observation(observation)
+
+    # send observation to device/gpu
+    observation = {key: observation[key].to(DEVICE, non_blocking=True) for key in observation}
+
+    # get the next action for the environment (also check that the observation batch is not modified)
+    observation_ = deepcopy(observation)
+    with torch.inference_mode():
+        action = policy.select_action(observation).cpu().numpy()
+    assert set(observation) == set(observation_), (
+        "Observation batch keys are not the same after a forward pass."
+    )
+    assert all(torch.equal(observation[k], observation_[k]) for k in observation), (
+        "Observation batch values are not the same after a forward pass."
+    )
+
+    # Test step through policy
+    env.step(action)
+
+
+# TODO(rcadene, aliberts): This test is quite end-to-end. Move this test in test_optimizer?
+def test_act_backbone_lr():
+    """
+    Test that the ACT policy can be instantiated with a different learning rate for the backbone.
+    """
+
+    cfg = TrainPipelineConfig(
+        # TODO(rcadene, aliberts): remove dataset download
+        dataset=DatasetConfig(repo_id="lerobot/aloha_sim_insertion_scripted", episodes=[0]),
+        policy=make_policy_config("act", optimizer_lr=0.01, optimizer_lr_backbone=0.001, push_to_hub=False),
+    )
+    cfg.policy.device = DEVICE
+    cfg.validate()  # Needed for auto-setting some parameters
+
+    assert cfg.policy.optimizer_lr == 0.01
+    assert cfg.policy.optimizer_lr_backbone == 0.001
+
+    dataset = make_dataset(cfg)
+    preprocessor, _ = make_pre_post_processors(cfg.policy, None)
+    policy = make_policy(cfg.policy, ds_meta=dataset.meta)
+    optimizer, _ = make_optimizer_and_scheduler(cfg, policy)
+    assert len(optimizer.param_groups) == 2
+    assert optimizer.param_groups[0]["lr"] == cfg.policy.optimizer_lr
+    assert optimizer.param_groups[1]["lr"] == cfg.policy.optimizer_lr_backbone
+    assert len(optimizer.param_groups[0]["params"]) == 133
+    assert len(optimizer.param_groups[1]["params"]) == 20
+
+
+@pytest.mark.parametrize("policy_name", available_policies)
+def test_policy_defaults(dummy_dataset_metadata, policy_name: str):
+    """Check that the policy can be instantiated with defaults."""
+    policy_cls = get_policy_class(policy_name)
+    policy_cfg = make_policy_config(policy_name)
+    features = dataset_to_policy_features(dummy_dataset_metadata.features)
+    policy_cfg.output_features = {key: ft for key, ft in features.items() if ft.type is FeatureType.ACTION}
+    policy_cfg.input_features = {
+        key: ft for key, ft in features.items() if key not in policy_cfg.output_features
+    }
+    policy_cls(policy_cfg)
+
+
+@pytest.mark.parametrize("policy_name", available_policies)
+def test_save_and_load_pretrained(dummy_dataset_metadata, tmp_path, policy_name: str):
+    policy_cls = get_policy_class(policy_name)
+    policy_cfg = make_policy_config(policy_name)
+    features = dataset_to_policy_features(dummy_dataset_metadata.features)
+    policy_cfg.output_features = {key: ft for key, ft in features.items() if ft.type is FeatureType.ACTION}
+    policy_cfg.input_features = {
+        key: ft for key, ft in features.items() if key not in policy_cfg.output_features
+    }
+    policy = policy_cls(policy_cfg)
+    policy.to(policy_cfg.device)
+    save_dir = tmp_path / f"test_save_and_load_pretrained_{policy_cls.__name__}"
+    policy.save_pretrained(save_dir)
+    loaded_policy = policy_cls.from_pretrained(save_dir, config=policy_cfg)
+    torch.testing.assert_close(list(policy.parameters()), list(loaded_policy.parameters()), rtol=0, atol=0)
+
+
+@pytest.mark.parametrize("multikey", [True, False])
+def test_multikey_construction(multikey: bool):
+    """
+    Asserts that multiple keys with type State/Action are correctly processed by the policy constructor,
+    preventing erroneous creation of the policy object.
+    """
+    input_features = {
+        OBS_STATE: PolicyFeature(
+            type=FeatureType.STATE,
+            shape=(10,),
+        ),
+    }
+    output_features = {
+        ACTION: PolicyFeature(
+            type=FeatureType.ACTION,
+            shape=(5,),
+        ),
+    }
+
+    if multikey:
+        """Simulates the complete state/action is constructed from more granular multiple
+        keys, of the same type as the overall state/action"""
+        input_features = {}
+        input_features[f"{OBS_STATE}.subset1"] = PolicyFeature(type=FeatureType.STATE, shape=(5,))
+        input_features[f"{OBS_STATE}.subset2"] = PolicyFeature(type=FeatureType.STATE, shape=(5,))
+        input_features[OBS_STATE] = PolicyFeature(type=FeatureType.STATE, shape=(10,))
+
+        output_features = {}
+        output_features["action.first_three_motors"] = PolicyFeature(type=FeatureType.ACTION, shape=(3,))
+        output_features["action.last_two_motors"] = PolicyFeature(type=FeatureType.ACTION, shape=(2,))
+        output_features[ACTION] = PolicyFeature(
+            type=FeatureType.ACTION,
+            shape=(5,),
+        )
+
+    config = ACTConfig(input_features=input_features, output_features=output_features)
+
+    state_condition = config.robot_state_feature == input_features[OBS_STATE]
+    action_condition = config.action_feature == output_features[ACTION]
+
+    assert state_condition, (
+        f"Discrepancy detected. Robot state feature is {config.robot_state_feature} but policy expects {input_features[OBS_STATE]}"
+    )
+    assert action_condition, (
+        f"Discrepancy detected. Action feature is {config.action_feature} but policy expects {output_features[ACTION]}"
+    )
+
+
+@pytest.mark.parametrize(
+    "ds_repo_id, policy_name, policy_kwargs, file_name_extra",
+    [
+        # TODO(alexander-soare): `policy.use_mpc=false` was previously the default in the config yaml but it
+        # was changed to true. For some reason, tests would pass locally, but not in CI. So here we override
+        # to test with `policy.use_mpc=false`.
+        # TODO(rcadene): the diffusion model was normalizing the image in mean=0.5 std=0.5 which is a hack supposed to
+        # to normalize the image at all. In our current codebase we dont normalize at all. But there is still a minor difference
+        # that fails the test. However, by testing to normalize the image with 0.5 0.5 in the current codebase, the test pass.
+        # Thus, we deactivate this test for now.
+        (
+            "lerobot/pusht",
+            "diffusion",
+            {
+                "n_action_steps": 8,
+                "num_inference_steps": 10,
+                "down_dims": [128, 256, 512],
+            },
+            "",
+        ),
+        ("lerobot/aloha_sim_insertion_human", "act", {"n_action_steps": 10}, ""),
+        (
+            "lerobot/aloha_sim_insertion_human",
+            "act",
+            {"n_action_steps": 1000, "chunk_size": 1000},
+            "1000_steps",
+        ),
+    ],
+)
+# As artifacts have been generated on an x86_64 kernel, this test won't
+# pass if it's run on another platform due to floating point errors
+@require_x86_64_kernel
+@require_cpu
+def test_backward_compatibility(ds_repo_id: str, policy_name: str, policy_kwargs: dict, file_name_extra: str):
+    """
+    NOTE: If this test does not pass, and you have intentionally changed something in the policy:
+        1. Inspect the differences in policy outputs and make sure you can account for them. Your PR should
+           include a report on what changed and how that affected the outputs.
+        2. Go to the `if __name__ == "__main__"` block of `tests/scripts/save_policy_to_safetensors.py` and
+           add the policies you want to update the test artifacts for.
+        3. Run `python tests/scripts/save_policy_to_safetensors.py`. The test artifact
+           should be updated.
+        4. Check that this test now passes.
+        5. Remember to restore `tests/scripts/save_policy_to_safetensors.py` to its original state.
+        6. Remember to stage and commit the resulting changes to `tests/artifacts`.
+
+    NOTE: If the test does not pass, and you don't change the policy, it is likely that the test artifact
+    is out of date. For example, some PyTorch versions have different randomness, see this PR:
+    https://github.com/huggingface/lerobot/pull/1127.
+    NOTE: If the test don't pass and you don't change the policy, and note the dependencies version,
+    and you changed your processor, you might have to update the test artifact.
+
+    """
+
+    # NOTE: ACT policy has different randomness, after PyTorch 2.7.0
+    if policy_name == "act" and version.parse(torch.__version__) < version.parse("2.7.0"):
+        pytest.skip(f"Skipping act policy test with PyTorch {torch.__version__}. Requires PyTorch >= 2.7.0")
+
+    ds_name = ds_repo_id.split("/")[-1]
+    artifact_dir = Path("tests/artifacts/policies") / f"{ds_name}_{policy_name}_{file_name_extra}"
+    saved_output_dict = load_file(artifact_dir / "output_dict.safetensors")
+    saved_grad_stats = load_file(artifact_dir / "grad_stats.safetensors")
+    saved_param_stats = load_file(artifact_dir / "param_stats.safetensors")
+    saved_actions = load_file(artifact_dir / "actions.safetensors")
+
+    output_dict, grad_stats, param_stats, actions = get_policy_stats(ds_repo_id, policy_name, policy_kwargs)
+
+    for key in saved_output_dict:
+        torch.testing.assert_close(output_dict[key], saved_output_dict[key])
+    for key in saved_grad_stats:
+        torch.testing.assert_close(grad_stats[key], saved_grad_stats[key])
+    for key in saved_param_stats:
+        torch.testing.assert_close(param_stats[key], saved_param_stats[key])
+    for key in saved_actions:
+        rtol, atol = (2e-3, 5e-6) if policy_name == "diffusion" else (None, None)  # HACK
+        torch.testing.assert_close(actions[key], saved_actions[key], rtol=rtol, atol=atol)
+
+
+def test_act_temporal_ensembler():
+    """Check that the online method in ACTTemporalEnsembler matches a simple offline calculation."""
+    temporal_ensemble_coeff = 0.01
+    chunk_size = 100
+    episode_length = 101
+    ensembler = ACTTemporalEnsembler(temporal_ensemble_coeff, chunk_size)
+    # An batch of arbitrary sequences of 1D actions we wish to compute the average over. We'll keep the
+    # "action space" in [-1, 1]. Apart from that, there is no real reason for the numbers chosen.
+    with seeded_context(0):
+        # Dimension is (batch, episode_length, chunk_size, action_dim(=1))
+        # Stepping through the episode_length dim is like running inference at each rollout step and getting
+        # a different action chunk.
+        batch_seq = torch.stack(
+            [
+                torch.rand(episode_length, chunk_size) * 0.05 - 0.6,
+                torch.rand(episode_length, chunk_size) * 0.02 - 0.01,
+                torch.rand(episode_length, chunk_size) * 0.2 + 0.3,
+            ],
+            dim=0,
+        ).unsqueeze(-1)  # unsqueeze for action dim
+    batch_size = batch_seq.shape[0]
+    # Exponential weighting (normalized). Unsqueeze once to match the position of the `episode_length`
+    # dimension of `batch_seq`.
+    weights = torch.exp(-temporal_ensemble_coeff * torch.arange(chunk_size)).unsqueeze(-1)
+
+    # Simulate stepping through a rollout and computing a batch of actions with model on each step.
+    for i in range(episode_length):
+        # Mock a batch of actions.
+        actions = torch.zeros(size=(batch_size, chunk_size, 1)) + batch_seq[:, i]
+        online_avg = ensembler.update(actions)
+        # Simple offline calculation: avg = Σ(aᵢ*wᵢ) / Σ(wᵢ).
+        # Note: The complicated bit here is the slicing. Think about the (episode_length, chunk_size) grid.
+        # What we want to do is take diagonal slices across it starting from the left.
+        #  eg: chunk_size=4, episode_length=6
+        #  ┌───────┐
+        #  │0 1 2 3│
+        #  │1 2 3 4│
+        #  │2 3 4 5│
+        #  │3 4 5 6│
+        #  │4 5 6 7│
+        #  │5 6 7 8│
+        #  └───────┘
+        chunk_indices = torch.arange(min(i, chunk_size - 1), -1, -1)
+        episode_step_indices = torch.arange(i + 1)[-len(chunk_indices) :]
+        seq_slice = batch_seq[:, episode_step_indices, chunk_indices]
+        offline_avg = (
+            einops.reduce(seq_slice * weights[: i + 1], "b s 1 -> b 1", "sum") / weights[: i + 1].sum()
+        )
+        # Sanity check. The average should be between the extrema.
+        assert torch.all(einops.reduce(seq_slice, "b s 1 -> b 1", "min") <= offline_avg)
+        assert torch.all(offline_avg <= einops.reduce(seq_slice, "b s 1 -> b 1", "max"))
+        # Selected atol=1e-4 keeping in mind actions in [-1, 1] and excepting 0.01% error.
+        torch.testing.assert_close(online_avg, offline_avg, rtol=1e-4, atol=1e-4)
+
+
+def test_vqbet_discretize_keeps_buffers_on_device():
+    """Regression test: VQBeTHead.discretize() must not move registered buffers off the model device.
+
+    Previously, `self.vqvae_model.discretized = torch.tensor(True)` replaced the
+    registered buffer with a new CPU tensor, causing DDP to crash with:
+        RuntimeError: No backend type associated with device type cpu
+    The fix uses `.fill_(True)` to update in-place, preserving device placement.
+    """
+    config = VQBeTConfig()
+    config.input_features = {
+        OBS_IMAGES: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 96, 96)),
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(6,)),
+    }
+    config.output_features = {
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(6,)),
+    }
+    # Tiny sizes for fast CPU/GPU execution.
+    config.n_vqvae_training_steps = 3
+    config.vqvae_n_embed = 8
+    config.vqvae_embedding_dim = 32
+    config.vqvae_enc_hidden_dim = 32
+    config.action_chunk_size = 2
+    config.crop_shape = (84, 84)
+
+    head = VQBeTHead(config).to(DEVICE)
+    vqvae = head.vqvae_model
+
+    dummy_actions = torch.randn(4, config.action_chunk_size, config.action_feature.shape[0], device=DEVICE)
+    n_steps = config.n_vqvae_training_steps
+    for _ in range(n_steps):
+        head.discretize(n_steps, dummy_actions)
+
+    assert vqvae.discretized.device.type == torch.device(DEVICE).type, (
+        "vqvae_model.discretized was moved off the model device after discretize(). "
+        "Use .fill_(True) instead of = torch.tensor(True) to keep the buffer on device."
+    )
+    assert vqvae.vq_layer.freeze_codebook.device.type == torch.device(DEVICE).type, (
+        "vq_layer.freeze_codebook was moved off the model device after discretize(). "
+        "Use .fill_(True) instead of = torch.tensor(True) to keep the buffer on device."
+    )
diff --git a/lerobot/tests/policies/test_sac_config.py b/lerobot/tests/policies/test_sac_config.py
new file mode 100644
index 0000000000000000000000000000000000000000..724c331fffca01e1e24defdab51c3158eda9e3d7
--- /dev/null
+++ b/lerobot/tests/policies/test_sac_config.py
@@ -0,0 +1,217 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import pytest
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.policies.sac.configuration_sac import (
+    ActorLearnerConfig,
+    ActorNetworkConfig,
+    ConcurrencyConfig,
+    CriticNetworkConfig,
+    PolicyConfig,
+    SACConfig,
+)
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_STATE
+
+
+def test_sac_config_default_initialization():
+    config = SACConfig()
+
+    assert config.normalization_mapping == {
+        "VISUAL": NormalizationMode.MEAN_STD,
+        "STATE": NormalizationMode.MIN_MAX,
+        "ENV": NormalizationMode.MIN_MAX,
+        "ACTION": NormalizationMode.MIN_MAX,
+    }
+    assert config.dataset_stats == {
+        OBS_IMAGE: {
+            "mean": [0.485, 0.456, 0.406],
+            "std": [0.229, 0.224, 0.225],
+        },
+        OBS_STATE: {
+            "min": [0.0, 0.0],
+            "max": [1.0, 1.0],
+        },
+        ACTION: {
+            "min": [0.0, 0.0, 0.0],
+            "max": [1.0, 1.0, 1.0],
+        },
+    }
+
+    # Basic parameters
+    assert config.device == "cpu"
+    assert config.storage_device == "cpu"
+    assert config.discount == 0.99
+    assert config.temperature_init == 1.0
+    assert config.num_critics == 2
+
+    # Architecture specifics
+    assert config.vision_encoder_name is None
+    assert config.freeze_vision_encoder is True
+    assert config.image_encoder_hidden_dim == 32
+    assert config.shared_encoder is True
+    assert config.num_discrete_actions is None
+    assert config.image_embedding_pooling_dim == 8
+
+    # Training parameters
+    assert config.online_steps == 1000000
+    assert config.online_buffer_capacity == 100000
+    assert config.offline_buffer_capacity == 100000
+    assert config.async_prefetch is False
+    assert config.online_step_before_learning == 100
+    assert config.policy_update_freq == 1
+
+    # SAC algorithm parameters
+    assert config.num_subsample_critics is None
+    assert config.critic_lr == 3e-4
+    assert config.actor_lr == 3e-4
+    assert config.temperature_lr == 3e-4
+    assert config.critic_target_update_weight == 0.005
+    assert config.utd_ratio == 1
+    assert config.state_encoder_hidden_dim == 256
+    assert config.latent_dim == 256
+    assert config.target_entropy is None
+    assert config.use_backup_entropy is True
+    assert config.grad_clip_norm == 40.0
+
+    # Dataset stats defaults
+    expected_dataset_stats = {
+        OBS_IMAGE: {
+            "mean": [0.485, 0.456, 0.406],
+            "std": [0.229, 0.224, 0.225],
+        },
+        OBS_STATE: {
+            "min": [0.0, 0.0],
+            "max": [1.0, 1.0],
+        },
+        ACTION: {
+            "min": [0.0, 0.0, 0.0],
+            "max": [1.0, 1.0, 1.0],
+        },
+    }
+    assert config.dataset_stats == expected_dataset_stats
+
+    # Critic network configuration
+    assert config.critic_network_kwargs.hidden_dims == [256, 256]
+    assert config.critic_network_kwargs.activate_final is True
+    assert config.critic_network_kwargs.final_activation is None
+
+    # Actor network configuration
+    assert config.actor_network_kwargs.hidden_dims == [256, 256]
+    assert config.actor_network_kwargs.activate_final is True
+
+    # Policy configuration
+    assert config.policy_kwargs.use_tanh_squash is True
+    assert config.policy_kwargs.std_min == 1e-5
+    assert config.policy_kwargs.std_max == 10.0
+    assert config.policy_kwargs.init_final == 0.05
+
+    # Discrete critic network configuration
+    assert config.discrete_critic_network_kwargs.hidden_dims == [256, 256]
+    assert config.discrete_critic_network_kwargs.activate_final is True
+    assert config.discrete_critic_network_kwargs.final_activation is None
+
+    # Actor learner configuration
+    assert config.actor_learner_config.learner_host == "127.0.0.1"
+    assert config.actor_learner_config.learner_port == 50051
+    assert config.actor_learner_config.policy_parameters_push_frequency == 4
+
+    # Concurrency configuration
+    assert config.concurrency.actor == "threads"
+    assert config.concurrency.learner == "threads"
+
+    assert isinstance(config.actor_network_kwargs, ActorNetworkConfig)
+    assert isinstance(config.critic_network_kwargs, CriticNetworkConfig)
+    assert isinstance(config.policy_kwargs, PolicyConfig)
+    assert isinstance(config.actor_learner_config, ActorLearnerConfig)
+    assert isinstance(config.concurrency, ConcurrencyConfig)
+
+
+def test_critic_network_kwargs():
+    config = CriticNetworkConfig()
+    assert config.hidden_dims == [256, 256]
+    assert config.activate_final is True
+    assert config.final_activation is None
+
+
+def test_actor_network_kwargs():
+    config = ActorNetworkConfig()
+    assert config.hidden_dims == [256, 256]
+    assert config.activate_final is True
+
+
+def test_policy_kwargs():
+    config = PolicyConfig()
+    assert config.use_tanh_squash is True
+    assert config.std_min == 1e-5
+    assert config.std_max == 10.0
+    assert config.init_final == 0.05
+
+
+def test_actor_learner_config():
+    config = ActorLearnerConfig()
+    assert config.learner_host == "127.0.0.1"
+    assert config.learner_port == 50051
+    assert config.policy_parameters_push_frequency == 4
+
+
+def test_concurrency_config():
+    config = ConcurrencyConfig()
+    assert config.actor == "threads"
+    assert config.learner == "threads"
+
+
+def test_sac_config_custom_initialization():
+    config = SACConfig(
+        device="cpu",
+        discount=0.95,
+        temperature_init=0.5,
+        num_critics=3,
+    )
+
+    assert config.device == "cpu"
+    assert config.discount == 0.95
+    assert config.temperature_init == 0.5
+    assert config.num_critics == 3
+
+
+def test_validate_features():
+    config = SACConfig(
+        input_features={OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(10,))},
+        output_features={ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(3,))},
+    )
+    config.validate_features()
+
+
+def test_validate_features_missing_observation():
+    config = SACConfig(
+        input_features={"wrong_key": PolicyFeature(type=FeatureType.STATE, shape=(10,))},
+        output_features={ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(3,))},
+    )
+    with pytest.raises(
+        ValueError, match="You must provide either 'observation.state' or an image observation"
+    ):
+        config.validate_features()
+
+
+def test_validate_features_missing_action():
+    config = SACConfig(
+        input_features={OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(10,))},
+        output_features={"wrong_key": PolicyFeature(type=FeatureType.ACTION, shape=(3,))},
+    )
+    with pytest.raises(ValueError, match="You must provide 'action' in the output features"):
+        config.validate_features()
diff --git a/lerobot/tests/policies/test_sac_policy.py b/lerobot/tests/policies/test_sac_policy.py
new file mode 100644
index 0000000000000000000000000000000000000000..11499ce305df46384226de40f6827a8abc2b1290
--- /dev/null
+++ b/lerobot/tests/policies/test_sac_policy.py
@@ -0,0 +1,546 @@
+# !/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import math
+
+import pytest
+import torch
+from torch import Tensor, nn
+
+from lerobot.configs.types import FeatureType, PolicyFeature
+from lerobot.policies.sac.configuration_sac import SACConfig
+from lerobot.policies.sac.modeling_sac import MLP, SACPolicy
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_STATE
+from lerobot.utils.random_utils import seeded_context, set_seed
+
+try:
+    import transformers  # noqa: F401
+
+    TRANSFORMERS_AVAILABLE = True
+except ImportError:
+    TRANSFORMERS_AVAILABLE = False
+
+
+@pytest.fixture(autouse=True)
+def set_random_seed():
+    seed = 42
+    set_seed(seed)
+
+
+def test_mlp_with_default_args():
+    mlp = MLP(input_dim=10, hidden_dims=[256, 256])
+
+    x = torch.randn(10)
+    y = mlp(x)
+    assert y.shape == (256,)
+
+
+def test_mlp_with_batch_dim():
+    mlp = MLP(input_dim=10, hidden_dims=[256, 256])
+    x = torch.randn(2, 10)
+    y = mlp(x)
+    assert y.shape == (2, 256)
+
+
+def test_forward_with_empty_hidden_dims():
+    mlp = MLP(input_dim=10, hidden_dims=[])
+    x = torch.randn(1, 10)
+    assert mlp(x).shape == (1, 10)
+
+
+def test_mlp_with_dropout():
+    mlp = MLP(input_dim=10, hidden_dims=[256, 256, 11], dropout_rate=0.1)
+    x = torch.randn(1, 10)
+    y = mlp(x)
+    assert y.shape == (1, 11)
+
+    drop_out_layers_count = sum(isinstance(layer, nn.Dropout) for layer in mlp.net)
+    assert drop_out_layers_count == 2
+
+
+def test_mlp_with_custom_final_activation():
+    mlp = MLP(input_dim=10, hidden_dims=[256, 256], final_activation=torch.nn.Tanh())
+    x = torch.randn(1, 10)
+    y = mlp(x)
+    assert y.shape == (1, 256)
+    assert (y >= -1).all() and (y <= 1).all()
+
+
+def test_sac_policy_with_default_args():
+    with pytest.raises(ValueError, match="should be an instance of class `PreTrainedConfig`"):
+        SACPolicy()
+
+
+def create_dummy_state(batch_size: int, state_dim: int = 10) -> Tensor:
+    return {
+        OBS_STATE: torch.randn(batch_size, state_dim),
+    }
+
+
+def create_dummy_with_visual_input(batch_size: int, state_dim: int = 10) -> Tensor:
+    return {
+        OBS_IMAGE: torch.randn(batch_size, 3, 84, 84),
+        OBS_STATE: torch.randn(batch_size, state_dim),
+    }
+
+
+def create_dummy_action(batch_size: int, action_dim: int = 10) -> Tensor:
+    return torch.randn(batch_size, action_dim)
+
+
+def create_default_train_batch(
+    batch_size: int = 8, state_dim: int = 10, action_dim: int = 10
+) -> dict[str, Tensor]:
+    return {
+        ACTION: create_dummy_action(batch_size, action_dim),
+        "reward": torch.randn(batch_size),
+        "state": create_dummy_state(batch_size, state_dim),
+        "next_state": create_dummy_state(batch_size, state_dim),
+        "done": torch.randn(batch_size),
+    }
+
+
+def create_train_batch_with_visual_input(
+    batch_size: int = 8, state_dim: int = 10, action_dim: int = 10
+) -> dict[str, Tensor]:
+    return {
+        ACTION: create_dummy_action(batch_size, action_dim),
+        "reward": torch.randn(batch_size),
+        "state": create_dummy_with_visual_input(batch_size, state_dim),
+        "next_state": create_dummy_with_visual_input(batch_size, state_dim),
+        "done": torch.randn(batch_size),
+    }
+
+
+def create_observation_batch(batch_size: int = 8, state_dim: int = 10) -> dict[str, Tensor]:
+    return {
+        OBS_STATE: torch.randn(batch_size, state_dim),
+    }
+
+
+def create_observation_batch_with_visual_input(batch_size: int = 8, state_dim: int = 10) -> dict[str, Tensor]:
+    return {
+        OBS_STATE: torch.randn(batch_size, state_dim),
+        OBS_IMAGE: torch.randn(batch_size, 3, 84, 84),
+    }
+
+
+def make_optimizers(policy: SACPolicy, has_discrete_action: bool = False) -> dict[str, torch.optim.Optimizer]:
+    """Create optimizers for the SAC policy."""
+    optimizer_actor = torch.optim.Adam(
+        # Handle the case of shared encoder where the encoder weights are not optimized with the actor gradient
+        params=[
+            p
+            for n, p in policy.actor.named_parameters()
+            if not policy.config.shared_encoder or not n.startswith("encoder")
+        ],
+        lr=policy.config.actor_lr,
+    )
+    optimizer_critic = torch.optim.Adam(
+        params=policy.critic_ensemble.parameters(),
+        lr=policy.config.critic_lr,
+    )
+    optimizer_temperature = torch.optim.Adam(
+        params=[policy.log_alpha],
+        lr=policy.config.critic_lr,
+    )
+
+    optimizers = {
+        "actor": optimizer_actor,
+        "critic": optimizer_critic,
+        "temperature": optimizer_temperature,
+    }
+
+    if has_discrete_action:
+        optimizers["discrete_critic"] = torch.optim.Adam(
+            params=policy.discrete_critic.parameters(),
+            lr=policy.config.critic_lr,
+        )
+
+    return optimizers
+
+
+def create_default_config(
+    state_dim: int, continuous_action_dim: int, has_discrete_action: bool = False
+) -> SACConfig:
+    action_dim = continuous_action_dim
+    if has_discrete_action:
+        action_dim += 1
+
+    config = SACConfig(
+        input_features={OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(state_dim,))},
+        output_features={ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(continuous_action_dim,))},
+        dataset_stats={
+            OBS_STATE: {
+                "min": [0.0] * state_dim,
+                "max": [1.0] * state_dim,
+            },
+            ACTION: {
+                "min": [0.0] * continuous_action_dim,
+                "max": [1.0] * continuous_action_dim,
+            },
+        },
+    )
+    config.validate_features()
+    return config
+
+
+def create_config_with_visual_input(
+    state_dim: int, continuous_action_dim: int, has_discrete_action: bool = False
+) -> SACConfig:
+    config = create_default_config(
+        state_dim=state_dim,
+        continuous_action_dim=continuous_action_dim,
+        has_discrete_action=has_discrete_action,
+    )
+    config.input_features[OBS_IMAGE] = PolicyFeature(type=FeatureType.VISUAL, shape=(3, 84, 84))
+    config.dataset_stats[OBS_IMAGE] = {
+        "mean": torch.randn(3, 1, 1),
+        "std": torch.randn(3, 1, 1),
+    }
+
+    # Let make tests a little bit faster
+    config.state_encoder_hidden_dim = 32
+    config.latent_dim = 32
+
+    config.validate_features()
+    return config
+
+
+@pytest.mark.parametrize("batch_size,state_dim,action_dim", [(2, 6, 6), (1, 10, 10)])
+def test_sac_policy_with_default_config(batch_size: int, state_dim: int, action_dim: int):
+    batch = create_default_train_batch(batch_size=batch_size, action_dim=action_dim, state_dim=state_dim)
+    config = create_default_config(state_dim=state_dim, continuous_action_dim=action_dim)
+
+    policy = SACPolicy(config=config)
+    policy.train()
+
+    optimizers = make_optimizers(policy)
+
+    cirtic_loss = policy.forward(batch, model="critic")["loss_critic"]
+    assert cirtic_loss.item() is not None
+    assert cirtic_loss.shape == ()
+    cirtic_loss.backward()
+    optimizers["critic"].step()
+
+    actor_loss = policy.forward(batch, model="actor")["loss_actor"]
+    assert actor_loss.item() is not None
+    assert actor_loss.shape == ()
+
+    actor_loss.backward()
+    optimizers["actor"].step()
+
+    temperature_loss = policy.forward(batch, model="temperature")["loss_temperature"]
+    assert temperature_loss.item() is not None
+    assert temperature_loss.shape == ()
+
+    temperature_loss.backward()
+    optimizers["temperature"].step()
+
+    policy.eval()
+    with torch.no_grad():
+        observation_batch = create_observation_batch(batch_size=batch_size, state_dim=state_dim)
+        selected_action = policy.select_action(observation_batch)
+        assert selected_action.shape == (batch_size, action_dim)
+
+
+@pytest.mark.parametrize("batch_size,state_dim,action_dim", [(2, 6, 6), (1, 10, 10)])
+def test_sac_policy_with_visual_input(batch_size: int, state_dim: int, action_dim: int):
+    config = create_config_with_visual_input(state_dim=state_dim, continuous_action_dim=action_dim)
+    policy = SACPolicy(config=config)
+
+    batch = create_train_batch_with_visual_input(
+        batch_size=batch_size, state_dim=state_dim, action_dim=action_dim
+    )
+
+    policy.train()
+
+    optimizers = make_optimizers(policy)
+
+    cirtic_loss = policy.forward(batch, model="critic")["loss_critic"]
+    assert cirtic_loss.item() is not None
+    assert cirtic_loss.shape == ()
+    cirtic_loss.backward()
+    optimizers["critic"].step()
+
+    actor_loss = policy.forward(batch, model="actor")["loss_actor"]
+    assert actor_loss.item() is not None
+    assert actor_loss.shape == ()
+
+    actor_loss.backward()
+    optimizers["actor"].step()
+
+    temperature_loss = policy.forward(batch, model="temperature")["loss_temperature"]
+    assert temperature_loss.item() is not None
+    assert temperature_loss.shape == ()
+
+    temperature_loss.backward()
+    optimizers["temperature"].step()
+
+    policy.eval()
+    with torch.no_grad():
+        observation_batch = create_observation_batch_with_visual_input(
+            batch_size=batch_size, state_dim=state_dim
+        )
+        selected_action = policy.select_action(observation_batch)
+        assert selected_action.shape == (batch_size, action_dim)
+
+
+# Let's check best candidates for pretrained encoders
+@pytest.mark.parametrize(
+    "batch_size,state_dim,action_dim,vision_encoder_name",
+    [(1, 6, 6, "helper2424/resnet10"), (1, 6, 6, "facebook/convnext-base-224")],
+)
+@pytest.mark.skipif(not TRANSFORMERS_AVAILABLE, reason="Transformers are not installed")
+@pytest.mark.skip(
+    reason="helper2424/resnet10 needs to be updated to work with the latest version of transformers"
+)
+def test_sac_policy_with_pretrained_encoder(
+    batch_size: int, state_dim: int, action_dim: int, vision_encoder_name: str
+):
+    config = create_config_with_visual_input(state_dim=state_dim, continuous_action_dim=action_dim)
+    config.vision_encoder_name = vision_encoder_name
+    policy = SACPolicy(config=config)
+    policy.train()
+
+    batch = create_train_batch_with_visual_input(
+        batch_size=batch_size, state_dim=state_dim, action_dim=action_dim
+    )
+
+    optimizers = make_optimizers(policy)
+
+    cirtic_loss = policy.forward(batch, model="critic")["loss_critic"]
+    assert cirtic_loss.item() is not None
+    assert cirtic_loss.shape == ()
+    cirtic_loss.backward()
+    optimizers["critic"].step()
+
+    actor_loss = policy.forward(batch, model="actor")["loss_actor"]
+    assert actor_loss.item() is not None
+    assert actor_loss.shape == ()
+
+
+def test_sac_policy_with_shared_encoder():
+    batch_size = 2
+    action_dim = 10
+    state_dim = 10
+    config = create_config_with_visual_input(state_dim=state_dim, continuous_action_dim=action_dim)
+    config.shared_encoder = True
+
+    policy = SACPolicy(config=config)
+    policy.train()
+
+    batch = create_train_batch_with_visual_input(
+        batch_size=batch_size, state_dim=state_dim, action_dim=action_dim
+    )
+
+    policy.train()
+
+    optimizers = make_optimizers(policy)
+
+    cirtic_loss = policy.forward(batch, model="critic")["loss_critic"]
+    assert cirtic_loss.item() is not None
+    assert cirtic_loss.shape == ()
+    cirtic_loss.backward()
+    optimizers["critic"].step()
+
+    actor_loss = policy.forward(batch, model="actor")["loss_actor"]
+    assert actor_loss.item() is not None
+    assert actor_loss.shape == ()
+
+    actor_loss.backward()
+    optimizers["actor"].step()
+
+
+def test_sac_policy_with_discrete_critic():
+    batch_size = 2
+    continuous_action_dim = 9
+    full_action_dim = continuous_action_dim + 1  # the last action is discrete
+    state_dim = 10
+    config = create_config_with_visual_input(
+        state_dim=state_dim, continuous_action_dim=continuous_action_dim, has_discrete_action=True
+    )
+
+    num_discrete_actions = 5
+    config.num_discrete_actions = num_discrete_actions
+
+    policy = SACPolicy(config=config)
+    policy.train()
+
+    batch = create_train_batch_with_visual_input(
+        batch_size=batch_size, state_dim=state_dim, action_dim=full_action_dim
+    )
+
+    policy.train()
+
+    optimizers = make_optimizers(policy, has_discrete_action=True)
+
+    cirtic_loss = policy.forward(batch, model="critic")["loss_critic"]
+    assert cirtic_loss.item() is not None
+    assert cirtic_loss.shape == ()
+    cirtic_loss.backward()
+    optimizers["critic"].step()
+
+    discrete_critic_loss = policy.forward(batch, model="discrete_critic")["loss_discrete_critic"]
+    assert discrete_critic_loss.item() is not None
+    assert discrete_critic_loss.shape == ()
+    discrete_critic_loss.backward()
+    optimizers["discrete_critic"].step()
+
+    actor_loss = policy.forward(batch, model="actor")["loss_actor"]
+    assert actor_loss.item() is not None
+    assert actor_loss.shape == ()
+
+    actor_loss.backward()
+    optimizers["actor"].step()
+
+    policy.eval()
+    with torch.no_grad():
+        observation_batch = create_observation_batch_with_visual_input(
+            batch_size=batch_size, state_dim=state_dim
+        )
+        selected_action = policy.select_action(observation_batch)
+        assert selected_action.shape == (batch_size, full_action_dim)
+
+        discrete_actions = selected_action[:, -1].long()
+        discrete_action_values = set(discrete_actions.tolist())
+
+        assert all(action in range(num_discrete_actions) for action in discrete_action_values), (
+            f"Discrete action {discrete_action_values} is not in range({num_discrete_actions})"
+        )
+
+
+def test_sac_policy_with_default_entropy():
+    config = create_default_config(continuous_action_dim=10, state_dim=10)
+    policy = SACPolicy(config=config)
+    assert policy.target_entropy == -5.0
+
+
+def test_sac_policy_default_target_entropy_with_discrete_action():
+    config = create_config_with_visual_input(state_dim=10, continuous_action_dim=6, has_discrete_action=True)
+    policy = SACPolicy(config=config)
+    assert policy.target_entropy == -3.0
+
+
+def test_sac_policy_with_predefined_entropy():
+    config = create_default_config(state_dim=10, continuous_action_dim=6)
+    config.target_entropy = -3.5
+
+    policy = SACPolicy(config=config)
+    assert policy.target_entropy == pytest.approx(-3.5)
+
+
+def test_sac_policy_update_temperature():
+    """Test that temperature property is always in sync with log_alpha."""
+    config = create_default_config(continuous_action_dim=10, state_dim=10)
+    policy = SACPolicy(config=config)
+
+    assert policy.temperature == pytest.approx(1.0)
+    policy.log_alpha.data = torch.tensor([math.log(0.1)])
+    # Temperature property automatically reflects log_alpha changes
+    assert policy.temperature == pytest.approx(0.1)
+
+
+def test_sac_policy_update_target_network():
+    config = create_default_config(state_dim=10, continuous_action_dim=6)
+    config.critic_target_update_weight = 1.0
+
+    policy = SACPolicy(config=config)
+    policy.train()
+
+    for p in policy.critic_ensemble.parameters():
+        p.data = torch.ones_like(p.data)
+
+    policy.update_target_networks()
+    for p in policy.critic_target.parameters():
+        assert torch.allclose(p.data, torch.ones_like(p.data)), (
+            f"Target network {p.data} is not equal to {torch.ones_like(p.data)}"
+        )
+
+
+@pytest.mark.parametrize("num_critics", [1, 3])
+def test_sac_policy_with_critics_number_of_heads(num_critics: int):
+    batch_size = 2
+    action_dim = 10
+    state_dim = 10
+    config = create_config_with_visual_input(state_dim=state_dim, continuous_action_dim=action_dim)
+    config.num_critics = num_critics
+
+    policy = SACPolicy(config=config)
+    policy.train()
+
+    assert len(policy.critic_ensemble.critics) == num_critics
+
+    batch = create_train_batch_with_visual_input(
+        batch_size=batch_size, state_dim=state_dim, action_dim=action_dim
+    )
+
+    policy.train()
+
+    optimizers = make_optimizers(policy)
+
+    cirtic_loss = policy.forward(batch, model="critic")["loss_critic"]
+    assert cirtic_loss.item() is not None
+    assert cirtic_loss.shape == ()
+    cirtic_loss.backward()
+    optimizers["critic"].step()
+
+
+def test_sac_policy_save_and_load(tmp_path):
+    root = tmp_path / "test_sac_save_and_load"
+
+    state_dim = 10
+    action_dim = 10
+    batch_size = 2
+
+    config = create_default_config(state_dim=state_dim, continuous_action_dim=action_dim)
+    policy = SACPolicy(config=config)
+    policy.eval()
+    policy.save_pretrained(root)
+    loaded_policy = SACPolicy.from_pretrained(root, config=config)
+    loaded_policy.eval()
+
+    batch = create_default_train_batch(batch_size=1, state_dim=10, action_dim=10)
+
+    with torch.no_grad():
+        with seeded_context(12):
+            # Collect policy values before saving
+            cirtic_loss = policy.forward(batch, model="critic")["loss_critic"]
+            actor_loss = policy.forward(batch, model="actor")["loss_actor"]
+            temperature_loss = policy.forward(batch, model="temperature")["loss_temperature"]
+
+            observation_batch = create_observation_batch(batch_size=batch_size, state_dim=state_dim)
+            actions = policy.select_action(observation_batch)
+
+        with seeded_context(12):
+            # Collect policy values after loading
+            loaded_cirtic_loss = loaded_policy.forward(batch, model="critic")["loss_critic"]
+            loaded_actor_loss = loaded_policy.forward(batch, model="actor")["loss_actor"]
+            loaded_temperature_loss = loaded_policy.forward(batch, model="temperature")["loss_temperature"]
+
+            loaded_observation_batch = create_observation_batch(batch_size=batch_size, state_dim=state_dim)
+            loaded_actions = loaded_policy.select_action(loaded_observation_batch)
+
+        assert policy.state_dict().keys() == loaded_policy.state_dict().keys()
+        for k in policy.state_dict():
+            assert torch.allclose(policy.state_dict()[k], loaded_policy.state_dict()[k], atol=1e-6)
+
+        # Compare values before and after saving and loading
+        # They should be the same
+        assert torch.allclose(cirtic_loss, loaded_cirtic_loss)
+        assert torch.allclose(actor_loss, loaded_actor_loss)
+        assert torch.allclose(temperature_loss, loaded_temperature_loss)
+        assert torch.allclose(actions, loaded_actions)
diff --git a/lerobot/tests/policies/test_sarm_processor.py b/lerobot/tests/policies/test_sarm_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..5b90784a66910151d2ffb71d61c541680ab94cd1
--- /dev/null
+++ b/lerobot/tests/policies/test_sarm_processor.py
@@ -0,0 +1,694 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import pytest
+
+pytest.importorskip("faker")
+
+from unittest.mock import MagicMock, patch
+
+import numpy as np
+import pandas as pd
+import pytest
+import torch
+
+from lerobot.types import TransitionKey
+
+
+class MockDatasetMeta:
+    """Mock dataset metadata for testing processor."""
+
+    def __init__(self, episodes: list[dict]):
+        self._episodes = episodes
+
+    @property
+    def episodes(self):
+        """Return episodes as a mock object with to_pandas() method."""
+        mock = MagicMock()
+        mock.__len__ = lambda s: len(self._episodes)
+        mock.__getitem__ = lambda s, idx: self._episodes[idx]
+        mock.to_pandas = lambda: pd.DataFrame(self._episodes)
+        return mock
+
+
+class MockConfig:
+    """Mock SARMConfig for testing processor methods."""
+
+    def __init__(
+        self,
+        n_obs_steps: int = 8,
+        max_rewind_steps: int = 4,
+        frame_gap: int = 30,
+        sparse_subtask_names: list = None,
+        sparse_temporal_proportions: list = None,
+        dense_subtask_names: list = None,
+        dense_temporal_proportions: list = None,
+        image_key: str = "observation.images.top",
+        state_key: str = "observation.state",
+        max_state_dim: int = 32,
+        device: str = None,
+        rewind_probability: float = 0.8,
+        language_perturbation_probability: float = 0.2,
+        annotation_mode: str = "dual",
+        clip_batch_size: int = 64,
+        text_dim: int = 512,
+    ):
+        self.n_obs_steps = n_obs_steps
+        self.max_rewind_steps = max_rewind_steps
+        self.frame_gap = frame_gap
+        self.sparse_subtask_names = sparse_subtask_names or ["task"]
+        self.sparse_temporal_proportions = sparse_temporal_proportions or [1.0]
+        self.dense_subtask_names = dense_subtask_names
+        self.dense_temporal_proportions = dense_temporal_proportions
+        self.uses_dual_heads = annotation_mode in ["dense_only", "dual"]
+        self.image_key = image_key
+        self.state_key = state_key
+        self.max_state_dim = max_state_dim
+        self.device = device
+        self.rewind_probability = rewind_probability
+        self.language_perturbation_probability = language_perturbation_probability
+        self.annotation_mode = annotation_mode
+        self.clip_batch_size = clip_batch_size
+        self.text_dim = text_dim
+
+        # Compute observation delta indices (same as config: bidirectional)
+        half_steps = self.n_obs_steps // 2
+        past_deltas = [-self.frame_gap * i for i in range(half_steps, 0, -1)]
+        future_deltas = [self.frame_gap * i for i in range(1, half_steps + 1)]
+        obs_deltas = past_deltas + [0] + future_deltas
+        rewind_deltas = [-self.frame_gap * (i + 1) for i in range(self.max_rewind_steps)]
+        self.observation_delta_indices = obs_deltas + rewind_deltas
+
+    @property
+    def num_frames(self) -> int:
+        return 1 + self.n_obs_steps + self.max_rewind_steps
+
+
+class TestSARMEncodingProcessorStepEndToEnd:
+    """End-to-end test for SARMEncodingProcessorStep with dummy batch data."""
+
+    @pytest.fixture
+    def mock_clip_model(self):
+        """Mock CLIP model to avoid loading real weights."""
+        with (
+            patch("lerobot.policies.sarm.processor_sarm.CLIPModel") as mock_model_cls,
+            patch("lerobot.policies.sarm.processor_sarm.CLIPProcessor") as mock_processor_cls,
+        ):
+            # Mock the CLIP model - return embeddings based on input batch size
+            mock_model = MagicMock()
+
+            def get_image_features_side_effect(**kwargs):
+                pixel_values = kwargs.get("pixel_values")
+                batch_size = pixel_values.shape[0] if pixel_values is not None else 1
+                return torch.randn(batch_size, 512)
+
+            mock_model.get_image_features.side_effect = get_image_features_side_effect
+            mock_model.get_text_features.return_value = torch.randn(1, 512)
+            mock_model.to.return_value = mock_model
+            mock_model_cls.from_pretrained.return_value = mock_model
+
+            # Mock the CLIP processor - return tensors based on input images
+            mock_processor = MagicMock()
+
+            def processor_side_effect(images=None, **kwargs):
+                num_images = len(images) if images is not None else 1
+                return {
+                    "pixel_values": torch.randn(num_images, 3, 224, 224),
+                }
+
+            mock_processor.side_effect = processor_side_effect
+            # Mock tokenizer for text encoding
+            mock_processor.tokenizer.return_value = {
+                "input_ids": torch.ones(1, 77, dtype=torch.long),
+                "attention_mask": torch.ones(1, 77, dtype=torch.long),
+            }
+            mock_processor_cls.from_pretrained.return_value = mock_processor
+
+            yield mock_model, mock_processor
+
+    @pytest.fixture
+    def processor_with_mocks(self, mock_clip_model):
+        """Create a processor with mocked CLIP and dataset metadata for dual mode."""
+        from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep
+
+        # Dual mode config with both sparse and dense annotations
+        config = MockConfig(
+            n_obs_steps=8,
+            max_rewind_steps=4,
+            frame_gap=30,
+            rewind_probability=0.0,  # Disable for deterministic test
+            language_perturbation_probability=0.0,  # Disable for deterministic test
+            annotation_mode="dual",
+            sparse_subtask_names=["reach", "grasp", "lift"],
+            sparse_temporal_proportions=[0.3, 0.4, 0.3],
+            dense_subtask_names=["approach", "contact", "close_gripper", "lift_up"],
+            dense_temporal_proportions=[0.25, 0.25, 0.25, 0.25],
+        )
+
+        # Create mock dataset metadata with one episode of 300 frames
+        # Include annotation columns for dual mode
+        episodes = [
+            {
+                "dataset_from_index": 0,
+                "dataset_to_index": 300,
+                "task": "pick up the cube",
+                "sparse_subtask_names": ["reach", "grasp", "lift"],
+                "sparse_subtask_start_frames": [0, 90, 210],
+                "sparse_subtask_end_frames": [90, 210, 300],
+                "dense_subtask_names": ["approach", "contact", "close_gripper", "lift_up"],
+                "dense_subtask_start_frames": [0, 75, 150, 225],
+                "dense_subtask_end_frames": [75, 150, 225, 300],
+            }
+        ]
+        dataset_meta = MockDatasetMeta(episodes)
+
+        processor = SARMEncodingProcessorStep(
+            config=config,
+            dataset_meta=dataset_meta,
+        )
+        processor.train(True)  # Use train() method, not direct assignment
+
+        return processor, config
+
+    def test_call_with_single_frame_batch(self, processor_with_mocks):
+        """Test processor __call__ with a single-frame batch."""
+        processor, config = processor_with_mocks
+
+        # Create dummy input transition
+        batch_size = 1
+        num_frames = config.num_frames  # 13 frames (9 obs + 4 rewind)
+
+        # Image: (T, C, H, W) format as expected by processor
+        dummy_image = np.random.rand(num_frames, 3, 224, 224).astype(np.float32)
+
+        # State: (T, D) format
+        dummy_state = np.random.rand(num_frames, 6).astype(np.float32)
+
+        transition = {
+            TransitionKey.OBSERVATION: {
+                config.image_key: dummy_image,
+                config.state_key: dummy_state,
+            },
+            TransitionKey.COMPLEMENTARY_DATA: {
+                "index": 150,  # Middle of episode
+                "episode_index": 0,
+                "task": "pick up the cube",
+            },
+        }
+
+        # Run processor
+        result = processor(transition)
+
+        # Verify output structure
+        obs = result[TransitionKey.OBSERVATION]
+
+        # Check video features exist and have correct shape
+        assert "video_features" in obs
+        video_features = obs["video_features"]
+        assert video_features.shape[0] == batch_size
+        assert video_features.shape[1] == num_frames
+        assert video_features.shape[2] == 512  # CLIP embedding dim
+
+        # Check state features exist and have correct shape
+        assert "state_features" in obs
+        state_features = obs["state_features"]
+        assert state_features.shape[0] == batch_size
+        assert state_features.shape[1] == num_frames
+        assert state_features.shape[2] == config.max_state_dim  # Padded to max_state_dim
+
+        # Check text features exist and have correct shape
+        assert "text_features" in obs
+        text_features = obs["text_features"]
+        assert text_features.shape[0] == batch_size
+        assert text_features.shape[1] == 512  # CLIP embedding dim
+
+        # Check lengths tensor
+        assert "lengths" in obs
+        lengths = obs["lengths"]
+        assert lengths.shape[0] == batch_size
+        assert lengths.dtype == torch.int32
+
+        # Check sparse_targets exist
+        assert "sparse_targets" in obs
+        sparse_targets = obs["sparse_targets"]
+        assert sparse_targets.shape == (batch_size, num_frames)
+        # All targets should be in [0, max_stages] range (stage.tau format)
+        assert (sparse_targets >= 0).all()
+
+        # Check dense_targets exist (for dual mode)
+        assert "dense_targets" in obs
+        dense_targets = obs["dense_targets"]
+        assert dense_targets.shape == (batch_size, num_frames)
+        assert (dense_targets >= 0).all()
+
+    def test_call_with_batched_input(self, mock_clip_model):
+        """Test processor __call__ with a batched input (multiple frames) in dual mode."""
+        from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep
+
+        config = MockConfig(
+            n_obs_steps=8,
+            max_rewind_steps=4,
+            frame_gap=30,
+            rewind_probability=0.0,
+            language_perturbation_probability=0.0,
+            annotation_mode="dual",
+            sparse_subtask_names=["reach", "grasp"],
+            sparse_temporal_proportions=[0.5, 0.5],
+            dense_subtask_names=["step1", "step2", "step3"],
+            dense_temporal_proportions=[0.33, 0.34, 0.33],
+        )
+
+        # Two episodes with different lengths, each with sparse+dense annotations
+        episodes = [
+            {
+                "dataset_from_index": 0,
+                "dataset_to_index": 200,
+                "task": "task A",
+                "sparse_subtask_names": ["reach", "grasp"],
+                "sparse_subtask_start_frames": [0, 100],
+                "sparse_subtask_end_frames": [100, 200],
+                "dense_subtask_names": ["step1", "step2", "step3"],
+                "dense_subtask_start_frames": [0, 66, 133],
+                "dense_subtask_end_frames": [66, 133, 200],
+            },
+            {
+                "dataset_from_index": 200,
+                "dataset_to_index": 500,
+                "task": "task B",
+                "sparse_subtask_names": ["reach", "grasp"],
+                "sparse_subtask_start_frames": [200, 350],
+                "sparse_subtask_end_frames": [350, 500],
+                "dense_subtask_names": ["step1", "step2", "step3"],
+                "dense_subtask_start_frames": [200, 300, 400],
+                "dense_subtask_end_frames": [300, 400, 500],
+            },
+        ]
+        dataset_meta = MockDatasetMeta(episodes)
+
+        processor = SARMEncodingProcessorStep(config=config, dataset_meta=dataset_meta)
+        processor.train(True)
+
+        batch_size = 2
+        num_frames = config.num_frames
+
+        # Image: (B, T, C, H, W) format
+        dummy_image = np.random.rand(batch_size, num_frames, 3, 224, 224).astype(np.float32)
+        dummy_state = np.random.rand(batch_size, num_frames, 6).astype(np.float32)
+
+        transition = {
+            TransitionKey.OBSERVATION: {
+                config.image_key: dummy_image,
+                config.state_key: dummy_state,
+            },
+            TransitionKey.COMPLEMENTARY_DATA: {
+                "index": np.array([100, 350]),  # One frame from each episode
+                "episode_index": np.array([0, 1]),
+                "task": ["task A", "task B"],
+            },
+        }
+
+        result = processor(transition)
+        obs = result[TransitionKey.OBSERVATION]
+
+        # Verify batch dimension is preserved for all outputs
+        assert obs["video_features"].shape[0] == batch_size
+        assert obs["state_features"].shape[0] == batch_size
+        assert obs["lengths"].shape[0] == batch_size
+        assert obs["sparse_targets"].shape[0] == batch_size
+        assert obs["dense_targets"].shape[0] == batch_size  # Dual mode has dense targets
+
+    def test_targets_increase_with_progress(self, mock_clip_model):
+        """Test that both sparse and dense targets increase as frame index progresses."""
+        from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep
+
+        config = MockConfig(
+            n_obs_steps=8,
+            max_rewind_steps=4,
+            frame_gap=30,
+            rewind_probability=0.0,
+            language_perturbation_probability=0.0,
+            annotation_mode="dual",
+            sparse_subtask_names=["phase1", "phase2"],
+            sparse_temporal_proportions=[0.5, 0.5],
+            dense_subtask_names=["a", "b", "c", "d"],
+            dense_temporal_proportions=[0.25, 0.25, 0.25, 0.25],
+        )
+
+        episodes = [
+            {
+                "dataset_from_index": 0,
+                "dataset_to_index": 300,
+                "task": "test task",
+                "sparse_subtask_names": ["phase1", "phase2"],
+                "sparse_subtask_start_frames": [0, 150],
+                "sparse_subtask_end_frames": [150, 300],
+                "dense_subtask_names": ["a", "b", "c", "d"],
+                "dense_subtask_start_frames": [0, 75, 150, 225],
+                "dense_subtask_end_frames": [75, 150, 225, 300],
+            }
+        ]
+        dataset_meta = MockDatasetMeta(episodes)
+
+        processor = SARMEncodingProcessorStep(config=config, dataset_meta=dataset_meta)
+        processor.train(True)
+
+        num_frames = config.num_frames
+
+        # Test at early, middle, and late points in episode
+        frame_indices = [30, 150, 270]
+        sparse_center_targets = []
+        dense_center_targets = []
+
+        for frame_idx in frame_indices:
+            dummy_image = np.random.rand(num_frames, 3, 224, 224).astype(np.float32)
+            dummy_state = np.random.rand(num_frames, 6).astype(np.float32)
+
+            transition = {
+                TransitionKey.OBSERVATION: {
+                    config.image_key: dummy_image,
+                    config.state_key: dummy_state,
+                },
+                TransitionKey.COMPLEMENTARY_DATA: {
+                    "index": frame_idx,
+                    "episode_index": 0,
+                    "task": "test task",
+                },
+            }
+
+            result = processor(transition)
+            obs = result[TransitionKey.OBSERVATION]
+            # Get target at center frame (index 4 in 9-frame observation window)
+            sparse_center_targets.append(obs["sparse_targets"][0, 4].item())
+            dense_center_targets.append(obs["dense_targets"][0, 4].item())
+
+        # Both sparse and dense targets should increase with frame index
+        assert sparse_center_targets[0] < sparse_center_targets[2], (
+            f"Early sparse target ({sparse_center_targets[0]}) should be < late ({sparse_center_targets[2]})"
+        )
+        assert dense_center_targets[0] < dense_center_targets[2], (
+            f"Early dense target ({dense_center_targets[0]}) should be < late ({dense_center_targets[2]})"
+        )
+
+    def test_progress_labels_exact_values(self, mock_clip_model):
+        """Test that progress labels (stage.tau) are computed correctly for known positions."""
+        from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep
+
+        # Simple setup: 2 sparse stages, 4 dense stages, 100 frame episode
+        config = MockConfig(
+            n_obs_steps=8,
+            max_rewind_steps=4,
+            frame_gap=10,  # Smaller gap for easier calculation
+            rewind_probability=0.0,
+            language_perturbation_probability=0.0,
+            annotation_mode="dual",
+            sparse_subtask_names=["A", "B"],
+            sparse_temporal_proportions=[0.5, 0.5],
+            dense_subtask_names=["d1", "d2", "d3", "d4"],
+            dense_temporal_proportions=[0.25, 0.25, 0.25, 0.25],
+        )
+
+        # Episode: frames 0-99, sparse stages at [0-49], [50-99]
+        # Dense stages at [0-24], [25-49], [50-74], [75-99]
+        episodes = [
+            {
+                "dataset_from_index": 0,
+                "dataset_to_index": 100,
+                "task": "test",
+                "sparse_subtask_names": ["A", "B"],
+                "sparse_subtask_start_frames": [0, 50],
+                "sparse_subtask_end_frames": [50, 100],
+                "dense_subtask_names": ["d1", "d2", "d3", "d4"],
+                "dense_subtask_start_frames": [0, 25, 50, 75],
+                "dense_subtask_end_frames": [25, 50, 75, 100],
+            }
+        ]
+        dataset_meta = MockDatasetMeta(episodes)
+
+        processor = SARMEncodingProcessorStep(config=config, dataset_meta=dataset_meta)
+        processor.train(True)
+
+        num_frames = config.num_frames
+
+        # Test at frame 50 (center of episode)
+        # With frame_gap=10, n_obs_steps=8:
+        # obs indices around frame 50: [10, 20, 30, 40, 50, 60, 70, 80, 90] (9 frames)
+        dummy_image = np.random.rand(num_frames, 3, 224, 224).astype(np.float32)
+        dummy_state = np.random.rand(num_frames, 6).astype(np.float32)
+
+        transition = {
+            TransitionKey.OBSERVATION: {
+                config.image_key: dummy_image,
+                config.state_key: dummy_state,
+            },
+            TransitionKey.COMPLEMENTARY_DATA: {
+                "index": 50,
+                "episode_index": 0,
+                "task": "test",
+            },
+        }
+
+        result = processor(transition)
+        obs = result[TransitionKey.OBSERVATION]
+        sparse_targets = obs["sparse_targets"][0]  # (13,)
+        dense_targets = obs["dense_targets"][0]  # (13,)
+
+        # First 9 frames are observation frames, last 4 are rewind placeholders (zeros when no rewind)
+        # Check that obs frames have non-zero targets
+        obs_sparse = sparse_targets[:9]
+        obs_dense = dense_targets[:9]
+
+        # Verify targets are monotonically increasing for observation frames
+        for i in range(1, 9):
+            assert obs_sparse[i] >= obs_sparse[i - 1], (
+                f"Sparse targets should be monotonic: {obs_sparse[i - 1].item():.3f} -> {obs_sparse[i].item():.3f}"
+            )
+            assert obs_dense[i] >= obs_dense[i - 1], (
+                f"Dense targets should be monotonic: {obs_dense[i - 1].item():.3f} -> {obs_dense[i].item():.3f}"
+            )
+
+        # Rewind slots should be zero when rewind is disabled
+        rewind_targets = sparse_targets[9:]
+        assert (rewind_targets == 0).all(), "Rewind slots should be zero when rewind is disabled"
+
+        # Check stage transitions: frame 50 is at boundary of sparse stage A->B
+        # Center frame (index 4) corresponds to actual frame 50
+        center_sparse = obs_sparse[4].item()
+        # At frame 50, sparse stage B starts, so target should be ~1.0 (stage 1 + tau 0)
+        assert 0.9 <= center_sparse <= 1.1, (
+            f"At sparse boundary, target should be ~1.0, got {center_sparse:.3f}"
+        )
+
+    def test_rewind_augmentation_applied(self, mock_clip_model):
+        """Test that rewind augmentation correctly extends sequence and generates targets."""
+        import random
+
+        from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep
+
+        config = MockConfig(
+            n_obs_steps=8,
+            max_rewind_steps=4,
+            frame_gap=10,
+            rewind_probability=1.0,  # Always apply rewind
+            language_perturbation_probability=0.0,
+            annotation_mode="dual",
+            sparse_subtask_names=["A", "B"],
+            sparse_temporal_proportions=[0.5, 0.5],
+            dense_subtask_names=["d1", "d2"],
+            dense_temporal_proportions=[0.5, 0.5],
+        )
+
+        episodes = [
+            {
+                "dataset_from_index": 0,
+                "dataset_to_index": 200,
+                "task": "test",
+                "sparse_subtask_names": ["A", "B"],
+                "sparse_subtask_start_frames": [0, 100],
+                "sparse_subtask_end_frames": [100, 200],
+                "dense_subtask_names": ["d1", "d2"],
+                "dense_subtask_start_frames": [0, 100],
+                "dense_subtask_end_frames": [100, 200],
+            }
+        ]
+        dataset_meta = MockDatasetMeta(episodes)
+
+        processor = SARMEncodingProcessorStep(config=config, dataset_meta=dataset_meta)
+        processor.train(True)
+
+        num_frames = config.num_frames  # 13
+
+        # Test at frame 150 (center of bidirectional window)
+        # With n_obs_steps=8, half_steps=4, frame_gap=10:
+        # - Earliest obs frame = 150 - 4*10 = 110
+        # - Rewind can go back from 110 to frames like 100, 90, 80, 70
+        # - History available = 110 - 0 = 110, so max rewind = 110/10 = 11 (capped at 4)
+        dummy_image = np.random.rand(num_frames, 3, 224, 224).astype(np.float32)
+        dummy_state = np.random.rand(num_frames, 6).astype(np.float32)
+
+        transition = {
+            TransitionKey.OBSERVATION: {
+                config.image_key: dummy_image,
+                config.state_key: dummy_state,
+            },
+            TransitionKey.COMPLEMENTARY_DATA: {
+                "index": 150,
+                "episode_index": 0,
+                "task": "test",
+            },
+        }
+
+        # Seed random for reproducibility
+        random.seed(42)
+        result = processor(transition)
+        obs = result[TransitionKey.OBSERVATION]
+
+        lengths = obs["lengths"][0].item()
+        sparse_targets = obs["sparse_targets"][0]
+
+        # With rewind_probability=1.0 and enough history, lengths should be > 9 (9 obs + some rewind)
+        assert lengths > 9, f"With rewind enabled, lengths should be > 9, got {lengths}"
+        assert lengths <= num_frames, f"Lengths should not exceed total frames {num_frames}, got {lengths}"
+
+        # Rewind targets should be non-zero for frames within valid length
+        n_obs_frames = 9
+        rewind_count = lengths - n_obs_frames
+
+        if rewind_count > 0:
+            # Check that rewind frames have targets
+            rewind_targets = sparse_targets[n_obs_frames : n_obs_frames + rewind_count]
+            # Rewind frames are from BEFORE the earliest obs frame (110)
+            # These frames (100, 90, 80, 70) are earlier in the episode
+            earliest_obs_target = sparse_targets[0].item()  # Frame 110
+
+            # Rewind targets should be less than earliest obs (they're from earlier frames)
+            for i, rt in enumerate(rewind_targets):
+                assert rt.item() < earliest_obs_target, (
+                    f"Rewind target {i} ({rt.item():.3f}) should be < earliest obs ({earliest_obs_target:.3f})"
+                )
+
+            # Rewind targets should be decreasing (going further back in time)
+            for i in range(1, len(rewind_targets)):
+                assert rewind_targets[i] <= rewind_targets[i - 1], (
+                    f"Rewind targets should decrease: {rewind_targets[i - 1].item():.3f} -> {rewind_targets[i].item():.3f}"
+                )
+
+    def test_full_sequence_target_consistency(self, mock_clip_model):
+        """Test that the full sequence of targets is consistent with frame positions."""
+        from lerobot.policies.sarm.processor_sarm import SARMEncodingProcessorStep
+        from lerobot.policies.sarm.sarm_utils import find_stage_and_tau
+
+        config = MockConfig(
+            n_obs_steps=8,
+            max_rewind_steps=4,
+            frame_gap=10,
+            rewind_probability=0.0,
+            language_perturbation_probability=0.0,
+            annotation_mode="dual",
+            sparse_subtask_names=["s1", "s2", "s3"],
+            sparse_temporal_proportions=[0.33, 0.34, 0.33],
+            dense_subtask_names=["d1", "d2"],
+            dense_temporal_proportions=[0.5, 0.5],
+        )
+
+        # 3 sparse stages: [0-33), [33-66), [66-99]
+        # 2 dense stages: [0-50), [50-100)
+        episodes = [
+            {
+                "dataset_from_index": 0,
+                "dataset_to_index": 100,
+                "task": "test",
+                "sparse_subtask_names": ["s1", "s2", "s3"],
+                "sparse_subtask_start_frames": [0, 33, 66],
+                "sparse_subtask_end_frames": [33, 66, 100],
+                "dense_subtask_names": ["d1", "d2"],
+                "dense_subtask_start_frames": [0, 50],
+                "dense_subtask_end_frames": [50, 100],
+            }
+        ]
+        dataset_meta = MockDatasetMeta(episodes)
+
+        processor = SARMEncodingProcessorStep(config=config, dataset_meta=dataset_meta)
+        processor.train(True)
+
+        num_frames = config.num_frames
+
+        # Test at frame 50 (middle of episode)
+        dummy_image = np.random.rand(num_frames, 3, 224, 224).astype(np.float32)
+        dummy_state = np.random.rand(num_frames, 6).astype(np.float32)
+
+        transition = {
+            TransitionKey.OBSERVATION: {
+                config.image_key: dummy_image,
+                config.state_key: dummy_state,
+            },
+            TransitionKey.COMPLEMENTARY_DATA: {
+                "index": 50,
+                "episode_index": 0,
+                "task": "test",
+            },
+        }
+
+        result = processor(transition)
+        obs = result[TransitionKey.OBSERVATION]
+        sparse_targets = obs["sparse_targets"][0]
+        dense_targets = obs["dense_targets"][0]
+
+        # Manually compute expected targets for observation frames
+        # With frame_gap=10, n_obs_steps=8, center at 50:
+        # obs frames: [10, 20, 30, 40, 50, 60, 70, 80, 90]
+        expected_obs_frames = [10, 20, 30, 40, 50, 60, 70, 80, 90]
+
+        sparse_names = ["s1", "s2", "s3"]
+        sparse_starts = [0, 33, 66]
+        sparse_ends = [33, 66, 100]
+        sparse_props = {"s1": 0.33, "s2": 0.34, "s3": 0.33}
+
+        dense_names = ["d1", "d2"]
+        dense_starts = [0, 50]
+        dense_ends = [50, 100]
+        dense_props = {"d1": 0.5, "d2": 0.5}
+
+        for i, frame in enumerate(expected_obs_frames):
+            expected_sparse = find_stage_and_tau(
+                frame,
+                100,
+                sparse_names,
+                sparse_starts,
+                sparse_ends,
+                sparse_names,
+                sparse_props,
+                return_combined=True,
+            )
+            expected_dense = find_stage_and_tau(
+                frame,
+                100,
+                dense_names,
+                dense_starts,
+                dense_ends,
+                dense_names,
+                dense_props,
+                return_combined=True,
+            )
+
+            actual_sparse = sparse_targets[i].item()
+            actual_dense = dense_targets[i].item()
+
+            assert abs(actual_sparse - expected_sparse) < 0.01, (
+                f"Frame {frame}: sparse mismatch {actual_sparse:.3f} vs expected {expected_sparse:.3f}"
+            )
+            assert abs(actual_dense - expected_dense) < 0.01, (
+                f"Frame {frame}: dense mismatch {actual_dense:.3f} vs expected {expected_dense:.3f}"
+            )
diff --git a/lerobot/tests/policies/test_sarm_subtask_annotations.py b/lerobot/tests/policies/test_sarm_subtask_annotations.py
new file mode 100644
index 0000000000000000000000000000000000000000..0dc087288e7adad4852f20f2de25cfce02815832
--- /dev/null
+++ b/lerobot/tests/policies/test_sarm_subtask_annotations.py
@@ -0,0 +1,134 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import pytest
+
+pytest.importorskip("transformers")
+
+from lerobot.data_processing.sarm_annotations.subtask_annotation import (
+    Subtask,
+    SubtaskAnnotation,
+    Timestamp,
+    compute_temporal_proportions,
+)
+
+
+def make_annotation(subtasks: list[tuple[str, int, int]]) -> SubtaskAnnotation:
+    """Helper to create SubtaskAnnotation from list of (name, start_sec, end_sec)."""
+    return SubtaskAnnotation(
+        subtasks=[
+            Subtask(
+                name=name,
+                timestamps=Timestamp(
+                    start=f"{start // 60:02d}:{start % 60:02d}", end=f"{end // 60:02d}:{end % 60:02d}"
+                ),
+            )
+            for name, start, end in subtasks
+        ]
+    )
+
+
+class TestComputeTemporalProportions:
+    """Tests for compute_temporal_proportions (SARM Paper Formula 1).
+
+    Formula: ᾱ_k = (1/M) × Σ_i (L_{i,k} / T_i)
+
+    Key insight: This averages the PROPORTION of each subtask within each trajectory,
+    giving equal weight to all trajectories regardless of absolute length.
+    """
+
+    def test_basic_two_trajectories_equal_proportions(self):
+        """Test with two trajectories that have equal proportions."""
+        # Both trajectories: subtask1 = 50%, subtask2 = 50%
+        # Traj 1: T=100s, subtask1=50s, subtask2=50s
+        # Traj 2: T=200s, subtask1=100s, subtask2=100s
+        annotations = {
+            0: make_annotation([("subtask1", 0, 50), ("subtask2", 50, 100)]),
+            1: make_annotation([("subtask1", 0, 100), ("subtask2", 100, 200)]),
+        }
+
+        result = compute_temporal_proportions(annotations)
+
+        # Both should be 0.5
+        assert abs(result["subtask1"] - 0.5) < 1e-6
+        assert abs(result["subtask2"] - 0.5) < 1e-6
+
+    def test_paper_example_different_from_avg_durations(self):
+        """Test that compute_temporal_proportions differs from naive average duration approach.
+
+        This is the key test showing the difference between:
+        - Paper formula: average of (L_i,k / T_i)
+        - Naive approach: mean(L_i,k) / sum(mean(L_i,j))
+        """
+        # Episode 1: T=100s, subtask1=80s, subtask2=20s (proportions: 0.8, 0.2)
+        # Episode 2: T=200s, subtask1=40s, subtask2=160s (proportions: 0.2, 0.8)
+        annotations = {
+            0: make_annotation([("subtask1", 0, 80), ("subtask2", 80, 100)]),
+            1: make_annotation([("subtask1", 0, 40), ("subtask2", 40, 200)]),
+        }
+
+        result = compute_temporal_proportions(annotations)
+
+        # Paper formula:
+        # ᾱ_1 = (1/2) × (80/100 + 40/200) = (1/2) × (0.8 + 0.2) = 0.5
+        # ᾱ_2 = (1/2) × (20/100 + 160/200) = (1/2) × (0.2 + 0.8) = 0.5
+        assert abs(result["subtask1"] - 0.5) < 1e-6
+        assert abs(result["subtask2"] - 0.5) < 1e-6
+
+    def test_single_trajectory(self):
+        """Test with a single trajectory."""
+        # T=100s, reach=30s, grasp=20s, lift=50s
+        annotations = {
+            0: make_annotation([("reach", 0, 30), ("grasp", 30, 50), ("lift", 50, 100)]),
+        }
+
+        result = compute_temporal_proportions(annotations)
+
+        assert abs(result["reach"] - 0.3) < 1e-6
+        assert abs(result["grasp"] - 0.2) < 1e-6
+        assert abs(result["lift"] - 0.5) < 1e-6
+
+    def test_sum_to_one(self):
+        """Test that proportions always sum to 1."""
+        # Three episodes with varying proportions
+        annotations = {
+            0: make_annotation([("a", 0, 10), ("b", 10, 50), ("c", 50, 100)]),  # 0.1, 0.4, 0.5
+            1: make_annotation([("a", 0, 20), ("b", 20, 70), ("c", 70, 100)]),  # 0.2, 0.5, 0.3
+            2: make_annotation([("a", 0, 30), ("b", 30, 90), ("c", 90, 100)]),  # 0.3, 0.6, 0.1
+        }
+
+        result = compute_temporal_proportions(annotations)
+
+        total = sum(result.values())
+        assert abs(total - 1.0) < 1e-6
+
+    def test_empty_annotations_returns_empty(self):
+        """Test that empty annotations returns empty dict."""
+        result = compute_temporal_proportions({})
+        assert result == {}
+
+    def test_uniform_proportions(self):
+        """Test with uniform proportions across subtasks."""
+        # Each subtask takes 25% of each episode
+        annotations = {
+            0: make_annotation([("a", 0, 25), ("b", 25, 50), ("c", 50, 75), ("d", 75, 100)]),
+            1: make_annotation([("a", 0, 50), ("b", 50, 100), ("c", 100, 150), ("d", 150, 200)]),
+        }
+
+        result = compute_temporal_proportions(annotations)
+
+        for name in ["a", "b", "c", "d"]:
+            assert abs(result[name] - 0.25) < 1e-6
diff --git a/lerobot/tests/policies/test_sarm_utils.py b/lerobot/tests/policies/test_sarm_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..510477ec8866eeacdd236a4da275a8798be9485e
--- /dev/null
+++ b/lerobot/tests/policies/test_sarm_utils.py
@@ -0,0 +1,615 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import numpy as np
+import pytest
+import torch
+
+from lerobot.policies.sarm.sarm_utils import (
+    apply_rewind_augmentation,
+    compute_absolute_indices,
+    compute_tau,
+    find_stage_and_tau,
+    normalize_stage_tau,
+    temporal_proportions_to_breakpoints,
+)
+
+
+class TestProgressLabelsWithModes:
+    """End-to-end tests for progress label generation in different modes."""
+
+    def test_sparse_mode_single_stage(self):
+        """Sparse mode with single stage should give linear progress."""
+        episode_length = 300
+        global_names = ["task"]
+        proportions = {"task": 1.0}
+
+        # Test at various frames
+        for frame in [0, 100, 200, 299]:
+            stage, tau = find_stage_and_tau(
+                frame, episode_length, None, None, None, global_names, proportions
+            )
+
+            expected_tau = frame / (episode_length - 1)
+            assert stage == 0
+            assert abs(tau - expected_tau) < 1e-5
+
+    def test_sparse_mode_multi_stage(self):
+        """Sparse mode with multiple stages."""
+        global_names = ["reach", "grasp", "lift", "place"]
+        proportions = {"reach": 0.2, "grasp": 0.2, "lift": 0.3, "place": 0.3}
+
+        subtask_names = ["reach", "grasp", "lift", "place"]
+        subtask_starts = [0, 60, 120, 210]
+        subtask_ends = [59, 119, 209, 299]
+
+        # Check stages are correctly identified
+        stage_at_30, _ = find_stage_and_tau(
+            30, 300, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage_at_30 == 0
+
+        stage_at_90, _ = find_stage_and_tau(
+            90, 300, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage_at_90 == 1
+
+        stage_at_150, _ = find_stage_and_tau(
+            150, 300, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage_at_150 == 2
+
+    def test_dense_mode_more_stages(self):
+        """Dense mode should work with more fine-grained stages."""
+        global_names = ["a", "b", "c", "d", "e", "f", "g", "h"]
+        proportions = dict.fromkeys(global_names, 1 / 8)
+
+        subtask_names = global_names
+        subtask_starts = [i * 50 for i in range(8)]
+        subtask_ends = [(i + 1) * 50 - 1 for i in range(8)]
+
+        # Each stage should occupy 50 frames
+        for stage_idx in range(8):
+            mid_frame = stage_idx * 50 + 25
+            stage, _ = find_stage_and_tau(
+                mid_frame, 400, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+            )
+            assert stage == stage_idx
+
+
+class TestComputeAbsoluteIndices:
+    """Tests for compute_absolute_indices (bidirectional sampling)."""
+
+    def test_no_clamping_when_in_middle(self):
+        """When frame is in middle of episode, no clamping should occur."""
+        frame_idx = 300
+        ep_start = 0
+        ep_end = 1000
+        n_obs_steps = 8
+        frame_gap = 30
+
+        indices, out_of_bounds = compute_absolute_indices(frame_idx, ep_start, ep_end, n_obs_steps, frame_gap)
+
+        # All should be valid (no out of bounds)
+        assert out_of_bounds.sum() == 0
+
+        # Check bidirectional indices: [-120, -90, -60, -30, 0, 30, 60, 90, 120] from center
+        half_steps = n_obs_steps // 2
+        expected = (
+            [frame_idx - frame_gap * i for i in range(half_steps, 0, -1)]
+            + [frame_idx]
+            + [frame_idx + frame_gap * i for i in range(1, half_steps + 1)]
+        )
+        assert indices.tolist() == expected
+
+        # Center frame (index 4) should be frame_idx
+        assert indices[half_steps] == frame_idx
+
+    def test_clamping_at_episode_start(self):
+        """Early frames should be clamped to episode start."""
+        frame_idx = 50  # Not enough history for full past window
+        ep_start = 0
+        ep_end = 1000
+        n_obs_steps = 8
+        frame_gap = 30
+
+        indices, out_of_bounds = compute_absolute_indices(frame_idx, ep_start, ep_end, n_obs_steps, frame_gap)
+
+        # Some past frames should be clamped (out_of_bounds = 1)
+        assert out_of_bounds.sum() > 0
+
+        # All indices should be >= ep_start
+        assert (indices >= ep_start).all()
+
+        # Center index should be frame_idx
+        half_steps = n_obs_steps // 2
+        assert indices[half_steps] == frame_idx
+
+    def test_clamping_at_episode_end(self):
+        """Late frames should be clamped to episode end."""
+        frame_idx = 950  # Not enough future for full window
+        ep_start = 0
+        ep_end = 1000
+        n_obs_steps = 8
+        frame_gap = 30
+
+        indices, out_of_bounds = compute_absolute_indices(frame_idx, ep_start, ep_end, n_obs_steps, frame_gap)
+
+        # Some future frames should be clamped
+        assert out_of_bounds.sum() > 0
+
+        # All indices should be < ep_end
+        assert (indices < ep_end).all()
+
+        # Center index should be frame_idx
+        half_steps = n_obs_steps // 2
+        assert indices[half_steps] == frame_idx
+
+    def test_sequence_is_monotonic(self):
+        """Frame indices should be monotonically increasing."""
+        for frame_idx in [50, 100, 300, 950]:
+            indices, _ = compute_absolute_indices(frame_idx, 0, 1000, 8, 30)
+
+            # Check monotonic (non-decreasing due to clamping)
+            diffs = indices[1:] - indices[:-1]
+            assert (diffs >= 0).all(), f"Non-monotonic at frame {frame_idx}"
+
+
+class TestComputeTau:
+    """Tests for compute_tau (within-subtask progress).
+
+    Formula: τ_t = (t - s_k) / (e_k - s_k) ∈ [0, 1]
+    """
+
+    def test_at_start(self):
+        """τ should be 0 at subtask start."""
+        tau = compute_tau(current_frame=10, subtask_start=10, subtask_end=50)
+        assert tau == 0.0
+
+    def test_at_end(self):
+        """τ should be 1 at subtask end."""
+        tau = compute_tau(current_frame=50, subtask_start=10, subtask_end=50)
+        assert tau == 1.0
+
+    def test_at_middle(self):
+        """τ should be 0.5 at subtask midpoint."""
+        tau = compute_tau(current_frame=30, subtask_start=10, subtask_end=50)
+        assert abs(tau - 0.5) < 1e-6
+
+    def test_quarter_progress(self):
+        """Test τ at 25% through subtask."""
+        tau = compute_tau(current_frame=20, subtask_start=0, subtask_end=80)
+        assert abs(tau - 0.25) < 1e-6
+
+    def test_zero_duration_subtask(self):
+        """τ should be 1.0 for zero-duration subtask."""
+        tau = compute_tau(current_frame=10, subtask_start=10, subtask_end=10)
+        assert tau == 1.0
+
+    def test_clamps_below_zero(self):
+        """τ should be clamped to 0 if frame is before subtask."""
+        tau = compute_tau(current_frame=5, subtask_start=10, subtask_end=50)
+        assert tau == 0.0
+
+    def test_clamps_above_one(self):
+        """τ should be clamped to 1 if frame is after subtask."""
+        tau = compute_tau(current_frame=60, subtask_start=10, subtask_end=50)
+        assert tau == 1.0
+
+    def test_float_inputs(self):
+        """Test with float frame indices (from interpolation)."""
+        tau = compute_tau(current_frame=25.5, subtask_start=10.0, subtask_end=50.0)
+        expected = (25.5 - 10.0) / (50.0 - 10.0)
+        assert abs(tau - expected) < 1e-6
+
+
+class TestFindStageAndTau:
+    """Tests for find_stage_and_tau logic.
+
+    This function is the core of progress label computation. It determines
+    which stage a frame belongs to and the within-stage progress (tau).
+    """
+
+    def test_single_stage_mode_linear_progress(self):
+        """Single-stage mode should give linear progress from 0 to 1."""
+        episode_length = 100
+
+        # Frame 0 -> tau = 0
+        stage, tau = find_stage_and_tau(0, episode_length, None, None, None, ["task"], {"task": 1.0})
+        assert stage == 0
+        assert abs(tau - 0.0) < 1e-6
+
+        # Frame 50 -> tau = 0.505 (50/99)
+        stage, tau = find_stage_and_tau(50, episode_length, None, None, None, ["task"], {"task": 1.0})
+        assert stage == 0
+        assert abs(tau - 50 / 99) < 1e-6
+
+        # Frame 99 -> tau = 1.0
+        stage, tau = find_stage_and_tau(99, episode_length, None, None, None, ["task"], {"task": 1.0})
+        assert stage == 0
+        assert abs(tau - 1.0) < 1e-6
+
+    def test_multi_stage_within_subtask(self):
+        """Test finding stage when frame is within a subtask."""
+        global_names = ["reach", "grasp", "lift"]
+        proportions = {"reach": 0.3, "grasp": 0.2, "lift": 0.5}
+
+        subtask_names = ["reach", "grasp", "lift"]
+        subtask_starts = [0, 30, 50]
+        subtask_ends = [29, 49, 99]
+
+        # Frame 15 in "reach" stage (index 0)
+        stage, tau = find_stage_and_tau(
+            15, 100, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage == 0
+        assert abs(tau - 15 / 29) < 1e-6
+
+        # Frame 40 in "grasp" stage (index 1)
+        stage, tau = find_stage_and_tau(
+            40, 100, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage == 1
+        # tau = (40 - 30) / (49 - 30) = 10/19
+        assert abs(tau - 10 / 19) < 1e-6
+
+        # Frame 75 in "lift" stage (index 2)
+        stage, tau = find_stage_and_tau(
+            75, 100, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage == 2
+        # tau = (75 - 50) / (99 - 50) = 25/49
+        assert abs(tau - 25 / 49) < 1e-6
+
+    def test_frame_at_subtask_boundaries(self):
+        """Test frames exactly at subtask boundaries."""
+        global_names = ["a", "b"]
+        proportions = {"a": 0.5, "b": 0.5}
+
+        subtask_names = ["a", "b"]
+        subtask_starts = [0, 50]
+        subtask_ends = [49, 99]
+
+        # Frame at start of first subtask
+        stage, tau = find_stage_and_tau(
+            0, 100, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage == 0
+        assert tau == 0.0
+
+        # Frame at end of first subtask
+        stage, tau = find_stage_and_tau(
+            49, 100, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage == 0
+        assert tau == 1.0
+
+        # Frame at start of second subtask
+        stage, tau = find_stage_and_tau(
+            50, 100, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage == 1
+        assert tau == 0.0
+
+    def test_frame_after_last_subtask(self):
+        """Frames after last subtask should return last stage with high tau."""
+        global_names = ["a", "b"]
+        proportions = {"a": 0.5, "b": 0.5}
+
+        subtask_names = ["a", "b"]
+        subtask_starts = [0, 30]
+        subtask_ends = [29, 59]
+
+        # Frame 80 is after last subtask
+        stage, tau = find_stage_and_tau(
+            80, 100, subtask_names, subtask_starts, subtask_ends, global_names, proportions
+        )
+        assert stage == 1  # Last stage
+        assert tau == 0.999  # Nearly complete
+
+
+class TestEndToEndProgressLabeling:
+    """End-to-end tests for progress label computation using normalize_stage_tau."""
+
+    def test_consistent_semantic_meaning(self):
+        """Test that same subtask completion maps to same progress across trajectories.
+
+        This is the key semantic property: "end of subtask 1" should always
+        mean the same progress value regardless of trajectory speed.
+        """
+        proportions = [0.3, 0.5, 0.2]
+
+        # Fast trajectory: subtask 1 ends at frame 30 (of 100)
+        tau_fast = compute_tau(30, 0, 30)  # = 1.0
+        y_fast = normalize_stage_tau(0 + tau_fast, temporal_proportions=proportions)
+
+        # Slow trajectory: subtask 1 ends at frame 90 (of 300)
+        tau_slow = compute_tau(90, 0, 90)  # = 1.0
+        y_slow = normalize_stage_tau(0 + tau_slow, temporal_proportions=proportions)
+
+        # Both should map to same progress (0.3 = end of subtask 1)
+        assert abs(y_fast - y_slow) < 1e-6
+        assert abs(y_fast - 0.3) < 1e-6
+
+    def test_monotonic_within_subtask(self):
+        """Test that progress is monotonically increasing within a subtask."""
+        proportions = [0.4, 0.6]
+
+        prev_y = -1
+        for tau in np.linspace(0, 1, 11):
+            y = normalize_stage_tau(0 + tau, temporal_proportions=proportions)
+            assert y > prev_y or (tau == 0 and y == 0)
+            prev_y = y
+
+    def test_continuous_across_subtasks(self):
+        """Test that progress is continuous at subtask boundaries."""
+        proportions = [0.3, 0.5, 0.2]
+
+        # End of subtask 0 (stage=0, tau=1.0) -> stage.tau = 1.0
+        y_end_0 = normalize_stage_tau(0 + 1.0, temporal_proportions=proportions)
+
+        # Start of subtask 1 (stage=1, tau=0.0) -> stage.tau = 1.0
+        y_start_1 = normalize_stage_tau(1 + 0.0, temporal_proportions=proportions)
+
+        # Should be equal (P_1 = 0.3)
+        assert abs(y_end_0 - y_start_1) < 1e-6
+
+        # End of subtask 1 (stage=1, tau=1.0) -> stage.tau = 2.0
+        y_end_1 = normalize_stage_tau(1 + 1.0, temporal_proportions=proportions)
+
+        # Start of subtask 2 (stage=2, tau=0.0) -> stage.tau = 2.0
+        y_start_2 = normalize_stage_tau(2 + 0.0, temporal_proportions=proportions)
+
+        # Should be equal (P_2 = 0.8)
+        assert abs(y_end_1 - y_start_2) < 1e-6
+
+
+class TestTemporalProportionsToBreakpoints:
+    """Tests for temporal_proportions_to_breakpoints.
+
+    Converts temporal proportions to cumulative breakpoints for normalization.
+    Example: [0.3, 0.5, 0.2] -> [0.0, 0.3, 0.8, 1.0]
+    """
+
+    def test_basic_conversion(self):
+        """Test basic conversion from proportions to breakpoints."""
+        proportions = [0.3, 0.5, 0.2]
+        breakpoints = temporal_proportions_to_breakpoints(proportions)
+
+        assert breakpoints is not None
+        assert len(breakpoints) == 4
+        assert breakpoints[0] == 0.0
+        assert abs(breakpoints[1] - 0.3) < 1e-6
+        assert abs(breakpoints[2] - 0.8) < 1e-6
+        assert breakpoints[3] == 1.0
+
+    def test_dict_input(self):
+        """Test with dict input."""
+        proportions = {"a": 0.25, "b": 0.25, "c": 0.5}
+        breakpoints = temporal_proportions_to_breakpoints(proportions)
+
+        assert breakpoints is not None
+        assert len(breakpoints) == 4
+        assert breakpoints[0] == 0.0
+        assert breakpoints[-1] == 1.0
+
+    def test_dict_with_subtask_names_order(self):
+        """Test that subtask_names determines order for dict input."""
+        proportions = {"c": 0.5, "a": 0.2, "b": 0.3}  # Dict order
+        subtask_names = ["a", "b", "c"]  # Different order
+
+        breakpoints = temporal_proportions_to_breakpoints(proportions, subtask_names)
+
+        # Breakpoints should follow subtask_names order: a=0.2, b=0.3, c=0.5
+        assert abs(breakpoints[1] - 0.2) < 1e-6  # a
+        assert abs(breakpoints[2] - 0.5) < 1e-6  # a + b = 0.5
+        assert breakpoints[3] == 1.0  # a + b + c = 1.0
+
+    def test_uniform_proportions(self):
+        """Test with uniform proportions."""
+        proportions = [0.25, 0.25, 0.25, 0.25]
+        breakpoints = temporal_proportions_to_breakpoints(proportions)
+
+        expected = [0.0, 0.25, 0.5, 0.75, 1.0]
+        for i, (bp, exp) in enumerate(zip(breakpoints, expected, strict=True)):
+            assert abs(bp - exp) < 1e-6, f"Breakpoint {i} mismatch"
+
+    def test_none_input(self):
+        """Test that None input returns None."""
+        result = temporal_proportions_to_breakpoints(None)
+        assert result is None
+
+    def test_normalization(self):
+        """Test that non-normalized proportions are normalized."""
+        # Proportions sum to 2.0, not 1.0
+        proportions = [0.6, 1.0, 0.4]
+        breakpoints = temporal_proportions_to_breakpoints(proportions)
+
+        # Should be normalized: [0.3, 0.5, 0.2] -> [0, 0.3, 0.8, 1.0]
+        assert breakpoints[-1] == 1.0
+        assert abs(breakpoints[1] - 0.3) < 1e-6
+
+
+class TestNormalizeStageTau:
+    """Tests for normalize_stage_tau.
+
+    Normalizes stage+tau values to [0, 1] using breakpoints.
+    """
+
+    def test_linear_fallback(self):
+        """Test linear normalization when only num_stages is provided."""
+        # 4 stages, linear: [0, 0.25, 0.5, 0.75, 1.0]
+
+        # Stage 0 start
+        assert normalize_stage_tau(0.0, num_stages=4) == 0.0
+
+        # Stage 0 end / Stage 1 start
+        assert abs(normalize_stage_tau(1.0, num_stages=4) - 0.25) < 1e-6
+
+        # Stage 1 middle
+        assert abs(normalize_stage_tau(1.5, num_stages=4) - 0.375) < 1e-6
+
+        # Stage 3 end
+        assert normalize_stage_tau(4.0, num_stages=4) == 1.0
+
+    def test_with_custom_breakpoints(self):
+        """Test with custom breakpoints."""
+        # Non-linear breakpoints
+        breakpoints = [0.0, 0.1, 0.5, 1.0]  # 3 stages
+
+        # Stage 0: maps [0, 1) to [0.0, 0.1)
+        assert abs(normalize_stage_tau(0.5, breakpoints=breakpoints) - 0.05) < 1e-6
+
+        # Stage 1: maps [1, 2) to [0.1, 0.5)
+        assert abs(normalize_stage_tau(1.5, breakpoints=breakpoints) - 0.3) < 1e-6
+
+        # Stage 2: maps [2, 3) to [0.5, 1.0)
+        assert abs(normalize_stage_tau(2.5, breakpoints=breakpoints) - 0.75) < 1e-6
+
+    def test_with_temporal_proportions(self):
+        """Test with temporal proportions (auto-computed breakpoints)."""
+        proportions = {"a": 0.2, "b": 0.3, "c": 0.5}
+        subtask_names = ["a", "b", "c"]
+
+        # Stage 0 end should map to 0.2
+        result = normalize_stage_tau(1.0, temporal_proportions=proportions, subtask_names=subtask_names)
+        assert abs(result - 0.2) < 1e-6
+
+        # Stage 1 end should map to 0.5
+        result = normalize_stage_tau(2.0, temporal_proportions=proportions, subtask_names=subtask_names)
+        assert abs(result - 0.5) < 1e-6
+
+    def test_tensor_input(self):
+        """Test with tensor input."""
+        x = torch.tensor([0.0, 0.5, 1.0, 1.5, 2.0])
+        breakpoints = [0.0, 0.3, 0.8, 1.0]  # 3 stages
+
+        result = normalize_stage_tau(x, breakpoints=breakpoints)
+
+        assert isinstance(result, torch.Tensor)
+        assert result.shape == x.shape
+        assert abs(result[0].item() - 0.0) < 1e-6
+        assert abs(result[2].item() - 0.3) < 1e-6  # End of stage 0
+        assert abs(result[4].item() - 0.8) < 1e-6  # End of stage 1
+
+    def test_clamping(self):
+        """Test that output is clamped to [0, 1]."""
+        # Below 0
+        assert normalize_stage_tau(-0.5, num_stages=4) == 0.0
+
+        # Above num_stages
+        assert normalize_stage_tau(5.0, num_stages=4) == 1.0
+
+    def test_batch_tensor(self):
+        """Test with batched tensor."""
+        x = torch.tensor([[0.0, 1.0, 2.0], [0.5, 1.5, 2.5]])  # (2, 3)
+
+        result = normalize_stage_tau(x, num_stages=3)
+
+        assert result.shape == (2, 3)
+        assert (result >= 0).all()
+        assert (result <= 1).all()
+
+    def test_requires_one_of_inputs(self):
+        """Test that at least one input method is required."""
+        with pytest.raises(ValueError):
+            normalize_stage_tau(1.0)
+
+
+class TestRewindAugmentation:
+    """Tests for rewind augmentation logic with bidirectional observation sampling.
+
+    Rewind appends frames before the earliest observation frame, going backwards.
+    With bidirectional sampling centered at frame_idx:
+    - Earliest obs frame = frame_idx - half_steps * frame_gap
+    - Rewind goes backwards from that point
+    """
+
+    def test_rewind_indices_go_backwards_from_earliest_obs(self):
+        """Rewind indices should go backwards from earliest observation frame."""
+        frame_idx = 300  # Center of bidirectional window
+        ep_start = 0
+        n_obs_steps = 4  # half_steps = 2
+        frame_gap = 30
+
+        # Earliest obs frame = 300 - 2*30 = 240
+        # Rewind goes backwards: 210, 180
+        rewind_step, rewind_indices = apply_rewind_augmentation(
+            frame_idx,
+            ep_start,
+            n_obs_steps=n_obs_steps,
+            max_rewind_steps=2,
+            frame_gap=frame_gap,
+            rewind_step=2,
+        )
+
+        assert rewind_step == 2
+        assert len(rewind_indices) == 2
+        # First rewind frame is closest to obs window, second is further back
+        assert rewind_indices[0] == 210  # 240 - 30
+        assert rewind_indices[1] == 180  # 240 - 60
+        assert rewind_indices[0] > rewind_indices[1], "Rewind should be descending"
+
+    def test_rewind_goes_backward_through_history(self):
+        """Rewind frames should go backward before the observation window."""
+        frame_idx = 450  # Center of bidirectional window
+        ep_start = 0
+        n_obs_steps = 8  # half_steps = 4
+        frame_gap = 30
+
+        # Earliest obs frame = 450 - 4*30 = 330
+        # Rewind from 330: [300, 270, 240]
+        rewind_step, rewind_indices = apply_rewind_augmentation(
+            frame_idx,
+            ep_start,
+            n_obs_steps=n_obs_steps,
+            max_rewind_steps=4,
+            frame_gap=frame_gap,
+            rewind_step=3,
+        )
+
+        assert rewind_step == 3
+        expected = [300, 270, 240]  # Going backwards from 330
+        assert rewind_indices == expected
+
+    def test_no_rewind_when_obs_window_at_episode_start(self):
+        """No rewind when observation window reaches episode start."""
+        frame_idx = 120  # Center of window
+        ep_start = 0
+        n_obs_steps = 8  # half_steps = 4
+        frame_gap = 30
+
+        # Earliest obs frame = 120 - 4*30 = 0 (at episode start)
+        rewind_step, rewind_indices = apply_rewind_augmentation(
+            frame_idx, ep_start, n_obs_steps=n_obs_steps, max_rewind_steps=4, frame_gap=frame_gap
+        )
+
+        # No room for rewind
+        assert rewind_step == 0
+        assert rewind_indices == []
+
+    def test_rewind_targets_are_decreasing(self):
+        """Progress targets for rewind frames should be decreasing."""
+        # Simulate progress values
+        obs_progress = [0.1, 0.2, 0.3, 0.4, 0.5]  # Forward progress
+
+        # Rewind reverses progress
+        rewind_indices = [4, 3, 2]  # Go backwards through indices
+        rewind_progress = [obs_progress[i] for i in rewind_indices]
+
+        # Should be decreasing
+        for i in range(len(rewind_progress) - 1):
+            assert rewind_progress[i] > rewind_progress[i + 1]
diff --git a/lerobot/tests/policies/wall_x/test_wallx.py b/lerobot/tests/policies/wall_x/test_wallx.py
new file mode 100644
index 0000000000000000000000000000000000000000..85656eca22ada8e2c6726b48008fb3591a08afa6
--- /dev/null
+++ b/lerobot/tests/policies/wall_x/test_wallx.py
@@ -0,0 +1,133 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test script to verify Wall-X policy integration with LeRobot"""
+
+import pytest
+import torch
+
+# Skip if required dependencies are not available
+pytest.importorskip("peft")
+pytest.importorskip("transformers")
+pytest.importorskip("torchdiffeq")
+
+from lerobot.policies.factory import make_policy_config  # noqa: E402
+from lerobot.policies.wall_x import WallXConfig  # noqa: E402
+from lerobot.policies.wall_x.modeling_wall_x import WallXPolicy  # noqa: E402
+from lerobot.policies.wall_x.processor_wall_x import make_wall_x_pre_post_processors  # noqa: E402
+from lerobot.utils.random_utils import set_seed  # noqa: E402
+from tests.utils import require_cuda, require_hf_token  # noqa: E402
+
+
+@require_cuda
+@require_hf_token
+def test_policy_instantiation():
+    # Create config
+    set_seed(42)
+    config = WallXConfig(device="cuda")
+
+    # Set up input_features and output_features in the config
+    from lerobot.configs.types import FeatureType, PolicyFeature
+
+    config.input_features = {
+        "observation.state": PolicyFeature(
+            type=FeatureType.STATE,
+            shape=(7,),
+        ),
+        "observation.images.face_view": PolicyFeature(
+            type=FeatureType.VISUAL,
+            shape=(3, 224, 224),
+        ),
+    }
+
+    config.output_features = {
+        "action": PolicyFeature(
+            type=FeatureType.ACTION,
+            shape=(7,),
+        ),
+    }
+
+    # Create dummy dataset stats
+    dataset_stats = {
+        "observation.state": {
+            "mean": torch.zeros(7),
+            "std": torch.ones(7),
+        },
+        "action": {
+            "mean": torch.zeros(7),
+            "std": torch.ones(7),
+        },
+        "observation.images.face_view": {
+            "mean": torch.zeros(3, 224, 224),
+            "std": torch.ones(3, 224, 224),
+        },
+    }
+
+    # Instantiate policy
+    policy = WallXPolicy(config)
+    preprocessor, postprocessor = make_wall_x_pre_post_processors(config=config, dataset_stats=dataset_stats)
+    # Test forward pass with dummy data
+    batch_size = 1
+    device = config.device
+    batch = {
+        "observation.state": torch.randn(batch_size, 7, dtype=torch.float32, device=device),
+        "action": torch.randn(batch_size, config.chunk_size, 7, dtype=torch.float32, device=device),
+        "observation.images.face_view": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),  # Use rand for [0,1] range
+        "task": ["Pick up the object"] * batch_size,
+    }
+    batch = preprocessor(batch)
+    try:
+        loss, loss_dict = policy.forward(batch)
+        print(f"Forward pass successful. Loss: {loss_dict['loss']:.4f}")
+    except Exception as e:
+        print(f"Forward pass failed: {e}")
+        raise
+
+    # Test inference
+    batch = {
+        "observation.state": torch.randn(batch_size, 7, dtype=torch.float32, device=device),
+        "observation.images.face_view": torch.rand(
+            batch_size, 3, 224, 224, dtype=torch.float32, device=device
+        ),  # Use rand for [0,1] range
+        "task": ["Pick up the object"] * batch_size,
+    }
+    batch = preprocessor(batch)
+    try:
+        with torch.no_grad():
+            action = policy.select_action(batch)
+            action = postprocessor(action)
+            print(f"Action: {action}")
+        print(f"Action prediction successful. Action shape: {action.shape}")
+    except Exception as e:
+        print(f"Action prediction failed: {e}")
+        raise
+
+
+@require_cuda
+@require_hf_token
+def test_config_creation():
+    """Test policy config creation through factory."""
+    try:
+        config = make_policy_config(
+            policy_type="wall_x",
+        )
+        print("Config created successfully through factory")
+        print(f"  Config type: {type(config).__name__}")
+    except Exception as e:
+        print(f"Config creation failed: {e}")
+        raise
diff --git a/lerobot/tests/policies/xvla/test_xvla_original_vs_lerobot.py b/lerobot/tests/policies/xvla/test_xvla_original_vs_lerobot.py
new file mode 100644
index 0000000000000000000000000000000000000000..3cea11329f7c6d4713d045cf30c8261b366cbd18
--- /dev/null
+++ b/lerobot/tests/policies/xvla/test_xvla_original_vs_lerobot.py
@@ -0,0 +1,319 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Test script to verify XVLA policy integration with LeRobot vs the original implementation"""
+# ruff: noqa: E402
+
+import random
+from copy import deepcopy
+from typing import Any
+
+import numpy as np
+import pytest
+import torch
+
+pytest.importorskip("transformers")
+
+from lerobot.policies.xvla.configuration_xvla import XVLAConfig
+from lerobot.policies.xvla.modeling_xvla import XVLAPolicy
+from lerobot.policies.xvla.processor_xvla import make_xvla_pre_post_processors
+from lerobot.processor import PolicyProcessorPipeline  # noqa: E402
+from lerobot.types import PolicyAction  # noqa: E402
+from lerobot.utils.constants import OBS_IMAGES, OBS_STATE  # noqa: E402
+from tests.utils import require_cuda  # noqa: E402
+
+# Constants
+DUMMY_ACTION_DIM = 7  # Standard robot arm action dimension
+DUMMY_STATE_DIM = 20  # Proprioceptive state dimension
+IMAGE_HEIGHT = 224
+IMAGE_WIDTH = 224
+NUM_VIEWS = 2  # Number of camera views
+DEVICE = "cuda" if torch.cuda.is_available() else "cpu"
+MODEL_PATH_LEROBOT = "lerobot/xvla-widowx"
+LIBERO_DOMAIN_ID = 0  # Domain ID for examples purposes
+
+# Expected values from original XVLA implementation (reference values)
+EXPECTED_ACTIONS_SHAPE = (30, 20)
+EXPECTED_ACTIONS_MEAN = 0.117606
+EXPECTED_ACTIONS_STD = 0.245411
+EXPECTED_ACTIONS_FIRST_5 = torch.tensor([0.2742, 0.4977, 0.0500, 0.7040, -0.2653])
+
+
+def set_seed_all(seed: int):
+    """Set random seed for all RNG sources to ensure reproducibility."""
+    random.seed(seed)
+    np.random.seed(seed)
+    torch.manual_seed(seed)
+
+    if torch.cuda.is_available():
+        torch.cuda.manual_seed(seed)
+        torch.cuda.manual_seed_all(seed)
+
+    # Set deterministic behavior
+    torch.backends.cudnn.deterministic = True
+    torch.backends.cudnn.benchmark = False
+    torch.use_deterministic_algorithms(True, warn_only=True)
+
+
+def instantiate_lerobot_xvla(
+    from_pretrained: bool = False,
+    model_path: str = MODEL_PATH_LEROBOT,
+) -> tuple[
+    Any,  # Policy
+    PolicyProcessorPipeline[dict[str, Any], dict[str, Any]],
+    PolicyProcessorPipeline[PolicyAction, PolicyAction],
+]:
+    """Instantiate LeRobot XVLA policy with preprocessor and postprocessor."""
+    if from_pretrained:
+        policy = XVLAPolicy.from_pretrained(
+            pretrained_name_or_path=model_path,
+            strict=False,
+        )
+    else:
+        config = XVLAConfig(
+            base_model_path=model_path,
+            n_action_steps=DUMMY_ACTION_DIM,
+            chunk_size=DUMMY_ACTION_DIM,
+            device=DEVICE,
+            num_image_views=NUM_VIEWS,
+        )  # add resize_imgs_with_padding=IMAGE_SIZE, IMAGE_SIZE?
+        policy = XVLAPolicy(config)
+
+    policy.to(DEVICE)
+    policy.config.device = DEVICE
+    preprocessor, postprocessor = make_xvla_pre_post_processors(
+        config=policy.config,
+        dataset_stats=None,  # Pass None for dataset_stats to disable normalization (original XVLA doesn't normalize)
+    )
+
+    return policy, preprocessor, postprocessor
+
+
+def create_dummy_data(device=DEVICE):
+    """Create dummy data for testing both implementations."""
+    batch_size = 1
+    prompt = "Pick up the red block and place it in the bin"
+
+    # Create random RGB images in [0, 255] uint8 range (as PIL images would be)
+    # Then convert to [0, 1] float32 range for LeRobot
+    def fake_rgb(h, w):
+        arr = np.random.randint(0, 255, (h, w, 3), dtype=np.uint8)
+        t = torch.from_numpy(arr).permute(2, 0, 1)  # CHW
+        return t
+
+    batch = {
+        f"{OBS_IMAGES}.image": torch.stack(
+            [fake_rgb(IMAGE_HEIGHT, IMAGE_WIDTH) for _ in range(batch_size)]
+        ).to(device),
+        f"{OBS_IMAGES}.image2": torch.stack(
+            [fake_rgb(IMAGE_HEIGHT, IMAGE_WIDTH) for _ in range(batch_size)]
+        ).to(device),
+        OBS_STATE: torch.randn(batch_size, DUMMY_STATE_DIM, dtype=torch.float32, device=device),
+        "task": [prompt for _ in range(batch_size)],
+    }
+
+    return batch
+
+
+# Pytest fixtures
+@pytest.fixture(scope="module")
+def xvla_components():
+    """Fixture to instantiate and provide all XVLA components for tests."""
+    print(f"\nTesting with DEVICE='{DEVICE}'")
+    print("\n[Setup] Instantiating LeRobot XVLA policy...")
+    policy_obj, preprocessor_obj, postprocessor_obj = instantiate_lerobot_xvla(from_pretrained=True)
+    print("✔️ Model loaded successfully")
+    yield policy_obj, preprocessor_obj, postprocessor_obj
+
+
+@pytest.fixture(scope="module")
+def policy(xvla_components):
+    """Fixture to provide the XVLA policy for tests."""
+    return xvla_components[0]
+
+
+@pytest.fixture(scope="module")
+def preprocessor(xvla_components):
+    """Fixture to provide the XVLA preprocessor for tests."""
+    return xvla_components[1]
+
+
+@require_cuda
+def test_xvla_preprocessor_alignment(policy, preprocessor):
+    """Test that LeRobot XVLA preprocessor produces expected outputs."""
+    print("\n" + "=" * 80)
+    print("Test: XVLA Preprocessor Outputs")
+    print("=" * 80)
+
+    set_seed_all(42)
+
+    print("\nCreating dummy data...")
+    batch = create_dummy_data()
+
+    print("\n[LeRobot] Preprocessing...")
+    lerobot_observation = preprocessor(deepcopy(batch))
+    lerobot_inputs = policy._build_model_inputs(lerobot_observation)
+
+    print("\nVerifying preprocessor outputs:")
+    print("-" * 80)
+
+    # Expected shapes from tester.txt
+    expected_shapes = {
+        "domain_id": (1,),
+        "input_ids": (1, 50),
+        "proprio": (1, 20),
+        "image_mask": (1, 2),
+        "image_input": (1, 2, 3, 224, 224),
+    }
+
+    for key, expected_shape in expected_shapes.items():
+        if key in lerobot_inputs:
+            actual_shape = tuple(lerobot_inputs[key].shape)
+            print(f"\nKey: {key}")
+            print(f"Expected shape: {expected_shape}")
+            print(f"Actual shape: {actual_shape}")
+
+            if actual_shape == expected_shape:
+                print("Shape matches!")
+            else:
+                print("Shape mismatch!")
+
+            assert actual_shape == expected_shape, f"Shape mismatch for {key}"
+        else:
+            print(f"\nKey '{key}' not found in inputs!")
+
+    print("\nAll preprocessor outputs have correct shapes!")
+
+
+@require_cuda
+def test_xvla_action_generation(policy, preprocessor):
+    """Test XVLA LeRobot implementation generates expected actions."""
+    print("\n" + "=" * 80)
+    print("Test: XVLA Action Generation Against Expected Values")
+    print("=" * 80)
+
+    set_seed_all(42)
+
+    print("\nCreating dummy data...")
+    batch = create_dummy_data()
+
+    print("\n[LeRobot] Running inference...")
+    lerobot_observation = preprocessor(deepcopy(batch))
+    lerobot_inputs = policy._build_model_inputs(lerobot_observation)
+
+    # Reset seed for inference
+    torch.manual_seed(42)
+    with torch.no_grad():
+        lerobot_actions = policy.model.generate_actions(**lerobot_inputs, steps=10)
+        lerobot_actions = lerobot_actions.squeeze(0).float().cpu()
+
+    print(f"LeRobot actions shape: {lerobot_actions.shape}")
+    print(f"LeRobot actions mean: {lerobot_actions.mean().item():.6f}")
+    print(f"LeRobot actions std: {lerobot_actions.std().item():.6f}")
+    print(f"LeRobot actions first 5: {lerobot_actions[0, :5]}")
+
+    print("\nExpected values (from original XVLA):")
+    print(f"Expected actions shape: {EXPECTED_ACTIONS_SHAPE}")
+    print(f"Expected actions mean: {EXPECTED_ACTIONS_MEAN:.6f}")
+    print(f"Expected actions std: {EXPECTED_ACTIONS_STD:.6f}")
+    print(f"Expected actions first 5: {EXPECTED_ACTIONS_FIRST_5}")
+
+    print("\nAction Comparison:")
+    print("-" * 80)
+
+    # Compare shapes
+    actual_shape = tuple(lerobot_actions.shape)
+    assert actual_shape == EXPECTED_ACTIONS_SHAPE, (
+        f"Shape mismatch: {actual_shape} vs {EXPECTED_ACTIONS_SHAPE}"
+    )
+    print(f"✔️ Shape matches: {actual_shape}")
+
+    # Compare statistics
+    actual_mean = lerobot_actions.mean().item()
+    actual_std = lerobot_actions.std().item()
+
+    mean_diff = abs(actual_mean - EXPECTED_ACTIONS_MEAN)
+    std_diff = abs(actual_std - EXPECTED_ACTIONS_STD)
+
+    print(f"\nMean: {actual_mean:.6f} (expected: {EXPECTED_ACTIONS_MEAN:.6f}, diff: {mean_diff:.6e})")
+    print(f"Std: {actual_std:.6f} (expected: {EXPECTED_ACTIONS_STD:.6f}, diff: {std_diff:.6e})")
+
+    # Compare first 5 actions
+    actual_first_5 = lerobot_actions[0, :5]
+    first_5_diff = torch.abs(actual_first_5 - EXPECTED_ACTIONS_FIRST_5)
+
+    print("\nFirst 5 actions comparison:")
+    print(f"  Actual:   {actual_first_5}")
+    print(f"  Expected: {EXPECTED_ACTIONS_FIRST_5}")
+    print(f"  Max diff: {first_5_diff.max().item():.6e}")
+    print(f"  Mean diff: {first_5_diff.mean().item():.6e}")
+
+    # Check with different tolerances
+    tolerances = [1e-5, 1e-4, 1e-3, 1e-2]
+    for tol in tolerances:
+        is_close = torch.allclose(actual_first_5, EXPECTED_ACTIONS_FIRST_5, atol=tol)
+        status = "Success" if is_close else "Failure"
+        print(f"{status}: First 5 actions close (atol={tol}): {is_close}")
+
+    # Assert with reasonable tolerance
+    tolerance = 1e-3
+    assert torch.allclose(actual_first_5, EXPECTED_ACTIONS_FIRST_5, atol=tolerance), (
+        f"First 5 actions differ by more than tolerance ({tolerance})"
+    )
+    print(f"\nSuccess: Actions match expected values within tolerance ({tolerance})!")
+
+
+@require_cuda
+def test_xvla_inference_reproducibility(policy, preprocessor):
+    """Test that XVLA inference is reproducible with the same seed."""
+    print("\n" + "=" * 80)
+    print("Test: XVLA Inference Reproducibility")
+    print("=" * 80)
+
+    print("\nCreating dummy data...")
+    batch = create_dummy_data()
+
+    # First inference
+    print("\n[Run 1] Running inference...")
+    set_seed_all(42)
+    lerobot_observation = preprocessor(deepcopy(batch))
+    lerobot_inputs = policy._build_model_inputs(lerobot_observation)
+    with torch.no_grad():
+        actions_1 = policy.model.generate_actions(**lerobot_inputs, steps=10)
+        actions_1 = actions_1.squeeze(0).float().cpu()
+
+    # Second inference with same seed
+    print("\n[Run 2] Running inference with same seed...")
+    set_seed_all(42)
+    lerobot_observation = preprocessor(deepcopy(batch))
+    lerobot_inputs = policy._build_model_inputs(lerobot_observation)
+    with torch.no_grad():
+        actions_2 = policy.model.generate_actions(**lerobot_inputs, steps=10)
+        actions_2 = actions_2.squeeze(0).float().cpu()
+
+    print("\nComparing two runs:")
+    print("-" * 80)
+    if torch.allclose(actions_1, actions_2, atol=1e-8):
+        print("Inference is perfectly reproducible!")
+    else:
+        diff = torch.abs(actions_1 - actions_2)
+        print("Small differences detected:")
+        print(f"  Max diff: {diff.max().item():.6e}")
+        print(f"  Mean diff: {diff.mean().item():.6e}")
+
+    assert torch.allclose(actions_1, actions_2, atol=1e-6), "Inference should be reproducible!"
+
+    print("\nInference is reproducible!")
diff --git a/lerobot/tests/processor/test_act_processor.py b/lerobot/tests/processor/test_act_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..134cff6849fe02337ab1a7d3124e67b81e41abaa
--- /dev/null
+++ b/lerobot/tests/processor/test_act_processor.py
@@ -0,0 +1,412 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tests for ACT policy processor."""
+
+import tempfile
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.policies.act.configuration_act import ACTConfig
+from lerobot.policies.act.processor_act import make_act_pre_post_processors
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DataProcessorPipeline,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    RenameObservationsProcessorStep,
+    TransitionKey,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import create_transition, transition_to_batch
+from lerobot.utils.constants import ACTION, OBS_STATE
+
+
+def create_default_config():
+    """Create a default ACT configuration for testing."""
+    config = ACTConfig()
+    config.input_features = {
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(7,)),
+    }
+    config.output_features = {
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(4,)),
+    }
+    config.normalization_mapping = {
+        FeatureType.STATE: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+    config.device = "cpu"
+    return config
+
+
+def create_default_stats():
+    """Create default dataset statistics for testing."""
+    return {
+        OBS_STATE: {"mean": torch.zeros(7), "std": torch.ones(7)},
+        ACTION: {"mean": torch.zeros(4), "std": torch.ones(4)},
+    }
+
+
+def test_make_act_processor_basic():
+    """Test basic creation of ACT processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_act_pre_post_processors(config, stats)
+
+    # Check processor names
+    assert preprocessor.name == "policy_preprocessor"
+    assert postprocessor.name == "policy_postprocessor"
+
+    # Check steps in preprocessor
+    assert len(preprocessor.steps) == 4
+    assert isinstance(preprocessor.steps[0], RenameObservationsProcessorStep)
+    assert isinstance(preprocessor.steps[1], AddBatchDimensionProcessorStep)
+    assert isinstance(preprocessor.steps[2], DeviceProcessorStep)
+    assert isinstance(preprocessor.steps[3], NormalizerProcessorStep)
+
+    # Check steps in postprocessor
+    assert len(postprocessor.steps) == 2
+    assert isinstance(postprocessor.steps[0], UnnormalizerProcessorStep)
+    assert isinstance(postprocessor.steps[1], DeviceProcessorStep)
+
+
+def test_act_processor_normalization():
+    """Test that ACT processor correctly normalizes and unnormalizes data."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_act_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create test data
+    observation = {OBS_STATE: torch.randn(7)}
+    action = torch.randn(4)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data is normalized and batched
+    assert processed[OBS_STATE].shape == (1, 7)
+    assert processed[TransitionKey.ACTION.value].shape == (1, 4)
+
+    # Process action through postprocessor
+    postprocessed = postprocessor(processed[TransitionKey.ACTION.value])
+
+    # Check that action is unnormalized
+    assert postprocessed.shape == (1, 4)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_act_processor_cuda():
+    """Test ACT processor with CUDA device."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_act_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create CPU data
+    observation = {OBS_STATE: torch.randn(7)}
+    action = torch.randn(4)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data is on CUDA
+    assert processed[OBS_STATE].device.type == "cuda"
+    assert processed[TransitionKey.ACTION.value].device.type == "cuda"
+
+    # Process through postprocessor
+    postprocessed = postprocessor(processed[TransitionKey.ACTION.value])
+
+    # Check that action is back on CPU
+    assert postprocessed.device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_act_processor_accelerate_scenario():
+    """Test ACT processor in simulated Accelerate scenario (data already on GPU)."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_act_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Simulate Accelerate: data already on GPU
+    device = torch.device("cuda:0")
+    observation = {OBS_STATE: torch.randn(1, 7).to(device)}  # Already batched and on GPU
+    action = torch.randn(1, 4).to(device)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data stays on same GPU (not moved unnecessarily)
+    assert processed[OBS_STATE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_act_processor_multi_gpu():
+    """Test ACT processor with multi-GPU setup."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_act_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Simulate data on different GPU (like in multi-GPU training)
+    device = torch.device("cuda:1")
+    observation = {OBS_STATE: torch.randn(1, 7).to(device)}
+    action = torch.randn(1, 4).to(device)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data stays on cuda:1 (not moved to cuda:0)
+    assert processed[OBS_STATE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+def test_act_processor_without_stats():
+    """Test ACT processor creation without dataset statistics."""
+    config = create_default_config()
+
+    preprocessor, postprocessor = make_act_pre_post_processors(
+        config,
+        dataset_stats=None,
+    )
+
+    # Should still create processors, but normalization won't have stats
+    assert preprocessor is not None
+    assert postprocessor is not None
+
+    # Process should still work (but won't normalize without stats)
+    observation = {OBS_STATE: torch.randn(7)}
+    action = torch.randn(4)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed is not None
+
+
+def test_act_processor_save_and_load():
+    """Test saving and loading ACT processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_act_pre_post_processors(
+        config,
+        stats,
+    )
+
+    with tempfile.TemporaryDirectory() as tmpdir:
+        # Save preprocessor
+        preprocessor.save_pretrained(tmpdir)
+
+        # Load preprocessor
+        loaded_preprocessor = DataProcessorPipeline.from_pretrained(
+            tmpdir, config_filename="policy_preprocessor.json"
+        )
+
+        # Test that loaded processor works
+        observation = {OBS_STATE: torch.randn(7)}
+        action = torch.randn(4)
+        transition = create_transition(observation, action)
+        batch = transition_to_batch(transition)
+
+        processed = loaded_preprocessor(batch)
+        assert processed[OBS_STATE].shape == (1, 7)
+        assert processed[TransitionKey.ACTION.value].shape == (1, 4)
+
+
+def test_act_processor_device_placement_preservation():
+    """Test that ACT processor preserves device placement correctly."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    # Test with CPU config
+    config.device = "cpu"
+    preprocessor, _ = make_act_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Process CPU data
+    observation = {OBS_STATE: torch.randn(7)}
+    action = torch.randn(4)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed[OBS_STATE].device.type == "cpu"
+    assert processed[TransitionKey.ACTION.value].device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_act_processor_mixed_precision():
+    """Test ACT processor with mixed precision (float16)."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    # Modify the device processor to use float16
+    preprocessor, postprocessor = make_act_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Replace DeviceProcessorStep with one that uses float16
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="float16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Update normalizer to use the same device as the device processor
+            norm_step = step  # Now type checker knows this is NormalizerProcessorStep
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=norm_step.features,
+                    norm_map=norm_step.norm_map,
+                    stats=norm_step.stats,
+                    device=config.device,
+                    dtype=torch.float16,  # Match the float16 dtype
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Create test data
+    observation = {OBS_STATE: torch.randn(7, dtype=torch.float32)}
+    action = torch.randn(4, dtype=torch.float32)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data is converted to float16
+    assert processed[OBS_STATE].dtype == torch.float16
+    assert processed[TransitionKey.ACTION.value].dtype == torch.float16
+
+
+def test_act_processor_batch_consistency():
+    """Test that ACT processor handles different batch sizes correctly."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_act_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Test single sample (unbatched)
+    observation = {OBS_STATE: torch.randn(7)}
+    action = torch.randn(4)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed[OBS_STATE].shape[0] == 1  # Batched
+
+    # Test already batched data
+    observation_batched = {OBS_STATE: torch.randn(8, 7)}  # Batch of 8
+    action_batched = torch.randn(8, 4)
+    transition_batched = create_transition(observation_batched, action_batched)
+    batch_batched = transition_to_batch(transition_batched)
+
+    processed_batched = preprocessor(batch_batched)
+    assert processed_batched[OBS_STATE].shape[0] == 8
+    assert processed_batched[TransitionKey.ACTION.value].shape[0] == 8
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_act_processor_bfloat16_device_float32_normalizer():
+    """Test: DeviceProcessor(bfloat16) + NormalizerProcessor(float32) → output bfloat16 via automatic adaptation"""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, _ = make_act_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Modify the pipeline to use bfloat16 device processor with float32 normalizer
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            # Device processor converts to bfloat16
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="bfloat16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Normalizer stays configured as float32 (will auto-adapt to bfloat16)
+            norm_step = step  # Now type checker knows this is NormalizerProcessorStep
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=norm_step.features,
+                    norm_map=norm_step.norm_map,
+                    stats=norm_step.stats,
+                    device=config.device,
+                    dtype=torch.float32,  # Deliberately configured as float32
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Verify initial normalizer configuration
+    normalizer_step = preprocessor.steps[3]  # NormalizerProcessorStep
+    assert normalizer_step.dtype == torch.float32
+
+    # Create test data
+    observation = {OBS_STATE: torch.randn(7, dtype=torch.float32)}  # Start with float32
+    action = torch.randn(4, dtype=torch.float32)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through full pipeline
+    processed = preprocessor(batch)
+
+    # Verify: DeviceProcessor → bfloat16, NormalizerProcessor adapts → final output is bfloat16
+    assert processed[OBS_STATE].dtype == torch.bfloat16
+    assert processed[TransitionKey.ACTION.value].dtype == torch.bfloat16
+
+    # Verify normalizer automatically adapted its internal state
+    assert normalizer_step.dtype == torch.bfloat16
+    for stat_tensor in normalizer_step._tensor_stats[OBS_STATE].values():
+        assert stat_tensor.dtype == torch.bfloat16
diff --git a/lerobot/tests/processor/test_batch_conversion.py b/lerobot/tests/processor/test_batch_conversion.py
new file mode 100644
index 0000000000000000000000000000000000000000..d589b6c5ec7163fa81628d9bde02cfd7068be4af
--- /dev/null
+++ b/lerobot/tests/processor/test_batch_conversion.py
@@ -0,0 +1,296 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import torch
+
+from lerobot.processor import DataProcessorPipeline
+from lerobot.processor.converters import batch_to_transition, transition_to_batch
+from lerobot.types import TransitionKey
+from lerobot.utils.constants import ACTION, DONE, OBS_IMAGE, OBS_PREFIX, OBS_STATE, REWARD, TRUNCATED
+
+
+def _dummy_batch():
+    """Create a dummy batch using the new format with observation.* and next.* keys."""
+    return {
+        f"{OBS_IMAGE}.left": torch.randn(1, 3, 128, 128),
+        f"{OBS_IMAGE}.right": torch.randn(1, 3, 128, 128),
+        OBS_STATE: torch.tensor([[0.1, 0.2, 0.3, 0.4]]),
+        ACTION: torch.tensor([[0.5]]),
+        REWARD: 1.0,
+        DONE: False,
+        TRUNCATED: False,
+        "info": {"key": "value"},
+    }
+
+
+def test_observation_grouping_roundtrip():
+    """Test that observation.* keys are properly grouped and ungrouped."""
+    proc = DataProcessorPipeline([])
+    batch_in = _dummy_batch()
+    batch_out = proc(batch_in)
+
+    # Check that all observation.* keys are preserved
+    original_obs_keys = {k: v for k, v in batch_in.items() if k.startswith(OBS_PREFIX)}
+    reconstructed_obs_keys = {k: v for k, v in batch_out.items() if k.startswith(OBS_PREFIX)}
+
+    assert set(original_obs_keys.keys()) == set(reconstructed_obs_keys.keys())
+
+    # Check tensor values
+    assert torch.allclose(batch_out[f"{OBS_IMAGE}.left"], batch_in[f"{OBS_IMAGE}.left"])
+    assert torch.allclose(batch_out[f"{OBS_IMAGE}.right"], batch_in[f"{OBS_IMAGE}.right"])
+    assert torch.allclose(batch_out[OBS_STATE], batch_in[OBS_STATE])
+
+    # Check other fields
+    assert torch.allclose(batch_out[ACTION], batch_in[ACTION])
+    assert batch_out[REWARD] == batch_in[REWARD]
+    assert batch_out[DONE] == batch_in[DONE]
+    assert batch_out[TRUNCATED] == batch_in[TRUNCATED]
+    assert batch_out["info"] == batch_in["info"]
+
+
+def test_batch_to_transition_observation_grouping():
+    """Test that batch_to_transition correctly groups observation.* keys."""
+    batch = {
+        f"{OBS_IMAGE}.top": torch.randn(1, 3, 128, 128),
+        f"{OBS_IMAGE}.left": torch.randn(1, 3, 128, 128),
+        OBS_STATE: [1, 2, 3, 4],
+        ACTION: torch.tensor([0.1, 0.2, 0.3, 0.4]),
+        REWARD: 1.5,
+        DONE: True,
+        TRUNCATED: False,
+        "info": {"episode": 42},
+    }
+
+    transition = batch_to_transition(batch)
+
+    # Check observation is a dict with all observation.* keys
+    assert isinstance(transition[TransitionKey.OBSERVATION], dict)
+    assert f"{OBS_IMAGE}.top" in transition[TransitionKey.OBSERVATION]
+    assert f"{OBS_IMAGE}.left" in transition[TransitionKey.OBSERVATION]
+    assert OBS_STATE in transition[TransitionKey.OBSERVATION]
+
+    # Check values are preserved
+    assert torch.allclose(
+        transition[TransitionKey.OBSERVATION][f"{OBS_IMAGE}.top"], batch[f"{OBS_IMAGE}.top"]
+    )
+    assert torch.allclose(
+        transition[TransitionKey.OBSERVATION][f"{OBS_IMAGE}.left"], batch[f"{OBS_IMAGE}.left"]
+    )
+    assert transition[TransitionKey.OBSERVATION][OBS_STATE] == [1, 2, 3, 4]
+
+    # Check other fields
+    assert torch.allclose(transition[TransitionKey.ACTION], torch.tensor([0.1, 0.2, 0.3, 0.4]))
+    assert transition[TransitionKey.REWARD] == 1.5
+    assert transition[TransitionKey.DONE]
+    assert not transition[TransitionKey.TRUNCATED]
+    assert transition[TransitionKey.INFO] == {"episode": 42}
+    assert transition[TransitionKey.COMPLEMENTARY_DATA] == {}
+
+
+def test_transition_to_batch_observation_flattening():
+    """Test that transition_to_batch correctly flattens observation dict."""
+    observation_dict = {
+        f"{OBS_IMAGE}.top": torch.randn(1, 3, 128, 128),
+        f"{OBS_IMAGE}.left": torch.randn(1, 3, 128, 128),
+        OBS_STATE: [1, 2, 3, 4],
+    }
+
+    transition = {
+        TransitionKey.OBSERVATION: observation_dict,
+        TransitionKey.ACTION: "action_data",
+        TransitionKey.REWARD: 1.5,
+        TransitionKey.DONE: True,
+        TransitionKey.TRUNCATED: False,
+        TransitionKey.INFO: {"episode": 42},
+        TransitionKey.COMPLEMENTARY_DATA: {},
+    }
+
+    batch = transition_to_batch(transition)
+
+    # Check that observation.* keys are flattened back to batch
+    assert f"{OBS_IMAGE}.top" in batch
+    assert f"{OBS_IMAGE}.left" in batch
+    assert OBS_STATE in batch
+
+    # Check values are preserved
+    assert torch.allclose(batch[f"{OBS_IMAGE}.top"], observation_dict[f"{OBS_IMAGE}.top"])
+    assert torch.allclose(batch[f"{OBS_IMAGE}.left"], observation_dict[f"{OBS_IMAGE}.left"])
+    assert batch[OBS_STATE] == [1, 2, 3, 4]
+
+    # Check other fields are mapped to next.* format
+    assert batch[ACTION] == "action_data"
+    assert batch[REWARD] == 1.5
+    assert batch[DONE]
+    assert not batch[TRUNCATED]
+    assert batch["info"] == {"episode": 42}
+
+
+def test_no_observation_keys():
+    """Test behavior when there are no observation.* keys."""
+    batch = {
+        ACTION: torch.tensor([1.0, 2.0]),
+        REWARD: 2.0,
+        DONE: False,
+        TRUNCATED: True,
+        "info": {"test": "no_obs"},
+    }
+
+    transition = batch_to_transition(batch)
+
+    # Observation should be None when no observation.* keys
+    assert transition[TransitionKey.OBSERVATION] is None
+
+    # Check other fields
+    assert torch.allclose(transition[TransitionKey.ACTION], torch.tensor([1.0, 2.0]))
+    assert transition[TransitionKey.REWARD] == 2.0
+    assert not transition[TransitionKey.DONE]
+    assert transition[TransitionKey.TRUNCATED]
+    assert transition[TransitionKey.INFO] == {"test": "no_obs"}
+
+    # Round trip should work
+    reconstructed_batch = transition_to_batch(transition)
+    assert torch.allclose(reconstructed_batch[ACTION], torch.tensor([1.0, 2.0]))
+    assert reconstructed_batch[REWARD] == 2.0
+    assert not reconstructed_batch[DONE]
+    assert reconstructed_batch[TRUNCATED]
+    assert reconstructed_batch["info"] == {"test": "no_obs"}
+
+
+def test_minimal_batch():
+    """Test with minimal batch containing only observation.* and action."""
+    batch = {OBS_STATE: "minimal_state", ACTION: torch.tensor([0.5])}
+
+    transition = batch_to_transition(batch)
+
+    # Check observation
+    assert transition[TransitionKey.OBSERVATION] == {OBS_STATE: "minimal_state"}
+    assert torch.allclose(transition[TransitionKey.ACTION], torch.tensor([0.5]))
+
+    # Check defaults
+    assert transition[TransitionKey.REWARD] == 0.0
+    assert not transition[TransitionKey.DONE]
+    assert not transition[TransitionKey.TRUNCATED]
+    assert transition[TransitionKey.INFO] == {}
+    assert transition[TransitionKey.COMPLEMENTARY_DATA] == {}
+
+    # Round trip
+    reconstructed_batch = transition_to_batch(transition)
+    assert reconstructed_batch[OBS_STATE] == "minimal_state"
+    assert torch.allclose(reconstructed_batch[ACTION], torch.tensor([0.5]))
+    assert reconstructed_batch[REWARD] == 0.0
+    assert not reconstructed_batch[DONE]
+    assert not reconstructed_batch[TRUNCATED]
+    assert reconstructed_batch["info"] == {}
+
+
+def test_empty_batch():
+    """Test behavior with empty batch."""
+    batch = {}
+
+    transition = batch_to_transition(batch)
+
+    # All fields should have defaults
+    assert transition[TransitionKey.OBSERVATION] is None
+    assert transition[TransitionKey.ACTION] is None
+    assert transition[TransitionKey.REWARD] == 0.0
+    assert not transition[TransitionKey.DONE]
+    assert not transition[TransitionKey.TRUNCATED]
+    assert transition[TransitionKey.INFO] == {}
+    assert transition[TransitionKey.COMPLEMENTARY_DATA] == {}
+
+    # Round trip
+    reconstructed_batch = transition_to_batch(transition)
+    assert reconstructed_batch[ACTION] is None
+    assert reconstructed_batch[REWARD] == 0.0
+    assert not reconstructed_batch[DONE]
+    assert not reconstructed_batch[TRUNCATED]
+    assert reconstructed_batch["info"] == {}
+
+
+def test_complex_nested_observation():
+    """Test with complex nested observation data."""
+    batch = {
+        f"{OBS_IMAGE}.top": {"image": torch.randn(1, 3, 128, 128), "timestamp": 1234567890},
+        f"{OBS_IMAGE}.left": {"image": torch.randn(1, 3, 128, 128), "timestamp": 1234567891},
+        OBS_STATE: torch.randn(7),
+        ACTION: torch.randn(8),
+        REWARD: 3.14,
+        DONE: False,
+        TRUNCATED: True,
+        "info": {"episode_length": 200, "success": True},
+    }
+
+    transition = batch_to_transition(batch)
+    reconstructed_batch = transition_to_batch(transition)
+
+    # Check that all observation keys are preserved
+    original_obs_keys = {k for k in batch if k.startswith(OBS_PREFIX)}
+    reconstructed_obs_keys = {k for k in reconstructed_batch if k.startswith(OBS_PREFIX)}
+
+    assert original_obs_keys == reconstructed_obs_keys
+
+    # Check tensor values
+    assert torch.allclose(batch[OBS_STATE], reconstructed_batch[OBS_STATE])
+
+    # Check nested dict with tensors
+    assert torch.allclose(
+        batch[f"{OBS_IMAGE}.top"]["image"], reconstructed_batch[f"{OBS_IMAGE}.top"]["image"]
+    )
+    assert torch.allclose(
+        batch[f"{OBS_IMAGE}.left"]["image"], reconstructed_batch[f"{OBS_IMAGE}.left"]["image"]
+    )
+
+    # Check action tensor
+    assert torch.allclose(batch[ACTION], reconstructed_batch[ACTION])
+
+    # Check other fields
+    assert batch[REWARD] == reconstructed_batch[REWARD]
+    assert batch[DONE] == reconstructed_batch[DONE]
+    assert batch[TRUNCATED] == reconstructed_batch[TRUNCATED]
+    assert batch["info"] == reconstructed_batch["info"]
+
+
+def test_custom_converter():
+    """Test that custom converters can still be used."""
+
+    def to_tr(batch):
+        # Custom converter that modifies the reward
+        tr = batch_to_transition(batch)
+        # Double the reward
+        reward = tr.get(TransitionKey.REWARD, 0.0)
+        new_tr = tr.copy()
+        new_tr[TransitionKey.REWARD] = reward * 2 if reward is not None else 0.0
+        return new_tr
+
+    def to_batch(tr):
+        batch = transition_to_batch(tr)
+        return batch
+
+    processor = DataProcessorPipeline(steps=[], to_transition=to_tr, to_output=to_batch)
+
+    batch = {
+        OBS_STATE: torch.randn(1, 4),
+        ACTION: torch.randn(1, 2),
+        REWARD: 1.0,
+        DONE: False,
+    }
+
+    result = processor(batch)
+
+    # Check the reward was doubled by our custom converter
+    assert result[REWARD] == 2.0
+    assert torch.allclose(result[OBS_STATE], batch[OBS_STATE])
+    assert torch.allclose(result[ACTION], batch[ACTION])
diff --git a/lerobot/tests/processor/test_batch_processor.py b/lerobot/tests/processor/test_batch_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..5c94b06570cbe146cfca3277581545f74e813279
--- /dev/null
+++ b/lerobot/tests/processor/test_batch_processor.py
@@ -0,0 +1,1184 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import tempfile
+from pathlib import Path
+
+import numpy as np
+import pytest
+import torch
+
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DataProcessorPipeline,
+    ProcessorStepRegistry,
+    TransitionKey,
+)
+from lerobot.processor.converters import create_transition, identity_transition
+from lerobot.utils.constants import OBS_ENV_STATE, OBS_IMAGE, OBS_IMAGES, OBS_STATE
+
+
+def test_state_1d_to_2d():
+    """Test that 1D state tensors get unsqueezed to 2D."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Test observation.state
+    state_1d = torch.randn(7)
+    observation = {OBS_STATE: state_1d}
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+
+    processed_state = result[TransitionKey.OBSERVATION][OBS_STATE]
+    assert processed_state.shape == (1, 7)
+    assert torch.allclose(processed_state.squeeze(0), state_1d)
+
+
+def test_env_state_1d_to_2d():
+    """Test that 1D environment state tensors get unsqueezed to 2D."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Test observation.environment_state
+    env_state_1d = torch.randn(10)
+    observation = {OBS_ENV_STATE: env_state_1d}
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+
+    processed_env_state = result[TransitionKey.OBSERVATION][OBS_ENV_STATE]
+    assert processed_env_state.shape == (1, 10)
+    assert torch.allclose(processed_env_state.squeeze(0), env_state_1d)
+
+
+def test_image_3d_to_4d():
+    """Test that 3D image tensors get unsqueezed to 4D."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Test observation.image
+    image_3d = torch.randn(224, 224, 3)
+    observation = {OBS_IMAGE: image_3d}
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+
+    processed_image = result[TransitionKey.OBSERVATION][OBS_IMAGE]
+    assert processed_image.shape == (1, 224, 224, 3)
+    assert torch.allclose(processed_image.squeeze(0), image_3d)
+
+
+def test_multiple_images_3d_to_4d():
+    """Test that 3D image tensors in observation.images.* get unsqueezed to 4D."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Test observation.images.camera1 and observation.images.camera2
+    image1_3d = torch.randn(64, 64, 3)
+    image2_3d = torch.randn(128, 128, 3)
+    observation = {
+        f"{OBS_IMAGES}.camera1": image1_3d,
+        f"{OBS_IMAGES}.camera2": image2_3d,
+    }
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+
+    processed_obs = result[TransitionKey.OBSERVATION]
+    processed_image1 = processed_obs[f"{OBS_IMAGES}.camera1"]
+    processed_image2 = processed_obs[f"{OBS_IMAGES}.camera2"]
+
+    assert processed_image1.shape == (1, 64, 64, 3)
+    assert processed_image2.shape == (1, 128, 128, 3)
+    assert torch.allclose(processed_image1.squeeze(0), image1_3d)
+    assert torch.allclose(processed_image2.squeeze(0), image2_3d)
+
+
+def test_already_batched_tensors_unchanged():
+    """Test that already batched tensors remain unchanged."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create already batched tensors
+    state_2d = torch.randn(1, 7)
+    env_state_2d = torch.randn(1, 10)
+    image_4d = torch.randn(1, 224, 224, 3)
+
+    observation = {
+        OBS_STATE: state_2d,
+        OBS_ENV_STATE: env_state_2d,
+        OBS_IMAGE: image_4d,
+    }
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Should remain unchanged
+    assert torch.allclose(processed_obs[OBS_STATE], state_2d)
+    assert torch.allclose(processed_obs[OBS_ENV_STATE], env_state_2d)
+    assert torch.allclose(processed_obs[OBS_IMAGE], image_4d)
+
+
+def test_higher_dimensional_tensors_unchanged():
+    """Test that tensors with more dimensions than expected remain unchanged."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create tensors with more dimensions
+    state_3d = torch.randn(2, 7, 5)  # More than 1D
+    image_5d = torch.randn(2, 3, 224, 224, 1)  # More than 3D
+
+    observation = {
+        OBS_STATE: state_3d,
+        OBS_IMAGE: image_5d,
+    }
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Should remain unchanged
+    assert torch.allclose(processed_obs[OBS_STATE], state_3d)
+    assert torch.allclose(processed_obs[OBS_IMAGE], image_5d)
+
+
+def test_non_tensor_values_unchanged():
+    """Test that non-tensor values in observations remain unchanged."""
+    processor = AddBatchDimensionProcessorStep()
+
+    observation = {
+        OBS_STATE: [1, 2, 3],  # List, not tensor
+        OBS_IMAGE: "not_a_tensor",  # String
+        "custom_key": 42,  # Integer
+        "another_key": {"nested": "dict"},  # Dict
+    }
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Should remain unchanged
+    assert processed_obs[OBS_STATE] == [1, 2, 3]
+    assert processed_obs[OBS_IMAGE] == "not_a_tensor"
+    assert processed_obs["custom_key"] == 42
+    assert processed_obs["another_key"] == {"nested": "dict"}
+
+
+def test_none_observation():
+    """Test processor handles None observation gracefully."""
+    processor = AddBatchDimensionProcessorStep()
+
+    transition = create_transition(observation={}, action=torch.empty(0))
+    result = processor(transition)
+
+    assert result[TransitionKey.OBSERVATION] == {}
+
+
+def test_empty_observation():
+    """Test processor handles empty observation dict."""
+    processor = AddBatchDimensionProcessorStep()
+
+    observation = {}
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+
+    assert result[TransitionKey.OBSERVATION] == {}
+
+
+def test_mixed_observation():
+    """Test processor with mixed observation containing various types and dimensions."""
+    processor = AddBatchDimensionProcessorStep()
+
+    state_1d = torch.randn(5)
+    env_state_2d = torch.randn(1, 8)  # Already batched
+    image_3d = torch.randn(32, 32, 3)
+    other_tensor = torch.randn(3, 3, 3, 3)  # 4D, should be unchanged
+
+    observation = {
+        OBS_STATE: state_1d,
+        OBS_ENV_STATE: env_state_2d,
+        OBS_IMAGE: image_3d,
+        f"{OBS_IMAGES}.front": torch.randn(64, 64, 3),  # 3D, should be batched
+        f"{OBS_IMAGES}.back": torch.randn(1, 64, 64, 3),  # 4D, should be unchanged
+        "other_tensor": other_tensor,
+        "non_tensor": "string_value",
+    }
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check transformations
+    assert processed_obs[OBS_STATE].shape == (1, 5)
+    assert processed_obs[OBS_ENV_STATE].shape == (1, 8)  # Unchanged
+    assert processed_obs[OBS_IMAGE].shape == (1, 32, 32, 3)
+    assert processed_obs[f"{OBS_IMAGES}.front"].shape == (1, 64, 64, 3)
+    assert processed_obs[f"{OBS_IMAGES}.back"].shape == (1, 64, 64, 3)  # Unchanged
+    assert processed_obs["other_tensor"].shape == (3, 3, 3, 3)  # Unchanged
+    assert processed_obs["non_tensor"] == "string_value"  # Unchanged
+
+
+def test_integration_with_robot_processor():
+    """Test AddBatchDimensionProcessorStep integration with RobotProcessor."""
+    to_batch_processor = AddBatchDimensionProcessorStep()
+    pipeline = DataProcessorPipeline(
+        [to_batch_processor], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    # Create unbatched observation
+    observation = {
+        OBS_STATE: torch.randn(7),
+        OBS_IMAGE: torch.randn(224, 224, 3),
+    }
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = pipeline(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    assert processed_obs[OBS_STATE].shape == (1, 7)
+    assert processed_obs[OBS_IMAGE].shape == (1, 224, 224, 3)
+
+
+def test_serialization_methods():
+    """Test get_config, state_dict, load_state_dict, and reset methods."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Test get_config
+    config = processor.get_config()
+    assert isinstance(config, dict)
+    assert config == {}
+
+    # Test state_dict
+    state = processor.state_dict()
+    assert isinstance(state, dict)
+    assert state == {}
+
+    # Test load_state_dict (should not raise an error)
+    processor.load_state_dict({})
+
+    # Test reset (should not raise an error)
+    processor.reset()
+
+
+def test_save_and_load_pretrained():
+    """Test saving and loading AddBatchDimensionProcessorStep with RobotProcessor."""
+    processor = AddBatchDimensionProcessorStep()
+    pipeline = DataProcessorPipeline(
+        [processor], name="BatchPipeline", to_transition=identity_transition, to_output=identity_transition
+    )
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save pipeline
+        pipeline.save_pretrained(tmp_dir)
+
+        # Check config file exists
+        config_path = Path(tmp_dir) / "batchpipeline.json"
+        assert config_path.exists()
+
+        # Load pipeline
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="batchpipeline.json",
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        assert loaded_pipeline.name == "BatchPipeline"
+        assert len(loaded_pipeline) == 1
+        assert isinstance(loaded_pipeline.steps[0], AddBatchDimensionProcessorStep)
+
+        # Test functionality of loaded processor
+        observation = {OBS_STATE: torch.randn(5)}
+        transition = create_transition(observation=observation, action=torch.empty(0))
+
+        result = loaded_pipeline(transition)
+        assert result[TransitionKey.OBSERVATION][OBS_STATE].shape == (1, 5)
+
+
+def test_registry_functionality():
+    """Test that AddBatchDimensionProcessorStep is properly registered."""
+    # Check that the processor is registered
+    registered_class = ProcessorStepRegistry.get("to_batch_processor")
+    assert registered_class is AddBatchDimensionProcessorStep
+
+    # Check that it's in the list of registered processors
+    assert "to_batch_processor" in ProcessorStepRegistry.list()
+
+
+def test_registry_based_save_load():
+    """Test saving and loading using registry name."""
+    processor = AddBatchDimensionProcessorStep()
+    pipeline = DataProcessorPipeline(
+        [processor], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="dataprocessorpipeline.json",
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        # Verify the loaded processor works
+        observation = {
+            OBS_STATE: torch.randn(3),
+            OBS_IMAGE: torch.randn(100, 100, 3),
+        }
+        transition = create_transition(observation=observation, action=torch.empty(0))
+
+        result = loaded_pipeline(transition)
+        processed_obs = result[TransitionKey.OBSERVATION]
+
+        assert processed_obs[OBS_STATE].shape == (1, 3)
+        assert processed_obs[OBS_IMAGE].shape == (1, 100, 100, 3)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_device_compatibility():
+    """Test processor works with tensors on different devices."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create tensors on GPU
+    state_1d = torch.randn(7, device="cuda")
+    image_3d = torch.randn(64, 64, 3, device="cuda")
+
+    observation = {
+        OBS_STATE: state_1d,
+        OBS_IMAGE: image_3d,
+    }
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check shapes and that tensors stayed on GPU
+    assert processed_obs[OBS_STATE].shape == (1, 7)
+    assert processed_obs[OBS_IMAGE].shape == (1, 64, 64, 3)
+    assert processed_obs[OBS_STATE].device.type == "cuda"
+    assert processed_obs[OBS_IMAGE].device.type == "cuda"
+
+
+def test_processor_preserves_other_transition_keys():
+    """Test that processor only modifies observation and preserves other transition keys."""
+    processor = AddBatchDimensionProcessorStep()
+
+    action = torch.randn(5)
+    reward = 1.5
+    done = True
+    truncated = False
+    info = {"step": 10}
+    comp_data = {"extra": "data"}
+
+    observation = {OBS_STATE: torch.randn(7)}
+
+    transition = create_transition(
+        observation=observation,
+        action=action,
+        reward=reward,
+        done=done,
+        truncated=truncated,
+        info=info,
+        complementary_data=comp_data,
+    )
+
+    result = processor(transition)
+
+    # Check that non-observation keys are preserved
+    assert torch.allclose(result[TransitionKey.ACTION], action)
+    assert result[TransitionKey.REWARD] == reward
+    assert result[TransitionKey.DONE] == done
+    assert result[TransitionKey.TRUNCATED] == truncated
+    assert result[TransitionKey.INFO] == info
+    assert result[TransitionKey.COMPLEMENTARY_DATA] == comp_data
+
+    # Check that observation was processed
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].shape == (1, 7)
+
+
+def test_edge_case_zero_dimensional_tensors():
+    """Test processor handles 0D tensors (scalars) correctly."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # 0D tensors should not be modified
+    scalar_tensor = torch.tensor(42.0)
+
+    observation = {
+        OBS_STATE: scalar_tensor,
+        "scalar_value": scalar_tensor,
+    }
+    transition = create_transition(observation=observation, action=torch.empty(0))
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # 0D tensors should remain unchanged
+    assert torch.allclose(processed_obs[OBS_STATE], scalar_tensor)
+    assert torch.allclose(processed_obs["scalar_value"], scalar_tensor)
+
+
+# Action-specific tests
+def test_action_1d_to_2d():
+    """Test that 1D action tensors get batch dimension added."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create 1D action tensor
+    action_1d = torch.randn(4)
+    transition = create_transition(observation={}, action=action_1d)
+
+    result = processor(transition)
+
+    # Should add batch dimension
+    assert result[TransitionKey.ACTION].shape == (1, 4)
+    assert torch.equal(result[TransitionKey.ACTION][0], action_1d)
+
+
+def test_action_already_batched():
+    """Test that already batched action tensors remain unchanged."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Test various batch sizes
+    action_batched_1 = torch.randn(1, 4)
+    action_batched_5 = torch.randn(5, 4)
+
+    # Single batch
+    transition = create_transition(action=action_batched_1, observation={})
+    result = processor(transition)
+    assert torch.equal(result[TransitionKey.ACTION], action_batched_1)
+
+    # Multiple batch
+    transition = create_transition(action=action_batched_5, observation={})
+    result = processor(transition)
+    assert torch.equal(result[TransitionKey.ACTION], action_batched_5)
+
+
+def test_action_higher_dimensional():
+    """Test that higher dimensional action tensors remain unchanged."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # 3D action tensor (e.g., sequence of actions)
+    action_3d = torch.randn(2, 4, 3)
+    transition = create_transition(action=action_3d, observation={})
+    result = processor(transition)
+    assert torch.equal(result[TransitionKey.ACTION], action_3d)
+
+    # 4D action tensor
+    action_4d = torch.randn(2, 10, 4, 3)
+    transition = create_transition(action=action_4d, observation={})
+    result = processor(transition)
+    assert torch.equal(result[TransitionKey.ACTION], action_4d)
+
+
+def test_action_scalar_tensor():
+    """Test that scalar (0D) action tensors remain unchanged."""
+    processor = AddBatchDimensionProcessorStep()
+
+    action_scalar = torch.tensor(1.5)
+    transition = create_transition(action=action_scalar, observation={})
+    result = processor(transition)
+
+    # Should remain scalar
+    assert result[TransitionKey.ACTION].dim() == 0
+    assert torch.equal(result[TransitionKey.ACTION], action_scalar)
+
+
+def test_action_non_tensor_raises_error():
+    """Test that non-tensor actions raise ValueError for PolicyAction processors."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # List action should raise error
+    action_list = [0.1, 0.2, 0.3, 0.4]
+    transition = create_transition(action=action_list)
+    with pytest.raises(ValueError, match="Action should be a PolicyAction type"):
+        processor(transition)
+
+    # Numpy array action should raise error
+    action_numpy = np.array([1, 2, 3, 4])
+    transition = create_transition(action=action_numpy)
+    with pytest.raises(ValueError, match="Action should be a PolicyAction type"):
+        processor(transition)
+
+    # String action should raise error
+    action_string = "forward"
+    transition = create_transition(action=action_string)
+    with pytest.raises(ValueError, match="Action should be a PolicyAction type"):
+        processor(transition)
+
+    # Dict action should raise error
+    action_dict = {"linear": [0.5, 0.0], "angular": 0.2}
+    transition = create_transition(action=action_dict)
+    with pytest.raises(ValueError, match="Action should be a PolicyAction type"):
+        processor(transition)
+
+
+def test_action_none():
+    """Test that empty action tensor is handled correctly."""
+    processor = AddBatchDimensionProcessorStep()
+
+    transition = create_transition(action=torch.empty(0), observation={})
+    result = processor(transition)
+    # Empty 1D tensor becomes empty 2D tensor with batch dimension
+    assert result[TransitionKey.ACTION].shape == (1, 0)
+
+
+def test_action_with_observation():
+    """Test action processing together with observation processing."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Both need batching
+    observation = {
+        OBS_STATE: torch.randn(7),
+        OBS_IMAGE: torch.randn(64, 64, 3),
+    }
+    action = torch.randn(4)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    # Both should be batched
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].shape == (1, 7)
+    assert result[TransitionKey.OBSERVATION][OBS_IMAGE].shape == (1, 64, 64, 3)
+    assert result[TransitionKey.ACTION].shape == (1, 4)
+
+
+def test_action_different_sizes():
+    """Test action processing with various action dimensions."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Different action sizes (robot with different DOF)
+    action_sizes = [1, 2, 4, 7, 10, 20]
+
+    for size in action_sizes:
+        action = torch.randn(size)
+        transition = create_transition(action=action, observation={})
+        result = processor(transition)
+
+        assert result[TransitionKey.ACTION].shape == (1, size)
+        assert torch.equal(result[TransitionKey.ACTION][0], action)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_action_device_compatibility():
+    """Test action processing on different devices."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # CUDA action
+    action_cuda = torch.randn(4, device="cuda")
+    transition = create_transition(action=action_cuda, observation={})
+    result = processor(transition)
+
+    assert result[TransitionKey.ACTION].shape == (1, 4)
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+
+    # CPU action
+    action_cpu = torch.randn(4, device="cpu")
+    transition = create_transition(action=action_cpu, observation={})
+    result = processor(transition)
+
+    assert result[TransitionKey.ACTION].shape == (1, 4)
+    assert result[TransitionKey.ACTION].device.type == "cpu"
+
+
+def test_action_dtype_preservation():
+    """Test that action dtype is preserved during processing."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Different dtypes
+    dtypes = [torch.float32, torch.float64, torch.int32, torch.int64]
+
+    for dtype in dtypes:
+        action = torch.randn(4).to(dtype)
+        transition = create_transition(action=action, observation={})
+        result = processor(transition)
+
+        assert result[TransitionKey.ACTION].dtype == dtype
+        assert result[TransitionKey.ACTION].shape == (1, 4)
+
+
+def test_empty_action_tensor():
+    """Test handling of empty action tensors."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Empty 1D tensor
+    action_empty = torch.tensor([])
+    transition = create_transition(action=action_empty, observation={})
+    result = processor(transition)
+
+    # Should add batch dimension even to empty tensor
+    assert result[TransitionKey.ACTION].shape == (1, 0)
+
+    # Empty 2D tensor (already batched)
+    action_empty_2d = torch.randn(1, 0)
+    transition = create_transition(action=action_empty_2d, observation={})
+    result = processor(transition)
+
+    # Should remain unchanged
+    assert result[TransitionKey.ACTION].shape == (1, 0)
+
+
+# Task-specific tests
+def test_task_string_to_list():
+    """Test that string tasks get wrapped in lists to add batch dimension."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create complementary data with string task
+    complementary_data = {"task": "pick_cube"}
+    transition = create_transition(
+        action=torch.empty(0), observation={}, complementary_data=complementary_data
+    )
+
+    result = processor(transition)
+
+    # String task should be wrapped in list
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["task"] == ["pick_cube"]
+    assert isinstance(processed_comp_data["task"], list)
+    assert len(processed_comp_data["task"]) == 1
+
+
+def test_task_string_validation():
+    """Test that only string and list of strings are valid task values."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Valid string task - should be converted to list
+    complementary_data = {"task": "valid_task"}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+    result = processor(transition)
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["task"] == ["valid_task"]
+
+    # Valid list of strings - should remain unchanged
+    complementary_data = {"task": ["task1", "task2"]}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+    result = processor(transition)
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["task"] == ["task1", "task2"]
+
+
+def test_task_list_of_strings():
+    """Test that lists of strings remain unchanged (already batched)."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Test various list of strings
+    test_lists = [
+        ["pick_cube"],  # Single string in list
+        ["pick_cube", "place_cube"],  # Multiple strings
+        ["task1", "task2", "task3"],  # Three strings
+        [],  # Empty list
+        [""],  # List with empty string
+        ["task with spaces", "task_with_underscores"],  # Mixed formats
+    ]
+
+    for task_list in test_lists:
+        complementary_data = {"task": task_list}
+        transition = create_transition(
+            complementary_data=complementary_data, observation={}, action=torch.empty(0)
+        )
+
+        result = processor(transition)
+
+        # Should remain unchanged since it's already a list
+        processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+        assert processed_comp_data["task"] == task_list
+        assert isinstance(processed_comp_data["task"], list)
+
+
+def test_complementary_data_none():
+    """Test processor handles None complementary_data gracefully."""
+    processor = AddBatchDimensionProcessorStep()
+
+    transition = create_transition(complementary_data=None, action=torch.empty(0), observation={})
+    result = processor(transition)
+
+    assert result[TransitionKey.COMPLEMENTARY_DATA] == {}
+
+
+def test_complementary_data_empty():
+    """Test processor handles empty complementary_data dict."""
+    processor = AddBatchDimensionProcessorStep()
+
+    complementary_data = {}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+
+    assert result[TransitionKey.COMPLEMENTARY_DATA] == {}
+
+
+def test_complementary_data_no_task():
+    """Test processor handles complementary_data without task field."""
+    processor = AddBatchDimensionProcessorStep()
+
+    complementary_data = {
+        "episode_id": 123,
+        "timestamp": 1234567890.0,
+        "extra_info": "some data",
+    }
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+
+    # Should remain unchanged
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data == complementary_data
+
+
+def test_complementary_data_mixed():
+    """Test processor with mixed complementary_data containing task and other fields."""
+    processor = AddBatchDimensionProcessorStep()
+
+    complementary_data = {
+        "task": "stack_blocks",
+        "episode_id": 456,
+        "difficulty": "hard",
+        "metadata": {"scene": "kitchen"},
+    }
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Task should be batched
+    assert processed_comp_data["task"] == ["stack_blocks"]
+
+    # Other fields should remain unchanged
+    assert processed_comp_data["episode_id"] == 456
+    assert processed_comp_data["difficulty"] == "hard"
+    assert processed_comp_data["metadata"] == {"scene": "kitchen"}
+
+
+def test_task_with_observation_and_action():
+    """Test task processing together with observation and action processing."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # All components need batching
+    observation = {
+        OBS_STATE: torch.randn(5),
+        OBS_IMAGE: torch.randn(32, 32, 3),
+    }
+    action = torch.randn(4)
+    complementary_data = {"task": "navigate_to_goal"}
+
+    transition = create_transition(
+        observation=observation, action=action, complementary_data=complementary_data
+    )
+
+    result = processor(transition)
+
+    # All should be batched
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].shape == (1, 5)
+    assert result[TransitionKey.OBSERVATION][OBS_IMAGE].shape == (1, 32, 32, 3)
+    assert result[TransitionKey.ACTION].shape == (1, 4)
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["task"] == ["navigate_to_goal"]
+
+
+def test_task_comprehensive_string_cases():
+    """Test task processing with comprehensive string cases and edge cases."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Test various string formats
+    string_tasks = [
+        "pick_and_place",
+        "navigate",
+        "open_drawer",
+        "",  # Empty string (valid but edge case)
+        "task with spaces",
+        "task_with_underscores",
+        "task-with-dashes",
+        "UPPERCASE_TASK",
+        "MixedCaseTask",
+        "task123",
+        "数字任务",  # Unicode task
+        "🤖 robot task",  # Emoji in task
+        "task\nwith\nnewlines",  # Special characters
+        "task\twith\ttabs",
+        "task with 'quotes'",
+        'task with "double quotes"',
+    ]
+
+    # Test that all string tasks get properly batched
+    for task in string_tasks:
+        complementary_data = {"task": task}
+        transition = create_transition(
+            complementary_data=complementary_data, observation={}, action=torch.empty(0)
+        )
+
+        result = processor(transition)
+
+        processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+        assert processed_comp_data["task"] == [task]
+        assert isinstance(processed_comp_data["task"], list)
+        assert len(processed_comp_data["task"]) == 1
+
+    # Test various list of strings (should remain unchanged)
+    list_tasks = [
+        ["single_task"],
+        ["task1", "task2"],
+        ["pick", "place", "navigate"],
+        [],  # Empty list
+        [""],  # List with empty string
+        ["task with spaces", "task_with_underscores", "UPPERCASE"],
+        ["🤖 task", "数字任务", "normal_task"],  # Mixed formats
+    ]
+
+    for task_list in list_tasks:
+        complementary_data = {"task": task_list}
+        transition = create_transition(
+            complementary_data=complementary_data, observation={}, action=torch.empty(0)
+        )
+
+        result = processor(transition)
+
+        processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+        assert processed_comp_data["task"] == task_list
+        assert isinstance(processed_comp_data["task"], list)
+
+
+def test_task_preserves_other_keys():
+    """Test that task processing preserves other keys in complementary_data."""
+    processor = AddBatchDimensionProcessorStep()
+
+    complementary_data = {
+        "task": "clean_table",
+        "robot_id": "robot_123",
+        "motor_id": "motor_456",
+        "config": {"speed": "slow", "precision": "high"},
+        "metrics": [1.0, 2.0, 3.0],
+    }
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Task should be processed
+    assert processed_comp_data["task"] == ["clean_table"]
+
+    # All other keys should be preserved exactly
+    assert processed_comp_data["robot_id"] == "robot_123"
+    assert processed_comp_data["motor_id"] == "motor_456"
+    assert processed_comp_data["config"] == {"speed": "slow", "precision": "high"}
+    assert processed_comp_data["metrics"] == [1.0, 2.0, 3.0]
+
+
+# Index and task_index specific tests
+def test_index_scalar_to_1d():
+    """Test that 0D index tensor gets unsqueezed to 1D."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create 0D index tensor (scalar)
+    index_0d = torch.tensor(42, dtype=torch.int64)
+    complementary_data = {"index": index_0d}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["index"].shape == (1,)
+    assert processed_comp_data["index"].dtype == torch.int64
+    assert processed_comp_data["index"][0] == 42
+
+
+def test_task_index_scalar_to_1d():
+    """Test that 0D task_index tensor gets unsqueezed to 1D."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create 0D task_index tensor (scalar)
+    task_index_0d = torch.tensor(7, dtype=torch.int64)
+    complementary_data = {"task_index": task_index_0d}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["task_index"].shape == (1,)
+    assert processed_comp_data["task_index"].dtype == torch.int64
+    assert processed_comp_data["task_index"][0] == 7
+
+
+def test_index_and_task_index_together():
+    """Test processing both index and task_index together."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create 0D tensors for both
+    index_0d = torch.tensor(100, dtype=torch.int64)
+    task_index_0d = torch.tensor(3, dtype=torch.int64)
+    complementary_data = {
+        "index": index_0d,
+        "task_index": task_index_0d,
+        "task": "pick_object",
+    }
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Check index
+    assert processed_comp_data["index"].shape == (1,)
+    assert processed_comp_data["index"][0] == 100
+
+    # Check task_index
+    assert processed_comp_data["task_index"].shape == (1,)
+    assert processed_comp_data["task_index"][0] == 3
+
+    # Check task is also processed
+    assert processed_comp_data["task"] == ["pick_object"]
+
+
+def test_index_already_batched():
+    """Test that already batched index tensors remain unchanged."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create already batched tensors
+    index_1d = torch.tensor([42], dtype=torch.int64)
+    index_2d = torch.tensor([[42, 43]], dtype=torch.int64)
+
+    # Test 1D (already batched)
+    complementary_data = {"index": index_1d}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+    result = processor(transition)
+    assert torch.equal(result[TransitionKey.COMPLEMENTARY_DATA]["index"], index_1d)
+
+    # Test 2D
+    complementary_data = {"index": index_2d}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+    result = processor(transition)
+    assert torch.equal(result[TransitionKey.COMPLEMENTARY_DATA]["index"], index_2d)
+
+
+def test_task_index_already_batched():
+    """Test that already batched task_index tensors remain unchanged."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create already batched tensors
+    task_index_1d = torch.tensor([7], dtype=torch.int64)
+    task_index_2d = torch.tensor([[7, 8]], dtype=torch.int64)
+
+    # Test 1D (already batched)
+    complementary_data = {"task_index": task_index_1d}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+    result = processor(transition)
+    assert torch.equal(result[TransitionKey.COMPLEMENTARY_DATA]["task_index"], task_index_1d)
+
+    # Test 2D
+    complementary_data = {"task_index": task_index_2d}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+    result = processor(transition)
+    assert torch.equal(result[TransitionKey.COMPLEMENTARY_DATA]["task_index"], task_index_2d)
+
+
+def test_index_non_tensor_unchanged():
+    """Test that non-tensor index values remain unchanged."""
+    processor = AddBatchDimensionProcessorStep()
+
+    complementary_data = {
+        "index": 42,  # Plain int, not tensor
+        "task_index": [1, 2, 3],  # List, not tensor
+    }
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["index"] == 42
+    assert processed_comp_data["task_index"] == [1, 2, 3]
+
+
+def test_index_dtype_preservation():
+    """Test that index and task_index dtype is preserved during processing."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Test different dtypes
+    dtypes = [torch.int32, torch.int64, torch.long]
+
+    for dtype in dtypes:
+        index_0d = torch.tensor(42, dtype=dtype)
+        task_index_0d = torch.tensor(7, dtype=dtype)
+        complementary_data = {
+            "index": index_0d,
+            "task_index": task_index_0d,
+        }
+        transition = create_transition(
+            complementary_data=complementary_data, observation={}, action=torch.empty(0)
+        )
+
+        result = processor(transition)
+
+        processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+        assert processed_comp_data["index"].dtype == dtype
+        assert processed_comp_data["task_index"].dtype == dtype
+
+
+def test_index_with_full_transition():
+    """Test index/task_index processing with full transition data."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create full transition with all components
+    observation = {
+        OBS_STATE: torch.randn(7),
+        OBS_IMAGE: torch.randn(64, 64, 3),
+    }
+    action = torch.randn(4)
+    complementary_data = {
+        "task": "navigate_to_goal",
+        "index": torch.tensor(1000, dtype=torch.int64),
+        "task_index": torch.tensor(5, dtype=torch.int64),
+        "episode_id": 123,
+    }
+
+    transition = create_transition(
+        observation=observation,
+        action=action,
+        reward=0.5,
+        done=False,
+        complementary_data=complementary_data,
+    )
+
+    result = processor(transition)
+
+    # Check all components are processed correctly
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].shape == (1, 7)
+    assert result[TransitionKey.OBSERVATION][OBS_IMAGE].shape == (1, 64, 64, 3)
+    assert result[TransitionKey.ACTION].shape == (1, 4)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["task"] == ["navigate_to_goal"]
+    assert processed_comp_data["index"].shape == (1,)
+    assert processed_comp_data["index"][0] == 1000
+    assert processed_comp_data["task_index"].shape == (1,)
+    assert processed_comp_data["task_index"][0] == 5
+    assert processed_comp_data["episode_id"] == 123  # Non-tensor field unchanged
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_index_device_compatibility():
+    """Test processor works with index/task_index tensors on different devices."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Create tensors on GPU
+    index_0d = torch.tensor(42, dtype=torch.int64, device="cuda")
+    task_index_0d = torch.tensor(7, dtype=torch.int64, device="cuda")
+
+    complementary_data = {
+        "index": index_0d,
+        "task_index": task_index_0d,
+    }
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Check shapes and that tensors stayed on GPU
+    assert processed_comp_data["index"].shape == (1,)
+    assert processed_comp_data["task_index"].shape == (1,)
+    assert processed_comp_data["index"].device.type == "cuda"
+    assert processed_comp_data["task_index"].device.type == "cuda"
+
+
+def test_empty_index_tensor():
+    """Test handling of empty index tensors."""
+    processor = AddBatchDimensionProcessorStep()
+
+    # Empty 0D tensor doesn't make sense, but test empty 1D
+    index_empty = torch.tensor([], dtype=torch.int64)
+    complementary_data = {"index": index_empty}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    result = processor(transition)
+
+    # Should remain unchanged (already 1D)
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["index"].shape == (0,)
+
+
+def test_action_processing_creates_new_transition():
+    """Test that the processor creates a new transition object with correctly processed action."""
+    processor = AddBatchDimensionProcessorStep()
+
+    action = torch.randn(4)
+    transition = create_transition(action=action, observation={})
+
+    # Store reference to original transition
+    original_transition = transition
+
+    # Process
+    result = processor(transition)
+
+    # Should be a different object (functional design, not in-place mutation)
+    assert result is not original_transition
+    # Original transition should remain unchanged
+    assert original_transition[TransitionKey.ACTION].shape == (4,)
+    # Result should have correctly processed action with batch dimension
+    assert result[TransitionKey.ACTION].shape == (1, 4)
+    assert torch.equal(result[TransitionKey.ACTION][0], action)
+
+
+def test_task_processing_creates_new_transition():
+    """Test that the processor creates a new transition object with correctly processed task."""
+    processor = AddBatchDimensionProcessorStep()
+
+    complementary_data = {"task": "sort_objects"}
+    transition = create_transition(
+        complementary_data=complementary_data, observation={}, action=torch.empty(0)
+    )
+
+    # Store reference to original transition and complementary_data
+    original_transition = transition
+    original_comp_data = complementary_data
+
+    # Process
+    result = processor(transition)
+
+    # Should be different transition object (functional design)
+    assert result is not original_transition
+    # The task should be processed correctly (wrapped in list)
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["task"] == ["sort_objects"]
+    # Original complementary data is also modified (current behavior)
+    assert original_comp_data["task"] == "sort_objects"
diff --git a/lerobot/tests/processor/test_classifier_processor.py b/lerobot/tests/processor/test_classifier_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..e1567bf29d06b07e622c97f0b805ff0d955c60be
--- /dev/null
+++ b/lerobot/tests/processor/test_classifier_processor.py
@@ -0,0 +1,362 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tests for Reward Classifier processor."""
+
+import tempfile
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.policies.sac.reward_model.configuration_classifier import RewardClassifierConfig
+from lerobot.policies.sac.reward_model.processor_classifier import make_classifier_processor
+from lerobot.processor import (
+    DataProcessorPipeline,
+    DeviceProcessorStep,
+    IdentityProcessorStep,
+    NormalizerProcessorStep,
+    TransitionKey,
+)
+from lerobot.processor.converters import create_transition, transition_to_batch
+from lerobot.utils.constants import OBS_IMAGE, OBS_STATE
+
+
+def create_default_config():
+    """Create a default Reward Classifier configuration for testing."""
+    config = RewardClassifierConfig()
+    config.input_features = {
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(10,)),
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        "reward": PolicyFeature(type=FeatureType.ACTION, shape=(1,)),  # Classifier output
+    }
+    config.normalization_mapping = {
+        FeatureType.STATE: NormalizationMode.MEAN_STD,
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,
+        FeatureType.ACTION: NormalizationMode.IDENTITY,  # No normalization for classifier output
+    }
+    config.device = "cpu"
+    return config
+
+
+def create_default_stats():
+    """Create default dataset statistics for testing."""
+    return {
+        OBS_STATE: {"mean": torch.zeros(10), "std": torch.ones(10)},
+        OBS_IMAGE: {},  # No normalization for images
+        "reward": {},  # No normalization for classifier output
+    }
+
+
+def test_make_classifier_processor_basic():
+    """Test basic creation of Classifier processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_classifier_processor(config, stats)
+
+    # Check processor names
+    assert preprocessor.name == "classifier_preprocessor"
+    assert postprocessor.name == "classifier_postprocessor"
+
+    # Check steps in preprocessor
+    assert len(preprocessor.steps) == 3
+    assert isinstance(preprocessor.steps[0], NormalizerProcessorStep)  # For input features
+    assert isinstance(preprocessor.steps[1], NormalizerProcessorStep)  # For output features
+    assert isinstance(preprocessor.steps[2], DeviceProcessorStep)
+
+    # Check steps in postprocessor
+    assert len(postprocessor.steps) == 2
+    assert isinstance(postprocessor.steps[0], DeviceProcessorStep)
+    assert isinstance(postprocessor.steps[1], IdentityProcessorStep)
+
+
+def test_classifier_processor_normalization():
+    """Test that Classifier processor correctly normalizes data."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_classifier_processor(
+        config,
+        stats,
+    )
+
+    # Create test data
+    observation = {
+        OBS_STATE: torch.randn(10),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(1)  # Dummy action/reward
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data is processed
+    assert processed[OBS_STATE].shape == (10,)
+    assert processed[OBS_IMAGE].shape == (3, 224, 224)
+    assert processed[TransitionKey.ACTION.value].shape == (1,)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_classifier_processor_cuda():
+    """Test Classifier processor with CUDA device."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_classifier_processor(
+        config,
+        stats,
+    )
+
+    # Create CPU data
+    observation = {
+        OBS_STATE: torch.randn(10),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(1)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is on CUDA
+    assert processed[OBS_STATE].device.type == "cuda"
+    assert processed[OBS_IMAGE].device.type == "cuda"
+    assert processed[TransitionKey.ACTION.value].device.type == "cuda"
+
+    # Process through postprocessor
+    postprocessed = postprocessor(processed[TransitionKey.ACTION.value])
+
+    # Check that output is back on CPU
+    assert postprocessed.device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_classifier_processor_accelerate_scenario():
+    """Test Classifier processor in simulated Accelerate scenario."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_classifier_processor(
+        config,
+        stats,
+    )
+
+    # Simulate Accelerate: data already on GPU
+    device = torch.device("cuda:0")
+    observation = {
+        OBS_STATE: torch.randn(10).to(device),
+        OBS_IMAGE: torch.randn(3, 224, 224).to(device),
+    }
+    action = torch.randn(1).to(device)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on same GPU
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_classifier_processor_multi_gpu():
+    """Test Classifier processor with multi-GPU setup."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_classifier_processor(config, stats)
+
+    # Simulate data on different GPU
+    device = torch.device("cuda:1")
+    observation = {
+        OBS_STATE: torch.randn(10).to(device),
+        OBS_IMAGE: torch.randn(3, 224, 224).to(device),
+    }
+    action = torch.randn(1).to(device)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on cuda:1
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+def test_classifier_processor_without_stats():
+    """Test Classifier processor creation without dataset statistics."""
+    config = create_default_config()
+
+    preprocessor, postprocessor = make_classifier_processor(config, dataset_stats=None)
+
+    # Should still create processors
+    assert preprocessor is not None
+    assert postprocessor is not None
+
+    # Process should still work
+    observation = {
+        OBS_STATE: torch.randn(10),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(1)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed is not None
+
+
+def test_classifier_processor_save_and_load():
+    """Test saving and loading Classifier processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_classifier_processor(config, stats)
+
+    with tempfile.TemporaryDirectory() as tmpdir:
+        # Save preprocessor
+        preprocessor.save_pretrained(tmpdir)
+
+        # Load preprocessor
+        loaded_preprocessor = DataProcessorPipeline.from_pretrained(
+            tmpdir, config_filename="classifier_preprocessor.json"
+        )
+
+        # Test that loaded processor works
+        observation = {
+            OBS_STATE: torch.randn(10),
+            OBS_IMAGE: torch.randn(3, 224, 224),
+        }
+        action = torch.randn(1)
+        transition = create_transition(observation, action)
+        batch = transition_to_batch(transition)
+
+        processed = loaded_preprocessor(batch)
+        assert processed[OBS_STATE].shape == (10,)
+        assert processed[OBS_IMAGE].shape == (3, 224, 224)
+        assert processed[TransitionKey.ACTION.value].shape == (1,)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_classifier_processor_mixed_precision():
+    """Test Classifier processor with mixed precision."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_classifier_processor(config, stats)
+
+    # Replace DeviceProcessorStep with one that uses float16
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="float16"))
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Create test data
+    observation = {
+        OBS_STATE: torch.randn(10, dtype=torch.float32),
+        OBS_IMAGE: torch.randn(3, 224, 224, dtype=torch.float32),
+    }
+    action = torch.randn(1, dtype=torch.float32)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is converted to float16
+    assert processed[OBS_STATE].dtype == torch.float16
+    assert processed[OBS_IMAGE].dtype == torch.float16
+    assert processed[TransitionKey.ACTION.value].dtype == torch.float16
+
+
+def test_classifier_processor_batch_data():
+    """Test Classifier processor with batched data."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_classifier_processor(
+        config,
+        stats,
+    )
+
+    # Test with batched data
+    batch_size = 16
+    observation = {
+        OBS_STATE: torch.randn(batch_size, 10),
+        OBS_IMAGE: torch.randn(batch_size, 3, 224, 224),
+    }
+    action = torch.randn(batch_size, 1)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that batch dimension is preserved
+    assert processed[OBS_STATE].shape == (batch_size, 10)
+    assert processed[OBS_IMAGE].shape == (batch_size, 3, 224, 224)
+    assert processed[TransitionKey.ACTION.value].shape == (batch_size, 1)
+
+
+def test_classifier_processor_postprocessor_identity():
+    """Test that Classifier postprocessor uses IdentityProcessor correctly."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_classifier_processor(
+        config,
+        stats,
+    )
+
+    # Create test data for postprocessor
+    reward = torch.tensor([[0.8], [0.3], [0.9]])  # Batch of rewards/predictions
+    transition = create_transition(action=reward)
+
+    _ = transition_to_batch(transition)
+
+    # Process through postprocessor
+    processed = postprocessor(reward)
+
+    # IdentityProcessor should leave values unchanged (except device)
+    assert torch.allclose(processed.cpu(), reward.cpu())
+    assert processed.device.type == "cpu"
diff --git a/lerobot/tests/processor/test_converters.py b/lerobot/tests/processor/test_converters.py
new file mode 100644
index 0000000000000000000000000000000000000000..91afdd0e524fca8bd7d10983e6e23c79c556fdce
--- /dev/null
+++ b/lerobot/tests/processor/test_converters.py
@@ -0,0 +1,309 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import numpy as np
+import pytest
+import torch
+
+from lerobot.processor.converters import (
+    batch_to_transition,
+    create_transition,
+    to_tensor,
+    transition_to_batch,
+)
+from lerobot.types import TransitionKey
+from lerobot.utils.constants import ACTION, DONE, OBS_STATE, OBS_STR, REWARD
+
+
+# Tests for the unified to_tensor function
+def test_to_tensor_numpy_arrays():
+    """Test to_tensor with various numpy arrays."""
+    # Regular numpy array
+    arr = np.array([1.0, 2.0, 3.0])
+    result = to_tensor(arr)
+    assert isinstance(result, torch.Tensor)
+    assert result.dtype == torch.float32
+    assert torch.allclose(result, torch.tensor([1.0, 2.0, 3.0]))
+
+    # Different numpy dtypes should convert to float32 by default
+    int_arr = np.array([1, 2, 3], dtype=np.int64)
+    result = to_tensor(int_arr)
+    assert isinstance(result, torch.Tensor)
+    assert result.dtype == torch.float32
+    assert torch.allclose(result, torch.tensor([1.0, 2.0, 3.0]))
+
+    # uint8 arrays (previously "preserved") should now convert
+    uint8_arr = np.array([100, 150, 200], dtype=np.uint8)
+    result = to_tensor(uint8_arr)
+    assert isinstance(result, torch.Tensor)
+    assert result.dtype == torch.float32
+    assert torch.allclose(result, torch.tensor([100.0, 150.0, 200.0]))
+
+
+def test_to_tensor_numpy_scalars():
+    """Test to_tensor with numpy scalars (0-dimensional arrays)."""
+    # numpy float32 scalar
+    scalar = np.float32(3.14)
+    result = to_tensor(scalar)
+    assert isinstance(result, torch.Tensor)
+    assert result.ndim == 0  # Should be 0-dimensional tensor
+    assert result.dtype == torch.float32
+    assert result.item() == pytest.approx(3.14)
+
+    # numpy int32 scalar
+    int_scalar = np.int32(42)
+    result = to_tensor(int_scalar)
+    assert isinstance(result, torch.Tensor)
+    assert result.ndim == 0
+    assert result.dtype == torch.float32
+    assert result.item() == pytest.approx(42.0)
+
+
+def test_to_tensor_python_scalars():
+    """Test to_tensor with Python scalars."""
+    # Python int
+    result = to_tensor(42)
+    assert isinstance(result, torch.Tensor)
+    assert result.dtype == torch.float32
+    assert result.item() == pytest.approx(42.0)
+
+    # Python float
+    result = to_tensor(3.14)
+    assert isinstance(result, torch.Tensor)
+    assert result.dtype == torch.float32
+    assert result.item() == pytest.approx(3.14)
+
+
+def test_to_tensor_sequences():
+    """Test to_tensor with lists and tuples."""
+    # List
+    result = to_tensor([1, 2, 3])
+    assert isinstance(result, torch.Tensor)
+    assert result.dtype == torch.float32
+    assert torch.allclose(result, torch.tensor([1.0, 2.0, 3.0]))
+
+    # Tuple
+    result = to_tensor((4.5, 5.5, 6.5))
+    assert isinstance(result, torch.Tensor)
+    assert result.dtype == torch.float32
+    assert torch.allclose(result, torch.tensor([4.5, 5.5, 6.5]))
+
+
+def test_to_tensor_existing_tensors():
+    """Test to_tensor with existing PyTorch tensors."""
+    # Tensor with same dtype should pass through with potential device change
+    tensor = torch.tensor([1.0, 2.0, 3.0], dtype=torch.float32)
+    result = to_tensor(tensor)
+    assert isinstance(result, torch.Tensor)
+    assert result.dtype == torch.float32
+    assert torch.allclose(result, tensor)
+
+    # Tensor with different dtype should convert
+    int_tensor = torch.tensor([1, 2, 3], dtype=torch.int64)
+    result = to_tensor(int_tensor)
+    assert isinstance(result, torch.Tensor)
+    assert result.dtype == torch.float32
+    assert torch.allclose(result, torch.tensor([1.0, 2.0, 3.0]))
+
+
+def test_to_tensor_dictionaries():
+    """Test to_tensor with nested dictionaries."""
+    # Simple dictionary
+    data = {"mean": [0.1, 0.2], "std": np.array([1.0, 2.0]), "count": 42}
+    result = to_tensor(data)
+    assert isinstance(result, dict)
+    assert isinstance(result["mean"], torch.Tensor)
+    assert isinstance(result["std"], torch.Tensor)
+    assert isinstance(result["count"], torch.Tensor)
+    assert torch.allclose(result["mean"], torch.tensor([0.1, 0.2]))
+    assert torch.allclose(result["std"], torch.tensor([1.0, 2.0]))
+    assert result["count"].item() == pytest.approx(42.0)
+
+    # Nested dictionary
+    nested = {
+        ACTION: {"mean": [0.1, 0.2], "std": [1.0, 2.0]},
+        OBS_STR: {"mean": np.array([0.5, 0.6]), "count": 10},
+    }
+    result = to_tensor(nested)
+    assert isinstance(result, dict)
+    assert isinstance(result[ACTION], dict)
+    assert isinstance(result[OBS_STR], dict)
+    assert isinstance(result[ACTION]["mean"], torch.Tensor)
+    assert isinstance(result[OBS_STR]["mean"], torch.Tensor)
+    assert torch.allclose(result[ACTION]["mean"], torch.tensor([0.1, 0.2]))
+    assert torch.allclose(result[OBS_STR]["mean"], torch.tensor([0.5, 0.6]))
+
+
+def test_to_tensor_none_filtering():
+    """Test that None values are filtered out from dictionaries."""
+    data = {"valid": [1, 2, 3], "none_value": None, "nested": {"valid": [4, 5], "also_none": None}}
+    result = to_tensor(data)
+    assert "none_value" not in result
+    assert "also_none" not in result["nested"]
+    assert "valid" in result
+    assert "valid" in result["nested"]
+    assert torch.allclose(result["valid"], torch.tensor([1.0, 2.0, 3.0]))
+
+
+def test_to_tensor_dtype_parameter():
+    """Test to_tensor with different dtype parameters."""
+    arr = np.array([1, 2, 3])
+
+    # Default dtype (float32)
+    result = to_tensor(arr)
+    assert result.dtype == torch.float32
+
+    # Explicit float32
+    result = to_tensor(arr, dtype=torch.float32)
+    assert result.dtype == torch.float32
+
+    # Float64
+    result = to_tensor(arr, dtype=torch.float64)
+    assert result.dtype == torch.float64
+
+    # Preserve original dtype
+    float64_arr = np.array([1.0, 2.0, 3.0], dtype=np.float64)
+    result = to_tensor(float64_arr, dtype=None)
+    assert result.dtype == torch.float64
+
+
+def test_to_tensor_device_parameter():
+    """Test to_tensor with device parameter."""
+    arr = np.array([1.0, 2.0, 3.0])
+
+    # CPU device (default)
+    result = to_tensor(arr, device="cpu")
+    assert result.device.type == "cpu"
+
+    # CUDA device (if available)
+    if torch.cuda.is_available():
+        result = to_tensor(arr, device="cuda")
+        assert result.device.type == "cuda"
+
+
+def test_to_tensor_empty_dict():
+    """Test to_tensor with empty dictionary."""
+    result = to_tensor({})
+    assert isinstance(result, dict)
+    assert len(result) == 0
+
+
+def test_to_tensor_unsupported_type():
+    """Test to_tensor with unsupported types raises TypeError."""
+    with pytest.raises(TypeError, match="Unsupported type for tensor conversion"):
+        to_tensor("unsupported_string")
+
+    with pytest.raises(TypeError, match="Unsupported type for tensor conversion"):
+        to_tensor(object())
+
+
+def test_batch_to_transition_with_index_fields():
+    """Test that batch_to_transition handles index and task_index fields correctly."""
+
+    # Create batch with index and task_index fields
+    batch = {
+        OBS_STATE: torch.randn(1, 7),
+        ACTION: torch.randn(1, 4),
+        REWARD: 1.5,
+        DONE: False,
+        "task": ["pick_cube"],
+        "index": torch.tensor([42], dtype=torch.int64),
+        "task_index": torch.tensor([3], dtype=torch.int64),
+    }
+
+    transition = batch_to_transition(batch)
+
+    # Check basic transition structure
+    assert TransitionKey.OBSERVATION in transition
+    assert TransitionKey.ACTION in transition
+    assert TransitionKey.COMPLEMENTARY_DATA in transition
+
+    # Check that index and task_index are in complementary_data
+    comp_data = transition[TransitionKey.COMPLEMENTARY_DATA]
+    assert "index" in comp_data
+    assert "task_index" in comp_data
+    assert "task" in comp_data
+
+    # Verify values
+    assert torch.equal(comp_data["index"], batch["index"])
+    assert torch.equal(comp_data["task_index"], batch["task_index"])
+    assert comp_data["task"] == batch["task"]
+
+
+def testtransition_to_batch_with_index_fields():
+    """Test that transition_to_batch handles index and task_index fields correctly."""
+
+    # Create transition with index and task_index in complementary_data
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(1, 7)},
+        action=torch.randn(1, 4),
+        reward=1.5,
+        done=False,
+        complementary_data={
+            "task": ["navigate"],
+            "index": torch.tensor([100], dtype=torch.int64),
+            "task_index": torch.tensor([5], dtype=torch.int64),
+        },
+    )
+
+    batch = transition_to_batch(transition)
+
+    # Check that index and task_index are in the batch
+    assert "index" in batch
+    assert "task_index" in batch
+    assert "task" in batch
+
+    # Verify values
+    assert torch.equal(batch["index"], transition[TransitionKey.COMPLEMENTARY_DATA]["index"])
+    assert torch.equal(batch["task_index"], transition[TransitionKey.COMPLEMENTARY_DATA]["task_index"])
+    assert batch["task"] == transition[TransitionKey.COMPLEMENTARY_DATA]["task"]
+
+
+def test_batch_to_transition_without_index_fields():
+    """Test that conversion works without index and task_index fields."""
+
+    # Batch without index/task_index
+    batch = {
+        OBS_STATE: torch.randn(1, 7),
+        ACTION: torch.randn(1, 4),
+        "task": ["pick_cube"],
+    }
+
+    transition = batch_to_transition(batch)
+    comp_data = transition[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Should have task but not index/task_index
+    assert "task" in comp_data
+    assert "index" not in comp_data
+    assert "task_index" not in comp_data
+
+
+def test_transition_to_batch_without_index_fields():
+    """Test that conversion works without index and task_index fields."""
+
+    # Transition without index/task_index
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(1, 7)},
+        action=torch.randn(1, 4),
+        complementary_data={"task": ["navigate"]},
+    )
+
+    batch = transition_to_batch(transition)
+
+    # Should have task but not index/task_index
+    assert "task" in batch
+    assert "index" not in batch
+    assert "task_index" not in batch
diff --git a/lerobot/tests/processor/test_device_processor.py b/lerobot/tests/processor/test_device_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..57b923076fd28d84bff143e6702b34c83729ef60
--- /dev/null
+++ b/lerobot/tests/processor/test_device_processor.py
@@ -0,0 +1,1159 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import tempfile
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from lerobot.processor import DataProcessorPipeline, DeviceProcessorStep
+from lerobot.processor.converters import create_transition, identity_transition
+from lerobot.types import TransitionKey
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_STATE
+
+
+def test_basic_functionality():
+    """Test basic device processor functionality on CPU."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    # Create a transition with CPU tensors
+    observation = {OBS_STATE: torch.randn(10), OBS_IMAGE: torch.randn(3, 224, 224)}
+    action = torch.randn(5)
+    reward = torch.tensor(1.0)
+    done = torch.tensor(False)
+    truncated = torch.tensor(False)
+
+    transition = create_transition(
+        observation=observation, action=action, reward=reward, done=done, truncated=truncated
+    )
+
+    result = processor(transition)
+
+    # Check that all tensors are on CPU
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cpu"
+    assert result[TransitionKey.OBSERVATION][OBS_IMAGE].device.type == "cpu"
+    assert result[TransitionKey.ACTION].device.type == "cpu"
+    assert result[TransitionKey.REWARD].device.type == "cpu"
+    assert result[TransitionKey.DONE].device.type == "cpu"
+    assert result[TransitionKey.TRUNCATED].device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_cuda_functionality():
+    """Test device processor functionality on CUDA."""
+    processor = DeviceProcessorStep(device="cuda")
+
+    # Create a transition with CPU tensors
+    observation = {OBS_STATE: torch.randn(10), OBS_IMAGE: torch.randn(3, 224, 224)}
+    action = torch.randn(5)
+    reward = torch.tensor(1.0)
+    done = torch.tensor(False)
+    truncated = torch.tensor(False)
+
+    transition = create_transition(
+        observation=observation, action=action, reward=reward, done=done, truncated=truncated
+    )
+
+    result = processor(transition)
+
+    # Check that all tensors are on CUDA
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cuda"
+    assert result[TransitionKey.OBSERVATION][OBS_IMAGE].device.type == "cuda"
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+    assert result[TransitionKey.REWARD].device.type == "cuda"
+    assert result[TransitionKey.DONE].device.type == "cuda"
+    assert result[TransitionKey.TRUNCATED].device.type == "cuda"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_specific_cuda_device():
+    """Test device processor with specific CUDA device."""
+    processor = DeviceProcessorStep(device="cuda:0")
+
+    observation = {OBS_STATE: torch.randn(10)}
+    action = torch.randn(5)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cuda"
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.index == 0
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+    assert result[TransitionKey.ACTION].device.index == 0
+
+
+def test_non_tensor_values():
+    """Test that non-tensor values are preserved."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    observation = {
+        OBS_STATE: torch.randn(10),
+        "observation.metadata": {"key": "value"},  # Non-tensor data
+        "observation.list": [1, 2, 3],  # Non-tensor data
+    }
+    action = torch.randn(5)
+    info = {"episode": 1, "step": 42}
+
+    transition = create_transition(observation=observation, action=action, info=info)
+
+    result = processor(transition)
+
+    # Check tensors are processed
+    assert isinstance(result[TransitionKey.OBSERVATION][OBS_STATE], torch.Tensor)
+    assert isinstance(result[TransitionKey.ACTION], torch.Tensor)
+
+    # Check non-tensor values are preserved
+    assert result[TransitionKey.OBSERVATION]["observation.metadata"] == {"key": "value"}
+    assert result[TransitionKey.OBSERVATION]["observation.list"] == [1, 2, 3]
+    assert result[TransitionKey.INFO] == {"episode": 1, "step": 42}
+
+
+def test_none_values():
+    """Test handling of None values."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    # Test with None observation
+    transition = create_transition(observation=None, action=torch.randn(5))
+    result = processor(transition)
+    assert result[TransitionKey.OBSERVATION] is None
+    assert result[TransitionKey.ACTION].device.type == "cpu"
+
+    # Test with None action
+    transition = create_transition(observation={OBS_STATE: torch.randn(10)}, action=None)
+    result = processor(transition)
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cpu"
+    assert result[TransitionKey.ACTION] is None
+
+
+def test_empty_observation():
+    """Test handling of empty observation dictionary."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    transition = create_transition(observation={}, action=torch.randn(5))
+    result = processor(transition)
+
+    assert result[TransitionKey.OBSERVATION] == {}
+    assert result[TransitionKey.ACTION].device.type == "cpu"
+
+
+def test_scalar_tensors():
+    """Test handling of scalar tensors."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    observation = {"observation.scalar": torch.tensor(1.5)}
+    action = torch.tensor(2.0)
+    reward = torch.tensor(0.5)
+
+    transition = create_transition(observation=observation, action=action, reward=reward)
+
+    result = processor(transition)
+
+    assert result[TransitionKey.OBSERVATION]["observation.scalar"].item() == 1.5
+    assert result[TransitionKey.ACTION].item() == 2.0
+    assert result[TransitionKey.REWARD].item() == 0.5
+
+
+def test_dtype_preservation():
+    """Test that tensor dtypes are preserved."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    observation = {
+        "observation.float32": torch.randn(5, dtype=torch.float32),
+        "observation.float64": torch.randn(5, dtype=torch.float64),
+        "observation.int32": torch.randint(0, 10, (5,), dtype=torch.int32),
+        "observation.bool": torch.tensor([True, False, True], dtype=torch.bool),
+    }
+    action = torch.randn(3, dtype=torch.float16)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    assert result[TransitionKey.OBSERVATION]["observation.float32"].dtype == torch.float32
+    assert result[TransitionKey.OBSERVATION]["observation.float64"].dtype == torch.float64
+    assert result[TransitionKey.OBSERVATION]["observation.int32"].dtype == torch.int32
+    assert result[TransitionKey.OBSERVATION]["observation.bool"].dtype == torch.bool
+    assert result[TransitionKey.ACTION].dtype == torch.float16
+
+
+def test_shape_preservation():
+    """Test that tensor shapes are preserved."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    observation = {
+        "observation.1d": torch.randn(10),
+        "observation.2d": torch.randn(5, 10),
+        "observation.3d": torch.randn(3, 224, 224),
+        "observation.4d": torch.randn(2, 3, 224, 224),
+    }
+    action = torch.randn(2, 5, 3)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    assert result[TransitionKey.OBSERVATION]["observation.1d"].shape == (10,)
+    assert result[TransitionKey.OBSERVATION]["observation.2d"].shape == (5, 10)
+    assert result[TransitionKey.OBSERVATION]["observation.3d"].shape == (3, 224, 224)
+    assert result[TransitionKey.OBSERVATION]["observation.4d"].shape == (2, 3, 224, 224)
+    assert result[TransitionKey.ACTION].shape == (2, 5, 3)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_mixed_devices():
+    """Test handling of tensors already on different devices."""
+    processor = DeviceProcessorStep(device="cuda")
+
+    # Create tensors on different devices
+    observation = {
+        "observation.cpu": torch.randn(5),  # CPU
+        "observation.cuda": torch.randn(5).cuda(),  # Already on CUDA
+    }
+    action = torch.randn(3).cuda()  # Already on CUDA
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    # All should be on CUDA
+    assert result[TransitionKey.OBSERVATION]["observation.cpu"].device.type == "cuda"
+    assert result[TransitionKey.OBSERVATION]["observation.cuda"].device.type == "cuda"
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+
+
+def test_non_blocking_flag():
+    """Test that non_blocking flag is set correctly."""
+    # CPU processor should have non_blocking=False
+    cpu_processor = DeviceProcessorStep(device="cpu")
+    assert cpu_processor.non_blocking is False
+
+    if torch.cuda.is_available():
+        # CUDA processor should have non_blocking=True
+        cuda_processor = DeviceProcessorStep(device="cuda")
+        assert cuda_processor.non_blocking is True
+
+        cuda_0_processor = DeviceProcessorStep(device="cuda:0")
+        assert cuda_0_processor.non_blocking is True
+
+
+def test_serialization_methods():
+    """Test get_config, state_dict, and load_state_dict methods."""
+    device = "cuda" if torch.cuda.is_available() else "cpu"
+    processor = DeviceProcessorStep(device=device)
+
+    # Test get_config
+    config = processor.get_config()
+    assert config == {"device": device, "float_dtype": None}
+
+    # Test state_dict (should be empty)
+    state = processor.state_dict()
+    assert state == {}
+
+    # Test load_state_dict (should be no-op)
+    processor.load_state_dict({})
+    assert processor.device == device
+
+    # Test reset (should be no-op)
+    processor.reset()
+    assert processor.device == device
+
+
+def test_features():
+    """Test that features returns features unchanged."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    features = {
+        PipelineFeatureType.OBSERVATION: {OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(10,))},
+        PipelineFeatureType.ACTION: {ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(5,))},
+    }
+
+    result = processor.transform_features(features)
+    assert result == features
+    assert result is features  # Should return the same object
+
+
+def test_integration_with_robot_processor():
+    """Test integration with RobotProcessor."""
+    from lerobot.processor import AddBatchDimensionProcessorStep
+    from lerobot.utils.constants import OBS_STATE
+
+    # Create a pipeline with DeviceProcessorStep
+    device_processor = DeviceProcessorStep(device="cpu")
+    batch_processor = AddBatchDimensionProcessorStep()
+
+    processor = DataProcessorPipeline(
+        steps=[batch_processor, device_processor],
+        name="test_pipeline",
+        to_transition=identity_transition,
+        to_output=identity_transition,
+    )
+
+    # Create test data
+    observation = {OBS_STATE: torch.randn(10)}
+    action = torch.randn(5)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    # Check that tensors are batched and on correct device
+    # The result has TransitionKey.OBSERVATION as the key, with observation.state inside
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].shape[0] == 1  # Batched
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cpu"
+    assert result[TransitionKey.ACTION].shape[0] == 1  # Batched
+    assert result[TransitionKey.ACTION].device.type == "cpu"
+
+
+def test_save_and_load_pretrained():
+    """Test saving and loading processor with DeviceProcessorStep."""
+    device = "cuda:0" if torch.cuda.is_available() else "cpu"
+    processor = DeviceProcessorStep(device=device, float_dtype="float16")
+    robot_processor = DataProcessorPipeline(steps=[processor], name="device_test_processor")
+
+    with tempfile.TemporaryDirectory() as tmpdir:
+        # Save
+        robot_processor.save_pretrained(tmpdir)
+
+        # Load
+        loaded_processor = DataProcessorPipeline.from_pretrained(
+            tmpdir, config_filename="device_test_processor.json"
+        )
+
+        assert len(loaded_processor.steps) == 1
+        loaded_device_processor = loaded_processor.steps[0]
+        assert isinstance(loaded_device_processor, DeviceProcessorStep)
+        # Use getattr to access attributes safely
+        assert (
+            getattr(loaded_device_processor, "device", None) == device.split(":")[0]
+        )  # Device normalizes cuda:0 to cuda
+        assert getattr(loaded_device_processor, "float_dtype", None) == "float16"
+
+
+def test_registry_functionality():
+    """Test that DeviceProcessorStep is properly registered."""
+    from lerobot.processor import ProcessorStepRegistry
+
+    # Check that DeviceProcessorStep is registered
+    registered_class = ProcessorStepRegistry.get("device_processor")
+    assert registered_class is DeviceProcessorStep
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_performance_with_large_tensors():
+    """Test performance with large tensors and non_blocking flag."""
+    processor = DeviceProcessorStep(device="cuda")
+
+    # Create large tensors
+    observation = {
+        "observation.large_image": torch.randn(10, 3, 512, 512),  # Large image batch
+        "observation.features": torch.randn(10, 2048),  # Large feature vector
+    }
+    action = torch.randn(10, 100)  # Large action space
+
+    transition = create_transition(observation=observation, action=action)
+
+    # Process should not raise any errors
+    result = processor(transition)
+
+    # Verify all tensors are on CUDA
+    assert result[TransitionKey.OBSERVATION]["observation.large_image"].device.type == "cuda"
+    assert result[TransitionKey.OBSERVATION]["observation.features"].device.type == "cuda"
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+
+
+def test_reward_done_truncated_types():
+    """Test handling of different types for reward, done, and truncated."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    # Test with scalar values (not tensors)
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(5)},
+        action=torch.randn(3),
+        reward=1.0,  # float
+        done=False,  # bool
+        truncated=True,  # bool
+    )
+
+    result = processor(transition)
+
+    # Non-tensor values should be preserved as-is
+    assert result[TransitionKey.REWARD] == 1.0
+    assert result[TransitionKey.DONE] is False
+    assert result[TransitionKey.TRUNCATED] is True
+
+    # Test with tensor values
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(5)},
+        action=torch.randn(3),
+        reward=torch.tensor(1.0),
+        done=torch.tensor(False),
+        truncated=torch.tensor(True),
+    )
+
+    result = processor(transition)
+
+    # Tensor values should be moved to device
+    assert isinstance(result[TransitionKey.REWARD], torch.Tensor)
+    assert isinstance(result[TransitionKey.DONE], torch.Tensor)
+    assert isinstance(result[TransitionKey.TRUNCATED], torch.Tensor)
+    assert result[TransitionKey.REWARD].device.type == "cpu"
+    assert result[TransitionKey.DONE].device.type == "cpu"
+    assert result[TransitionKey.TRUNCATED].device.type == "cpu"
+
+
+def test_complementary_data_preserved():
+    """Test that complementary_data is preserved unchanged."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    complementary_data = {
+        "task": "pick_object",
+        "episode_id": 42,
+        "metadata": {"sensor": "camera_1"},
+        "observation_is_pad": torch.tensor([False, False, True]),  # This should be moved to device
+    }
+
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(5)}, complementary_data=complementary_data
+    )
+
+    result = processor(transition)
+
+    # Check that complementary_data is preserved
+    assert TransitionKey.COMPLEMENTARY_DATA in result
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["task"] == "pick_object"
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["episode_id"] == 42
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["metadata"] == {"sensor": "camera_1"}
+    # Note: Currently DeviceProcessorStep doesn't process tensors in complementary_data
+    # This is intentional as complementary_data is typically metadata
+
+
+def test_float_dtype_conversion():
+    """Test float dtype conversion functionality."""
+    processor = DeviceProcessorStep(device="cpu", float_dtype="float16")
+
+    # Create tensors of different types
+    observation = {
+        "observation.float32": torch.randn(5, dtype=torch.float32),
+        "observation.float64": torch.randn(5, dtype=torch.float64),
+        "observation.int32": torch.randint(0, 10, (5,), dtype=torch.int32),
+        "observation.int64": torch.randint(0, 10, (5,), dtype=torch.int64),
+        "observation.bool": torch.tensor([True, False, True], dtype=torch.bool),
+    }
+    action = torch.randn(3, dtype=torch.float32)
+    reward = torch.tensor(1.0, dtype=torch.float32)
+
+    transition = create_transition(observation=observation, action=action, reward=reward)
+    result = processor(transition)
+
+    # Check that float tensors are converted to float16
+    assert result[TransitionKey.OBSERVATION]["observation.float32"].dtype == torch.float16
+    assert result[TransitionKey.OBSERVATION]["observation.float64"].dtype == torch.float16
+    assert result[TransitionKey.ACTION].dtype == torch.float16
+    assert result[TransitionKey.REWARD].dtype == torch.float16
+
+    # Check that non-float tensors are preserved
+    assert result[TransitionKey.OBSERVATION]["observation.int32"].dtype == torch.int32
+    assert result[TransitionKey.OBSERVATION]["observation.int64"].dtype == torch.int64
+    assert result[TransitionKey.OBSERVATION]["observation.bool"].dtype == torch.bool
+
+
+def test_float_dtype_none():
+    """Test that when float_dtype is None, no dtype conversion occurs."""
+    processor = DeviceProcessorStep(device="cpu", float_dtype=None)
+
+    observation = {
+        "observation.float32": torch.randn(5, dtype=torch.float32),
+        "observation.float64": torch.randn(5, dtype=torch.float64),
+        "observation.int32": torch.randint(0, 10, (5,), dtype=torch.int32),
+    }
+    action = torch.randn(3, dtype=torch.float64)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    # Check that dtypes are preserved when float_dtype is None
+    assert result[TransitionKey.OBSERVATION]["observation.float32"].dtype == torch.float32
+    assert result[TransitionKey.OBSERVATION]["observation.float64"].dtype == torch.float64
+    assert result[TransitionKey.OBSERVATION]["observation.int32"].dtype == torch.int32
+    assert result[TransitionKey.ACTION].dtype == torch.float64
+
+
+def test_float_dtype_bfloat16():
+    """Test conversion to bfloat16."""
+    processor = DeviceProcessorStep(device="cpu", float_dtype="bfloat16")
+
+    observation = {OBS_STATE: torch.randn(5, dtype=torch.float32)}
+    action = torch.randn(3, dtype=torch.float64)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].dtype == torch.bfloat16
+    assert result[TransitionKey.ACTION].dtype == torch.bfloat16
+
+
+def test_float_dtype_float64():
+    """Test conversion to float64."""
+    processor = DeviceProcessorStep(device="cpu", float_dtype="float64")
+
+    observation = {OBS_STATE: torch.randn(5, dtype=torch.float16)}
+    action = torch.randn(3, dtype=torch.float32)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].dtype == torch.float64
+    assert result[TransitionKey.ACTION].dtype == torch.float64
+
+
+def test_float_dtype_invalid():
+    """Test that invalid float_dtype raises ValueError."""
+    with pytest.raises(ValueError, match="Invalid float_dtype 'invalid_dtype'"):
+        DeviceProcessorStep(device="cpu", float_dtype="invalid_dtype")
+
+
+def test_float_dtype_aliases():
+    """Test that dtype aliases work correctly."""
+    # Test 'half' alias for float16
+    processor_half = DeviceProcessorStep(device="cpu", float_dtype="half")
+    assert processor_half._target_float_dtype == torch.float16
+
+    # Test 'float' alias for float32
+    processor_float = DeviceProcessorStep(device="cpu", float_dtype="float")
+    assert processor_float._target_float_dtype == torch.float32
+
+    # Test 'double' alias for float64
+    processor_double = DeviceProcessorStep(device="cpu", float_dtype="double")
+    assert processor_double._target_float_dtype == torch.float64
+
+
+def test_float_dtype_with_mixed_tensors():
+    """Test float dtype conversion with mixed tensor types."""
+    processor = DeviceProcessorStep(device="cpu", float_dtype="float32")
+
+    observation = {
+        OBS_IMAGE: torch.randint(0, 255, (3, 64, 64), dtype=torch.uint8),  # Should not convert
+        OBS_STATE: torch.randn(10, dtype=torch.float64),  # Should convert
+        "observation.mask": torch.tensor([True, False, True], dtype=torch.bool),  # Should not convert
+        "observation.indices": torch.tensor([1, 2, 3], dtype=torch.long),  # Should not convert
+    }
+    action = torch.randn(5, dtype=torch.float16)  # Should convert
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    # Check conversions
+    assert result[TransitionKey.OBSERVATION][OBS_IMAGE].dtype == torch.uint8  # Unchanged
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].dtype == torch.float32  # Converted
+    assert result[TransitionKey.OBSERVATION]["observation.mask"].dtype == torch.bool  # Unchanged
+    assert result[TransitionKey.OBSERVATION]["observation.indices"].dtype == torch.long  # Unchanged
+    assert result[TransitionKey.ACTION].dtype == torch.float32  # Converted
+
+
+def test_float_dtype_serialization():
+    """Test that float_dtype is properly serialized in get_config."""
+    device = "cuda" if torch.cuda.is_available() else "cpu"
+    processor = DeviceProcessorStep(device=device, float_dtype="float16")
+    config = processor.get_config()
+
+    assert config == {"device": device, "float_dtype": "float16"}
+
+    # Test with None float_dtype
+    processor_none = DeviceProcessorStep(device="cpu", float_dtype=None)
+    config_none = processor_none.get_config()
+
+    assert config_none == {"device": "cpu", "float_dtype": None}
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_float_dtype_with_cuda():
+    """Test float dtype conversion combined with CUDA device."""
+    processor = DeviceProcessorStep(device="cuda", float_dtype="float16")
+
+    # Create tensors on CPU with different dtypes
+    observation = {
+        "observation.float32": torch.randn(5, dtype=torch.float32),
+        "observation.int64": torch.tensor([1, 2, 3], dtype=torch.int64),
+    }
+    action = torch.randn(3, dtype=torch.float64)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    # Check that tensors are on CUDA and float types are converted
+    assert result[TransitionKey.OBSERVATION]["observation.float32"].device.type == "cuda"
+    assert result[TransitionKey.OBSERVATION]["observation.float32"].dtype == torch.float16
+
+    assert result[TransitionKey.OBSERVATION]["observation.int64"].device.type == "cuda"
+    assert result[TransitionKey.OBSERVATION]["observation.int64"].dtype == torch.int64  # Unchanged
+
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+    assert result[TransitionKey.ACTION].dtype == torch.float16
+
+
+def test_complementary_data_index_fields():
+    """Test processing of index and task_index fields in complementary_data."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    # Create transition with index and task_index in complementary_data
+    complementary_data = {
+        "task": ["pick_cube"],
+        "index": torch.tensor([42], dtype=torch.int64),
+        "task_index": torch.tensor([3], dtype=torch.int64),
+        "episode_id": 123,  # Non-tensor field
+    }
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(1, 7)},
+        action=torch.randn(1, 4),
+        complementary_data=complementary_data,
+    )
+
+    result = processor(transition)
+
+    # Check that tensors in complementary_data are processed
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Check index tensor
+    assert isinstance(processed_comp_data["index"], torch.Tensor)
+    assert processed_comp_data["index"].device.type == "cpu"
+    assert torch.equal(processed_comp_data["index"], complementary_data["index"])
+
+    # Check task_index tensor
+    assert isinstance(processed_comp_data["task_index"], torch.Tensor)
+    assert processed_comp_data["task_index"].device.type == "cpu"
+    assert torch.equal(processed_comp_data["task_index"], complementary_data["task_index"])
+
+    # Check non-tensor fields remain unchanged
+    assert processed_comp_data["task"] == ["pick_cube"]
+    assert processed_comp_data["episode_id"] == 123
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_complementary_data_index_fields_cuda():
+    """Test moving index and task_index fields to CUDA."""
+    processor = DeviceProcessorStep(device="cuda:0")
+
+    # Create CPU tensors
+    complementary_data = {
+        "index": torch.tensor([100, 101], dtype=torch.int64),
+        "task_index": torch.tensor([5], dtype=torch.int64),
+    }
+    transition = create_transition(complementary_data=complementary_data)
+
+    result = processor(transition)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Check tensors moved to CUDA
+    assert processed_comp_data["index"].device.type == "cuda"
+    assert processed_comp_data["index"].device.index == 0
+    assert processed_comp_data["task_index"].device.type == "cuda"
+    assert processed_comp_data["task_index"].device.index == 0
+
+
+def test_complementary_data_without_index_fields():
+    """Test that complementary_data without index/task_index fields works correctly."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    complementary_data = {
+        "task": ["navigate"],
+        "episode_id": 456,
+    }
+    transition = create_transition(complementary_data=complementary_data)
+
+    result = processor(transition)
+
+    # Should process without errors and preserve non-tensor fields
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["task"] == ["navigate"]
+    assert processed_comp_data["episode_id"] == 456
+
+
+def test_complementary_data_mixed_tensors():
+    """Test complementary_data with mix of tensors and non-tensors."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    complementary_data = {
+        "task": ["pick_and_place"],
+        "index": torch.tensor([42], dtype=torch.int64),
+        "task_index": torch.tensor([3], dtype=torch.int64),
+        "metrics": [1.0, 2.0, 3.0],  # List, not tensor
+        "config": {"speed": "fast"},  # Dict
+        "episode_id": 789,  # Int
+    }
+    transition = create_transition(complementary_data=complementary_data)
+
+    result = processor(transition)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Check tensors are processed
+    assert isinstance(processed_comp_data["index"], torch.Tensor)
+    assert isinstance(processed_comp_data["task_index"], torch.Tensor)
+
+    # Check non-tensors remain unchanged
+    assert processed_comp_data["task"] == ["pick_and_place"]
+    assert processed_comp_data["metrics"] == [1.0, 2.0, 3.0]
+    assert processed_comp_data["config"] == {"speed": "fast"}
+    assert processed_comp_data["episode_id"] == 789
+
+
+def test_complementary_data_float_dtype_conversion():
+    """Test that float dtype conversion doesn't affect int tensors in complementary_data."""
+    processor = DeviceProcessorStep(device="cpu", float_dtype="float16")
+
+    complementary_data = {
+        "index": torch.tensor([42], dtype=torch.int64),
+        "task_index": torch.tensor([3], dtype=torch.int64),
+        "float_tensor": torch.tensor([1.5, 2.5], dtype=torch.float32),  # Should be converted
+    }
+    transition = create_transition(complementary_data=complementary_data)
+
+    result = processor(transition)
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Int tensors should keep their dtype
+    assert processed_comp_data["index"].dtype == torch.int64
+    assert processed_comp_data["task_index"].dtype == torch.int64
+
+    # Float tensor should be converted
+    assert processed_comp_data["float_tensor"].dtype == torch.float16
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_complementary_data_full_pipeline_cuda():
+    """Test full transition with complementary_data on CUDA."""
+    processor = DeviceProcessorStep(device="cuda:0", float_dtype="float16")
+
+    # Create full transition with mixed CPU tensors
+    observation = {OBS_STATE: torch.randn(1, 7, dtype=torch.float32)}
+    action = torch.randn(1, 4, dtype=torch.float32)
+    reward = torch.tensor(1.5, dtype=torch.float32)
+    done = torch.tensor(False)
+    complementary_data = {
+        "task": ["reach_target"],
+        "index": torch.tensor([1000], dtype=torch.int64),
+        "task_index": torch.tensor([10], dtype=torch.int64),
+    }
+
+    transition = create_transition(
+        observation=observation,
+        action=action,
+        reward=reward,
+        done=done,
+        complementary_data=complementary_data,
+    )
+
+    result = processor(transition)
+
+    # Check all components moved to CUDA
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cuda"
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+    assert result[TransitionKey.REWARD].device.type == "cuda"
+    assert result[TransitionKey.DONE].device.type == "cuda"
+
+    # Check complementary_data tensors
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["index"].device.type == "cuda"
+    assert processed_comp_data["task_index"].device.type == "cuda"
+
+    # Check float conversion happened for float tensors
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].dtype == torch.float16
+    assert result[TransitionKey.ACTION].dtype == torch.float16
+    assert result[TransitionKey.REWARD].dtype == torch.float16
+
+    # Check int tensors kept their dtype
+    assert processed_comp_data["index"].dtype == torch.int64
+    assert processed_comp_data["task_index"].dtype == torch.int64
+
+
+def test_complementary_data_empty():
+    """Test empty complementary_data handling."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(1, 7)},
+        complementary_data={},
+    )
+
+    result = processor(transition)
+
+    # Should have empty dict
+    assert result[TransitionKey.COMPLEMENTARY_DATA] == {}
+
+
+def test_complementary_data_none():
+    """Test None complementary_data handling."""
+    processor = DeviceProcessorStep(device="cpu")
+
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(1, 7)},
+        complementary_data=None,
+    )
+
+    result = processor(transition)
+
+    # Complementary data should not be in the result (same as input)
+    assert result[TransitionKey.COMPLEMENTARY_DATA] == {}
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_preserves_gpu_placement():
+    """Test that DeviceProcessorStep preserves GPU placement when tensor is already on GPU."""
+    processor = DeviceProcessorStep(device="cuda:0")
+
+    # Create tensors already on GPU
+    observation = {
+        OBS_STATE: torch.randn(10).cuda(),  # Already on GPU
+        OBS_IMAGE: torch.randn(3, 224, 224).cuda(),  # Already on GPU
+    }
+    action = torch.randn(5).cuda()  # Already on GPU
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    # Check that tensors remain on their original GPU
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cuda"
+    assert result[TransitionKey.OBSERVATION][OBS_IMAGE].device.type == "cuda"
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+
+    # Verify no unnecessary copies were made (same data pointer)
+    assert torch.equal(result[TransitionKey.OBSERVATION][OBS_STATE], observation[OBS_STATE])
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_multi_gpu_preservation():
+    """Test that DeviceProcessorStep preserves placement on different GPUs in multi-GPU setup."""
+    # Test 1: GPU-to-GPU preservation (cuda:0 config, cuda:1 input)
+    processor_gpu = DeviceProcessorStep(device="cuda:0")
+
+    # Create tensors on cuda:1 (simulating Accelerate placement)
+    cuda1_device = torch.device("cuda:1")
+    observation = {
+        OBS_STATE: torch.randn(10).to(cuda1_device),
+        OBS_IMAGE: torch.randn(3, 224, 224).to(cuda1_device),
+    }
+    action = torch.randn(5).to(cuda1_device)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor_gpu(transition)
+
+    # Check that tensors remain on cuda:1 (not moved to cuda:0)
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device == cuda1_device
+    assert result[TransitionKey.OBSERVATION][OBS_IMAGE].device == cuda1_device
+    assert result[TransitionKey.ACTION].device == cuda1_device
+
+    # Test 2: GPU-to-CPU should move to CPU (not preserve GPU)
+    processor_cpu = DeviceProcessorStep(device="cpu")
+
+    transition_gpu = create_transition(
+        observation={OBS_STATE: torch.randn(10).cuda()}, action=torch.randn(5).cuda()
+    )
+    result_cpu = processor_cpu(transition_gpu)
+
+    # Check that tensors are moved to CPU
+    assert result_cpu[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cpu"
+    assert result_cpu[TransitionKey.ACTION].device.type == "cpu"
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_multi_gpu_with_cpu_tensors():
+    """Test that CPU tensors are moved to configured device even in multi-GPU context."""
+    # Processor configured for cuda:1
+    processor = DeviceProcessorStep(device="cuda:1")
+
+    # Mix of CPU and GPU tensors
+    observation = {
+        "observation.cpu": torch.randn(10),  # CPU tensor
+        "observation.gpu0": torch.randn(10).cuda(0),  # Already on cuda:0
+        "observation.gpu1": torch.randn(10).cuda(1),  # Already on cuda:1
+    }
+    action = torch.randn(5)  # CPU tensor
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    # CPU tensor should move to configured device (cuda:1)
+    assert result[TransitionKey.OBSERVATION]["observation.cpu"].device.type == "cuda"
+    assert result[TransitionKey.OBSERVATION]["observation.cpu"].device.index == 1
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+    assert result[TransitionKey.ACTION].device.index == 1
+
+    # GPU tensors should stay on their original devices
+    assert result[TransitionKey.OBSERVATION]["observation.gpu0"].device.index == 0
+    assert result[TransitionKey.OBSERVATION]["observation.gpu1"].device.index == 1
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_multi_gpu_with_float_dtype():
+    """Test float dtype conversion works correctly with multi-GPU preservation."""
+    processor = DeviceProcessorStep(device="cuda:0", float_dtype="float16")
+
+    # Create float tensors on different GPUs
+    observation = {
+        "observation.gpu0": torch.randn(5, dtype=torch.float32).cuda(0),
+        "observation.gpu1": torch.randn(5, dtype=torch.float32).cuda(1),
+        "observation.cpu": torch.randn(5, dtype=torch.float32),  # CPU
+    }
+
+    transition = create_transition(observation=observation)
+    result = processor(transition)
+
+    # Check device placement
+    assert result[TransitionKey.OBSERVATION]["observation.gpu0"].device.index == 0
+    assert result[TransitionKey.OBSERVATION]["observation.gpu1"].device.index == 1
+    assert result[TransitionKey.OBSERVATION]["observation.cpu"].device.index == 0  # Moved to cuda:0
+
+    # Check dtype conversion happened for all
+    assert result[TransitionKey.OBSERVATION]["observation.gpu0"].dtype == torch.float16
+    assert result[TransitionKey.OBSERVATION]["observation.gpu1"].dtype == torch.float16
+    assert result[TransitionKey.OBSERVATION]["observation.cpu"].dtype == torch.float16
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_simulated_accelerate_scenario():
+    """Test a scenario simulating how Accelerate would use the processor."""
+    # Simulate different processes getting different GPU assignments
+    for gpu_id in range(min(torch.cuda.device_count(), 2)):
+        # Each "process" has a processor configured for cuda:0
+        # but data comes in already placed on the process's GPU
+        processor = DeviceProcessorStep(device="cuda:0")
+
+        # Simulate data already placed by Accelerate
+        device = torch.device(f"cuda:{gpu_id}")
+        observation = {OBS_STATE: torch.randn(1, 10).to(device)}
+        action = torch.randn(1, 5).to(device)
+
+        transition = create_transition(observation=observation, action=action)
+        result = processor(transition)
+
+        # Verify data stays on the GPU where Accelerate placed it
+        assert result[TransitionKey.OBSERVATION][OBS_STATE].device == device
+        assert result[TransitionKey.ACTION].device == device
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_policy_processor_integration():
+    """Test integration with policy processors - input on GPU, output on CPU."""
+    from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+    from lerobot.processor import (
+        AddBatchDimensionProcessorStep,
+        NormalizerProcessorStep,
+        UnnormalizerProcessorStep,
+    )
+    from lerobot.utils.constants import ACTION, OBS_STATE
+
+    # Create features and stats
+    features = {
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(10,)),
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(5,)),
+    }
+
+    stats = {
+        OBS_STATE: {"mean": torch.zeros(10), "std": torch.ones(10)},
+        ACTION: {"mean": torch.zeros(5), "std": torch.ones(5)},
+    }
+
+    norm_map = {FeatureType.STATE: NormalizationMode.MEAN_STD, FeatureType.ACTION: NormalizationMode.MEAN_STD}
+
+    # Create input processor (preprocessor) that moves to GPU
+    input_processor = DataProcessorPipeline(
+        steps=[
+            NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats),
+            AddBatchDimensionProcessorStep(),
+            DeviceProcessorStep(device="cuda"),
+        ],
+        name="test_preprocessor",
+        to_transition=identity_transition,
+        to_output=identity_transition,
+    )
+
+    # Create output processor (postprocessor) that moves to CPU
+    output_processor = DataProcessorPipeline(
+        steps=[
+            DeviceProcessorStep(device="cpu"),
+            UnnormalizerProcessorStep(features={ACTION: features[ACTION]}, norm_map=norm_map, stats=stats),
+        ],
+        name="test_postprocessor",
+        to_transition=identity_transition,
+        to_output=identity_transition,
+    )
+
+    # Test data on CPU
+    observation = {OBS_STATE: torch.randn(10)}
+    action = torch.randn(5)
+    transition = create_transition(observation=observation, action=action)
+
+    # Process through input processor
+    input_result = input_processor(transition)
+
+    # Verify tensors are on GPU and batched
+    # The result has TransitionKey.OBSERVATION as the key, with observation.state inside
+    assert input_result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cuda"
+    assert input_result[TransitionKey.OBSERVATION][OBS_STATE].shape[0] == 1
+    assert input_result[TransitionKey.ACTION].device.type == "cuda"
+    assert input_result[TransitionKey.ACTION].shape[0] == 1
+
+    # Simulate model output on GPU
+    model_output = create_transition(action=torch.randn(1, 5).cuda())
+
+    # Process through output processor
+    output_result = output_processor(model_output)
+
+    # Verify action is back on CPU and unnormalized
+    assert output_result[TransitionKey.ACTION].device.type == "cpu"
+    assert output_result[TransitionKey.ACTION].shape == (1, 5)
+
+
+@pytest.mark.skipif(not torch.backends.mps.is_available(), reason="MPS not available")
+def test_mps_float64_compatibility():
+    """Test MPS device compatibility with float64 tensors (automatic conversion to float32)."""
+    processor = DeviceProcessorStep(device="mps")
+
+    # Create tensors with different dtypes, including float64 which MPS doesn't support
+    observation = {
+        "observation.float64": torch.randn(5, dtype=torch.float64),  # Should be converted to float32
+        "observation.float32": torch.randn(5, dtype=torch.float32),  # Should remain float32
+        "observation.float16": torch.randn(5, dtype=torch.float16),  # Should remain float16
+        "observation.int64": torch.randint(0, 10, (5,), dtype=torch.int64),  # Should remain int64
+        "observation.bool": torch.tensor([True, False, True], dtype=torch.bool),  # Should remain bool
+    }
+    action = torch.randn(3, dtype=torch.float64)  # Should be converted to float32
+    reward = torch.tensor(1.0, dtype=torch.float64)  # Should be converted to float32
+    done = torch.tensor(False, dtype=torch.bool)  # Should remain bool
+    truncated = torch.tensor(True, dtype=torch.bool)  # Should remain bool
+
+    transition = create_transition(
+        observation=observation, action=action, reward=reward, done=done, truncated=truncated
+    )
+
+    result = processor(transition)
+
+    # Check that all tensors are on MPS device
+    assert result[TransitionKey.OBSERVATION]["observation.float64"].device.type == "mps"
+    assert result[TransitionKey.OBSERVATION]["observation.float32"].device.type == "mps"
+    assert result[TransitionKey.OBSERVATION]["observation.float16"].device.type == "mps"
+    assert result[TransitionKey.OBSERVATION]["observation.int64"].device.type == "mps"
+    assert result[TransitionKey.OBSERVATION]["observation.bool"].device.type == "mps"
+    assert result[TransitionKey.ACTION].device.type == "mps"
+    assert result[TransitionKey.REWARD].device.type == "mps"
+    assert result[TransitionKey.DONE].device.type == "mps"
+    assert result[TransitionKey.TRUNCATED].device.type == "mps"
+
+    # Check that float64 tensors were automatically converted to float32
+    assert result[TransitionKey.OBSERVATION]["observation.float64"].dtype == torch.float32
+    assert result[TransitionKey.ACTION].dtype == torch.float32
+    assert result[TransitionKey.REWARD].dtype == torch.float32
+
+    # Check that other dtypes were preserved
+    assert result[TransitionKey.OBSERVATION]["observation.float32"].dtype == torch.float32
+    assert result[TransitionKey.OBSERVATION]["observation.float16"].dtype == torch.float16
+    assert result[TransitionKey.OBSERVATION]["observation.int64"].dtype == torch.int64
+    assert result[TransitionKey.OBSERVATION]["observation.bool"].dtype == torch.bool
+    assert result[TransitionKey.DONE].dtype == torch.bool
+    assert result[TransitionKey.TRUNCATED].dtype == torch.bool
+
+
+@pytest.mark.skipif(not torch.backends.mps.is_available(), reason="MPS not available")
+def test_mps_float64_with_complementary_data():
+    """Test MPS float64 conversion with complementary_data tensors."""
+    processor = DeviceProcessorStep(device="mps")
+
+    # Create complementary_data with float64 tensors
+    complementary_data = {
+        "task": ["pick_object"],
+        "index": torch.tensor([42], dtype=torch.int64),  # Should remain int64
+        "task_index": torch.tensor([3], dtype=torch.int64),  # Should remain int64
+        "float64_tensor": torch.tensor([1.5, 2.5], dtype=torch.float64),  # Should convert to float32
+        "float32_tensor": torch.tensor([3.5], dtype=torch.float32),  # Should remain float32
+    }
+
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(5, dtype=torch.float64)},
+        action=torch.randn(3, dtype=torch.float64),
+        complementary_data=complementary_data,
+    )
+
+    result = processor(transition)
+
+    # Check that all tensors are on MPS device
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "mps"
+    assert result[TransitionKey.ACTION].device.type == "mps"
+
+    processed_comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+    assert processed_comp_data["index"].device.type == "mps"
+    assert processed_comp_data["task_index"].device.type == "mps"
+    assert processed_comp_data["float64_tensor"].device.type == "mps"
+    assert processed_comp_data["float32_tensor"].device.type == "mps"
+
+    # Check dtype conversions
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].dtype == torch.float32  # Converted
+    assert result[TransitionKey.ACTION].dtype == torch.float32  # Converted
+    assert processed_comp_data["float64_tensor"].dtype == torch.float32  # Converted
+    assert processed_comp_data["float32_tensor"].dtype == torch.float32  # Unchanged
+    assert processed_comp_data["index"].dtype == torch.int64  # Unchanged
+    assert processed_comp_data["task_index"].dtype == torch.int64  # Unchanged
+
+    # Check non-tensor data preserved
+    assert processed_comp_data["task"] == ["pick_object"]
+
+
+@pytest.mark.skipif(not torch.backends.mps.is_available(), reason="MPS not available")
+def test_mps_with_explicit_float_dtype():
+    """Test MPS device with explicit float_dtype setting."""
+    # Test that explicit float_dtype still works on MPS
+    processor = DeviceProcessorStep(device="mps", float_dtype="float16")
+
+    observation = {
+        "observation.float64": torch.randn(
+            5, dtype=torch.float64
+        ),  # First converted to float32, then to float16
+        "observation.float32": torch.randn(5, dtype=torch.float32),  # Converted to float16
+        "observation.int32": torch.randint(0, 10, (5,), dtype=torch.int32),  # Should remain int32
+    }
+    action = torch.randn(3, dtype=torch.float64)
+
+    transition = create_transition(observation=observation, action=action)
+    result = processor(transition)
+
+    # Check device placement
+    assert result[TransitionKey.OBSERVATION]["observation.float64"].device.type == "mps"
+    assert result[TransitionKey.OBSERVATION]["observation.float32"].device.type == "mps"
+    assert result[TransitionKey.OBSERVATION]["observation.int32"].device.type == "mps"
+    assert result[TransitionKey.ACTION].device.type == "mps"
+
+    # Check that all float tensors end up as float16 (the target dtype)
+    assert result[TransitionKey.OBSERVATION]["observation.float64"].dtype == torch.float16
+    assert result[TransitionKey.OBSERVATION]["observation.float32"].dtype == torch.float16
+    assert result[TransitionKey.ACTION].dtype == torch.float16
+
+    # Check that non-float tensors are preserved
+    assert result[TransitionKey.OBSERVATION]["observation.int32"].dtype == torch.int32
+
+
+@pytest.mark.skipif(not torch.backends.mps.is_available(), reason="MPS not available")
+def test_mps_serialization():
+    """Test that MPS device processor can be serialized and loaded correctly."""
+    processor = DeviceProcessorStep(device="mps", float_dtype="float32")
+
+    # Test get_config
+    config = processor.get_config()
+    assert config == {"device": "mps", "float_dtype": "float32"}
+
+    # Test state_dict (should be empty)
+    state = processor.state_dict()
+    assert state == {}
+
+    # Test load_state_dict (should be no-op)
+    processor.load_state_dict({})
+    assert processor.device == "mps"
diff --git a/lerobot/tests/processor/test_diffusion_processor.py b/lerobot/tests/processor/test_diffusion_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..67981c70d5cc6725c8d5c503c64f0332ee7dd2a2
--- /dev/null
+++ b/lerobot/tests/processor/test_diffusion_processor.py
@@ -0,0 +1,398 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tests for Diffusion policy processor."""
+
+import tempfile
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.policies.diffusion.configuration_diffusion import DiffusionConfig
+from lerobot.policies.diffusion.processor_diffusion import make_diffusion_pre_post_processors
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DataProcessorPipeline,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    RenameObservationsProcessorStep,
+    TransitionKey,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import create_transition, transition_to_batch
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_STATE
+
+
+def create_default_config():
+    """Create a default Diffusion configuration for testing."""
+    config = DiffusionConfig()
+    config.input_features = {
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(7,)),
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(6,)),
+    }
+    config.normalization_mapping = {
+        FeatureType.STATE: NormalizationMode.MEAN_STD,
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,
+        FeatureType.ACTION: NormalizationMode.MIN_MAX,
+    }
+    config.device = "cpu"
+    return config
+
+
+def create_default_stats():
+    """Create default dataset statistics for testing."""
+    return {
+        OBS_STATE: {"mean": torch.zeros(7), "std": torch.ones(7)},
+        OBS_IMAGE: {},  # No normalization for images
+        ACTION: {"min": torch.full((6,), -1.0), "max": torch.ones(6)},
+    }
+
+
+def test_make_diffusion_processor_basic():
+    """Test basic creation of Diffusion processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_diffusion_pre_post_processors(config, stats)
+
+    # Check processor names
+    assert preprocessor.name == "policy_preprocessor"
+    assert postprocessor.name == "policy_postprocessor"
+
+    # Check steps in preprocessor
+    assert len(preprocessor.steps) == 4
+    assert isinstance(preprocessor.steps[0], RenameObservationsProcessorStep)
+    assert isinstance(preprocessor.steps[1], AddBatchDimensionProcessorStep)
+    assert isinstance(preprocessor.steps[2], DeviceProcessorStep)
+    assert isinstance(preprocessor.steps[3], NormalizerProcessorStep)
+
+    # Check steps in postprocessor
+    assert len(postprocessor.steps) == 2
+    assert isinstance(postprocessor.steps[0], UnnormalizerProcessorStep)
+    assert isinstance(postprocessor.steps[1], DeviceProcessorStep)
+
+
+def test_diffusion_processor_with_images():
+    """Test Diffusion processor with image observations."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_diffusion_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create test data with images
+    observation = {
+        OBS_STATE: torch.randn(7),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(6)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is batched
+    assert processed[OBS_STATE].shape == (1, 7)
+    assert processed[OBS_IMAGE].shape == (1, 3, 224, 224)
+    assert processed[TransitionKey.ACTION.value].shape == (1, 6)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_diffusion_processor_cuda():
+    """Test Diffusion processor with CUDA device."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_diffusion_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create CPU data
+    observation = {
+        OBS_STATE: torch.randn(7),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(6)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is on CUDA
+    assert processed[OBS_STATE].device.type == "cuda"
+    assert processed[OBS_IMAGE].device.type == "cuda"
+    assert processed[TransitionKey.ACTION.value].device.type == "cuda"
+
+    # Process through postprocessor
+    postprocessed = postprocessor(processed[TransitionKey.ACTION.value])
+
+    # Check that action is back on CPU
+    assert postprocessed.device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_diffusion_processor_accelerate_scenario():
+    """Test Diffusion processor in simulated Accelerate scenario."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_diffusion_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Simulate Accelerate: data already on GPU
+    device = torch.device("cuda:0")
+    observation = {
+        OBS_STATE: torch.randn(1, 7).to(device),
+        OBS_IMAGE: torch.randn(1, 3, 224, 224).to(device),
+    }
+    action = torch.randn(1, 6).to(device)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on same GPU
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_diffusion_processor_multi_gpu():
+    """Test Diffusion processor with multi-GPU setup."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_diffusion_pre_post_processors(config, stats)
+
+    # Simulate data on different GPU
+    device = torch.device("cuda:1")
+    observation = {
+        OBS_STATE: torch.randn(1, 7).to(device),
+        OBS_IMAGE: torch.randn(1, 3, 224, 224).to(device),
+    }
+    action = torch.randn(1, 6).to(device)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on cuda:1
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+def test_diffusion_processor_without_stats():
+    """Test Diffusion processor creation without dataset statistics."""
+    config = create_default_config()
+
+    preprocessor, postprocessor = make_diffusion_pre_post_processors(
+        config,
+        dataset_stats=None,
+    )
+
+    # Should still create processors
+    assert preprocessor is not None
+    assert postprocessor is not None
+
+    # Process should still work
+    observation = {
+        OBS_STATE: torch.randn(7),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(6)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed is not None
+
+
+def test_diffusion_processor_save_and_load():
+    """Test saving and loading Diffusion processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_diffusion_pre_post_processors(config, stats)
+
+    with tempfile.TemporaryDirectory() as tmpdir:
+        # Save preprocessor
+        preprocessor.save_pretrained(tmpdir)
+
+        # Load preprocessor
+        loaded_preprocessor = DataProcessorPipeline.from_pretrained(
+            tmpdir, config_filename="policy_preprocessor.json"
+        )
+
+        # Test that loaded processor works
+        observation = {
+            OBS_STATE: torch.randn(7),
+            OBS_IMAGE: torch.randn(3, 224, 224),
+        }
+        action = torch.randn(6)
+        transition = create_transition(observation, action)
+        batch = transition_to_batch(transition)
+
+        processed = loaded_preprocessor(batch)
+        assert processed[OBS_STATE].shape == (1, 7)
+        assert processed[OBS_IMAGE].shape == (1, 3, 224, 224)
+        assert processed[TransitionKey.ACTION.value].shape == (1, 6)
+
+
+def test_diffusion_processor_identity_normalization():
+    """Test that images with IDENTITY normalization are not normalized."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_diffusion_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create test data
+    image_value = torch.rand(3, 224, 224) * 255  # Large values
+    observation = {
+        OBS_STATE: torch.randn(7),
+        OBS_IMAGE: image_value.clone(),
+    }
+    action = torch.randn(6)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Image should not be normalized (IDENTITY mode)
+    # Just batched
+    assert torch.allclose(processed[OBS_IMAGE][0], image_value, rtol=1e-5)
+
+
+def test_diffusion_processor_batch_consistency():
+    """Test Diffusion processor with different batch sizes."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_diffusion_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Test with different batch sizes
+    for batch_size in [1, 8, 32]:
+        observation = {
+            OBS_STATE: torch.randn(batch_size, 7) if batch_size > 1 else torch.randn(7),
+            OBS_IMAGE: torch.randn(batch_size, 3, 224, 224) if batch_size > 1 else torch.randn(3, 224, 224),
+        }
+        action = torch.randn(batch_size, 6) if batch_size > 1 else torch.randn(6)
+        transition = create_transition(observation, action)
+
+        batch = transition_to_batch(transition)
+
+        processed = preprocessor(batch)
+
+        # Check correct batch size
+        expected_batch = batch_size if batch_size > 1 else 1
+        assert processed[OBS_STATE].shape[0] == expected_batch
+        assert processed[OBS_IMAGE].shape[0] == expected_batch
+        assert processed[TransitionKey.ACTION.value].shape[0] == expected_batch
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_diffusion_processor_bfloat16_device_float32_normalizer():
+    """Test: DeviceProcessor(bfloat16) + NormalizerProcessor(float32) → output bfloat16 via automatic adaptation"""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, _ = make_diffusion_pre_post_processors(config, stats)
+
+    # Modify the pipeline to use bfloat16 device processor with float32 normalizer
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            # Device processor converts to bfloat16
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="bfloat16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Normalizer stays configured as float32 (will auto-adapt to bfloat16)
+            norm_step = step  # Now type checker knows this is NormalizerProcessorStep
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=norm_step.features,
+                    norm_map=norm_step.norm_map,
+                    stats=norm_step.stats,
+                    device=config.device,
+                    dtype=torch.float32,  # Deliberately configured as float32
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Verify initial normalizer configuration
+    normalizer_step = preprocessor.steps[3]  # NormalizerProcessorStep
+    assert normalizer_step.dtype == torch.float32
+
+    # Create test data with both state and visual observations
+    observation = {
+        OBS_STATE: torch.randn(7, dtype=torch.float32),
+        OBS_IMAGE: torch.randn(3, 224, 224, dtype=torch.float32),
+    }
+    action = torch.randn(6, dtype=torch.float32)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through full pipeline
+    processed = preprocessor(batch)
+
+    # Verify: DeviceProcessor → bfloat16, NormalizerProcessor adapts → final output is bfloat16
+    assert processed[OBS_STATE].dtype == torch.bfloat16
+    assert processed[OBS_IMAGE].dtype == torch.bfloat16  # IDENTITY normalization still gets dtype conversion
+    assert processed[TransitionKey.ACTION.value].dtype == torch.bfloat16
+
+    # Verify normalizer automatically adapted its internal state
+    assert normalizer_step.dtype == torch.bfloat16
+    # Check state stats (has normalization)
+    for stat_tensor in normalizer_step._tensor_stats[OBS_STATE].values():
+        assert stat_tensor.dtype == torch.bfloat16
+    # OBS_IMAGE uses IDENTITY normalization, so no stats to check
diff --git a/lerobot/tests/processor/test_libero_processor.py b/lerobot/tests/processor/test_libero_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..2bd788e6b46911592073ee2e98274ef5c4ec0ab7
--- /dev/null
+++ b/lerobot/tests/processor/test_libero_processor.py
@@ -0,0 +1,72 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import numpy as np
+import torch
+
+from lerobot.envs.utils import preprocess_observation
+from lerobot.processor.env_processor import LiberoProcessorStep
+from lerobot.processor.pipeline import PolicyProcessorPipeline
+
+seed = 42
+np.random.seed(seed)
+
+B = 5
+obs1 = {
+    "pixels": {
+        "image": (np.random.rand(B, 256, 256, 3) * 255).astype(np.uint8),
+        "image2": (np.random.rand(B, 256, 256, 3) * 255).astype(np.uint8),
+    },
+    "robot_state": {
+        "eef": {
+            "pos": np.random.randn(B, 3),
+            "quat": np.random.randn(B, 4),
+            "mat": np.random.randn(B, 3, 3),
+        },
+        "gripper": {
+            "qpos": np.random.randn(B, 2),
+            "qvel": np.random.randn(B, 2),
+        },
+        "joints": {
+            "pos": np.random.randn(B, 7),
+            "vel": np.random.randn(B, 7),
+        },
+    },
+}
+
+observation = preprocess_observation(obs1)
+libero_preprocessor = PolicyProcessorPipeline(
+    steps=[
+        LiberoProcessorStep(),
+    ]
+)
+processed_obs = libero_preprocessor(observation)
+assert "observation.state" in processed_obs
+state = processed_obs["observation.state"]
+assert isinstance(state, torch.Tensor)
+assert state.dtype == torch.float32
+
+assert state.shape[0] == B
+assert state.shape[1] == 8
+
+assert "observation.images.image" in processed_obs
+assert "observation.images.image2" in processed_obs
+
+assert isinstance(processed_obs["observation.images.image"], torch.Tensor)
+assert isinstance(processed_obs["observation.images.image2"], torch.Tensor)
+
+assert processed_obs["observation.images.image"].shape == (B, 3, 256, 256)
+assert processed_obs["observation.images.image2"].shape == (B, 3, 256, 256)
diff --git a/lerobot/tests/processor/test_migration_detection.py b/lerobot/tests/processor/test_migration_detection.py
new file mode 100644
index 0000000000000000000000000000000000000000..1ddc87d1ece23e842332b5b53f76cd2109d83f56
--- /dev/null
+++ b/lerobot/tests/processor/test_migration_detection.py
@@ -0,0 +1,342 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Tests for processor migration detection functionality.
+"""
+
+import json
+import tempfile
+from pathlib import Path
+
+import pytest
+
+from lerobot.processor.pipeline import DataProcessorPipeline, ProcessorMigrationError
+from lerobot.utils.constants import ACTION, OBS_STATE
+
+
+def test_is_processor_config_valid_configs():
+    """Test processor config detection with valid configurations."""
+    valid_configs = [
+        {"steps": []},  # Empty steps
+        {"steps": [{"class": "MyClass"}]},  # Class-based step
+        {"steps": [{"registry_name": "my_step"}]},  # Registry-based step
+        {"steps": [{"class": "A"}, {"registry_name": "B"}]},  # Mixed
+        {"name": "Test", "steps": [{"class": "MyClass"}]},  # With name
+    ]
+
+    for i, config in enumerate(valid_configs):
+        assert DataProcessorPipeline._is_processor_config(config), (
+            f"Valid config {i} should be detected as processor config: {config}"
+        )
+
+
+def test_is_processor_config_invalid_configs():
+    """Test processor config detection with invalid configurations."""
+    invalid_configs = [
+        {},  # No steps field
+        {"steps": "not a list"},  # Steps is not a list
+        {"steps": [{}]},  # Step without class or registry_name
+        {"steps": ["not a dict"]},  # Step is not a dict
+        {"steps": [{"other_field": "value"}]},  # Step with wrong fields
+        {"other_field": "value"},  # Completely different structure
+    ]
+
+    for i, config in enumerate(invalid_configs):
+        assert not DataProcessorPipeline._is_processor_config(config), (
+            f"Invalid config {i} should not be detected as processor config: {config}"
+        )
+
+
+def test_should_suggest_migration_with_processor_config():
+    """Test that migration is NOT suggested when processor config exists."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create a valid processor config
+        processor_config = {
+            "name": "TestProcessor",
+            "steps": [
+                {
+                    "class": "lerobot.processor.normalize.NormalizeStep",
+                    "config": {"mean": 0.0, "std": 1.0},
+                }
+            ],
+        }
+
+        with open(tmp_path / "processor.json", "w") as f:
+            json.dump(processor_config, f)
+
+        # Should NOT suggest migration (processor config exists)
+        result = DataProcessorPipeline._should_suggest_migration(tmp_path)
+        assert not result
+
+
+def test_should_suggest_migration_with_empty_processor_config():
+    """Test that migration is NOT suggested when empty processor config exists."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create an empty processor config
+        empty_processor_config = {
+            "name": "EmptyProcessor",
+            "steps": [],  # Empty steps is valid
+        }
+
+        with open(tmp_path / "empty_processor.json", "w") as f:
+            json.dump(empty_processor_config, f)
+
+        # Should NOT suggest migration (processor config exists, even if empty)
+        result = DataProcessorPipeline._should_suggest_migration(tmp_path)
+        assert not result
+
+
+def test_should_suggest_migration_with_model_config_only():
+    """Test that migration IS suggested when only model config exists."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create a model config (like old LeRobot format)
+        model_config = {
+            "type": "act",
+            "input_features": {OBS_STATE: {"shape": [7]}},
+            "output_features": {ACTION: {"shape": [7]}},
+            "hidden_dim": 256,
+            "n_obs_steps": 1,
+            "n_action_steps": 1,
+        }
+
+        with open(tmp_path / "config.json", "w") as f:
+            json.dump(model_config, f)
+
+        # SHOULD suggest migration (model config exists but no processor)
+        result = DataProcessorPipeline._should_suggest_migration(tmp_path)
+        assert result
+
+
+def test_should_suggest_migration_no_json_files():
+    """Test that migration is NOT suggested when no JSON files exist."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create some non-JSON files
+        with open(tmp_path / "model.safetensors", "w") as f:
+            f.write("fake model data")
+
+        with open(tmp_path / "README.md", "w") as f:
+            f.write("# Model README")
+
+        # Should NOT suggest migration (no JSON files)
+        result = DataProcessorPipeline._should_suggest_migration(tmp_path)
+        assert not result
+
+
+def test_should_suggest_migration_random_json_files():
+    """Test that migration IS suggested when JSON files exist but none are processor configs."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create some random JSON file (not a processor config)
+        random_config = {"some_field": "some_value", "another_field": 123}
+
+        with open(tmp_path / "random.json", "w") as f:
+            json.dump(random_config, f)
+
+        # SHOULD suggest migration (JSON files exist but none are processor configs)
+        result = DataProcessorPipeline._should_suggest_migration(tmp_path)
+        assert result
+
+
+def test_should_suggest_migration_mixed_configs():
+    """Test that migration is NOT suggested when processor config exists alongside other configs."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create both a processor config and a model config
+        processor_config = {"name": "TestProcessor", "steps": [{"registry_name": "normalize_step"}]}
+
+        model_config = {"type": "diffusion", "hidden_dim": 512}
+
+        with open(tmp_path / "processor.json", "w") as f:
+            json.dump(processor_config, f)
+
+        with open(tmp_path / "config.json", "w") as f:
+            json.dump(model_config, f)
+
+        # Should NOT suggest migration (processor config exists)
+        result = DataProcessorPipeline._should_suggest_migration(tmp_path)
+        assert not result
+
+
+def test_should_suggest_migration_invalid_json():
+    """Test that invalid JSON is handled gracefully."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create an invalid JSON file
+        with open(tmp_path / "invalid.json", "w") as f:
+            f.write("{ invalid json")
+
+        # Create a valid non-processor config
+        model_config = {"type": "act"}
+        with open(tmp_path / "model.json", "w") as f:
+            json.dump(model_config, f)
+
+        # SHOULD suggest migration (invalid JSON is ignored, but we have non-processor JSON)
+        result = DataProcessorPipeline._should_suggest_migration(tmp_path)
+        assert result
+
+
+def test_from_pretrained_multiple_json_files_migration_error():
+    """Test that multiple JSON files trigger ProcessorMigrationError."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create multiple non-processor configs
+        model_config = {"type": "act", "hidden_dim": 128}
+        train_config = {"batch_size": 32, "lr": 0.001}
+
+        with open(tmp_path / "config.json", "w") as f:
+            json.dump(model_config, f)
+
+        with open(tmp_path / "train_config.json", "w") as f:
+            json.dump(train_config, f)
+
+        # Should raise ProcessorMigrationError
+        with pytest.raises(ProcessorMigrationError) as exc_info:
+            DataProcessorPipeline.from_pretrained(tmp_path, config_filename="config.json")
+
+        # Check the error details
+        error = exc_info.value
+        assert str(tmp_path) in str(error.model_path)
+        assert "migrate_policy_normalization.py" in error.migration_command
+        assert "not a valid processor configuration" in error.original_error
+
+
+def test_from_pretrained_no_processor_config_migration_error():
+    """Test that missing processor config triggers ProcessorMigrationError."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create a model config but no processor
+        model_config = {"type": "diffusion", "hidden_dim": 256}
+
+        with open(tmp_path / "config.json", "w") as f:
+            json.dump(model_config, f)
+
+        # Should raise ProcessorMigrationError
+        with pytest.raises(ProcessorMigrationError) as exc_info:
+            DataProcessorPipeline.from_pretrained(tmp_path, config_filename="config.json")
+
+        # Check the error details
+        error = exc_info.value
+        assert str(tmp_path) in str(error.model_path)
+        assert "migrate_policy_normalization.py" in error.migration_command
+        assert "not a valid processor configuration" in error.original_error
+
+
+def test_from_pretrained_valid_processor_no_migration_error():
+    """Test that valid processor config does NOT trigger migration error."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create a valid processor config
+        processor_config = {
+            "name": "TestProcessor",
+            "steps": [],  # Empty is valid
+        }
+
+        with open(tmp_path / "processor.json", "w") as f:
+            json.dump(processor_config, f)
+
+        # Should succeed and create pipeline
+        pipeline = DataProcessorPipeline.from_pretrained(tmp_path, config_filename="processor.json")
+        assert pipeline is not None
+        assert pipeline.name == "TestProcessor"
+        assert len(pipeline) == 0
+
+
+def test_from_pretrained_no_json_files_no_migration_error():
+    """Test that directories with no JSON files don't trigger migration errors."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create some non-JSON files
+        with open(tmp_path / "model.safetensors", "w") as f:
+            f.write("fake model data")
+
+        # Should raise FileNotFoundError (config file not found)
+        with pytest.raises(FileNotFoundError, match="not found in directory"):
+            DataProcessorPipeline.from_pretrained(tmp_path, config_filename="processor.json")
+
+
+def test_processor_migration_error_creation():
+    """Test that ProcessorMigrationError is created correctly."""
+    model_path = "/path/to/model"
+    migration_command = "python migrate.py --path /path/to/model"
+    original_error = "Config not found"
+
+    error = ProcessorMigrationError(model_path, migration_command, original_error)
+
+    assert error.model_path == model_path
+    assert error.migration_command == migration_command
+    assert error.original_error == original_error
+    assert model_path in str(error)
+    assert migration_command in str(error)
+    assert original_error in str(error)
+
+
+def test_processor_migration_error_attributes():
+    """Test that ProcessorMigrationError has correct attributes."""
+    model_path = Path("/test/path")
+    migration_command = "python test.py"
+    original_error = "Test error"
+
+    error = ProcessorMigrationError(model_path, migration_command, original_error)
+
+    # Test that attributes are accessible
+    assert hasattr(error, "model_path")
+    assert hasattr(error, "migration_command")
+    assert hasattr(error, "original_error")
+
+    # Test that it's still an Exception
+    assert isinstance(error, Exception)
+
+
+def test_migration_suggestion_raises_error():
+    """Test that migration suggestion always raises ProcessorMigrationError."""
+    with pytest.raises(ProcessorMigrationError) as exc_info:
+        DataProcessorPipeline._suggest_processor_migration("/test/path", "Test error")
+
+    error = exc_info.value
+    assert "/test/path" in str(error.model_path)
+    assert "Test error" in error.original_error
+    assert "migrate_policy_normalization.py" in error.migration_command
+
+
+def test_migration_error_always_raised_for_invalid_configs():
+    """Test that ProcessorMigrationError is always raised for invalid configs."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create a model config
+        model_config = {"type": "test", "param": "value"}
+        with open(tmp_path / "config.json", "w") as f:
+            json.dump(model_config, f)
+
+        # Should always raise ProcessorMigrationError
+        with pytest.raises(ProcessorMigrationError):
+            DataProcessorPipeline.from_pretrained(tmp_path, config_filename="config.json")
diff --git a/lerobot/tests/processor/test_normalize_processor.py b/lerobot/tests/processor/test_normalize_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..cd5c7500529733cfdece371970212bada39b9c12
--- /dev/null
+++ b/lerobot/tests/processor/test_normalize_processor.py
@@ -0,0 +1,2130 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from unittest.mock import Mock
+
+import numpy as np
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.processor import (
+    DataProcessorPipeline,
+    IdentityProcessorStep,
+    NormalizerProcessorStep,
+    TransitionKey,
+    UnnormalizerProcessorStep,
+    hotswap_stats,
+)
+from lerobot.processor.converters import create_transition, identity_transition, to_tensor
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_STATE, OBS_STR
+from lerobot.utils.device_utils import auto_select_torch_device
+
+
+def test_numpy_conversion():
+    stats = {
+        OBS_IMAGE: {
+            "mean": np.array([0.5, 0.5, 0.5]),
+            "std": np.array([0.2, 0.2, 0.2]),
+        }
+    }
+    tensor_stats = to_tensor(stats)
+
+    assert isinstance(tensor_stats[OBS_IMAGE]["mean"], torch.Tensor)
+    assert isinstance(tensor_stats[OBS_IMAGE]["std"], torch.Tensor)
+    assert torch.allclose(tensor_stats[OBS_IMAGE]["mean"], torch.tensor([0.5, 0.5, 0.5]))
+    assert torch.allclose(tensor_stats[OBS_IMAGE]["std"], torch.tensor([0.2, 0.2, 0.2]))
+
+
+def test_tensor_conversion():
+    stats = {
+        ACTION: {
+            "mean": torch.tensor([0.0, 0.0]),
+            "std": torch.tensor([1.0, 1.0]),
+        }
+    }
+    tensor_stats = to_tensor(stats)
+
+    assert tensor_stats[ACTION]["mean"].dtype == torch.float32
+    assert tensor_stats[ACTION]["std"].dtype == torch.float32
+
+
+def test_scalar_conversion():
+    stats = {
+        "reward": {
+            "mean": 0.5,
+            "std": 0.1,
+        }
+    }
+    tensor_stats = to_tensor(stats)
+
+    assert torch.allclose(tensor_stats["reward"]["mean"], torch.tensor(0.5))
+    assert torch.allclose(tensor_stats["reward"]["std"], torch.tensor(0.1))
+
+
+def test_list_conversion():
+    stats = {
+        OBS_STATE: {
+            "min": [0.0, -1.0, -2.0],
+            "max": [1.0, 1.0, 2.0],
+        }
+    }
+    tensor_stats = to_tensor(stats)
+
+    assert torch.allclose(tensor_stats[OBS_STATE]["min"], torch.tensor([0.0, -1.0, -2.0]))
+    assert torch.allclose(tensor_stats[OBS_STATE]["max"], torch.tensor([1.0, 1.0, 2.0]))
+
+
+def test_unsupported_type():
+    stats = {
+        "bad_key": {
+            "mean": "string_value",
+        }
+    }
+    with pytest.raises(TypeError, match="Unsupported type"):
+        to_tensor(stats)
+
+
+# Helper functions to create feature maps and norm maps
+def _create_observation_features():
+    return {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96)),
+        OBS_STATE: PolicyFeature(FeatureType.STATE, (2,)),
+    }
+
+
+def _create_observation_norm_map():
+    return {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.STATE: NormalizationMode.MIN_MAX,
+    }
+
+
+# Fixtures for observation normalisation tests using NormalizerProcessorStep
+@pytest.fixture
+def observation_stats():
+    return {
+        OBS_IMAGE: {
+            "mean": np.array([0.5, 0.5, 0.5]),
+            "std": np.array([0.2, 0.2, 0.2]),
+        },
+        OBS_STATE: {
+            "min": np.array([0.0, -1.0]),
+            "max": np.array([1.0, 1.0]),
+        },
+    }
+
+
+@pytest.fixture
+def observation_normalizer(observation_stats):
+    """Return a NormalizerProcessorStep that only has observation stats (no action)."""
+    features = _create_observation_features()
+    norm_map = _create_observation_norm_map()
+    return NormalizerProcessorStep(features=features, norm_map=norm_map, stats=observation_stats)
+
+
+def test_mean_std_normalization(observation_normalizer):
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([0.5, 0.0]),
+    }
+    transition = create_transition(observation=observation)
+
+    normalized_transition = observation_normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    # Check mean/std normalization
+    expected_image = (torch.tensor([0.7, 0.5, 0.3]) - 0.5) / 0.2
+    assert torch.allclose(normalized_obs[OBS_IMAGE], expected_image)
+
+
+def test_min_max_normalization(observation_normalizer):
+    observation = {
+        OBS_STATE: torch.tensor([0.5, 0.0]),
+    }
+    transition = create_transition(observation=observation)
+
+    normalized_transition = observation_normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    # Check min/max normalization to [-1, 1]
+    # For state[0]: 2 * (0.5 - 0.0) / (1.0 - 0.0) - 1 = 0.0
+    # For state[1]: 2 * (0.0 - (-1.0)) / (1.0 - (-1.0)) - 1 = 0.0
+    expected_state = torch.tensor([0.0, 0.0])
+    assert torch.allclose(normalized_obs[OBS_STATE], expected_state, atol=1e-6)
+
+
+def test_quantile_normalization():
+    """Test QUANTILES mode using 1st-99th percentiles."""
+    features = {
+        "observation.state": PolicyFeature(FeatureType.STATE, (2,)),
+    }
+    norm_map = {
+        FeatureType.STATE: NormalizationMode.QUANTILES,
+    }
+    stats = {
+        "observation.state": {
+            "q01": np.array([0.1, -0.8]),  # 1st percentile
+            "q99": np.array([0.9, 0.8]),  # 99th percentile
+        },
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {
+        "observation.state": torch.tensor([0.5, 0.0]),
+    }
+    transition = create_transition(observation=observation)
+
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    # Check quantile normalization to [-1, 1]
+    # For state[0]: 2 * (0.5 - 0.1) / (0.9 - 0.1) - 1 = 2 * 0.4 / 0.8 - 1 = 0.0
+    # For state[1]: 2 * (0.0 - (-0.8)) / (0.8 - (-0.8)) - 1 = 2 * 0.8 / 1.6 - 1 = 0.0
+    expected_state = torch.tensor([0.0, 0.0])
+    assert torch.allclose(normalized_obs["observation.state"], expected_state, atol=1e-6)
+
+
+def test_quantile10_normalization():
+    """Test QUANTILE10 mode using 10th-90th percentiles."""
+    features = {
+        "observation.state": PolicyFeature(FeatureType.STATE, (2,)),
+    }
+    norm_map = {
+        FeatureType.STATE: NormalizationMode.QUANTILE10,
+    }
+    stats = {
+        "observation.state": {
+            "q10": np.array([0.2, -0.6]),  # 10th percentile
+            "q90": np.array([0.8, 0.6]),  # 90th percentile
+        },
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {
+        "observation.state": torch.tensor([0.5, 0.0]),
+    }
+    transition = create_transition(observation=observation)
+
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    # Check quantile normalization to [-1, 1]
+    # For state[0]: 2 * (0.5 - 0.2) / (0.8 - 0.2) - 1 = 2 * 0.3 / 0.6 - 1 = 0.0
+    # For state[1]: 2 * (0.0 - (-0.6)) / (0.6 - (-0.6)) - 1 = 2 * 0.6 / 1.2 - 1 = 0.0
+    expected_state = torch.tensor([0.0, 0.0])
+    assert torch.allclose(normalized_obs["observation.state"], expected_state, atol=1e-6)
+
+
+def test_quantile_unnormalization():
+    """Test that quantile normalization can be reversed properly."""
+    features = {
+        "action": PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+    norm_map = {
+        FeatureType.ACTION: NormalizationMode.QUANTILES,
+    }
+    stats = {
+        "action": {
+            "q01": np.array([0.1, -0.8]),
+            "q99": np.array([0.9, 0.8]),
+        },
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+    unnormalizer = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    # Test round-trip normalization
+    original_action = torch.tensor([0.5, 0.0])
+    transition = create_transition(action=original_action)
+
+    # Normalize then unnormalize
+    normalized = normalizer(transition)
+    unnormalized = unnormalizer(normalized)
+
+    # Should recover original values
+    recovered_action = unnormalized[TransitionKey.ACTION]
+    assert torch.allclose(recovered_action, original_action, atol=1e-6)
+
+
+def test_quantile_division_by_zero():
+    """Test quantile normalization handles edge case where q01 == q99."""
+    features = {
+        "observation.state": PolicyFeature(FeatureType.STATE, (1,)),
+    }
+    norm_map = {
+        FeatureType.STATE: NormalizationMode.QUANTILES,
+    }
+    stats = {
+        "observation.state": {
+            "q01": np.array([0.5]),  # Same value
+            "q99": np.array([0.5]),  # Same value -> division by zero case
+        },
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {
+        "observation.state": torch.tensor([0.5]),
+    }
+    transition = create_transition(observation=observation)
+
+    # Should not crash and should handle gracefully
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    # When quantiles are identical, should normalize to 0 (due to epsilon handling)
+    assert torch.isfinite(normalized_obs["observation.state"]).all()
+
+
+def test_quantile_partial_stats():
+    """Test that quantile normalization handles missing quantile stats by raising."""
+    features = {
+        "observation.state": PolicyFeature(FeatureType.STATE, (2,)),
+    }
+    norm_map = {
+        FeatureType.STATE: NormalizationMode.QUANTILES,
+    }
+
+    # Missing q99 - should pass through unchanged
+    stats_partial = {
+        "observation.state": {
+            "q01": np.array([0.1, -0.8]),  # Only q01, missing q99
+        },
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats_partial)
+
+    observation = {
+        "observation.state": torch.tensor([0.5, 0.0]),
+    }
+    transition = create_transition(observation=observation)
+
+    with pytest.raises(ValueError, match="QUANTILES normalization mode requires q01 and q99 stats"):
+        _ = normalizer(transition)
+
+
+def test_quantile_mixed_with_other_modes():
+    """Test quantile normalization mixed with other normalization modes."""
+    features = {
+        "observation.image": PolicyFeature(FeatureType.VISUAL, (3,)),
+        "observation.state": PolicyFeature(FeatureType.STATE, (2,)),
+        "action": PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,  # Standard normalization
+        FeatureType.STATE: NormalizationMode.QUANTILES,  # Quantile normalization
+        FeatureType.ACTION: NormalizationMode.QUANTILE10,  # Different quantile mode
+    }
+    stats = {
+        "observation.image": {"mean": [0.5, 0.5, 0.5], "std": [0.2, 0.2, 0.2]},
+        "observation.state": {"q01": [0.1, -0.8], "q99": [0.9, 0.8]},
+        "action": {"q10": [0.2, -0.6], "q90": [0.8, 0.6]},
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {
+        "observation.image": torch.tensor([0.7, 0.5, 0.3]),
+        "observation.state": torch.tensor([0.5, 0.0]),  # Should use QUANTILES
+    }
+    action = torch.tensor([0.5, 0.0])  # Should use QUANTILE10
+    transition = create_transition(observation=observation, action=action)
+
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+    normalized_action = normalized_transition[TransitionKey.ACTION]
+
+    # Image should be mean/std normalized: (0.7 - 0.5) / 0.2 = 1.0, etc.
+    expected_image = (torch.tensor([0.7, 0.5, 0.3]) - 0.5) / 0.2
+    assert torch.allclose(normalized_obs["observation.image"], expected_image)
+
+    # State should be quantile normalized: 2 * (0.5 - 0.1) / (0.9 - 0.1) - 1 = 0.0, etc.
+    expected_state = torch.tensor([0.0, 0.0])
+    assert torch.allclose(normalized_obs["observation.state"], expected_state, atol=1e-6)
+
+    # Action should be quantile10 normalized: 2 * (0.5 - 0.2) / (0.8 - 0.2) - 1 = 0.0, etc.
+    expected_action = torch.tensor([0.0, 0.0])
+    assert torch.allclose(normalized_action, expected_action, atol=1e-6)
+
+
+def test_quantile_with_missing_stats():
+    """Test that quantile normalization handles completely missing stats gracefully."""
+    features = {
+        "observation.state": PolicyFeature(FeatureType.STATE, (2,)),
+    }
+    norm_map = {
+        FeatureType.STATE: NormalizationMode.QUANTILES,
+    }
+    stats = {}  # No stats provided
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {
+        "observation.state": torch.tensor([0.5, 0.0]),
+    }
+    transition = create_transition(observation=observation)
+
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    # Should pass through unchanged when no stats available
+    assert torch.allclose(normalized_obs["observation.state"], observation["observation.state"])
+
+
+def test_selective_normalization(observation_stats):
+    features = _create_observation_features()
+    norm_map = _create_observation_norm_map()
+    normalizer = NormalizerProcessorStep(
+        features=features,
+        norm_map=norm_map,
+        stats=observation_stats,
+        normalize_observation_keys={OBS_IMAGE},
+    )
+
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([0.5, 0.0]),
+    }
+    transition = create_transition(observation=observation)
+
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    # Only image should be normalized
+    assert torch.allclose(normalized_obs[OBS_IMAGE], (torch.tensor([0.7, 0.5, 0.3]) - 0.5) / 0.2)
+    # State should remain unchanged
+    assert torch.allclose(normalized_obs[OBS_STATE], observation[OBS_STATE])
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_device_compatibility(observation_stats):
+    features = _create_observation_features()
+    norm_map = _create_observation_norm_map()
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=observation_stats)
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]).cuda(),
+    }
+    transition = create_transition(observation=observation)
+
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    assert normalized_obs[OBS_IMAGE].device.type == "cuda"
+
+
+def test_from_lerobot_dataset():
+    # Mock dataset
+    mock_dataset = Mock()
+    mock_dataset.meta.stats = {
+        OBS_IMAGE: {"mean": [0.5], "std": [0.2]},
+        ACTION: {"mean": [0.0], "std": [1.0]},
+    }
+
+    features = {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96)),
+        ACTION: PolicyFeature(FeatureType.ACTION, (1,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+
+    normalizer = NormalizerProcessorStep.from_lerobot_dataset(mock_dataset, features, norm_map)
+
+    # Both observation and action statistics should be present in tensor stats
+    assert OBS_IMAGE in normalizer._tensor_stats
+    assert ACTION in normalizer._tensor_stats
+
+
+def test_state_dict_save_load(observation_normalizer):
+    # Save state
+    state_dict = observation_normalizer.state_dict()
+    print("State dict:", state_dict)
+
+    # Create new normalizer and load state
+    features = _create_observation_features()
+    norm_map = _create_observation_norm_map()
+    new_normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats={})
+    new_normalizer.load_state_dict(state_dict)
+
+    # Test that it works the same
+    observation = {OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3])}
+    transition = create_transition(observation=observation)
+
+    result1 = observation_normalizer(transition)[TransitionKey.OBSERVATION]
+    result2 = new_normalizer(transition)[TransitionKey.OBSERVATION]
+
+    assert torch.allclose(result1[OBS_IMAGE], result2[OBS_IMAGE])
+
+
+# Fixtures for ActionUnnormalizer tests
+@pytest.fixture
+def action_stats_mean_std():
+    return {
+        "mean": np.array([0.0, 0.0, 0.0]),
+        "std": np.array([1.0, 2.0, 0.5]),
+    }
+
+
+@pytest.fixture
+def action_stats_min_max():
+    return {
+        "min": np.array([-1.0, -2.0, 0.0]),
+        "max": np.array([1.0, 2.0, 1.0]),
+    }
+
+
+def _create_action_features():
+    return {
+        ACTION: PolicyFeature(FeatureType.ACTION, (3,)),
+    }
+
+
+def _create_action_norm_map_mean_std():
+    return {
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+
+
+def _create_action_norm_map_min_max():
+    return {
+        FeatureType.ACTION: NormalizationMode.MIN_MAX,
+    }
+
+
+def test_mean_std_unnormalization(action_stats_mean_std):
+    features = _create_action_features()
+    norm_map = _create_action_norm_map_mean_std()
+    unnormalizer = UnnormalizerProcessorStep(
+        features=features, norm_map=norm_map, stats={ACTION: action_stats_mean_std}
+    )
+
+    normalized_action = torch.tensor([1.0, -0.5, 2.0])
+    transition = create_transition(action=normalized_action)
+
+    unnormalized_transition = unnormalizer(transition)
+    unnormalized_action = unnormalized_transition[TransitionKey.ACTION]
+
+    # action * std + mean
+    expected = torch.tensor([1.0 * 1.0 + 0.0, -0.5 * 2.0 + 0.0, 2.0 * 0.5 + 0.0])
+    assert torch.allclose(unnormalized_action, expected)
+
+
+def test_min_max_unnormalization(action_stats_min_max):
+    features = _create_action_features()
+    norm_map = _create_action_norm_map_min_max()
+    unnormalizer = UnnormalizerProcessorStep(
+        features=features, norm_map=norm_map, stats={ACTION: action_stats_min_max}
+    )
+
+    # Actions in [-1, 1]
+    normalized_action = torch.tensor([0.0, -1.0, 1.0])
+    transition = create_transition(action=normalized_action)
+
+    unnormalized_transition = unnormalizer(transition)
+    unnormalized_action = unnormalized_transition[TransitionKey.ACTION]
+
+    # Map from [-1, 1] to [min, max]
+    # (action + 1) / 2 * (max - min) + min
+    expected = torch.tensor(
+        [
+            (0.0 + 1) / 2 * (1.0 - (-1.0)) + (-1.0),  # 0.0
+            (-1.0 + 1) / 2 * (2.0 - (-2.0)) + (-2.0),  # -2.0
+            (1.0 + 1) / 2 * (1.0 - 0.0) + 0.0,  # 1.0
+        ]
+    )
+    assert torch.allclose(unnormalized_action, expected)
+
+
+def test_tensor_action_input(action_stats_mean_std):
+    features = _create_action_features()
+    norm_map = _create_action_norm_map_mean_std()
+    unnormalizer = UnnormalizerProcessorStep(
+        features=features, norm_map=norm_map, stats={ACTION: action_stats_mean_std}
+    )
+
+    normalized_action = torch.tensor([1.0, -0.5, 2.0], dtype=torch.float32)
+    transition = create_transition(action=normalized_action)
+
+    unnormalized_transition = unnormalizer(transition)
+    unnormalized_action = unnormalized_transition[TransitionKey.ACTION]
+
+    assert isinstance(unnormalized_action, torch.Tensor)
+    expected = torch.tensor([1.0, -1.0, 1.0])
+    assert torch.allclose(unnormalized_action, expected)
+
+
+def test_none_action(action_stats_mean_std):
+    features = _create_action_features()
+    norm_map = _create_action_norm_map_mean_std()
+    unnormalizer = UnnormalizerProcessorStep(
+        features=features, norm_map=norm_map, stats={ACTION: action_stats_mean_std}
+    )
+
+    transition = create_transition()
+    result = unnormalizer(transition)
+
+    # Should return transition unchanged
+    assert result == transition
+
+
+def test_action_from_lerobot_dataset():
+    mock_dataset = Mock()
+    mock_dataset.meta.stats = {ACTION: {"mean": [0.0], "std": [1.0]}}
+    features = {ACTION: PolicyFeature(FeatureType.ACTION, (1,))}
+    norm_map = {FeatureType.ACTION: NormalizationMode.MEAN_STD}
+    unnormalizer = UnnormalizerProcessorStep.from_lerobot_dataset(mock_dataset, features, norm_map)
+    assert "mean" in unnormalizer._tensor_stats[ACTION]
+
+
+# Fixtures for NormalizerProcessorStep tests
+@pytest.fixture
+def full_stats():
+    return {
+        OBS_IMAGE: {
+            "mean": np.array([0.5, 0.5, 0.5]),
+            "std": np.array([0.2, 0.2, 0.2]),
+        },
+        OBS_STATE: {
+            "min": np.array([0.0, -1.0]),
+            "max": np.array([1.0, 1.0]),
+        },
+        ACTION: {
+            "mean": np.array([0.0, 0.0]),
+            "std": np.array([1.0, 2.0]),
+        },
+    }
+
+
+def _create_full_features():
+    return {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96)),
+        OBS_STATE: PolicyFeature(FeatureType.STATE, (2,)),
+        ACTION: PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+
+
+def _create_full_norm_map():
+    return {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.STATE: NormalizationMode.MIN_MAX,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+
+
+@pytest.fixture
+def normalizer_processor(full_stats):
+    features = _create_full_features()
+    norm_map = _create_full_norm_map()
+    return NormalizerProcessorStep(features=features, norm_map=norm_map, stats=full_stats)
+
+
+def test_combined_normalization(normalizer_processor):
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([0.5, 0.0]),
+    }
+    action = torch.tensor([1.0, -0.5])
+    transition = create_transition(
+        observation=observation,
+        action=action,
+        reward=1.0,
+        done=False,
+        truncated=False,
+        info={},
+        complementary_data={},
+    )
+
+    processed_transition = normalizer_processor(transition)
+
+    # Check normalized observations
+    processed_obs = processed_transition[TransitionKey.OBSERVATION]
+    expected_image = (torch.tensor([0.7, 0.5, 0.3]) - 0.5) / 0.2
+    assert torch.allclose(processed_obs[OBS_IMAGE], expected_image)
+
+    # Check normalized action
+    processed_action = processed_transition[TransitionKey.ACTION]
+    expected_action = torch.tensor([(1.0 - 0.0) / 1.0, (-0.5 - 0.0) / 2.0])
+    assert torch.allclose(processed_action, expected_action)
+
+    # Check other fields remain unchanged
+    assert processed_transition[TransitionKey.REWARD] == 1.0
+    assert not processed_transition[TransitionKey.DONE]
+
+
+def test_processor_from_lerobot_dataset(full_stats):
+    # Mock dataset
+    mock_dataset = Mock()
+    mock_dataset.meta.stats = full_stats
+
+    features = _create_full_features()
+    norm_map = _create_full_norm_map()
+
+    processor = NormalizerProcessorStep.from_lerobot_dataset(
+        mock_dataset, features, norm_map, normalize_observation_keys={OBS_IMAGE}
+    )
+
+    assert processor.normalize_observation_keys == {OBS_IMAGE}
+    assert OBS_IMAGE in processor._tensor_stats
+    assert ACTION in processor._tensor_stats
+
+
+def test_get_config(full_stats):
+    features = _create_full_features()
+    norm_map = _create_full_norm_map()
+    processor = NormalizerProcessorStep(
+        features=features,
+        norm_map=norm_map,
+        stats=full_stats,
+        normalize_observation_keys={OBS_IMAGE},
+        eps=1e-6,
+    )
+
+    config = processor.get_config()
+    expected_config = {
+        "normalize_observation_keys": [OBS_IMAGE],
+        "eps": 1e-6,
+        "features": {
+            OBS_IMAGE: {"type": "VISUAL", "shape": (3, 96, 96)},
+            OBS_STATE: {"type": "STATE", "shape": (2,)},
+            ACTION: {"type": "ACTION", "shape": (2,)},
+        },
+        "norm_map": {
+            "VISUAL": "MEAN_STD",
+            "STATE": "MIN_MAX",
+            "ACTION": "MEAN_STD",
+        },
+    }
+    assert config == expected_config
+
+
+def test_integration_with_robot_processor(normalizer_processor):
+    """Test integration with RobotProcessor pipeline"""
+    robot_processor = DataProcessorPipeline(
+        [normalizer_processor], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([0.5, 0.0]),
+    }
+    action = torch.tensor([1.0, -0.5])
+    transition = create_transition(
+        observation=observation,
+        action=action,
+        reward=1.0,
+        done=False,
+        truncated=False,
+        info={},
+        complementary_data={},
+    )
+
+    processed_transition = robot_processor(transition)
+
+    # Verify the processing worked
+    assert isinstance(processed_transition[TransitionKey.OBSERVATION], dict)
+    assert isinstance(processed_transition[TransitionKey.ACTION], torch.Tensor)
+
+
+# Edge case tests
+def test_empty_observation():
+    stats = {OBS_IMAGE: {"mean": [0.5], "std": [0.2]}}
+    features = {OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96))}
+    norm_map = {FeatureType.VISUAL: NormalizationMode.MEAN_STD}
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    transition = create_transition()
+    result = normalizer(transition)
+
+    assert result == transition
+
+
+def test_empty_stats():
+    features = {OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96))}
+    norm_map = {FeatureType.VISUAL: NormalizationMode.MEAN_STD}
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats={})
+    observation = {OBS_IMAGE: torch.tensor([0.5])}
+    transition = create_transition(observation=observation)
+
+    result = normalizer(transition)
+    # Should return observation unchanged since no stats are available
+    assert torch.allclose(result[TransitionKey.OBSERVATION][OBS_IMAGE], observation[OBS_IMAGE])
+
+
+def test_partial_stats():
+    """If statistics are incomplete, we should raise."""
+    stats = {OBS_IMAGE: {"mean": [0.5]}}  # Missing std / (min,max)
+    features = {OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96))}
+    norm_map = {FeatureType.VISUAL: NormalizationMode.MEAN_STD}
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+    observation = {OBS_IMAGE: torch.tensor([0.7])}
+    transition = create_transition(observation=observation)
+
+    with pytest.raises(ValueError, match="MEAN_STD normalization mode requires mean and std stats"):
+        _ = normalizer(transition)[TransitionKey.OBSERVATION]
+
+
+def test_missing_action_stats_no_error():
+    mock_dataset = Mock()
+    mock_dataset.meta.stats = {OBS_IMAGE: {"mean": [0.5], "std": [0.2]}}
+
+    features = {OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96))}
+    norm_map = {FeatureType.VISUAL: NormalizationMode.MEAN_STD}
+
+    processor = UnnormalizerProcessorStep.from_lerobot_dataset(mock_dataset, features, norm_map)
+    # The tensor stats should not contain the 'action' key
+    assert ACTION not in processor._tensor_stats
+
+
+def test_serialization_roundtrip(full_stats):
+    """Test that features and norm_map can be serialized and deserialized correctly."""
+    features = _create_full_features()
+    norm_map = _create_full_norm_map()
+    original_processor = NormalizerProcessorStep(
+        features=features,
+        norm_map=norm_map,
+        stats=full_stats,
+        normalize_observation_keys={OBS_IMAGE},
+        eps=1e-6,
+    )
+
+    # Get config (serialization)
+    config = original_processor.get_config()
+
+    # Create a new processor from the config (deserialization)
+    new_processor = NormalizerProcessorStep(
+        features=config["features"],
+        norm_map=config["norm_map"],
+        stats=full_stats,
+        normalize_observation_keys=set(config["normalize_observation_keys"]),
+        eps=config["eps"],
+    )
+
+    # Test that both processors work the same way
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([0.5, 0.0]),
+    }
+    action = torch.tensor([1.0, -0.5])
+    transition = create_transition(
+        observation=observation,
+        action=action,
+        reward=1.0,
+        done=False,
+        truncated=False,
+        info={},
+        complementary_data={},
+    )
+
+    result1 = original_processor(transition)
+    result2 = new_processor(transition)
+
+    # Compare results
+    assert torch.allclose(
+        result1[TransitionKey.OBSERVATION][OBS_IMAGE],
+        result2[TransitionKey.OBSERVATION][OBS_IMAGE],
+    )
+    assert torch.allclose(result1[TransitionKey.ACTION], result2[TransitionKey.ACTION])
+
+    # Verify features and norm_map are correctly reconstructed
+    assert (
+        new_processor.transform_features(features).keys()
+        == original_processor.transform_features(features).keys()
+    )
+    for key in new_processor.transform_features(features):
+        assert (
+            new_processor.transform_features(features)[key].type
+            == original_processor.transform_features(features)[key].type
+        )
+        assert (
+            new_processor.transform_features(features)[key].shape
+            == original_processor.transform_features(features)[key].shape
+        )
+
+    assert new_processor.norm_map == original_processor.norm_map
+
+
+# Identity normalization tests
+def test_identity_normalization_observations():
+    """Test that IDENTITY mode skips normalization for observations."""
+    features = {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96)),
+        OBS_STATE: PolicyFeature(FeatureType.STATE, (2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,  # IDENTITY mode
+        FeatureType.STATE: NormalizationMode.MEAN_STD,  # Normal mode for comparison
+    }
+    stats = {
+        OBS_IMAGE: {"mean": [0.5, 0.5, 0.5], "std": [0.2, 0.2, 0.2]},
+        OBS_STATE: {"mean": [0.0, 0.0], "std": [1.0, 1.0]},
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([1.0, -0.5]),
+    }
+    transition = create_transition(observation=observation)
+
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    # Image should remain unchanged (IDENTITY)
+    assert torch.allclose(normalized_obs[OBS_IMAGE], observation[OBS_IMAGE])
+
+    # State should be normalized (MEAN_STD)
+    expected_state = (torch.tensor([1.0, -0.5]) - torch.tensor([0.0, 0.0])) / torch.tensor([1.0, 1.0])
+    assert torch.allclose(normalized_obs[OBS_STATE], expected_state)
+
+
+def test_identity_normalization_actions():
+    """Test that IDENTITY mode skips normalization for actions."""
+    features = {ACTION: PolicyFeature(FeatureType.ACTION, (2,))}
+    norm_map = {FeatureType.ACTION: NormalizationMode.IDENTITY}
+    stats = {ACTION: {"mean": [0.0, 0.0], "std": [1.0, 2.0]}}
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    action = torch.tensor([1.0, -0.5])
+    transition = create_transition(action=action)
+
+    normalized_transition = normalizer(transition)
+
+    # Action should remain unchanged
+    assert torch.allclose(normalized_transition[TransitionKey.ACTION], action)
+
+
+def test_identity_unnormalization_observations():
+    """Test that IDENTITY mode skips unnormalization for observations."""
+    features = {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96)),
+        OBS_STATE: PolicyFeature(FeatureType.STATE, (2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,  # IDENTITY mode
+        FeatureType.STATE: NormalizationMode.MIN_MAX,  # Normal mode for comparison
+    }
+    stats = {
+        OBS_IMAGE: {"mean": [0.5, 0.5, 0.5], "std": [0.2, 0.2, 0.2]},
+        OBS_STATE: {"min": [-1.0, -1.0], "max": [1.0, 1.0]},
+    }
+
+    unnormalizer = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([0.0, -1.0]),  # Normalized values in [-1, 1]
+    }
+    transition = create_transition(observation=observation)
+
+    unnormalized_transition = unnormalizer(transition)
+    unnormalized_obs = unnormalized_transition[TransitionKey.OBSERVATION]
+
+    # Image should remain unchanged (IDENTITY)
+    assert torch.allclose(unnormalized_obs[OBS_IMAGE], observation[OBS_IMAGE])
+
+    # State should be unnormalized (MIN_MAX)
+    # (0.0 + 1) / 2 * (1.0 - (-1.0)) + (-1.0) = 0.0
+    # (-1.0 + 1) / 2 * (1.0 - (-1.0)) + (-1.0) = -1.0
+    expected_state = torch.tensor([0.0, -1.0])
+    assert torch.allclose(unnormalized_obs[OBS_STATE], expected_state)
+
+
+def test_identity_unnormalization_actions():
+    """Test that IDENTITY mode skips unnormalization for actions."""
+    features = {ACTION: PolicyFeature(FeatureType.ACTION, (2,))}
+    norm_map = {FeatureType.ACTION: NormalizationMode.IDENTITY}
+    stats = {ACTION: {"min": [-1.0, -2.0], "max": [1.0, 2.0]}}
+
+    unnormalizer = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    action = torch.tensor([0.5, -0.8])  # Normalized values
+    transition = create_transition(action=action)
+
+    unnormalized_transition = unnormalizer(transition)
+
+    # Action should remain unchanged
+    assert torch.allclose(unnormalized_transition[TransitionKey.ACTION], action)
+
+
+def test_identity_with_missing_stats():
+    """Test that IDENTITY mode works even when stats are missing."""
+    features = {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96)),
+        ACTION: PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,
+        FeatureType.ACTION: NormalizationMode.IDENTITY,
+    }
+    stats = {}  # No stats provided
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+    unnormalizer = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3])}
+    action = torch.tensor([1.0, -0.5])
+    transition = create_transition(observation=observation, action=action)
+
+    # Both should work without errors and return unchanged data
+    normalized_transition = normalizer(transition)
+    unnormalized_transition = unnormalizer(transition)
+
+    assert torch.allclose(
+        normalized_transition[TransitionKey.OBSERVATION][OBS_IMAGE],
+        observation[OBS_IMAGE],
+    )
+    assert torch.allclose(normalized_transition[TransitionKey.ACTION], action)
+    assert torch.allclose(
+        unnormalized_transition[TransitionKey.OBSERVATION][OBS_IMAGE],
+        observation[OBS_IMAGE],
+    )
+    assert torch.allclose(unnormalized_transition[TransitionKey.ACTION], action)
+
+
+def test_identity_mixed_with_other_modes():
+    """Test IDENTITY mode mixed with other normalization modes."""
+    features = {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3,)),
+        OBS_STATE: PolicyFeature(FeatureType.STATE, (2,)),
+        ACTION: PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,
+        FeatureType.STATE: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MIN_MAX,
+    }
+    stats = {
+        OBS_IMAGE: {"mean": [0.5, 0.5, 0.5], "std": [0.2, 0.2, 0.2]},  # Will be ignored
+        OBS_STATE: {"mean": [0.0, 0.0], "std": [1.0, 1.0]},
+        ACTION: {"min": [-1.0, -1.0], "max": [1.0, 1.0]},
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([1.0, -0.5]),
+    }
+    action = torch.tensor([0.5, 0.0])
+    transition = create_transition(observation=observation, action=action)
+
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+    normalized_action = normalized_transition[TransitionKey.ACTION]
+
+    # Image should remain unchanged (IDENTITY)
+    assert torch.allclose(normalized_obs[OBS_IMAGE], observation[OBS_IMAGE])
+
+    # State should be normalized (MEAN_STD)
+    expected_state = torch.tensor([1.0, -0.5])  # (x - 0) / 1 = x
+    assert torch.allclose(normalized_obs[OBS_STATE], expected_state)
+
+    # Action should be normalized (MIN_MAX) to [-1, 1]
+    # 2 * (0.5 - (-1)) / (1 - (-1)) - 1 = 2 * 1.5 / 2 - 1 = 0.5
+    # 2 * (0.0 - (-1)) / (1 - (-1)) - 1 = 2 * 1.0 / 2 - 1 = 0.0
+    expected_action = torch.tensor([0.5, 0.0])
+    assert torch.allclose(normalized_action, expected_action)
+
+
+def test_identity_defaults_when_not_in_norm_map():
+    """Test that IDENTITY is used as default when feature type not in norm_map."""
+    features = {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3,)),
+        OBS_STATE: PolicyFeature(FeatureType.STATE, (2,)),
+    }
+    norm_map = {
+        FeatureType.STATE: NormalizationMode.MEAN_STD,
+        # VISUAL not specified, should default to IDENTITY
+    }
+    stats = {
+        OBS_IMAGE: {"mean": [0.5, 0.5, 0.5], "std": [0.2, 0.2, 0.2]},
+        OBS_STATE: {"mean": [0.0, 0.0], "std": [1.0, 1.0]},
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([1.0, -0.5]),
+    }
+    transition = create_transition(observation=observation)
+
+    normalized_transition = normalizer(transition)
+    normalized_obs = normalized_transition[TransitionKey.OBSERVATION]
+
+    # Image should remain unchanged (defaults to IDENTITY)
+    assert torch.allclose(normalized_obs[OBS_IMAGE], observation[OBS_IMAGE])
+
+    # State should be normalized (explicitly MEAN_STD)
+    expected_state = torch.tensor([1.0, -0.5])
+    assert torch.allclose(normalized_obs[OBS_STATE], expected_state)
+
+
+def test_identity_roundtrip():
+    """Test that IDENTITY normalization and unnormalization are true inverses."""
+    features = {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3,)),
+        ACTION: PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,
+        FeatureType.ACTION: NormalizationMode.IDENTITY,
+    }
+    stats = {
+        OBS_IMAGE: {"mean": [0.5, 0.5, 0.5], "std": [0.2, 0.2, 0.2]},
+        ACTION: {"min": [-1.0, -1.0], "max": [1.0, 1.0]},
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+    unnormalizer = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    original_observation = {OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3])}
+    original_action = torch.tensor([0.5, -0.2])
+    original_transition = create_transition(observation=original_observation, action=original_action)
+
+    # Normalize then unnormalize
+    normalized = normalizer(original_transition)
+    roundtrip = unnormalizer(normalized)
+
+    # Should be identical to original
+    assert torch.allclose(roundtrip[TransitionKey.OBSERVATION][OBS_IMAGE], original_observation[OBS_IMAGE])
+    assert torch.allclose(roundtrip[TransitionKey.ACTION], original_action)
+
+
+def test_identity_config_serialization():
+    """Test that IDENTITY mode is properly saved and loaded in config."""
+    features = {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3,)),
+        ACTION: PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+    stats = {
+        OBS_IMAGE: {"mean": [0.5], "std": [0.2]},
+        ACTION: {"mean": [0.0, 0.0], "std": [1.0, 1.0]},
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    # Get config
+    config = normalizer.get_config()
+
+    # Check that IDENTITY is properly serialized
+    assert config["norm_map"]["VISUAL"] == "IDENTITY"
+    assert config["norm_map"]["ACTION"] == "MEAN_STD"
+
+    # Create new processor from config (simulating load)
+    new_normalizer = NormalizerProcessorStep(
+        features=config["features"],
+        norm_map=config["norm_map"],
+        stats=stats,
+        eps=config["eps"],
+    )
+
+    # Test that both work the same way
+    observation = {OBS_IMAGE: torch.tensor([0.7])}
+    action = torch.tensor([1.0, -0.5])
+    transition = create_transition(observation=observation, action=action)
+
+    result1 = normalizer(transition)
+    result2 = new_normalizer(transition)
+
+    # Results should be identical
+    assert torch.allclose(
+        result1[TransitionKey.OBSERVATION][OBS_IMAGE],
+        result2[TransitionKey.OBSERVATION][OBS_IMAGE],
+    )
+    assert torch.allclose(result1[TransitionKey.ACTION], result2[TransitionKey.ACTION])
+
+
+# def test_unsupported_normalization_mode_error():
+#     """Test that unsupported normalization modes raise appropriate errors."""
+#     features = {OBS_STATE: PolicyFeature(FeatureType.STATE, (2,))}
+
+#     # Create an invalid norm_map (this would never happen in practice, but tests error handling)
+#     from enum import Enum
+
+#     class InvalidMode(str, Enum):
+#         INVALID = "INVALID"
+
+#     # We can't actually pass an invalid enum to the processor due to type checking,
+#     # but we can test the error by manipulating the norm_map after creation
+#     norm_map = {FeatureType.STATE: NormalizationMode.MEAN_STD}
+#     stats = {OBS_STATE: {"mean": [0.0, 0.0], "std": [1.0, 1.0]}}
+
+#     normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+#     # Manually inject an invalid mode to test error handling
+#     normalizer.norm_map[FeatureType.STATE] = "INVALID_MODE"
+
+#     observation = {OBS_STATE: torch.tensor([1.0, -0.5])}
+#     transition = create_transition(observation=observation)
+
+#     with pytest.raises(ValueError, match="Unsupported normalization mode"):
+#         normalizer(transition)
+
+
+def test_hotswap_stats_basic_functionality():
+    """Test that hotswap_stats correctly updates stats in normalizer/unnormalizer steps."""
+    # Create initial stats
+    initial_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5, 0.5, 0.5]), "std": np.array([0.2, 0.2, 0.2])},
+        ACTION: {"mean": np.array([0.0, 0.0]), "std": np.array([1.0, 1.0])},
+    }
+
+    # Create new stats for hotswapping
+    new_stats = {
+        OBS_IMAGE: {"mean": np.array([0.3, 0.3, 0.3]), "std": np.array([0.1, 0.1, 0.1])},
+        ACTION: {"mean": np.array([0.1, 0.1]), "std": np.array([0.5, 0.5])},
+    }
+
+    # Create features and norm_map
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+
+    # Create processors
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+    unnormalizer = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+    identity = IdentityProcessorStep()
+
+    # Create robot processor
+    robot_processor = DataProcessorPipeline(steps=[normalizer, unnormalizer, identity])
+
+    # Hotswap stats
+    new_processor = hotswap_stats(robot_processor, new_stats)
+
+    # Check that normalizer and unnormalizer have new stats
+    assert new_processor.steps[0].stats == new_stats
+    assert new_processor.steps[1].stats == new_stats
+
+    # Check that tensor stats are updated correctly
+    expected_tensor_stats = to_tensor(new_stats)
+    for key in expected_tensor_stats:
+        for stat_name in expected_tensor_stats[key]:
+            torch.testing.assert_close(
+                new_processor.steps[0]._tensor_stats[key][stat_name], expected_tensor_stats[key][stat_name]
+            )
+            torch.testing.assert_close(
+                new_processor.steps[1]._tensor_stats[key][stat_name], expected_tensor_stats[key][stat_name]
+            )
+
+
+def test_hotswap_stats_deep_copy():
+    """Test that hotswap_stats creates a deep copy and doesn't modify the original processor."""
+    initial_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5, 0.5, 0.5]), "std": np.array([0.2, 0.2, 0.2])},
+    }
+
+    new_stats = {
+        OBS_IMAGE: {"mean": np.array([0.3, 0.3, 0.3]), "std": np.array([0.1, 0.1, 0.1])},
+    }
+
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+    }
+    norm_map = {FeatureType.VISUAL: NormalizationMode.MEAN_STD}
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+    original_processor = DataProcessorPipeline(steps=[normalizer])
+
+    # Store reference to original stats
+    original_stats_reference = original_processor.steps[0].stats
+    original_tensor_stats_reference = original_processor.steps[0]._tensor_stats
+
+    # Hotswap stats
+    new_processor = hotswap_stats(original_processor, new_stats)
+
+    # Original processor should be unchanged
+    assert original_processor.steps[0].stats is original_stats_reference
+    assert original_processor.steps[0]._tensor_stats is original_tensor_stats_reference
+    assert original_processor.steps[0].stats == initial_stats
+
+    # New processor should have new stats
+    assert new_processor.steps[0].stats == new_stats
+    assert new_processor.steps[0].stats is not original_stats_reference
+
+    # Processors should be different objects
+    assert new_processor is not original_processor
+    assert new_processor.steps[0] is not original_processor.steps[0]
+
+
+def test_hotswap_stats_only_affects_normalizer_steps():
+    """Test that hotswap_stats only modifies NormalizerProcessorStep and UnnormalizerProcessorStep steps."""
+    stats = {
+        OBS_IMAGE: {"mean": np.array([0.5]), "std": np.array([0.2])},
+    }
+
+    new_stats = {
+        OBS_IMAGE: {"mean": np.array([0.3]), "std": np.array([0.1])},
+    }
+
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+    }
+    norm_map = {FeatureType.VISUAL: NormalizationMode.MEAN_STD}
+
+    # Create mixed steps
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+    unnormalizer = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+    identity = IdentityProcessorStep()
+
+    robot_processor = DataProcessorPipeline(steps=[normalizer, identity, unnormalizer])
+
+    # Hotswap stats
+    new_processor = hotswap_stats(robot_processor, new_stats)
+
+    # Check that only normalizer and unnormalizer steps are affected
+    assert new_processor.steps[0].stats == new_stats  # normalizer
+    assert new_processor.steps[2].stats == new_stats  # unnormalizer
+
+    # Identity processor should remain unchanged (and it doesn't have stats attribute)
+    assert not hasattr(new_processor.steps[1], "stats")
+
+
+def test_hotswap_stats_empty_stats():
+    """Test hotswap_stats with empty stats dictionary."""
+    initial_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5]), "std": np.array([0.2])},
+    }
+
+    empty_stats = {}
+
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+    }
+    norm_map = {FeatureType.VISUAL: NormalizationMode.MEAN_STD}
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+    robot_processor = DataProcessorPipeline(steps=[normalizer])
+
+    # Hotswap with empty stats
+    new_processor = hotswap_stats(robot_processor, empty_stats)
+
+    # Should update to empty stats
+    assert new_processor.steps[0].stats == empty_stats
+    assert new_processor.steps[0]._tensor_stats == {}
+
+
+def test_hotswap_stats_no_normalizer_steps():
+    """Test hotswap_stats with a processor that has no normalizer/unnormalizer steps."""
+    stats = {
+        OBS_IMAGE: {"mean": np.array([0.5]), "std": np.array([0.2])},
+    }
+
+    # Create processor with only identity steps
+    robot_processor = DataProcessorPipeline(steps=[IdentityProcessorStep(), IdentityProcessorStep()])
+
+    # Hotswap stats - should work without error
+    new_processor = hotswap_stats(robot_processor, stats)
+
+    # Should return a different object (deep copy)
+    assert new_processor is not robot_processor
+
+    # Steps should be deep copied but unchanged
+    assert len(new_processor.steps) == len(robot_processor.steps)
+    for i, step in enumerate(new_processor.steps):
+        assert step is not robot_processor.steps[i]  # Different objects
+        assert isinstance(step, type(robot_processor.steps[i]))  # Same type
+
+
+def test_hotswap_stats_preserves_other_attributes():
+    """Test that hotswap_stats preserves other processor attributes like features and norm_map."""
+    initial_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5]), "std": np.array([0.2])},
+    }
+
+    new_stats = {
+        OBS_IMAGE: {"mean": np.array([0.3]), "std": np.array([0.1])},
+    }
+
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+    }
+    norm_map = {FeatureType.VISUAL: NormalizationMode.MEAN_STD}
+    normalize_observation_keys = {OBS_IMAGE}
+    eps = 1e-6
+
+    normalizer = NormalizerProcessorStep(
+        features=features,
+        norm_map=norm_map,
+        stats=initial_stats,
+        normalize_observation_keys=normalize_observation_keys,
+        eps=eps,
+    )
+    robot_processor = DataProcessorPipeline(steps=[normalizer])
+
+    # Hotswap stats
+    new_processor = hotswap_stats(robot_processor, new_stats)
+
+    # Check that other attributes are preserved
+    new_normalizer = new_processor.steps[0]
+    assert new_normalizer.features == features
+    assert new_normalizer.norm_map == norm_map
+    assert new_normalizer.normalize_observation_keys == normalize_observation_keys
+    assert new_normalizer.eps == eps
+
+    # But stats should be updated
+    assert new_normalizer.stats == new_stats
+
+
+def test_hotswap_stats_multiple_normalizer_types():
+    """Test hotswap_stats with multiple normalizer and unnormalizer steps."""
+    initial_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5]), "std": np.array([0.2])},
+        ACTION: {"min": np.array([-1.0]), "max": np.array([1.0])},
+    }
+
+    new_stats = {
+        OBS_IMAGE: {"mean": np.array([0.3]), "std": np.array([0.1])},
+        ACTION: {"min": np.array([-2.0]), "max": np.array([2.0])},
+    }
+
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(1,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MIN_MAX,
+    }
+
+    # Create multiple normalizers and unnormalizers
+    normalizer1 = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+    normalizer2 = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+    unnormalizer1 = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+    unnormalizer2 = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+
+    robot_processor = DataProcessorPipeline(steps=[normalizer1, unnormalizer1, normalizer2, unnormalizer2])
+
+    # Hotswap stats
+    new_processor = hotswap_stats(robot_processor, new_stats)
+
+    # All normalizer/unnormalizer steps should be updated
+    for step in new_processor.steps:
+        assert step.stats == new_stats
+
+        # Check tensor stats conversion
+        expected_tensor_stats = to_tensor(new_stats)
+        for key in expected_tensor_stats:
+            for stat_name in expected_tensor_stats[key]:
+                torch.testing.assert_close(
+                    step._tensor_stats[key][stat_name], expected_tensor_stats[key][stat_name]
+                )
+
+
+def test_hotswap_stats_with_different_data_types():
+    """Test hotswap_stats with various data types in stats."""
+    initial_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5]), "std": np.array([0.2])},
+    }
+
+    # New stats with different data types (int, float, list, tuple)
+    new_stats = {
+        OBS_IMAGE: {
+            "mean": [0.3, 0.4, 0.5],  # list
+            "std": (0.1, 0.2, 0.3),  # tuple
+            "min": 0,  # int
+            "max": 1.0,  # float
+        },
+        ACTION: {
+            "mean": np.array([0.1, 0.2]),  # numpy array
+            "std": torch.tensor([0.5, 0.6]),  # torch tensor
+        },
+    }
+
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+    robot_processor = DataProcessorPipeline(steps=[normalizer])
+
+    # Hotswap stats
+    new_processor = hotswap_stats(robot_processor, new_stats)
+
+    # Check that stats are updated
+    assert new_processor.steps[0].stats == new_stats
+
+    # Check that tensor conversion worked correctly
+    tensor_stats = new_processor.steps[0]._tensor_stats
+    assert isinstance(tensor_stats[OBS_IMAGE]["mean"], torch.Tensor)
+    assert isinstance(tensor_stats[OBS_IMAGE]["std"], torch.Tensor)
+    assert isinstance(tensor_stats[OBS_IMAGE]["min"], torch.Tensor)
+    assert isinstance(tensor_stats[OBS_IMAGE]["max"], torch.Tensor)
+    assert isinstance(tensor_stats[ACTION]["mean"], torch.Tensor)
+    assert isinstance(tensor_stats[ACTION]["std"], torch.Tensor)
+
+    # Check values
+    torch.testing.assert_close(tensor_stats[OBS_IMAGE]["mean"], torch.tensor([0.3, 0.4, 0.5]))
+    torch.testing.assert_close(tensor_stats[OBS_IMAGE]["std"], torch.tensor([0.1, 0.2, 0.3]))
+    torch.testing.assert_close(tensor_stats[OBS_IMAGE]["min"], torch.tensor(0.0))
+    torch.testing.assert_close(tensor_stats[OBS_IMAGE]["max"], torch.tensor(1.0))
+
+
+def test_hotswap_stats_functional_test():
+    """Test that hotswapped processor actually works functionally."""
+    # Create test data
+    observation = {
+        OBS_IMAGE: torch.tensor([[[0.6, 0.7], [0.8, 0.9]], [[0.5, 0.6], [0.7, 0.8]]]),
+    }
+    action = torch.tensor([0.5, -0.5])
+    transition = create_transition(observation=observation, action=action)
+
+    # Initial stats
+    initial_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5, 0.4]), "std": np.array([0.2, 0.3])},
+        ACTION: {"mean": np.array([0.0, 0.0]), "std": np.array([1.0, 1.0])},
+    }
+
+    # New stats
+    new_stats = {
+        OBS_IMAGE: {"mean": np.array([0.3, 0.2]), "std": np.array([0.1, 0.2])},
+        ACTION: {"mean": np.array([0.1, -0.1]), "std": np.array([0.5, 0.5])},
+    }
+
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(2, 2, 2)),
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+
+    # Create original processor
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=initial_stats)
+    original_processor = DataProcessorPipeline(
+        steps=[normalizer], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    # Process with original stats
+    original_result = original_processor(transition)
+
+    # Hotswap stats
+    new_processor = hotswap_stats(original_processor, new_stats)
+
+    # Process with new stats
+    new_result = new_processor(transition)
+
+    # Results should be different since normalization changed
+    assert not torch.allclose(
+        original_result[OBS_STR][OBS_IMAGE],
+        new_result[OBS_STR][OBS_IMAGE],
+        rtol=1e-3,
+        atol=1e-3,
+    )
+    assert not torch.allclose(original_result[ACTION], new_result[ACTION], rtol=1e-3, atol=1e-3)
+
+    # Verify that the new processor is actually using the new stats by checking internal state
+    assert new_processor.steps[0].stats == new_stats
+    assert torch.allclose(new_processor.steps[0]._tensor_stats[OBS_IMAGE]["mean"], torch.tensor([0.3, 0.2]))
+    assert torch.allclose(new_processor.steps[0]._tensor_stats[OBS_IMAGE]["std"], torch.tensor([0.1, 0.2]))
+    assert torch.allclose(new_processor.steps[0]._tensor_stats[ACTION]["mean"], torch.tensor([0.1, -0.1]))
+    assert torch.allclose(new_processor.steps[0]._tensor_stats[ACTION]["std"], torch.tensor([0.5, 0.5]))
+
+    # Test that normalization actually happens (output should not equal input)
+    assert not torch.allclose(new_result[OBS_STR][OBS_IMAGE], observation[OBS_IMAGE])
+    assert not torch.allclose(new_result[ACTION], action)
+
+
+def test_zero_std_uses_eps():
+    """When std == 0, (x-mean)/(std+eps) is well-defined; x==mean should map to 0."""
+    features = {OBS_STATE: PolicyFeature(FeatureType.STATE, (1,))}
+    norm_map = {FeatureType.STATE: NormalizationMode.MEAN_STD}
+    stats = {OBS_STATE: {"mean": np.array([0.5]), "std": np.array([0.0])}}
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats, eps=1e-6)
+
+    observation = {OBS_STATE: torch.tensor([0.5])}  # equals mean
+    out = normalizer(create_transition(observation=observation))
+    assert torch.allclose(out[TransitionKey.OBSERVATION][OBS_STATE], torch.tensor([0.0]))
+
+
+def test_min_equals_max_maps_to_minus_one():
+    """When min == max, MIN_MAX path maps to -1 after [-1,1] scaling for x==min."""
+    features = {OBS_STATE: PolicyFeature(FeatureType.STATE, (1,))}
+    norm_map = {FeatureType.STATE: NormalizationMode.MIN_MAX}
+    stats = {OBS_STATE: {"min": np.array([2.0]), "max": np.array([2.0])}}
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats, eps=1e-6)
+
+    observation = {OBS_STATE: torch.tensor([2.0])}
+    out = normalizer(create_transition(observation=observation))
+    assert torch.allclose(out[TransitionKey.OBSERVATION][OBS_STATE], torch.tensor([-1.0]))
+
+
+def test_action_normalized_despite_normalize_observation_keys():
+    """Action normalization is independent of normalize_observation_keys filter for observations."""
+    features = {
+        OBS_STATE: PolicyFeature(FeatureType.STATE, (1,)),
+        ACTION: PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+    norm_map = {FeatureType.STATE: NormalizationMode.IDENTITY, FeatureType.ACTION: NormalizationMode.MEAN_STD}
+    stats = {ACTION: {"mean": np.array([1.0, -1.0]), "std": np.array([2.0, 4.0])}}
+    normalizer = NormalizerProcessorStep(
+        features=features, norm_map=norm_map, stats=stats, normalize_observation_keys={OBS_STATE}
+    )
+
+    transition = create_transition(
+        observation={OBS_STATE: torch.tensor([3.0])}, action=torch.tensor([3.0, 3.0])
+    )
+    out = normalizer(transition)
+    # (3-1)/2 = 1.0 ; (3-(-1))/4 = 1.0
+    assert torch.allclose(out[TransitionKey.ACTION], torch.tensor([1.0, 1.0]))
+
+
+def test_unnormalize_observations_mean_std_and_min_max():
+    features = {
+        "observation.ms": PolicyFeature(FeatureType.STATE, (2,)),
+        "observation.mm": PolicyFeature(FeatureType.STATE, (2,)),
+    }
+    # Build two processors: one mean/std and one min/max
+    unnorm_ms = UnnormalizerProcessorStep(
+        features={"observation.ms": features["observation.ms"]},
+        norm_map={FeatureType.STATE: NormalizationMode.MEAN_STD},
+        stats={"observation.ms": {"mean": np.array([1.0, -1.0]), "std": np.array([2.0, 4.0])}},
+    )
+    unnorm_mm = UnnormalizerProcessorStep(
+        features={"observation.mm": features["observation.mm"]},
+        norm_map={FeatureType.STATE: NormalizationMode.MIN_MAX},
+        stats={"observation.mm": {"min": np.array([0.0, -2.0]), "max": np.array([2.0, 2.0])}},
+    )
+
+    tr = create_transition(
+        observation={
+            "observation.ms": torch.tensor([0.0, 0.0]),  # → mean
+            "observation.mm": torch.tensor([0.0, 0.0]),  # → mid-point
+        }
+    )
+    out_ms = unnorm_ms(tr)[TransitionKey.OBSERVATION]["observation.ms"]
+    out_mm = unnorm_mm(tr)[TransitionKey.OBSERVATION]["observation.mm"]
+    assert torch.allclose(out_ms, torch.tensor([1.0, -1.0]))
+    assert torch.allclose(out_mm, torch.tensor([1.0, 0.0]))  # mid of [0,2] and [-2,2]
+
+
+def test_unknown_observation_keys_ignored():
+    features = {OBS_STATE: PolicyFeature(FeatureType.STATE, (1,))}
+    norm_map = {FeatureType.STATE: NormalizationMode.MEAN_STD}
+    stats = {OBS_STATE: {"mean": np.array([0.0]), "std": np.array([1.0])}}
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    obs = {OBS_STATE: torch.tensor([1.0]), "observation.unknown": torch.tensor([5.0])}
+    tr = create_transition(observation=obs)
+    out = normalizer(tr)
+
+    # Unknown key should pass through unchanged and not be tracked
+    assert torch.allclose(out[TransitionKey.OBSERVATION]["observation.unknown"], obs["observation.unknown"])
+
+
+def test_batched_action_normalization():
+    features = {ACTION: PolicyFeature(FeatureType.ACTION, (2,))}
+    norm_map = {FeatureType.ACTION: NormalizationMode.MEAN_STD}
+    stats = {ACTION: {"mean": np.array([1.0, -1.0]), "std": np.array([2.0, 4.0])}}
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    actions = torch.tensor([[1.0, -1.0], [3.0, 3.0]])  # first equals mean → zeros; second → [1, 1]
+    out = normalizer(create_transition(action=actions))[TransitionKey.ACTION]
+    expected = torch.tensor([[0.0, 0.0], [1.0, 1.0]])
+    assert torch.allclose(out, expected)
+
+
+def test_complementary_data_preservation():
+    features = {OBS_STATE: PolicyFeature(FeatureType.STATE, (1,))}
+    norm_map = {FeatureType.STATE: NormalizationMode.MEAN_STD}
+    stats = {OBS_STATE: {"mean": np.array([0.0]), "std": np.array([1.0])}}
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    comp = {"existing": 123}
+    tr = create_transition(observation={OBS_STATE: torch.tensor([1.0])}, complementary_data=comp)
+    out = normalizer(tr)
+    new_comp = out[TransitionKey.COMPLEMENTARY_DATA]
+    assert new_comp["existing"] == 123
+
+
+def test_roundtrip_normalize_unnormalize_non_identity():
+    features = {
+        OBS_STATE: PolicyFeature(FeatureType.STATE, (2,)),
+        ACTION: PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+    norm_map = {FeatureType.STATE: NormalizationMode.MEAN_STD, FeatureType.ACTION: NormalizationMode.MIN_MAX}
+    stats = {
+        OBS_STATE: {"mean": np.array([1.0, -1.0]), "std": np.array([2.0, 4.0])},
+        ACTION: {"min": np.array([-2.0, 0.0]), "max": np.array([2.0, 4.0])},
+    }
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+    unnormalizer = UnnormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    # Add a time dimension in action for broadcasting check (B,T,D)
+    obs = {OBS_STATE: torch.tensor([[3.0, 3.0], [1.0, -1.0]])}
+    act = torch.tensor([[[0.0, -1.0], [1.0, 1.0]]])  # shape (1,2,2) already in [-1,1]
+
+    tr = create_transition(observation=obs, action=act)
+    out = unnormalizer(normalizer(tr))
+
+    assert torch.allclose(out[TransitionKey.OBSERVATION][OBS_STATE], obs[OBS_STATE], atol=1e-5)
+    assert torch.allclose(out[TransitionKey.ACTION], act, atol=1e-5)
+
+
+def test_dtype_adaptation_bfloat16_input_float32_normalizer():
+    """Test automatic dtype adaptation: NormalizerProcessor(float32) adapts to bfloat16 input → bfloat16 output"""
+    features = {OBS_STATE: PolicyFeature(FeatureType.STATE, (5,))}
+    norm_map = {FeatureType.STATE: NormalizationMode.MEAN_STD}
+    stats = {
+        OBS_STATE: {
+            "mean": np.array([0.0, 0.0, 0.0, 0.0, 0.0]),
+            "std": np.array([1.0, 1.0, 1.0, 1.0, 1.0]),
+        }
+    }
+
+    # Create normalizer configured with float32 dtype
+    normalizer = NormalizerProcessorStep(
+        features=features, norm_map=norm_map, stats=stats, dtype=torch.float32
+    )
+
+    # Verify initial configuration
+    assert normalizer.dtype == torch.float32
+    for stat_tensor in normalizer._tensor_stats[OBS_STATE].values():
+        assert stat_tensor.dtype == torch.float32
+
+    # Create bfloat16 input tensor
+    observation = {OBS_STATE: torch.tensor([1.0, 2.0, 3.0, 4.0, 5.0], dtype=torch.bfloat16)}
+    transition = create_transition(observation=observation)
+
+    # Process the transition
+    result = normalizer(transition)
+
+    # Verify that:
+    # 1. Stats were automatically adapted to bfloat16
+    assert normalizer.dtype == torch.bfloat16
+    for stat_tensor in normalizer._tensor_stats[OBS_STATE].values():
+        assert stat_tensor.dtype == torch.bfloat16
+
+    # 2. Output is in bfloat16
+    output_tensor = result[TransitionKey.OBSERVATION][OBS_STATE]
+    assert output_tensor.dtype == torch.bfloat16
+
+    # 3. Normalization was applied correctly (mean should be close to original - mean) / std
+    expected = (
+        torch.tensor([1.0, 2.0, 3.0, 4.0, 5.0], dtype=torch.bfloat16)
+        - torch.tensor([0.0, 0.0, 0.0, 0.0, 0.0], dtype=torch.bfloat16)
+    ) / torch.tensor([1.0, 1.0, 1.0, 1.0, 1.0], dtype=torch.bfloat16)
+    assert torch.allclose(output_tensor, expected, atol=1e-2)  # bfloat16 has lower precision
+
+
+def test_stats_override_preservation_in_load_state_dict():
+    """
+    Test that explicitly provided stats are preserved during load_state_dict.
+
+    This tests the fix for the bug where stats provided via overrides were
+    being overwritten when load_state_dict was called.
+    """
+    # Create original stats
+    original_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5, 0.5, 0.5]), "std": np.array([0.2, 0.2, 0.2])},
+        ACTION: {"mean": np.array([0.0, 0.0]), "std": np.array([1.0, 1.0])},
+    }
+
+    # Create override stats (what user wants to use)
+    override_stats = {
+        OBS_IMAGE: {"mean": np.array([0.3, 0.3, 0.3]), "std": np.array([0.1, 0.1, 0.1])},
+        ACTION: {"mean": np.array([0.1, 0.1]), "std": np.array([0.5, 0.5])},
+    }
+
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+
+    # Create a normalizer with original stats and save its state
+    original_normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=original_stats)
+    saved_state_dict = original_normalizer.state_dict()
+
+    # Create a new normalizer with override stats (simulating from_pretrained with overrides)
+    override_normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=override_stats)
+
+    # Verify that the override stats are initially set correctly
+    assert set(override_normalizer.stats.keys()) == set(override_stats.keys())
+    for key in override_stats:
+        assert set(override_normalizer.stats[key].keys()) == set(override_stats[key].keys())
+        for stat_name in override_stats[key]:
+            np.testing.assert_array_equal(
+                override_normalizer.stats[key][stat_name], override_stats[key][stat_name]
+            )
+    assert override_normalizer._stats_explicitly_provided is True
+
+    # This is the critical test: load_state_dict should NOT overwrite the override stats
+    override_normalizer.load_state_dict(saved_state_dict)
+
+    # After loading state_dict, stats should still be the override stats, not the original stats
+    # Check that loaded stats match override stats
+    assert set(override_normalizer.stats.keys()) == set(override_stats.keys())
+    for key in override_stats:
+        assert set(override_normalizer.stats[key].keys()) == set(override_stats[key].keys())
+        for stat_name in override_stats[key]:
+            np.testing.assert_array_equal(
+                override_normalizer.stats[key][stat_name], override_stats[key][stat_name]
+            )
+    # Compare individual arrays to avoid numpy array comparison ambiguity
+    for key in override_stats:
+        for stat_name in override_stats[key]:
+            assert not np.array_equal(
+                override_normalizer.stats[key][stat_name], original_stats[key][stat_name]
+            ), f"Stats for {key}.{stat_name} should not match original stats"
+
+    # Verify that _tensor_stats are also correctly set to match the override stats
+    expected_tensor_stats = to_tensor(override_stats)
+    for key in expected_tensor_stats:
+        for stat_name in expected_tensor_stats[key]:
+            if isinstance(expected_tensor_stats[key][stat_name], torch.Tensor):
+                torch.testing.assert_close(
+                    override_normalizer._tensor_stats[key][stat_name], expected_tensor_stats[key][stat_name]
+                )
+
+
+def test_stats_without_override_loads_normally():
+    """
+    Test that when stats are not explicitly provided (normal case),
+    load_state_dict works as before.
+    """
+    original_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5, 0.5, 0.5]), "std": np.array([0.2, 0.2, 0.2])},
+        ACTION: {"mean": np.array([0.0, 0.0]), "std": np.array([1.0, 1.0])},
+    }
+
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+
+    # Create a normalizer with original stats and save its state
+    original_normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=original_stats)
+    saved_state_dict = original_normalizer.state_dict()
+
+    # Create a new normalizer without stats (simulating normal from_pretrained)
+    new_normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats={})
+
+    # Verify that stats are not explicitly provided
+    assert new_normalizer._stats_explicitly_provided is False
+
+    # Load state dict - this should work normally and load the saved stats
+    new_normalizer.load_state_dict(saved_state_dict)
+
+    # Stats should now match the original stats (normal behavior)
+    # Check that all keys and values match
+    assert set(new_normalizer.stats.keys()) == set(original_stats.keys())
+    for key in original_stats:
+        assert set(new_normalizer.stats[key].keys()) == set(original_stats[key].keys())
+        for stat_name in original_stats[key]:
+            np.testing.assert_allclose(
+                new_normalizer.stats[key][stat_name], original_stats[key][stat_name], rtol=1e-6, atol=1e-6
+            )
+
+
+def test_stats_explicit_provided_flag_detection():
+    """Test that the _stats_explicitly_provided flag is set correctly in different scenarios."""
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 128, 128)),
+    }
+    norm_map = {FeatureType.VISUAL: NormalizationMode.MEAN_STD}
+
+    # Test 1: Explicitly provided stats (non-empty dict)
+    stats = {OBS_IMAGE: {"mean": [0.5], "std": [0.2]}}
+    normalizer1 = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+    assert normalizer1._stats_explicitly_provided is True
+
+    # Test 2: Empty stats dict
+    normalizer2 = NormalizerProcessorStep(features=features, norm_map=norm_map, stats={})
+    assert normalizer2._stats_explicitly_provided is False
+
+    # Test 3: None stats
+    normalizer3 = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=None)
+    assert normalizer3._stats_explicitly_provided is False
+
+    # Test 4: Stats not provided (defaults to None)
+    normalizer4 = NormalizerProcessorStep(features=features, norm_map=norm_map)
+    assert normalizer4._stats_explicitly_provided is False
+
+
+def test_pipeline_from_pretrained_with_stats_overrides():
+    """
+    Test the actual use case: DataProcessorPipeline.from_pretrained with stat overrides.
+
+    This is an integration test that verifies the fix works in the real scenario
+    where users provide stat overrides when loading a pipeline.
+    """
+    import tempfile
+
+    # Create test data
+    features = {
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 32, 32)),
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+
+    original_stats = {
+        OBS_IMAGE: {"mean": np.array([0.5, 0.5, 0.5]), "std": np.array([0.2, 0.2, 0.2])},
+        ACTION: {"mean": np.array([0.0, 0.0]), "std": np.array([1.0, 1.0])},
+    }
+
+    override_stats = {
+        OBS_IMAGE: {"mean": np.array([0.3, 0.3, 0.3]), "std": np.array([0.1, 0.1, 0.1])},
+        ACTION: {"mean": np.array([0.1, 0.1]), "std": np.array([0.5, 0.5])},
+    }
+
+    # Create and save a pipeline with the original stats
+    normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=original_stats)
+    identity = IdentityProcessorStep()
+    original_pipeline = DataProcessorPipeline(steps=[normalizer, identity], name="test_pipeline")
+
+    with tempfile.TemporaryDirectory() as temp_dir:
+        # Save the pipeline
+        original_pipeline.save_pretrained(temp_dir)
+
+        # Load the pipeline with stat overrides
+        overrides = {"normalizer_processor": {"stats": override_stats}}
+
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            temp_dir, config_filename="test_pipeline.json", overrides=overrides
+        )
+
+        # The critical test: the loaded pipeline should use override stats, not original stats
+        loaded_normalizer = loaded_pipeline.steps[0]
+        assert isinstance(loaded_normalizer, NormalizerProcessorStep)
+
+        # Check that loaded stats match override stats
+        assert set(loaded_normalizer.stats.keys()) == set(override_stats.keys())
+        for key in override_stats:
+            assert set(loaded_normalizer.stats[key].keys()) == set(override_stats[key].keys())
+            for stat_name in override_stats[key]:
+                np.testing.assert_array_equal(
+                    loaded_normalizer.stats[key][stat_name], override_stats[key][stat_name]
+                )
+
+        # Verify stats don't match original stats
+        for key in override_stats:
+            for stat_name in override_stats[key]:
+                assert not np.array_equal(
+                    loaded_normalizer.stats[key][stat_name], original_stats[key][stat_name]
+                ), f"Stats for {key}.{stat_name} should not match original stats"
+
+        # Test that the override stats are actually used in processing
+        observation = {
+            OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        }
+        action = torch.tensor([1.0, -0.5])
+        transition = create_transition(observation=observation, action=action)
+
+        # Process with override pipeline
+        override_result = loaded_pipeline(transition)
+
+        # Create a reference pipeline with override stats for comparison
+        reference_normalizer = NormalizerProcessorStep(
+            features=features, norm_map=norm_map, stats=override_stats
+        )
+        reference_pipeline = DataProcessorPipeline(
+            steps=[reference_normalizer, identity],
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+        _ = reference_pipeline(transition)
+
+        # The critical part was verified above: loaded_normalizer.stats == override_stats
+        # This confirms that override stats are preserved during load_state_dict.
+        # Let's just verify the pipeline processes data successfully.
+        assert ACTION in override_result
+        assert isinstance(override_result[ACTION], torch.Tensor)
+
+
+def test_dtype_adaptation_device_processor_bfloat16_normalizer_float32():
+    """Test policy pipeline scenario: DeviceProcessor(bfloat16) + NormalizerProcessor(float32) → bfloat16 output"""
+    from lerobot.processor import DeviceProcessorStep
+
+    features = {OBS_STATE: PolicyFeature(FeatureType.STATE, (3,))}
+    norm_map = {FeatureType.STATE: NormalizationMode.MEAN_STD}
+    stats = {OBS_STATE: {"mean": np.array([0.0, 0.0, 0.0]), "std": np.array([1.0, 1.0, 1.0])}}
+
+    # Create pipeline: DeviceProcessor(bfloat16) → NormalizerProcessor(float32)
+    device_processor = DeviceProcessorStep(device=str(auto_select_torch_device()), float_dtype="bfloat16")
+    normalizer = NormalizerProcessorStep(
+        features=features, norm_map=norm_map, stats=stats, dtype=torch.float32
+    )
+
+    # Verify initial normalizer configuration
+    assert normalizer.dtype == torch.float32
+
+    # Create CPU input
+    observation = {OBS_STATE: torch.tensor([1.0, 2.0, 3.0], dtype=torch.float32)}
+    transition = create_transition(observation=observation)
+
+    # Step 1: DeviceProcessor converts to bfloat16 + moves to CUDA
+    processed_1 = device_processor(transition)
+    intermediate_tensor = processed_1[TransitionKey.OBSERVATION][OBS_STATE]
+    assert intermediate_tensor.dtype == torch.bfloat16
+    assert intermediate_tensor.device.type == str(auto_select_torch_device())
+
+    # Step 2: NormalizerProcessor receives bfloat16 input and adapts
+    final_result = normalizer(processed_1)
+    final_tensor = final_result[TransitionKey.OBSERVATION][OBS_STATE]
+
+    # Verify final output is bfloat16 (automatic adaptation worked)
+    assert final_tensor.dtype == torch.bfloat16
+    assert final_tensor.device.type == str(auto_select_torch_device())
+
+    # Verify normalizer adapted its internal state
+    assert normalizer.dtype == torch.bfloat16
+    for stat_tensor in normalizer._tensor_stats[OBS_STATE].values():
+        assert stat_tensor.dtype == torch.bfloat16
+        assert stat_tensor.device.type == str(auto_select_torch_device())
+
+
+def test_stats_reconstruction_after_load_state_dict():
+    """
+    Test that stats dict is properly reconstructed from _tensor_stats after loading.
+
+    This test ensures the bug where stats became empty after loading is fixed.
+    The bug occurred when:
+    1. Only _tensor_stats were saved via state_dict()
+    2. stats field became empty {} after loading
+    3. Calling to() method or hotswap_stats would fail because they depend on self.stats
+    """
+
+    # Create normalizer with stats
+    features = {
+        OBS_IMAGE: PolicyFeature(FeatureType.VISUAL, (3, 96, 96)),
+        OBS_STATE: PolicyFeature(FeatureType.STATE, (2,)),
+        ACTION: PolicyFeature(FeatureType.ACTION, (2,)),
+    }
+    norm_map = {
+        FeatureType.VISUAL: NormalizationMode.MEAN_STD,
+        FeatureType.STATE: NormalizationMode.MIN_MAX,
+        FeatureType.ACTION: NormalizationMode.MEAN_STD,
+    }
+    stats = {
+        OBS_IMAGE: {
+            "mean": np.array([0.5, 0.5, 0.5]),
+            "std": np.array([0.2, 0.2, 0.2]),
+        },
+        OBS_STATE: {
+            "min": np.array([0.0, -1.0]),
+            "max": np.array([1.0, 1.0]),
+        },
+        ACTION: {
+            "mean": np.array([0.0, 0.0]),
+            "std": np.array([1.0, 2.0]),
+        },
+    }
+
+    original_normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats=stats)
+
+    # Save state dict (simulating save/load)
+    state_dict = original_normalizer.state_dict()
+
+    # Create new normalizer with empty stats (simulating load)
+    new_normalizer = NormalizerProcessorStep(features=features, norm_map=norm_map, stats={})
+
+    # Before fix: this would cause stats to remain empty
+    new_normalizer.load_state_dict(state_dict)
+
+    # Verify that stats dict is properly reconstructed from _tensor_stats
+    assert new_normalizer.stats is not None
+    assert new_normalizer.stats != {}
+
+    # Check that all expected keys are present
+    assert OBS_IMAGE in new_normalizer.stats
+    assert OBS_STATE in new_normalizer.stats
+    assert ACTION in new_normalizer.stats
+
+    # Check that values are correct (converted back from tensors)
+    np.testing.assert_allclose(new_normalizer.stats[OBS_IMAGE]["mean"], [0.5, 0.5, 0.5])
+    np.testing.assert_allclose(new_normalizer.stats[OBS_IMAGE]["std"], [0.2, 0.2, 0.2])
+    np.testing.assert_allclose(new_normalizer.stats[OBS_STATE]["min"], [0.0, -1.0])
+    np.testing.assert_allclose(new_normalizer.stats[OBS_STATE]["max"], [1.0, 1.0])
+    np.testing.assert_allclose(new_normalizer.stats[ACTION]["mean"], [0.0, 0.0])
+    np.testing.assert_allclose(new_normalizer.stats[ACTION]["std"], [1.0, 2.0])
+
+    # Test that methods that depend on self.stats work correctly after loading
+    # This would fail before the bug fix because self.stats was empty
+
+    # Test 1: to() method should work without crashing
+    try:
+        new_normalizer.to(device="cpu", dtype=torch.float32)
+        # If we reach here, the bug is fixed
+    except (KeyError, AttributeError) as e:
+        pytest.fail(f"to() method failed after loading state_dict: {e}")
+
+    # Test 2: hotswap_stats should work
+    new_stats = {
+        OBS_IMAGE: {"mean": [0.3, 0.3, 0.3], "std": [0.1, 0.1, 0.1]},
+        OBS_STATE: {"min": [-1.0, -2.0], "max": [2.0, 2.0]},
+        ACTION: {"mean": [0.1, 0.1], "std": [0.5, 0.5]},
+    }
+
+    pipeline = DataProcessorPipeline([new_normalizer])
+    try:
+        new_pipeline = hotswap_stats(pipeline, new_stats)
+        # If we reach here, hotswap_stats worked correctly
+        assert new_pipeline.steps[0].stats == new_stats
+    except (KeyError, AttributeError) as e:
+        pytest.fail(f"hotswap_stats failed after loading state_dict: {e}")
+
+    # Test 3: The normalizer should work functionally the same as the original
+    observation = {
+        OBS_IMAGE: torch.tensor([0.7, 0.5, 0.3]),
+        OBS_STATE: torch.tensor([0.5, 0.0]),
+    }
+    action = torch.tensor([1.0, -0.5])
+    transition = create_transition(observation=observation, action=action)
+
+    original_result = original_normalizer(transition)
+    new_result = new_normalizer(transition)
+
+    # Results should be identical (within floating point precision)
+    torch.testing.assert_close(
+        original_result[TransitionKey.OBSERVATION][OBS_IMAGE],
+        new_result[TransitionKey.OBSERVATION][OBS_IMAGE],
+    )
+    torch.testing.assert_close(
+        original_result[TransitionKey.OBSERVATION][OBS_STATE],
+        new_result[TransitionKey.OBSERVATION][OBS_STATE],
+    )
+    torch.testing.assert_close(original_result[TransitionKey.ACTION], new_result[TransitionKey.ACTION])
diff --git a/lerobot/tests/processor/test_observation_processor.py b/lerobot/tests/processor/test_observation_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..9230592108af8e92df0c9ce512aebec31fbe4271
--- /dev/null
+++ b/lerobot/tests/processor/test_observation_processor.py
@@ -0,0 +1,528 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import numpy as np
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType
+from lerobot.processor import VanillaObservationProcessorStep
+from lerobot.processor.converters import create_transition
+from lerobot.types import TransitionKey
+from lerobot.utils.constants import OBS_ENV_STATE, OBS_IMAGE, OBS_IMAGES, OBS_STATE
+from tests.conftest import assert_contract_is_typed
+
+
+def test_process_single_image():
+    """Test processing a single image."""
+    processor = VanillaObservationProcessorStep()
+
+    # Create a mock image (H, W, C) format, uint8
+    image = np.random.randint(0, 256, size=(64, 64, 3), dtype=np.uint8)
+
+    observation = {"pixels": image}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that the image was processed correctly
+    assert OBS_IMAGE in processed_obs
+    processed_img = processed_obs[OBS_IMAGE]
+
+    # Check shape: should be (1, 3, 64, 64) - batch, channels, height, width
+    assert processed_img.shape == (1, 3, 64, 64)
+
+    # Check dtype and range
+    assert processed_img.dtype == torch.float32
+    assert processed_img.min() >= 0.0
+    assert processed_img.max() <= 1.0
+
+
+def test_process_image_dict():
+    """Test processing multiple images in a dictionary."""
+    processor = VanillaObservationProcessorStep()
+
+    # Create mock images
+    image1 = np.random.randint(0, 256, size=(32, 32, 3), dtype=np.uint8)
+    image2 = np.random.randint(0, 256, size=(48, 48, 3), dtype=np.uint8)
+
+    observation = {"pixels": {"camera1": image1, "camera2": image2}}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that both images were processed
+    assert f"{OBS_IMAGES}.camera1" in processed_obs
+    assert f"{OBS_IMAGES}.camera2" in processed_obs
+
+    # Check shapes
+    assert processed_obs[f"{OBS_IMAGES}.camera1"].shape == (1, 3, 32, 32)
+    assert processed_obs[f"{OBS_IMAGES}.camera2"].shape == (1, 3, 48, 48)
+
+
+def test_process_batched_image():
+    """Test processing already batched images."""
+    processor = VanillaObservationProcessorStep()
+
+    # Create a batched image (B, H, W, C)
+    image = np.random.randint(0, 256, size=(2, 64, 64, 3), dtype=np.uint8)
+
+    observation = {"pixels": image}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that batch dimension is preserved
+    assert processed_obs[OBS_IMAGE].shape == (2, 3, 64, 64)
+
+
+def test_invalid_image_format():
+    """Test error handling for invalid image formats."""
+    processor = VanillaObservationProcessorStep()
+
+    # Test wrong channel order (channels first)
+    image = np.random.randint(0, 256, size=(3, 64, 64), dtype=np.uint8)
+    observation = {"pixels": image}
+    transition = create_transition(observation=observation)
+
+    with pytest.raises(ValueError, match="Expected channel-last images"):
+        processor(transition)
+
+
+def test_invalid_image_dtype():
+    """Test error handling for invalid image dtype."""
+    processor = VanillaObservationProcessorStep()
+
+    # Test wrong dtype
+    image = np.random.rand(64, 64, 3).astype(np.float32)
+    observation = {"pixels": image}
+    transition = create_transition(observation=observation)
+
+    with pytest.raises(ValueError, match="Expected torch.uint8 images"):
+        processor(transition)
+
+
+def test_no_pixels_in_observation():
+    """Test processor when no pixels are in observation."""
+    processor = VanillaObservationProcessorStep()
+
+    observation = {"other_data": np.array([1, 2, 3])}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Should preserve other data unchanged
+    assert "other_data" in processed_obs
+    np.testing.assert_array_equal(processed_obs["other_data"], np.array([1, 2, 3]))
+
+
+def test_none_observation():
+    """Test processor with None observation."""
+    processor = VanillaObservationProcessorStep()
+
+    transition = create_transition(observation={})
+    result = processor(transition)
+
+    assert result == transition
+
+
+def test_serialization_methods():
+    """Test serialization methods."""
+    processor = VanillaObservationProcessorStep()
+
+    # Test get_config
+    config = processor.get_config()
+    assert isinstance(config, dict)
+
+    # Test state_dict
+    state = processor.state_dict()
+    assert isinstance(state, dict)
+
+    # Test load_state_dict (should not raise)
+    processor.load_state_dict(state)
+
+    # Test reset (should not raise)
+    processor.reset()
+
+
+def test_process_environment_state():
+    """Test processing environment_state."""
+    processor = VanillaObservationProcessorStep()
+
+    env_state = np.array([1.0, 2.0, 3.0], dtype=np.float32)
+    observation = {"environment_state": env_state}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that environment_state was renamed and processed
+    assert OBS_ENV_STATE in processed_obs
+    assert "environment_state" not in processed_obs
+
+    processed_state = processed_obs[OBS_ENV_STATE]
+    assert processed_state.shape == (1, 3)  # Batch dimension added
+    assert processed_state.dtype == torch.float32
+    torch.testing.assert_close(processed_state, torch.tensor([[1.0, 2.0, 3.0]]))
+
+
+def test_process_agent_pos():
+    """Test processing agent_pos."""
+    processor = VanillaObservationProcessorStep()
+
+    agent_pos = np.array([0.5, -0.5, 1.0], dtype=np.float32)
+    observation = {"agent_pos": agent_pos}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that agent_pos was renamed and processed
+    assert OBS_STATE in processed_obs
+    assert "agent_pos" not in processed_obs
+
+    processed_state = processed_obs[OBS_STATE]
+    assert processed_state.shape == (1, 3)  # Batch dimension added
+    assert processed_state.dtype == torch.float32
+    torch.testing.assert_close(processed_state, torch.tensor([[0.5, -0.5, 1.0]]))
+
+
+def test_process_batched_states():
+    """Test processing already batched states."""
+    processor = VanillaObservationProcessorStep()
+
+    env_state = np.array([[1.0, 2.0], [3.0, 4.0]], dtype=np.float32)
+    agent_pos = np.array([[0.5, -0.5], [1.0, -1.0]], dtype=np.float32)
+
+    observation = {"environment_state": env_state, "agent_pos": agent_pos}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that batch dimensions are preserved
+    assert processed_obs[OBS_ENV_STATE].shape == (2, 2)
+    assert processed_obs[OBS_STATE].shape == (2, 2)
+
+
+def test_process_both_states():
+    """Test processing both environment_state and agent_pos."""
+    processor = VanillaObservationProcessorStep()
+
+    env_state = np.array([1.0, 2.0], dtype=np.float32)
+    agent_pos = np.array([0.5, -0.5], dtype=np.float32)
+
+    observation = {"environment_state": env_state, "agent_pos": agent_pos, "other_data": "keep_me"}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that both states were processed
+    assert OBS_ENV_STATE in processed_obs
+    assert OBS_STATE in processed_obs
+
+    # Check that original keys were removed
+    assert "environment_state" not in processed_obs
+    assert "agent_pos" not in processed_obs
+
+    # Check that other data was preserved
+    assert processed_obs["other_data"] == "keep_me"
+
+
+def test_no_states_in_observation():
+    """Test processor when no states are in observation."""
+    processor = VanillaObservationProcessorStep()
+
+    observation = {"other_data": np.array([1, 2, 3])}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Should preserve data unchanged
+    np.testing.assert_array_equal(processed_obs, observation)
+
+
+def test_complete_observation_processing():
+    """Test processing a complete observation with both images and states."""
+    processor = VanillaObservationProcessorStep()
+
+    # Create mock data
+    image = np.random.randint(0, 256, size=(32, 32, 3), dtype=np.uint8)
+    env_state = np.array([1.0, 2.0, 3.0], dtype=np.float32)
+    agent_pos = np.array([0.5, -0.5, 1.0], dtype=np.float32)
+
+    observation = {
+        "pixels": image,
+        "environment_state": env_state,
+        "agent_pos": agent_pos,
+        "other_data": "preserve_me",
+    }
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that image was processed
+    assert OBS_IMAGE in processed_obs
+    assert processed_obs[OBS_IMAGE].shape == (1, 3, 32, 32)
+
+    # Check that states were processed
+    assert OBS_ENV_STATE in processed_obs
+    assert OBS_STATE in processed_obs
+
+    # Check that original keys were removed
+    assert "pixels" not in processed_obs
+    assert "environment_state" not in processed_obs
+    assert "agent_pos" not in processed_obs
+
+    # Check that other data was preserved
+    assert processed_obs["other_data"] == "preserve_me"
+
+
+def test_image_only_processing():
+    """Test processing observation with only images."""
+    processor = VanillaObservationProcessorStep()
+
+    image = np.random.randint(0, 256, size=(64, 64, 3), dtype=np.uint8)
+    observation = {"pixels": image}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    assert OBS_IMAGE in processed_obs
+    assert len(processed_obs) == 1
+
+
+def test_state_only_processing():
+    """Test processing observation with only states."""
+    processor = VanillaObservationProcessorStep()
+
+    agent_pos = np.array([1.0, 2.0], dtype=np.float32)
+    observation = {"agent_pos": agent_pos}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    assert OBS_STATE in processed_obs
+    assert "agent_pos" not in processed_obs
+
+
+def test_empty_observation():
+    """Test processing empty observation."""
+    processor = VanillaObservationProcessorStep()
+
+    observation = {}
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    assert processed_obs == {}
+
+
+def test_equivalent_to_original_function():
+    """Test that ObservationProcessor produces equivalent results to preprocess_observation."""
+    # Import the original function for comparison
+    from lerobot.envs.utils import preprocess_observation
+
+    processor = VanillaObservationProcessorStep()
+
+    # Create test data similar to what the original function expects
+    image = np.random.randint(0, 256, size=(64, 64, 3), dtype=np.uint8)
+    env_state = np.array([1.0, 2.0, 3.0], dtype=np.float32)
+    agent_pos = np.array([0.5, -0.5, 1.0], dtype=np.float32)
+
+    observation = {"pixels": image, "environment_state": env_state, "agent_pos": agent_pos}
+
+    # Process with original function
+    original_result = preprocess_observation(observation)
+
+    # Process with new processor
+    transition = create_transition(observation=observation)
+    processor_result = processor(transition)[TransitionKey.OBSERVATION]
+
+    # Compare results
+    assert set(original_result.keys()) == set(processor_result.keys())
+
+    for key in original_result:
+        torch.testing.assert_close(original_result[key], processor_result[key])
+
+
+def test_equivalent_with_image_dict():
+    """Test equivalence with dictionary of images."""
+    from lerobot.envs.utils import preprocess_observation
+
+    processor = VanillaObservationProcessorStep()
+
+    # Create test data with multiple cameras
+    image1 = np.random.randint(0, 256, size=(32, 32, 3), dtype=np.uint8)
+    image2 = np.random.randint(0, 256, size=(48, 48, 3), dtype=np.uint8)
+    agent_pos = np.array([1.0, 2.0], dtype=np.float32)
+
+    observation = {"pixels": {"cam1": image1, "cam2": image2}, "agent_pos": agent_pos}
+
+    # Process with original function
+    original_result = preprocess_observation(observation)
+
+    # Process with new processor
+    transition = create_transition(observation=observation)
+    processor_result = processor(transition)[TransitionKey.OBSERVATION]
+
+    # Compare results
+    assert set(original_result.keys()) == set(processor_result.keys())
+
+    for key in original_result:
+        torch.testing.assert_close(original_result[key], processor_result[key])
+
+
+def test_image_processor_features_pixels_to_image(policy_feature_factory):
+    processor = VanillaObservationProcessorStep()
+    features = {
+        PipelineFeatureType.OBSERVATION: {
+            "pixels": policy_feature_factory(FeatureType.VISUAL, (3, 64, 64)),
+            "keep": policy_feature_factory(FeatureType.ENV, (1,)),
+        },
+    }
+    out = processor.transform_features(features.copy())
+
+    assert (
+        OBS_IMAGE in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION][OBS_IMAGE]
+        == features[PipelineFeatureType.OBSERVATION]["pixels"]
+    )
+    assert "pixels" not in out[PipelineFeatureType.OBSERVATION]
+    assert out[PipelineFeatureType.OBSERVATION]["keep"] == features[PipelineFeatureType.OBSERVATION]["keep"]
+    assert_contract_is_typed(out)
+
+
+def test_image_processor_features_observation_pixels_to_image(policy_feature_factory):
+    processor = VanillaObservationProcessorStep()
+    features = {
+        PipelineFeatureType.OBSERVATION: {
+            "observation.pixels": policy_feature_factory(FeatureType.VISUAL, (3, 64, 64)),
+            "keep": policy_feature_factory(FeatureType.ENV, (1,)),
+        },
+    }
+    out = processor.transform_features(features.copy())
+
+    assert (
+        OBS_IMAGE in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION][OBS_IMAGE]
+        == features[PipelineFeatureType.OBSERVATION]["observation.pixels"]
+    )
+    assert "observation.pixels" not in out[PipelineFeatureType.OBSERVATION]
+    assert out[PipelineFeatureType.OBSERVATION]["keep"] == features[PipelineFeatureType.OBSERVATION]["keep"]
+    assert_contract_is_typed(out)
+
+
+def test_image_processor_features_multi_camera_and_prefixed(policy_feature_factory):
+    processor = VanillaObservationProcessorStep()
+    features = {
+        PipelineFeatureType.OBSERVATION: {
+            "pixels.front": policy_feature_factory(FeatureType.VISUAL, (3, 64, 64)),
+            "pixels.wrist": policy_feature_factory(FeatureType.VISUAL, (3, 64, 64)),
+            "observation.pixels.rear": policy_feature_factory(FeatureType.VISUAL, (3, 64, 64)),
+            "keep": policy_feature_factory(FeatureType.ENV, (7,)),
+        },
+    }
+    out = processor.transform_features(features.copy())
+
+    assert (
+        f"{OBS_IMAGES}.front" in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION][f"{OBS_IMAGES}.front"]
+        == features[PipelineFeatureType.OBSERVATION]["pixels.front"]
+    )
+    assert (
+        f"{OBS_IMAGES}.wrist" in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION][f"{OBS_IMAGES}.wrist"]
+        == features[PipelineFeatureType.OBSERVATION]["pixels.wrist"]
+    )
+    assert (
+        f"{OBS_IMAGES}.rear" in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION][f"{OBS_IMAGES}.rear"]
+        == features[PipelineFeatureType.OBSERVATION]["observation.pixels.rear"]
+    )
+    assert (
+        "pixels.front" not in out[PipelineFeatureType.OBSERVATION]
+        and "pixels.wrist" not in out[PipelineFeatureType.OBSERVATION]
+        and "observation.pixels.rear" not in out[PipelineFeatureType.OBSERVATION]
+    )
+    assert out[PipelineFeatureType.OBSERVATION]["keep"] == features[PipelineFeatureType.OBSERVATION]["keep"]
+    assert_contract_is_typed(out)
+
+
+def test_state_processor_features_environment_and_agent_pos(policy_feature_factory):
+    processor = VanillaObservationProcessorStep()
+    features = {
+        PipelineFeatureType.OBSERVATION: {
+            "environment_state": policy_feature_factory(FeatureType.STATE, (3,)),
+            "agent_pos": policy_feature_factory(FeatureType.STATE, (7,)),
+            "keep": policy_feature_factory(FeatureType.ENV, (1,)),
+        },
+    }
+    out = processor.transform_features(features.copy())
+
+    assert (
+        OBS_ENV_STATE in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION][OBS_ENV_STATE]
+        == features[PipelineFeatureType.OBSERVATION]["environment_state"]
+    )
+    assert (
+        OBS_STATE in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION][OBS_STATE]
+        == features[PipelineFeatureType.OBSERVATION]["agent_pos"]
+    )
+    assert (
+        "environment_state" not in out[PipelineFeatureType.OBSERVATION]
+        and "agent_pos" not in out[PipelineFeatureType.OBSERVATION]
+    )
+    assert out[PipelineFeatureType.OBSERVATION]["keep"] == features[PipelineFeatureType.OBSERVATION]["keep"]
+    assert_contract_is_typed(out)
+
+
+def test_state_processor_features_prefixed_inputs(policy_feature_factory):
+    proc = VanillaObservationProcessorStep()
+    features = {
+        PipelineFeatureType.OBSERVATION: {
+            OBS_ENV_STATE: policy_feature_factory(FeatureType.STATE, (2,)),
+            "observation.agent_pos": policy_feature_factory(FeatureType.STATE, (4,)),
+        },
+    }
+    out = proc.transform_features(features.copy())
+
+    assert (
+        OBS_ENV_STATE in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION][OBS_ENV_STATE]
+        == features[PipelineFeatureType.OBSERVATION][OBS_ENV_STATE]
+    )
+    assert (
+        OBS_STATE in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION][OBS_STATE]
+        == features[PipelineFeatureType.OBSERVATION]["observation.agent_pos"]
+    )
+    assert (
+        "environment_state" not in out[PipelineFeatureType.OBSERVATION]
+        and "agent_pos" not in out[PipelineFeatureType.OBSERVATION]
+    )
+    assert_contract_is_typed(out)
diff --git a/lerobot/tests/processor/test_pipeline.py b/lerobot/tests/processor/test_pipeline.py
new file mode 100644
index 0000000000000000000000000000000000000000..a335c2b4b92b8b5a8932a8e3d831b72ebffcf2ce
--- /dev/null
+++ b/lerobot/tests/processor/test_pipeline.py
@@ -0,0 +1,2198 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import json
+import tempfile
+from collections.abc import Callable
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import Any
+
+import pytest
+import torch
+import torch.nn as nn
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features
+from lerobot.processor import (
+    DataProcessorPipeline,
+    EnvTransition,
+    ProcessorStep,
+    ProcessorStepRegistry,
+    TransitionKey,
+)
+from lerobot.processor.converters import create_transition, identity_transition
+from lerobot.utils.constants import ACTION, DONE, OBS_IMAGE, OBS_IMAGES, OBS_STATE, REWARD, TRUNCATED
+from tests.conftest import assert_contract_is_typed
+
+
+@dataclass
+class MockStep(ProcessorStep):
+    """Mock pipeline step for testing - demonstrates best practices.
+
+    This example shows the proper separation:
+    - JSON-serializable attributes (name, counter) go in get_config()
+    - Only torch tensors go in state_dict()
+
+    Note: The counter is part of the configuration, so it will be restored
+    when the step is recreated from config during loading.
+    """
+
+    name: str = "mock_step"
+    counter: int = 0
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Add a counter to the complementary_data."""
+        comp_data = transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+        comp_data = {} if comp_data is None else dict(comp_data)  # Make a copy
+
+        comp_data[f"{self.name}_counter"] = self.counter
+        self.counter += 1
+
+        # Create a new transition with updated complementary_data
+        new_transition = transition.copy()
+        new_transition[TransitionKey.COMPLEMENTARY_DATA] = comp_data
+        return new_transition
+
+    def get_config(self) -> dict[str, Any]:
+        # Return all JSON-serializable attributes that should be persisted
+        # These will be passed to __init__ when loading
+        return {"name": self.name, "counter": self.counter}
+
+    def state_dict(self) -> dict[str, torch.Tensor]:
+        # Only return torch tensors (empty in this case since we have no tensor state)
+        return {}
+
+    def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+        # No tensor state to load
+        pass
+
+    def reset(self) -> None:
+        self.counter = 0
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We do not test features here
+        return features
+
+
+@dataclass
+class MockStepWithoutOptionalMethods(ProcessorStep):
+    """Mock step that only implements the required __call__ method."""
+
+    multiplier: float = 2.0
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Multiply reward by multiplier."""
+        reward = transition.get(TransitionKey.REWARD)
+
+        if reward is not None:
+            new_transition = transition.copy()
+            new_transition[TransitionKey.REWARD] = reward * self.multiplier
+            return new_transition
+
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We do not test features here
+        return features
+
+
+@dataclass
+class MockStepWithTensorState(ProcessorStep):
+    """Mock step demonstrating mixed JSON attributes and tensor state."""
+
+    name: str = "tensor_step"
+    learning_rate: float = 0.01
+    window_size: int = 10
+
+    def __init__(self, name: str = "tensor_step", learning_rate: float = 0.01, window_size: int = 10):
+        self.name = name
+        self.learning_rate = learning_rate
+        self.window_size = window_size
+        # Tensor state
+        self.running_mean = torch.zeros(window_size)
+        self.running_count = torch.tensor(0)
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Update running statistics."""
+        reward = transition.get(TransitionKey.REWARD)
+
+        if reward is not None:
+            # Update running mean
+            idx = self.running_count % self.window_size
+            self.running_mean[idx] = reward
+            self.running_count += 1
+
+        return transition
+
+    def get_config(self) -> dict[str, Any]:
+        # Only JSON-serializable attributes
+        return {
+            "name": self.name,
+            "learning_rate": self.learning_rate,
+            "window_size": self.window_size,
+        }
+
+    def state_dict(self) -> dict[str, torch.Tensor]:
+        # Only tensor state
+        return {
+            "running_mean": self.running_mean,
+            "running_count": self.running_count,
+        }
+
+    def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+        self.running_mean = state["running_mean"]
+        self.running_count = state["running_count"]
+
+    def reset(self) -> None:
+        self.running_mean.zero_()
+        self.running_count.zero_()
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We do not test features here
+        return features
+
+
+def test_empty_pipeline():
+    """Test pipeline with no steps."""
+    pipeline = DataProcessorPipeline([], to_transition=identity_transition, to_output=identity_transition)
+
+    transition = create_transition()
+    result = pipeline(transition)
+
+    assert result == transition
+    assert len(pipeline) == 0
+
+
+def test_single_step_pipeline():
+    """Test pipeline with a single step."""
+    step = MockStep("test_step")
+    pipeline = DataProcessorPipeline([step], to_transition=identity_transition, to_output=identity_transition)
+
+    transition = create_transition()
+    result = pipeline(transition)
+
+    assert len(pipeline) == 1
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["test_step_counter"] == 0
+
+    # Call again to test counter increment
+    result = pipeline(transition)
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["test_step_counter"] == 1
+
+
+def test_multiple_steps_pipeline():
+    """Test pipeline with multiple steps."""
+    step1 = MockStep("step1")
+    step2 = MockStep("step2")
+    pipeline = DataProcessorPipeline(
+        [step1, step2], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    transition = create_transition()
+    result = pipeline(transition)
+
+    assert len(pipeline) == 2
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["step1_counter"] == 0
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["step2_counter"] == 0
+
+
+def test_invalid_transition_format():
+    """Test pipeline with invalid transition format."""
+    pipeline = DataProcessorPipeline([MockStep()])
+
+    # Test with wrong type (tuple instead of dict)
+    with pytest.raises(ValueError, match="EnvTransition must be a dictionary"):
+        pipeline((None, None, 0.0, False, False, {}, {}))  # Tuple instead of dict
+
+    # Test with wrong type (string)
+    with pytest.raises(ValueError, match="EnvTransition must be a dictionary"):
+        pipeline("not a dict")
+
+
+def test_step_through():
+    """Test step_through method with dict input."""
+    step1 = MockStep("step1")
+    step2 = MockStep("step2")
+    pipeline = DataProcessorPipeline([step1, step2])
+
+    transition = create_transition()
+
+    results = list(pipeline.step_through(transition))
+
+    assert len(results) == 3  # Original + 2 steps
+    assert results[0] == transition  # Original
+    assert "step1_counter" in results[1][TransitionKey.COMPLEMENTARY_DATA]  # After step1
+    assert "step2_counter" in results[2][TransitionKey.COMPLEMENTARY_DATA]  # After step2
+
+    # Ensure all results are dicts (same format as input)
+    for result in results:
+        assert isinstance(result, dict)
+        assert all(isinstance(k, TransitionKey) for k in result)
+
+
+def test_step_through_with_dict():
+    """Test step_through method with dict input."""
+    step1 = MockStep("step1")
+    step2 = MockStep("step2")
+    pipeline = DataProcessorPipeline([step1, step2])
+
+    batch = {
+        OBS_IMAGE: None,
+        ACTION: None,
+        REWARD: 0.0,
+        DONE: False,
+        TRUNCATED: False,
+        "info": {},
+    }
+
+    results = list(pipeline.step_through(batch))
+
+    assert len(results) == 3  # Original + 2 steps
+
+    # Ensure all results are EnvTransition dicts (regardless of input format)
+    for result in results:
+        assert isinstance(result, dict)
+        # Check that keys are TransitionKey enums or at least valid transition keys
+        for key in result:
+            assert key in [
+                TransitionKey.OBSERVATION,
+                TransitionKey.ACTION,
+                TransitionKey.REWARD,
+                TransitionKey.DONE,
+                TransitionKey.TRUNCATED,
+                TransitionKey.INFO,
+                TransitionKey.COMPLEMENTARY_DATA,
+            ]
+
+    # Check that the processing worked - verify step counters in complementary_data
+    assert results[1].get(TransitionKey.COMPLEMENTARY_DATA, {}).get("step1_counter") == 0
+    assert results[2].get(TransitionKey.COMPLEMENTARY_DATA, {}).get("step1_counter") == 0
+    assert results[2].get(TransitionKey.COMPLEMENTARY_DATA, {}).get("step2_counter") == 0
+
+
+def test_step_through_no_hooks():
+    """Test that step_through doesn't execute hooks."""
+    step = MockStep("test_step")
+    pipeline = DataProcessorPipeline([step])
+
+    hook_calls = []
+
+    def tracking_hook(idx: int, transition: EnvTransition):
+        hook_calls.append(f"hook_called_step_{idx}")
+
+    # Register hooks
+    pipeline.register_before_step_hook(tracking_hook)
+    pipeline.register_after_step_hook(tracking_hook)
+
+    # Use step_through
+    transition = create_transition()
+    results = list(pipeline.step_through(transition))
+
+    # Verify step was executed (counter should increment)
+    assert len(results) == 2  # Initial + 1 step
+    assert results[1][TransitionKey.COMPLEMENTARY_DATA]["test_step_counter"] == 0
+
+    # Verify hooks were NOT called
+    assert len(hook_calls) == 0
+
+    # Now use __call__ to verify hooks ARE called there
+    hook_calls.clear()
+    pipeline(transition)
+
+    # Verify hooks were called (before and after for 1 step = 2 calls)
+    assert len(hook_calls) == 2
+    assert hook_calls == ["hook_called_step_0", "hook_called_step_0"]
+
+
+def test_indexing():
+    """Test pipeline indexing."""
+    step1 = MockStep("step1")
+    step2 = MockStep("step2")
+    pipeline = DataProcessorPipeline([step1, step2])
+
+    # Test integer indexing
+    assert pipeline[0] is step1
+    assert pipeline[1] is step2
+
+    # Test slice indexing
+    sub_pipeline = pipeline[0:1]
+    assert isinstance(sub_pipeline, DataProcessorPipeline)
+    assert len(sub_pipeline) == 1
+    assert sub_pipeline[0] is step1
+
+
+def test_hooks():
+    """Test before/after step hooks."""
+    step = MockStep("test_step")
+    pipeline = DataProcessorPipeline([step])
+
+    before_calls = []
+    after_calls = []
+
+    def before_hook(idx: int, transition: EnvTransition):
+        before_calls.append(idx)
+
+    def after_hook(idx: int, transition: EnvTransition):
+        after_calls.append(idx)
+
+    pipeline.register_before_step_hook(before_hook)
+    pipeline.register_after_step_hook(after_hook)
+
+    transition = create_transition()
+    pipeline(transition)
+
+    assert before_calls == [0]
+    assert after_calls == [0]
+
+
+def test_unregister_hooks():
+    """Test unregistering hooks from the pipeline."""
+    step = MockStep("test_step")
+    pipeline = DataProcessorPipeline([step])
+
+    # Test before_step_hook
+    before_calls = []
+
+    def before_hook(idx: int, transition: EnvTransition):
+        before_calls.append(idx)
+
+    pipeline.register_before_step_hook(before_hook)
+
+    # Verify hook is registered
+    transition = create_transition()
+    pipeline(transition)
+    assert len(before_calls) == 1
+
+    # Unregister and verify it's no longer called
+    pipeline.unregister_before_step_hook(before_hook)
+    before_calls.clear()
+    pipeline(transition)
+    assert len(before_calls) == 0
+
+    # Test after_step_hook
+    after_calls = []
+
+    def after_hook(idx: int, transition: EnvTransition):
+        after_calls.append(idx)
+
+    pipeline.register_after_step_hook(after_hook)
+    pipeline(transition)
+    assert len(after_calls) == 1
+
+    pipeline.unregister_after_step_hook(after_hook)
+    after_calls.clear()
+    pipeline(transition)
+    assert len(after_calls) == 0
+
+
+def test_unregister_nonexistent_hook():
+    """Test error handling when unregistering hooks that don't exist."""
+    pipeline = DataProcessorPipeline([MockStep()])
+
+    def some_hook(idx: int, transition: EnvTransition):
+        pass
+
+    def reset_hook():
+        pass
+
+    # Test unregistering hooks that were never registered
+    with pytest.raises(ValueError, match="not found in before_step_hooks"):
+        pipeline.unregister_before_step_hook(some_hook)
+
+    with pytest.raises(ValueError, match="not found in after_step_hooks"):
+        pipeline.unregister_after_step_hook(some_hook)
+
+
+def test_multiple_hooks_and_selective_unregister():
+    """Test registering multiple hooks and selectively unregistering them."""
+    pipeline = DataProcessorPipeline([MockStep("step1"), MockStep("step2")])
+
+    calls_1 = []
+    calls_2 = []
+    calls_3 = []
+
+    def hook1(idx: int, transition: EnvTransition):
+        calls_1.append(f"hook1_step{idx}")
+
+    def hook2(idx: int, transition: EnvTransition):
+        calls_2.append(f"hook2_step{idx}")
+
+    def hook3(idx: int, transition: EnvTransition):
+        calls_3.append(f"hook3_step{idx}")
+
+    # Register multiple hooks
+    pipeline.register_before_step_hook(hook1)
+    pipeline.register_before_step_hook(hook2)
+    pipeline.register_before_step_hook(hook3)
+
+    # Run pipeline - all hooks should be called for both steps
+    transition = create_transition()
+    pipeline(transition)
+
+    assert calls_1 == ["hook1_step0", "hook1_step1"]
+    assert calls_2 == ["hook2_step0", "hook2_step1"]
+    assert calls_3 == ["hook3_step0", "hook3_step1"]
+
+    # Clear calls
+    calls_1.clear()
+    calls_2.clear()
+    calls_3.clear()
+
+    # Unregister middle hook
+    pipeline.unregister_before_step_hook(hook2)
+
+    # Run again - only hook1 and hook3 should be called
+    pipeline(transition)
+
+    assert calls_1 == ["hook1_step0", "hook1_step1"]
+    assert calls_2 == []  # hook2 was unregistered
+    assert calls_3 == ["hook3_step0", "hook3_step1"]
+
+
+def test_hook_execution_order_documentation():
+    """Test and document that hooks are executed sequentially in registration order."""
+    pipeline = DataProcessorPipeline([MockStep("step")])
+
+    execution_order = []
+
+    def hook_a(idx: int, transition: EnvTransition):
+        execution_order.append("A")
+
+    def hook_b(idx: int, transition: EnvTransition):
+        execution_order.append("B")
+
+    def hook_c(idx: int, transition: EnvTransition):
+        execution_order.append("C")
+
+    # Register in specific order: A, B, C
+    pipeline.register_before_step_hook(hook_a)
+    pipeline.register_before_step_hook(hook_b)
+    pipeline.register_before_step_hook(hook_c)
+
+    transition = create_transition()
+    pipeline(transition)
+
+    # Verify execution order matches registration order
+    assert execution_order == ["A", "B", "C"]
+
+    # Test that after unregistering B and re-registering it, it goes to the end
+    pipeline.unregister_before_step_hook(hook_b)
+    execution_order.clear()
+
+    pipeline(transition)
+    assert execution_order == ["A", "C"]  # B is gone
+
+    # Re-register B - it should now be at the end
+    pipeline.register_before_step_hook(hook_b)
+    execution_order.clear()
+
+    pipeline(transition)
+    assert execution_order == ["A", "C", "B"]  # B is now last
+
+
+def test_save_and_load_pretrained():
+    """Test saving and loading pipeline.
+
+    This test demonstrates that JSON-serializable attributes (like counter)
+    are saved in the config and restored when the step is recreated.
+    """
+    step1 = MockStep("step1")
+    step2 = MockStep("step2")
+
+    # Increment counters to have some state
+    step1.counter = 5
+    step2.counter = 10
+
+    pipeline = DataProcessorPipeline([step1, step2], name="TestPipeline")
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save pipeline
+        pipeline.save_pretrained(tmp_dir)
+
+        # Check files were created
+        config_path = Path(tmp_dir) / "testpipeline.json"  # Based on name="TestPipeline"
+        assert config_path.exists()
+
+        # Check config content
+        with open(config_path) as f:
+            config = json.load(f)
+
+        assert config["name"] == "TestPipeline"
+        assert len(config["steps"]) == 2
+
+        # Verify counters are saved in config, not in separate state files
+        assert config["steps"][0]["config"]["counter"] == 5
+        assert config["steps"][1]["config"]["counter"] == 10
+
+        # Load pipeline
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="testpipeline.json")
+
+        assert loaded_pipeline.name == "TestPipeline"
+        assert len(loaded_pipeline) == 2
+
+        # Check that counter was restored from config
+        assert loaded_pipeline.steps[0].counter == 5
+        assert loaded_pipeline.steps[1].counter == 10
+
+
+def test_step_without_optional_methods():
+    """Test pipeline with steps that don't implement optional methods."""
+    step = MockStepWithoutOptionalMethods(multiplier=3.0)
+    pipeline = DataProcessorPipeline(
+        [step], to_transition=identity_transition, to_output=identity_transition
+    )  # Identity for EnvTransition input/output
+
+    transition = create_transition(reward=2.0)
+    result = pipeline(transition)
+
+    assert result[TransitionKey.REWARD] == 6.0  # 2.0 * 3.0
+
+    # Reset should work even if step doesn't implement reset
+    pipeline.reset()
+
+    # Save/load should work even without optional methods
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir, config_filename="dataprocessorpipeline.json"
+        )
+        assert len(loaded_pipeline) == 1
+
+
+def test_mixed_json_and_tensor_state():
+    """Test step with both JSON attributes and tensor state."""
+    step = MockStepWithTensorState(name="stats", learning_rate=0.05, window_size=5)
+    pipeline = DataProcessorPipeline([step])
+
+    # Process some transitions with rewards
+    for i in range(10):
+        transition = create_transition(reward=float(i))
+        pipeline(transition)
+
+    # Check state
+    assert step.running_count.item() == 10
+    assert step.learning_rate == 0.05
+
+    # Save and load
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Check that both config and state files were created
+        config_path = Path(tmp_dir) / "dataprocessorpipeline.json"  # Default name is "RobotProcessor"
+        state_path = Path(tmp_dir) / "dataprocessorpipeline_step_0.safetensors"
+        assert config_path.exists()
+        assert state_path.exists()
+
+        # Load and verify
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir, config_filename="dataprocessorpipeline.json"
+        )
+        loaded_step = loaded_pipeline.steps[0]
+
+        # Check JSON attributes were restored
+        assert loaded_step.name == "stats"
+        assert loaded_step.learning_rate == 0.05
+        assert loaded_step.window_size == 5
+
+        # Check tensor state was restored
+        assert loaded_step.running_count.item() == 10
+        assert torch.allclose(loaded_step.running_mean, step.running_mean)
+
+
+class MockModuleStep(ProcessorStep, nn.Module):
+    """Mock step that inherits from nn.Module to test state_dict handling of module parameters."""
+
+    def __init__(self, input_dim: int = 10, hidden_dim: int = 5):
+        super().__init__()
+        self.input_dim = input_dim
+        self.hidden_dim = hidden_dim
+        self.linear = nn.Linear(input_dim, hidden_dim)
+        self.running_mean = nn.Parameter(torch.zeros(hidden_dim), requires_grad=False)
+        self.counter = 0  # Non-tensor state
+
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        return self.linear(x)
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Process transition and update running mean."""
+        obs = transition.get(TransitionKey.OBSERVATION)
+
+        if obs is not None and isinstance(obs, torch.Tensor):
+            # Process observation through linear layer
+            processed = self.forward(obs[:, : self.input_dim])
+
+            # Update running mean in-place (don't reassign the parameter)
+            with torch.no_grad():
+                self.running_mean.mul_(0.9).add_(processed.mean(dim=0), alpha=0.1)
+
+            self.counter += 1
+
+        return transition
+
+    def get_config(self) -> dict[str, Any]:
+        return {
+            "input_dim": self.input_dim,
+            "hidden_dim": self.hidden_dim,
+            "counter": self.counter,
+        }
+
+    def state_dict(self) -> dict[str, torch.Tensor]:
+        """Override to return all module parameters and buffers."""
+        # Get the module's state dict (includes all parameters and buffers)
+        return nn.Module.state_dict(self)
+
+    def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+        """Override to load all module parameters and buffers."""
+        # Use the module's load_state_dict
+        nn.Module.load_state_dict(self, state)
+
+    def reset(self) -> None:
+        self.running_mean.zero_()
+        self.counter = 0
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We do not test features here
+        return features
+
+
+class MockNonModuleStepWithState(ProcessorStep):
+    """Mock step that explicitly does NOT inherit from nn.Module but has tensor state.
+
+    This tests the state_dict/load_state_dict path for regular classes.
+    """
+
+    def __init__(self, name: str = "non_module_step", feature_dim: int = 10):
+        self.name = name
+        self.feature_dim = feature_dim
+
+        # Initialize tensor state - these are regular tensors, not nn.Parameters
+        self.weights = torch.randn(feature_dim, feature_dim)
+        self.bias = torch.zeros(feature_dim)
+        self.running_stats = torch.zeros(feature_dim)
+        self.step_count = torch.tensor(0)
+
+        # Non-tensor state
+        self.config_value = 42
+        self.history = []
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        """Process transition using tensor operations."""
+        obs = transition.get(TransitionKey.OBSERVATION)
+        comp_data = transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+
+        if obs is not None and isinstance(obs, torch.Tensor) and obs.numel() >= self.feature_dim:
+            # Perform some tensor operations
+            flat_obs = obs.flatten()[: self.feature_dim]
+
+            # Simple linear transformation (ensure dimensions match for matmul)
+            output = torch.matmul(self.weights.T, flat_obs) + self.bias
+
+            # Update running stats
+            self.running_stats = 0.9 * self.running_stats + 0.1 * output
+            self.step_count += 1
+
+            # Add to complementary data
+            comp_data = {} if comp_data is None else dict(comp_data)
+            comp_data[f"{self.name}_mean_output"] = output.mean().item()
+            comp_data[f"{self.name}_steps"] = self.step_count.item()
+
+            # Return updated transition
+            new_transition = transition.copy()
+            new_transition[TransitionKey.COMPLEMENTARY_DATA] = comp_data
+            return new_transition
+
+        return transition
+
+    def get_config(self) -> dict[str, Any]:
+        return {
+            "name": self.name,
+            "feature_dim": self.feature_dim,
+            "config_value": self.config_value,
+        }
+
+    def state_dict(self) -> dict[str, torch.Tensor]:
+        """Return only tensor state."""
+        return {
+            "weights": self.weights,
+            "bias": self.bias,
+            "running_stats": self.running_stats,
+            "step_count": self.step_count,
+        }
+
+    def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+        """Load tensor state."""
+        self.weights = state["weights"]
+        self.bias = state["bias"]
+        self.running_stats = state["running_stats"]
+        self.step_count = state["step_count"]
+
+    def reset(self) -> None:
+        """Reset statistics but keep learned parameters."""
+        self.running_stats.zero_()
+        self.step_count.zero_()
+        self.history.clear()
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We do not test features here
+        return features
+
+
+# Tests for overrides functionality
+@dataclass
+class MockStepWithNonSerializableParam(ProcessorStep):
+    """Mock step that requires a non-serializable parameter."""
+
+    def __init__(self, name: str = "mock_env_step", multiplier: float = 1.0, env: Any = None):
+        self.name = name
+        # Add type validation for multiplier
+        if isinstance(multiplier, str):
+            raise ValueError(f"multiplier must be a number, got string '{multiplier}'")
+        if not isinstance(multiplier, (int | float)):
+            raise TypeError(f"multiplier must be a number, got {type(multiplier).__name__}")
+        self.multiplier = float(multiplier)
+        self.env = env  # Non-serializable parameter (like gym.Env)
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        reward = transition.get(TransitionKey.REWARD)
+        comp_data = transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+
+        # Use the env parameter if provided
+        if self.env is not None:
+            comp_data = {} if comp_data is None else dict(comp_data)
+            comp_data[f"{self.name}_env_info"] = str(self.env)
+
+        # Apply multiplier to reward
+        new_transition = transition.copy()
+        if reward is not None:
+            new_transition[TransitionKey.REWARD] = reward * self.multiplier
+
+        if comp_data:
+            new_transition[TransitionKey.COMPLEMENTARY_DATA] = comp_data
+
+        return new_transition
+
+    def get_config(self) -> dict[str, Any]:
+        # Note: env is intentionally NOT included here as it's not serializable
+        return {
+            "name": self.name,
+            "multiplier": self.multiplier,
+        }
+
+    def state_dict(self) -> dict[str, torch.Tensor]:
+        return {}
+
+    def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+        pass
+
+    def reset(self) -> None:
+        pass
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We do not test features here
+        return features
+
+
+@ProcessorStepRegistry.register("registered_mock_step")
+@dataclass
+class RegisteredMockStep(ProcessorStep):
+    """Mock step registered in the registry."""
+
+    value: int = 42
+    device: str = "cpu"
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        comp_data = transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+
+        comp_data = {} if comp_data is None else dict(comp_data)
+        comp_data["registered_step_value"] = self.value
+        comp_data["registered_step_device"] = self.device
+
+        new_transition = transition.copy()
+        new_transition[TransitionKey.COMPLEMENTARY_DATA] = comp_data
+        return new_transition
+
+    def get_config(self) -> dict[str, Any]:
+        return {
+            "value": self.value,
+            "device": self.device,
+        }
+
+    def state_dict(self) -> dict[str, torch.Tensor]:
+        return {}
+
+    def load_state_dict(self, state: dict[str, torch.Tensor]) -> None:
+        pass
+
+    def reset(self) -> None:
+        pass
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # We do not test features here
+        return features
+
+
+class MockEnvironment:
+    """Mock environment for testing non-serializable parameters."""
+
+    def __init__(self, name: str):
+        self.name = name
+
+    def __str__(self):
+        return f"MockEnvironment({self.name})"
+
+
+def test_from_pretrained_with_overrides():
+    """Test loading processor with parameter overrides."""
+    # Create a processor with steps that need overrides
+    env_step = MockStepWithNonSerializableParam(name="env_step", multiplier=2.0)
+    registered_step = RegisteredMockStep(value=100, device="cpu")
+
+    pipeline = DataProcessorPipeline([env_step, registered_step], name="TestOverrides")
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save the pipeline
+        pipeline.save_pretrained(tmp_dir)
+
+        # Create a mock environment for override
+        mock_env = MockEnvironment("test_env")
+
+        # Load with overrides
+        overrides = {
+            "MockStepWithNonSerializableParam": {
+                "env": mock_env,
+                "multiplier": 3.0,  # Override the multiplier too
+            },
+            "registered_mock_step": {"device": "cuda", "value": 200},
+        }
+
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="testoverrides.json",
+            overrides=overrides,
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        # Verify the pipeline was loaded correctly
+        assert len(loaded_pipeline) == 2
+        assert loaded_pipeline.name == "TestOverrides"
+
+        # Test the loaded steps
+        transition = create_transition(reward=1.0)
+        result = loaded_pipeline(transition)
+
+        # Check that overrides were applied
+        comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+        assert "env_step_env_info" in comp_data
+        assert comp_data["env_step_env_info"] == "MockEnvironment(test_env)"
+        assert comp_data["registered_step_value"] == 200
+        assert comp_data["registered_step_device"] == "cuda"
+
+        # Check that multiplier override was applied
+        assert result[TransitionKey.REWARD] == 3.0  # 1.0 * 3.0 (overridden multiplier)
+
+
+def test_from_pretrained_with_partial_overrides():
+    """Test loading processor with overrides for only some steps."""
+    step1 = MockStepWithNonSerializableParam(name="step1", multiplier=1.0)
+    step2 = MockStepWithNonSerializableParam(name="step2", multiplier=2.0)
+
+    pipeline = DataProcessorPipeline([step1, step2])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Override only one step
+        overrides = {"MockStepWithNonSerializableParam": {"multiplier": 5.0}}
+
+        # The current implementation applies overrides to ALL steps with the same class name
+        # Both steps will get the override
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="dataprocessorpipeline.json",
+            overrides=overrides,
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        transition = create_transition(reward=1.0)
+        result = loaded_pipeline(transition)
+
+        # The reward should be affected by both steps, both getting the override
+        # First step: 1.0 * 5.0 = 5.0 (overridden)
+        # Second step: 5.0 * 5.0 = 25.0 (also overridden)
+        assert result[TransitionKey.REWARD] == 25.0
+
+
+def test_from_pretrained_invalid_override_key():
+    """Test that invalid override keys raise KeyError."""
+    step = MockStepWithNonSerializableParam()
+    pipeline = DataProcessorPipeline([step])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Try to override a non-existent step
+        overrides = {"NonExistentStep": {"param": "value"}}
+
+        with pytest.raises(KeyError, match="Override keys.*do not match any step"):
+            DataProcessorPipeline.from_pretrained(
+                tmp_dir, config_filename="dataprocessorpipeline.json", overrides=overrides
+            )
+
+
+def test_from_pretrained_multiple_invalid_override_keys():
+    """Test that multiple invalid override keys are reported."""
+    step = MockStepWithNonSerializableParam()
+    pipeline = DataProcessorPipeline([step])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Try to override multiple non-existent steps
+        overrides = {"NonExistentStep1": {"param": "value1"}, "NonExistentStep2": {"param": "value2"}}
+
+        with pytest.raises(KeyError) as exc_info:
+            DataProcessorPipeline.from_pretrained(
+                tmp_dir, config_filename="dataprocessorpipeline.json", overrides=overrides
+            )
+
+        error_msg = str(exc_info.value)
+        assert "NonExistentStep1" in error_msg
+        assert "NonExistentStep2" in error_msg
+        assert "Available step keys" in error_msg
+
+
+def test_from_pretrained_registered_step_override():
+    """Test overriding registered steps using registry names."""
+    registered_step = RegisteredMockStep(value=50, device="cpu")
+    pipeline = DataProcessorPipeline([registered_step])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Override using registry name
+        overrides = {"registered_mock_step": {"value": 999, "device": "cuda"}}
+
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="dataprocessorpipeline.json",
+            overrides=overrides,
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        # Test that overrides were applied
+        transition = create_transition()
+        result = loaded_pipeline(transition)
+
+        comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+        assert comp_data["registered_step_value"] == 999
+        assert comp_data["registered_step_device"] == "cuda"
+
+
+def test_from_pretrained_mixed_registered_and_unregistered():
+    """Test overriding both registered and unregistered steps."""
+    unregistered_step = MockStepWithNonSerializableParam(name="unregistered", multiplier=1.0)
+    registered_step = RegisteredMockStep(value=10, device="cpu")
+
+    pipeline = DataProcessorPipeline([unregistered_step, registered_step])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        mock_env = MockEnvironment("mixed_test")
+
+        overrides = {
+            "MockStepWithNonSerializableParam": {"env": mock_env, "multiplier": 4.0},
+            "registered_mock_step": {"value": 777},
+        }
+
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="dataprocessorpipeline.json",
+            overrides=overrides,
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        # Test both steps
+        transition = create_transition(reward=2.0)
+        result = loaded_pipeline(transition)
+
+        comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+        assert comp_data["unregistered_env_info"] == "MockEnvironment(mixed_test)"
+        assert comp_data["registered_step_value"] == 777
+        assert result[TransitionKey.REWARD] == 8.0  # 2.0 * 4.0
+
+
+def test_from_pretrained_no_overrides():
+    """Test that from_pretrained works without overrides (backward compatibility)."""
+    step = MockStepWithNonSerializableParam(name="no_override", multiplier=3.0)
+    pipeline = DataProcessorPipeline([step])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Load without overrides
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="dataprocessorpipeline.json",
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        assert len(loaded_pipeline) == 1
+
+        # Test that the step works (env will be None)
+        transition = create_transition(reward=1.0)
+        result = loaded_pipeline(transition)
+
+        assert result[TransitionKey.REWARD] == 3.0  # 1.0 * 3.0
+
+
+def test_from_pretrained_empty_overrides():
+    """Test that from_pretrained works with empty overrides dict."""
+    step = MockStepWithNonSerializableParam(multiplier=2.0)
+    pipeline = DataProcessorPipeline([step])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Load with empty overrides
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="dataprocessorpipeline.json",
+            overrides={},
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        assert len(loaded_pipeline) == 1
+
+        # Test that the step works normally
+        transition = create_transition(reward=1.0)
+        result = loaded_pipeline(transition)
+
+        assert result[TransitionKey.REWARD] == 2.0
+
+
+def test_from_pretrained_override_instantiation_error():
+    """Test that instantiation errors with overrides are properly reported."""
+    step = MockStepWithNonSerializableParam(multiplier=1.0)
+    pipeline = DataProcessorPipeline([step])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Try to override with invalid parameter type
+        overrides = {
+            "MockStepWithNonSerializableParam": {
+                "multiplier": "invalid_type"  # Should be float, not string
+            }
+        }
+
+        with pytest.raises(ValueError, match="Failed to instantiate processor step"):
+            DataProcessorPipeline.from_pretrained(
+                tmp_dir, config_filename="dataprocessorpipeline.json", overrides=overrides
+            )
+
+
+def test_from_pretrained_with_state_and_overrides():
+    """Test that overrides work correctly with steps that have tensor state."""
+    step = MockStepWithTensorState(name="tensor_step", learning_rate=0.01, window_size=5)
+    pipeline = DataProcessorPipeline([step])
+
+    # Process some data to create state
+    for i in range(10):
+        transition = create_transition(reward=float(i))
+        pipeline(transition)
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Load with overrides
+        overrides = {
+            "MockStepWithTensorState": {
+                "learning_rate": 0.05,  # Override learning rate
+                "window_size": 3,  # Override window size
+            }
+        }
+
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir, config_filename="dataprocessorpipeline.json", overrides=overrides
+        )
+        loaded_step = loaded_pipeline.steps[0]
+
+        # Check that config overrides were applied
+        assert loaded_step.learning_rate == 0.05
+        assert loaded_step.window_size == 3
+
+        # Check that tensor state was preserved
+        assert loaded_step.running_count.item() == 10
+
+        # The running_mean should still have the original window_size (5) from saved state
+        # but the new step will use window_size=3 for future operations
+        assert loaded_step.running_mean.shape[0] == 5  # From saved state
+
+
+def test_from_pretrained_override_error_messages():
+    """Test that error messages for override failures are helpful."""
+    step1 = MockStepWithNonSerializableParam(name="step1")
+    step2 = RegisteredMockStep()
+    pipeline = DataProcessorPipeline([step1, step2])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Test with invalid override key
+        overrides = {"WrongStepName": {"param": "value"}}
+
+        with pytest.raises(KeyError) as exc_info:
+            DataProcessorPipeline.from_pretrained(
+                tmp_dir, config_filename="dataprocessorpipeline.json", overrides=overrides
+            )
+
+        error_msg = str(exc_info.value)
+        assert "WrongStepName" in error_msg
+        assert "Available step keys" in error_msg
+        assert "MockStepWithNonSerializableParam" in error_msg
+        assert "registered_mock_step" in error_msg
+
+
+def test_repr_empty_processor():
+    """Test __repr__ with empty processor."""
+    pipeline = DataProcessorPipeline()
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='DataProcessorPipeline', steps=0: [])"
+    assert repr_str == expected
+
+
+def test_repr_single_step():
+    """Test __repr__ with single step."""
+    step = MockStep("test_step")
+    pipeline = DataProcessorPipeline([step])
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='DataProcessorPipeline', steps=1: [MockStep])"
+    assert repr_str == expected
+
+
+def test_repr_multiple_steps_under_limit():
+    """Test __repr__ with 2-3 steps (all shown)."""
+    step1 = MockStep("step1")
+    step2 = MockStepWithoutOptionalMethods()
+    pipeline = DataProcessorPipeline([step1, step2])
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='DataProcessorPipeline', steps=2: [MockStep, MockStepWithoutOptionalMethods])"
+    assert repr_str == expected
+
+    # Test with 3 steps (boundary case)
+    step3 = MockStepWithTensorState()
+    pipeline = DataProcessorPipeline([step1, step2, step3])
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='DataProcessorPipeline', steps=3: [MockStep, MockStepWithoutOptionalMethods, MockStepWithTensorState])"
+    assert repr_str == expected
+
+
+def test_repr_many_steps_truncated():
+    """Test __repr__ with more than 3 steps (truncated with ellipsis)."""
+    step1 = MockStep("step1")
+    step2 = MockStepWithoutOptionalMethods()
+    step3 = MockStepWithTensorState()
+    step4 = MockModuleStep()
+    step5 = MockNonModuleStepWithState()
+
+    pipeline = DataProcessorPipeline([step1, step2, step3, step4, step5])
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='DataProcessorPipeline', steps=5: [MockStep, MockStepWithoutOptionalMethods, ..., MockNonModuleStepWithState])"
+    assert repr_str == expected
+
+
+def test_repr_with_custom_name():
+    """Test __repr__ with custom processor name."""
+    step = MockStep("test_step")
+    pipeline = DataProcessorPipeline([step], name="CustomProcessor")
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='CustomProcessor', steps=1: [MockStep])"
+    assert repr_str == expected
+
+
+def test_repr_with_seed():
+    """Test __repr__ with seed parameter."""
+    step = MockStep("test_step")
+    pipeline = DataProcessorPipeline([step])
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='DataProcessorPipeline', steps=1: [MockStep])"
+    assert repr_str == expected
+
+
+def test_repr_with_custom_name_and_seed():
+    """Test __repr__ with both custom name and seed."""
+    step1 = MockStep("step1")
+    step2 = MockStepWithoutOptionalMethods()
+    pipeline = DataProcessorPipeline([step1, step2], name="MyProcessor")
+    repr_str = repr(pipeline)
+
+    expected = (
+        "DataProcessorPipeline(name='MyProcessor', steps=2: [MockStep, MockStepWithoutOptionalMethods])"
+    )
+    assert repr_str == expected
+
+
+def test_repr_without_seed():
+    """Test __repr__ when seed is explicitly None (should not show seed)."""
+    step = MockStep("test_step")
+    pipeline = DataProcessorPipeline([step], name="TestProcessor")
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='TestProcessor', steps=1: [MockStep])"
+    assert repr_str == expected
+
+
+def test_repr_various_step_types():
+    """Test __repr__ with different types of steps to verify class name extraction."""
+    step1 = MockStep()
+    step2 = MockStepWithTensorState()
+    step3 = MockModuleStep()
+    step4 = MockNonModuleStepWithState()
+
+    pipeline = DataProcessorPipeline([step1, step2, step3, step4], name="MixedSteps")
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='MixedSteps', steps=4: [MockStep, MockStepWithTensorState, ..., MockNonModuleStepWithState])"
+    assert repr_str == expected
+
+
+def test_repr_edge_case_long_names():
+    """Test __repr__ handles steps with long class names properly."""
+    step1 = MockStepWithNonSerializableParam()
+    step2 = MockStepWithoutOptionalMethods()
+    step3 = MockStepWithTensorState()
+    step4 = MockNonModuleStepWithState()
+
+    pipeline = DataProcessorPipeline([step1, step2, step3, step4], name="LongNames")
+    repr_str = repr(pipeline)
+
+    expected = "DataProcessorPipeline(name='LongNames', steps=4: [MockStepWithNonSerializableParam, MockStepWithoutOptionalMethods, ..., MockNonModuleStepWithState])"
+    assert repr_str == expected
+
+
+# Tests for config filename features and multiple processors
+def test_save_with_custom_config_filename():
+    """Test saving processor with custom config filename."""
+    step = MockStep("test")
+    pipeline = DataProcessorPipeline([step], name="TestProcessor")
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save with custom filename
+        pipeline.save_pretrained(tmp_dir, config_filename="my_custom_config.json")
+
+        # Check file exists
+        config_path = Path(tmp_dir) / "my_custom_config.json"
+        assert config_path.exists()
+
+        # Check content
+        with open(config_path) as f:
+            config = json.load(f)
+        assert config["name"] == "TestProcessor"
+
+        # Load with specific filename
+        loaded = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="my_custom_config.json")
+        assert loaded.name == "TestProcessor"
+
+
+def test_multiple_processors_same_directory():
+    """Test saving multiple processors to the same directory with different config files."""
+    # Create different processors
+    preprocessor = DataProcessorPipeline([MockStep("pre1"), MockStep("pre2")], name="preprocessor")
+
+    postprocessor = DataProcessorPipeline(
+        [MockStepWithoutOptionalMethods(multiplier=0.5)], name="postprocessor"
+    )
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save both to same directory
+        preprocessor.save_pretrained(tmp_dir)
+        postprocessor.save_pretrained(tmp_dir)
+
+        # Check both config files exist
+        assert (Path(tmp_dir) / "preprocessor.json").exists()
+        assert (Path(tmp_dir) / "postprocessor.json").exists()
+
+        # Load them back
+        loaded_pre = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="preprocessor.json")
+        loaded_post = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="postprocessor.json")
+
+        assert loaded_pre.name == "preprocessor"
+        assert loaded_post.name == "postprocessor"
+        assert len(loaded_pre) == 2
+        assert len(loaded_post) == 1
+
+
+def test_explicit_config_filename_loading():
+    """Test explicit config filename loading (no more auto-detection)."""
+    step = MockStepWithTensorState()
+    pipeline = DataProcessorPipeline([step], name="SingleConfig")
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Load with explicit config_filename (now required)
+        loaded = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="singleconfig.json")
+        assert loaded.name == "SingleConfig"
+
+
+def test_explicit_config_selection_with_multiple_configs():
+    """Test explicit config selection when multiple configs exist."""
+    proc1 = DataProcessorPipeline([MockStep()], name="processor1")
+    proc2 = DataProcessorPipeline([MockStep()], name="processor2")
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        proc1.save_pretrained(tmp_dir)
+        proc2.save_pretrained(tmp_dir)
+
+        # Can load specific configs explicitly
+        loaded1 = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="processor1.json")
+        loaded2 = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="processor2.json")
+
+        assert loaded1.name == "processor1"
+        assert loaded2.name == "processor2"
+
+
+def test_state_file_naming_with_indices():
+    """Test that state files include pipeline name and step indices to avoid conflicts."""
+    # Create multiple steps of same type with state
+    step1 = MockStepWithTensorState(name="norm1", window_size=5)
+    step2 = MockStepWithTensorState(name="norm2", window_size=10)
+    step3 = MockModuleStep(input_dim=5)
+
+    pipeline = DataProcessorPipeline([step1, step2, step3])
+
+    # Process some data to create state
+    for i in range(5):
+        transition = create_transition(observation=torch.randn(2, 5), reward=float(i))
+        pipeline(transition)
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Check state files have indices
+        state_files = sorted(Path(tmp_dir).glob("*.safetensors"))
+        assert len(state_files) == 3
+
+        # Files should be named with pipeline name prefix and indices
+        expected_names = [
+            "dataprocessorpipeline_step_0.safetensors",
+            "dataprocessorpipeline_step_1.safetensors",
+            "dataprocessorpipeline_step_2.safetensors",
+        ]
+        actual_names = [f.name for f in state_files]
+        assert actual_names == expected_names
+
+
+def test_state_file_naming_with_registry():
+    """Test state file naming for registered steps includes pipeline name, index and registry name."""
+
+    # Register a test step
+    @ProcessorStepRegistry.register("test_stateful_step")
+    @dataclass
+    class TestStatefulStep(ProcessorStep):
+        value: int = 0
+
+        def __init__(self, value: int = 0):
+            self.value = value
+            self.state_tensor = torch.randn(3, 3)
+
+        def __call__(self, transition: EnvTransition) -> EnvTransition:
+            return transition
+
+        def get_config(self):
+            return {"value": self.value}
+
+        def state_dict(self):
+            return {"state_tensor": self.state_tensor}
+
+        def load_state_dict(self, state):
+            self.state_tensor = state["state_tensor"]
+
+        def transform_features(
+            self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+        ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+            # We do not test features here
+            return features
+
+    try:
+        # Create pipeline with registered steps
+        step1 = TestStatefulStep(1)
+        step2 = TestStatefulStep(2)
+        pipeline = DataProcessorPipeline([step1, step2])
+
+        with tempfile.TemporaryDirectory() as tmp_dir:
+            pipeline.save_pretrained(tmp_dir)
+
+            # Check state files
+            state_files = sorted(Path(tmp_dir).glob("*.safetensors"))
+            assert len(state_files) == 2
+
+            # Should include pipeline name, index and registry name
+            expected_names = [
+                "dataprocessorpipeline_step_0_test_stateful_step.safetensors",
+                "dataprocessorpipeline_step_1_test_stateful_step.safetensors",
+            ]
+            actual_names = [f.name for f in state_files]
+            assert actual_names == expected_names
+
+    finally:
+        # Cleanup registry
+        ProcessorStepRegistry.unregister("test_stateful_step")
+
+
+# More comprehensive override tests
+def test_override_with_nested_config():
+    """Test overrides with nested configuration dictionaries."""
+
+    @ProcessorStepRegistry.register("complex_config_step")
+    @dataclass
+    class ComplexConfigStep(ProcessorStep):
+        name: str = "complex"
+        simple_param: int = 42
+        nested_config: dict = None
+
+        def __post_init__(self):
+            if self.nested_config is None:
+                self.nested_config = {"level1": {"level2": "default"}}
+
+        def __call__(self, transition: EnvTransition) -> EnvTransition:
+            comp_data = transition.get(TransitionKey.COMPLEMENTARY_DATA, {})
+            comp_data = dict(comp_data)
+            comp_data["config_value"] = self.nested_config.get("level1", {}).get("level2", "missing")
+
+            new_transition = transition.copy()
+            new_transition[TransitionKey.COMPLEMENTARY_DATA] = comp_data
+            return new_transition
+
+        def get_config(self):
+            return {"name": self.name, "simple_param": self.simple_param, "nested_config": self.nested_config}
+
+        def transform_features(
+            self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+        ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+            # We do not test features here
+            return features
+
+    try:
+        step = ComplexConfigStep()
+        pipeline = DataProcessorPipeline([step])
+
+        with tempfile.TemporaryDirectory() as tmp_dir:
+            pipeline.save_pretrained(tmp_dir)
+
+            # Load with nested override
+            loaded = DataProcessorPipeline.from_pretrained(
+                tmp_dir,
+                config_filename="dataprocessorpipeline.json",
+                overrides={"complex_config_step": {"nested_config": {"level1": {"level2": "overridden"}}}},
+                to_transition=identity_transition,
+                to_output=identity_transition,
+            )
+
+            # Test that override worked
+            transition = create_transition()
+            result = loaded(transition)
+            assert result[TransitionKey.COMPLEMENTARY_DATA]["config_value"] == "overridden"
+    finally:
+        ProcessorStepRegistry.unregister("complex_config_step")
+
+
+def test_override_preserves_defaults():
+    """Test that overrides only affect specified parameters."""
+    step = MockStepWithNonSerializableParam(name="test", multiplier=2.0)
+    pipeline = DataProcessorPipeline([step])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Override only one parameter
+        loaded = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="dataprocessorpipeline.json",
+            overrides={
+                "MockStepWithNonSerializableParam": {
+                    "multiplier": 5.0  # Only override multiplier
+                }
+            },
+        )
+
+        # Check that name was preserved from saved config
+        loaded_step = loaded.steps[0]
+        assert loaded_step.name == "test"  # Original value
+        assert loaded_step.multiplier == 5.0  # Overridden value
+
+
+def test_override_type_validation():
+    """Test that type errors in overrides are caught properly."""
+    step = MockStepWithTensorState(learning_rate=0.01)
+    pipeline = DataProcessorPipeline([step])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Try to override with wrong type
+        overrides = {
+            "MockStepWithTensorState": {
+                "window_size": "not_an_int"  # Should be int
+            }
+        }
+
+        with pytest.raises(ValueError, match="Failed to instantiate"):
+            DataProcessorPipeline.from_pretrained(
+                tmp_dir, config_filename="dataprocessorpipeline.json", overrides=overrides
+            )
+
+
+def test_override_with_callables():
+    """Test overriding with callable objects."""
+
+    @ProcessorStepRegistry.register("callable_step")
+    @dataclass
+    class CallableStep(ProcessorStep):
+        name: str = "callable_step"
+        transform_fn: Any = None
+
+        def __call__(self, transition: EnvTransition) -> EnvTransition:
+            obs = transition.get(TransitionKey.OBSERVATION)
+            if obs is not None and self.transform_fn is not None:
+                processed_obs = {}
+                for k, v in obs.items():
+                    processed_obs[k] = self.transform_fn(v)
+
+                new_transition = transition.copy()
+                new_transition[TransitionKey.OBSERVATION] = processed_obs
+                return new_transition
+            return transition
+
+        def get_config(self):
+            return {"name": self.name}
+
+        def transform_features(
+            self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+        ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+            # We do not test features here
+            return features
+
+    try:
+        step = CallableStep()
+        pipeline = DataProcessorPipeline([step])
+
+        with tempfile.TemporaryDirectory() as tmp_dir:
+            pipeline.save_pretrained(tmp_dir)
+
+            # Define a transform function
+            def double_values(x):
+                if isinstance(x, (int | float | torch.Tensor)):
+                    return x * 2
+                return x
+
+            # Load with callable override
+            loaded = DataProcessorPipeline.from_pretrained(
+                tmp_dir,
+                config_filename="dataprocessorpipeline.json",
+                overrides={"callable_step": {"transform_fn": double_values}},
+                to_transition=identity_transition,
+                to_output=identity_transition,
+            )
+
+            # Test it works
+            transition = create_transition(observation={"value": torch.tensor(5.0)})
+            result = loaded(transition)
+            assert result[TransitionKey.OBSERVATION]["value"].item() == 10.0
+    finally:
+        ProcessorStepRegistry.unregister("callable_step")
+
+
+def test_override_multiple_same_class_warning():
+    """Test behavior when multiple steps of same class exist."""
+    step1 = MockStepWithNonSerializableParam(name="step1", multiplier=1.0)
+    step2 = MockStepWithNonSerializableParam(name="step2", multiplier=2.0)
+    pipeline = DataProcessorPipeline([step1, step2])
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Override affects all instances of the class
+        loaded = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="dataprocessorpipeline.json",
+            overrides={"MockStepWithNonSerializableParam": {"multiplier": 10.0}},
+        )
+
+        # Both steps get the same override
+        assert loaded.steps[0].multiplier == 10.0
+        assert loaded.steps[1].multiplier == 10.0
+
+        # But original names are preserved
+        assert loaded.steps[0].name == "step1"
+        assert loaded.steps[1].name == "step2"
+
+
+def test_config_filename_special_characters():
+    """Test config filenames with special characters are sanitized."""
+    # Processor name with special characters
+    pipeline = DataProcessorPipeline([MockStep()], name="My/Processor\\With:Special*Chars")
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Check that filename was sanitized
+        json_files = list(Path(tmp_dir).glob("*.json"))
+        assert len(json_files) == 1
+
+        # Should have replaced special chars with underscores
+        expected_name = "my_processor_with_special_chars.json"
+        assert json_files[0].name == expected_name
+
+
+def test_state_file_naming_with_multiple_processors():
+    """Test that state files are properly prefixed with pipeline names to avoid conflicts."""
+    # Create two processors with state
+    step1 = MockStepWithTensorState(name="norm", window_size=5)
+    preprocessor = DataProcessorPipeline([step1], name="PreProcessor")
+
+    step2 = MockStepWithTensorState(name="norm", window_size=10)
+    postprocessor = DataProcessorPipeline([step2], name="PostProcessor")
+
+    # Process some data to create state
+    for i in range(3):
+        transition = create_transition(reward=float(i))
+        preprocessor(transition)
+        postprocessor(transition)
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save both processors to the same directory
+        preprocessor.save_pretrained(tmp_dir)
+        postprocessor.save_pretrained(tmp_dir)
+
+        # Check that all files exist and are distinct
+        assert (Path(tmp_dir) / "preprocessor.json").exists()
+        assert (Path(tmp_dir) / "postprocessor.json").exists()
+        assert (Path(tmp_dir) / "preprocessor_step_0.safetensors").exists()
+        assert (Path(tmp_dir) / "postprocessor_step_0.safetensors").exists()
+
+        # Load both back and verify they work correctly
+        loaded_pre = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="preprocessor.json")
+        loaded_post = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="postprocessor.json")
+
+        assert loaded_pre.name == "PreProcessor"
+        assert loaded_post.name == "PostProcessor"
+        assert loaded_pre.steps[0].window_size == 5
+        assert loaded_post.steps[0].window_size == 10
+
+
+def test_override_with_device_strings():
+    """Test overriding device parameters with string values."""
+
+    @ProcessorStepRegistry.register("device_aware_step")
+    @dataclass
+    class DeviceAwareStep(ProcessorStep):
+        device: str = "cpu"
+
+        def __init__(self, device: str = "cpu"):
+            self.device = device
+            self.buffer = torch.zeros(10, device=device)
+
+        def __call__(self, transition: EnvTransition) -> EnvTransition:
+            return transition
+
+        def get_config(self):
+            return {"device": str(self.device)}
+
+        def state_dict(self):
+            return {"buffer": self.buffer}
+
+        def load_state_dict(self, state):
+            self.buffer = state["buffer"]
+
+        def transform_features(
+            self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+        ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+            # We do not test features here
+            return features
+
+    try:
+        step = DeviceAwareStep(device="cpu")
+        pipeline = DataProcessorPipeline([step])
+
+        with tempfile.TemporaryDirectory() as tmp_dir:
+            pipeline.save_pretrained(tmp_dir)
+
+            # Override device
+            if torch.cuda.is_available():
+                loaded = DataProcessorPipeline.from_pretrained(
+                    tmp_dir,
+                    config_filename="dataprocessorpipeline.json",
+                    overrides={"device_aware_step": {"device": "cuda:0"}},
+                )
+
+                loaded_step = loaded.steps[0]
+                assert loaded_step.device == "cuda:0"
+                # Note: buffer will still be on CPU from saved state
+                # until .to() is called on the processor
+
+    finally:
+        ProcessorStepRegistry.unregister("device_aware_step")
+
+
+def test_from_pretrained_nonexistent_path():
+    """Test error handling when loading from non-existent sources."""
+    from huggingface_hub.errors import HfHubHTTPError
+
+    # Test with an invalid local path - should raise FileNotFoundError
+    with pytest.raises(FileNotFoundError):
+        DataProcessorPipeline.from_pretrained("/path/that/does/not/exist", config_filename="processor.json")
+
+    # Test with a path that doesn't exist as a directory
+    with pytest.raises(FileNotFoundError):
+        DataProcessorPipeline.from_pretrained("user/repo/extra/path", config_filename="processor.json")
+
+    # Test with a non-existent Hub repo
+    with pytest.raises((FileNotFoundError, HfHubHTTPError)):
+        DataProcessorPipeline.from_pretrained(
+            "nonexistent-user/nonexistent-repo", config_filename="processor.json"
+        )
+
+    # Test with a local directory that exists but has no config files
+    with tempfile.TemporaryDirectory() as tmp_dir, pytest.raises(FileNotFoundError):
+        # Since the directory exists but has no config, it will raise FileNotFoundError
+        DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="processor.json")
+
+
+def test_save_load_with_custom_converter_functions():
+    """Test that custom to_transition and to_output functions are NOT saved."""
+
+    def custom_to_transition(batch):
+        # Custom conversion logic
+        return {
+            TransitionKey.OBSERVATION: batch.get("obs"),
+            TransitionKey.ACTION: batch.get("act"),
+            TransitionKey.REWARD: batch.get("rew", 0.0),
+            TransitionKey.DONE: batch.get("done", False),
+            TransitionKey.TRUNCATED: batch.get("truncated", False),
+            TransitionKey.INFO: {},
+            TransitionKey.COMPLEMENTARY_DATA: {},
+        }
+
+    def custom_to_output(transition):
+        # Custom output format
+        return {
+            "obs": transition.get(TransitionKey.OBSERVATION),
+            "act": transition.get(TransitionKey.ACTION),
+            "rew": transition.get(TransitionKey.REWARD),
+            "done": transition.get(TransitionKey.DONE),
+            "truncated": transition.get(TransitionKey.TRUNCATED),
+        }
+
+    # Create processor with custom converters
+    pipeline = DataProcessorPipeline(
+        [MockStep()], to_transition=custom_to_transition, to_output=custom_to_output
+    )
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        pipeline.save_pretrained(tmp_dir)
+
+        # Load - should use default converters
+        loaded = DataProcessorPipeline.from_pretrained(tmp_dir, config_filename="dataprocessorpipeline.json")
+
+        # Verify it uses default converters by checking with standard batch format
+        batch = {
+            OBS_IMAGE: torch.randn(1, 3, 32, 32),
+            ACTION: torch.randn(1, 7),
+            REWARD: torch.tensor([1.0]),
+            DONE: torch.tensor([False]),
+            TRUNCATED: torch.tensor([False]),
+            "info": {},
+        }
+
+        # Should work with standard format (wouldn't work with custom converter)
+        result = loaded(batch)
+        # With new behavior, default to_output is _default_transition_to_batch, so result is batch dict
+        assert OBS_IMAGE in result
+
+
+class NonCompliantStep:
+    """Intentionally non-compliant: missing features."""
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        return transition
+
+
+class NonCallableStep(ProcessorStep):
+    """Intentionally non-compliant: missing __call__."""
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return features
+
+
+def test_construction_rejects_step_without_call():
+    """Test that DataProcessorPipeline rejects steps that don't inherit from ProcessorStep."""
+    with pytest.raises(TypeError, match=r"Can't instantiate abstract class NonCallableStep"):
+        DataProcessorPipeline([NonCallableStep()])
+
+    with pytest.raises(TypeError, match=r"must inherit from ProcessorStep"):
+        DataProcessorPipeline([NonCompliantStep()])
+
+
+@dataclass
+class FeatureContractAddStep(ProcessorStep):
+    """Adds a PolicyFeature"""
+
+    key: str = "a"
+    value: PolicyFeature = field(default_factory=lambda: PolicyFeature(type=FeatureType.STATE, shape=(1,)))
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        features[PipelineFeatureType.OBSERVATION][self.key] = self.value
+        return features
+
+
+@dataclass
+class FeatureContractMutateStep(ProcessorStep):
+    """Mutates a PolicyFeature"""
+
+    key: str = "a"
+    fn: Callable[[PolicyFeature | None], PolicyFeature] = identity_transition  # noqa: E731
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        features[PipelineFeatureType.OBSERVATION][self.key] = self.fn(
+            features[PipelineFeatureType.OBSERVATION].get(self.key)
+        )
+        return features
+
+
+@dataclass
+class FeatureContractBadReturnStep(ProcessorStep):
+    """Returns a non-dict"""
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        return ["not-a-dict"]
+
+
+@dataclass
+class FeatureContractRemoveStep(ProcessorStep):
+    """Removes a PolicyFeature"""
+
+    key: str
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        return transition
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        features[PipelineFeatureType.OBSERVATION].pop(self.key, None)
+        return features
+
+
+def test_features_orders_and_merges(policy_feature_factory):
+    p = DataProcessorPipeline(
+        [
+            FeatureContractAddStep("a", policy_feature_factory(FeatureType.STATE, (1,))),
+            FeatureContractMutateStep("a", lambda v: PolicyFeature(type=v.type, shape=(3,))),
+            FeatureContractAddStep("b", policy_feature_factory(FeatureType.ENV, (2,))),
+        ]
+    )
+    out = p.transform_features({PipelineFeatureType.OBSERVATION: {}})
+    assert out[PipelineFeatureType.OBSERVATION]["a"].type == FeatureType.STATE and out[
+        PipelineFeatureType.OBSERVATION
+    ]["a"].shape == (3,)
+    assert out[PipelineFeatureType.OBSERVATION]["b"].type == FeatureType.ENV and out[
+        PipelineFeatureType.OBSERVATION
+    ]["b"].shape == (2,)
+    assert_contract_is_typed(out)
+
+
+def test_features_respects_initial_without_mutation(policy_feature_factory):
+    initial = {
+        PipelineFeatureType.OBSERVATION: {
+            "seed": policy_feature_factory(FeatureType.STATE, (7,)),
+            "nested": policy_feature_factory(FeatureType.ENV, (0,)),
+        }
+    }
+    p = DataProcessorPipeline(
+        [
+            FeatureContractMutateStep("seed", lambda v: PolicyFeature(type=v.type, shape=(v.shape[0] + 1,))),
+            FeatureContractMutateStep(
+                "nested", lambda v: PolicyFeature(type=v.type, shape=(v.shape[0] + 5,))
+            ),
+        ]
+    )
+    out = p.transform_features(initial_features=initial)
+
+    assert out[PipelineFeatureType.OBSERVATION]["seed"].shape == (8,)
+    assert out[PipelineFeatureType.OBSERVATION]["nested"].shape == (5,)
+    # Initial dict must be preserved
+    assert initial[PipelineFeatureType.OBSERVATION]["seed"].shape == (7,)
+    assert initial[PipelineFeatureType.OBSERVATION]["nested"].shape == (0,)
+
+    assert_contract_is_typed(out)
+
+
+def test_features_execution_order_tracking():
+    class Track(ProcessorStep):
+        def __init__(self, label):
+            self.label = label
+
+        def __call__(self, transition: EnvTransition) -> EnvTransition:
+            return transition
+
+        def transform_features(
+            self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+        ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+            code = {"A": 1, "B": 2, "C": 3}[self.label]
+            pf = features[PipelineFeatureType.OBSERVATION].get(
+                "order", PolicyFeature(type=FeatureType.ENV, shape=())
+            )
+            features[PipelineFeatureType.OBSERVATION]["order"] = PolicyFeature(
+                type=pf.type, shape=pf.shape + (code,)
+            )
+            return features
+
+    out = DataProcessorPipeline([Track("A"), Track("B"), Track("C")]).transform_features(
+        initial_features={PipelineFeatureType.OBSERVATION: {}}
+    )
+    assert out[PipelineFeatureType.OBSERVATION]["order"].shape == (1, 2, 3)
+
+
+def test_features_remove_key(policy_feature_factory):
+    p = DataProcessorPipeline(
+        [
+            FeatureContractAddStep("a", policy_feature_factory(FeatureType.STATE, (1,))),
+            FeatureContractRemoveStep("a"),
+        ]
+    )
+    out = p.transform_features({PipelineFeatureType.OBSERVATION: {}})
+    assert "a" not in out[PipelineFeatureType.OBSERVATION]
+
+
+def test_features_remove_from_initial(policy_feature_factory):
+    initial = {
+        PipelineFeatureType.OBSERVATION: {
+            "keep": policy_feature_factory(FeatureType.STATE, (1,)),
+            "drop": policy_feature_factory(FeatureType.STATE, (1,)),
+        },
+    }
+    p = DataProcessorPipeline([FeatureContractRemoveStep("drop")])
+    out = p.transform_features(initial_features=initial)
+    assert (
+        "drop" not in out[PipelineFeatureType.OBSERVATION]
+        and out[PipelineFeatureType.OBSERVATION]["keep"] == initial[PipelineFeatureType.OBSERVATION]["keep"]
+    )
+
+
+@dataclass
+class AddActionEEAndJointFeatures(ProcessorStep):
+    """Adds both EE and JOINT action features."""
+
+    def __call__(self, tr):
+        return tr
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # EE features
+        features[PipelineFeatureType.ACTION]["action.ee.x"] = float
+        features[PipelineFeatureType.ACTION]["action.ee.y"] = float
+        # JOINT features
+        features[PipelineFeatureType.ACTION]["action.j1.pos"] = float
+        features[PipelineFeatureType.ACTION]["action.j2.pos"] = float
+        return features
+
+
+@dataclass
+class AddObservationStateFeatures(ProcessorStep):
+    """Adds state features (and optionally an image spec to test precedence)."""
+
+    add_front_image: bool = False
+    front_image_shape: tuple = (240, 320, 3)
+
+    def __call__(self, tr):
+        return tr
+
+    def transform_features(
+        self, features: dict[PipelineFeatureType, dict[str, PolicyFeature]]
+    ) -> dict[PipelineFeatureType, dict[str, PolicyFeature]]:
+        # State features (mix EE and a joint state)
+        features[PipelineFeatureType.OBSERVATION][f"{OBS_STATE}.ee.x"] = float
+        features[PipelineFeatureType.OBSERVATION][f"{OBS_STATE}.j1.pos"] = float
+        if self.add_front_image:
+            features[PipelineFeatureType.OBSERVATION][f"{OBS_IMAGES}.front"] = self.front_image_shape
+        return features
+
+
+def test_aggregate_joint_action_only():
+    rp = DataProcessorPipeline([AddActionEEAndJointFeatures()])
+    initial = {PipelineFeatureType.OBSERVATION: {"front": (480, 640, 3)}, PipelineFeatureType.ACTION: {}}
+
+    out = aggregate_pipeline_dataset_features(
+        pipeline=rp,
+        initial_features=initial,
+        use_videos=True,
+        patterns=["action.j1.pos", "action.j2.pos"],
+    )
+
+    # Expect only ACTION with joint names
+    assert ACTION in out and OBS_STATE not in out
+    assert out[ACTION]["dtype"] == "float32"
+    assert set(out[ACTION]["names"]) == {"j1.pos", "j2.pos"}
+    assert out[ACTION]["shape"] == (len(out[ACTION]["names"]),)
+
+
+def test_aggregate_ee_action_and_observation_with_videos():
+    rp = DataProcessorPipeline([AddActionEEAndJointFeatures(), AddObservationStateFeatures()])
+    initial = {"front": (480, 640, 3), "side": (720, 1280, 3)}
+
+    out = aggregate_pipeline_dataset_features(
+        pipeline=rp,
+        initial_features={PipelineFeatureType.OBSERVATION: initial, PipelineFeatureType.ACTION: {}},
+        use_videos=True,
+        patterns=["action.ee", OBS_STATE],
+    )
+
+    # Action should pack only EE names
+    assert ACTION in out
+    assert set(out[ACTION]["names"]) == {"ee.x", "ee.y"}
+    assert out[ACTION]["dtype"] == "float32"
+
+    # Observation state should pack both ee.x and j1.pos as a vector
+    assert OBS_STATE in out
+    assert set(out[OBS_STATE]["names"]) == {"ee.x", "j1.pos"}
+    assert out[OBS_STATE]["dtype"] == "float32"
+
+    # Cameras from initial_features appear as videos
+    for cam in ("front", "side"):
+        key = f"{OBS_IMAGES}.{cam}"
+        assert key in out
+        assert out[key]["dtype"] == "video"
+        assert out[key]["shape"] == initial[cam]
+        assert out[key]["names"] == ["height", "width", "channels"]
+
+
+def test_aggregate_both_action_types():
+    rp = DataProcessorPipeline([AddActionEEAndJointFeatures()])
+    out = aggregate_pipeline_dataset_features(
+        pipeline=rp,
+        initial_features={PipelineFeatureType.ACTION: {}, PipelineFeatureType.OBSERVATION: {}},
+        use_videos=True,
+        patterns=["action.ee", "action.j1", "action.j2.pos"],
+    )
+
+    assert ACTION in out
+    expected = {"ee.x", "ee.y", "j1.pos", "j2.pos"}
+    assert set(out[ACTION]["names"]) == expected
+    assert out[ACTION]["shape"] == (len(expected),)
+
+
+def test_aggregate_images_when_use_videos_false():
+    rp = DataProcessorPipeline([AddObservationStateFeatures(add_front_image=True)])
+    initial = {"back": (480, 640, 3)}
+
+    out = aggregate_pipeline_dataset_features(
+        pipeline=rp,
+        initial_features={PipelineFeatureType.ACTION: {}, PipelineFeatureType.OBSERVATION: initial},
+        use_videos=False,  # expect "image" dtype
+        patterns=None,
+    )
+
+    key = f"{OBS_IMAGES}.back"
+    key_front = f"{OBS_IMAGES}.front"
+    assert key not in out
+    assert key_front not in out
+
+
+def test_aggregate_images_when_use_videos_true():
+    rp = DataProcessorPipeline([AddObservationStateFeatures(add_front_image=True)])
+    initial = {"back": (480, 640, 3)}
+
+    out = aggregate_pipeline_dataset_features(
+        pipeline=rp,
+        initial_features={PipelineFeatureType.OBSERVATION: initial, PipelineFeatureType.ACTION: {}},
+        use_videos=True,
+        patterns=None,
+    )
+
+    key = f"{OBS_IMAGES}.front"
+    key_back = f"{OBS_IMAGES}.back"
+    assert key in out
+    assert key_back in out
+    assert out[key]["dtype"] == "video"
+    assert out[key_back]["dtype"] == "video"
+    assert out[key_back]["shape"] == initial["back"]
+
+
+def test_initial_camera_not_overridden_by_step_image():
+    # Step explicitly sets a different front image shape; initial has another shape.
+    # aggregate_pipeline_dataset_features should keep the step's value (setdefault behavior on initial cams).
+    rp = DataProcessorPipeline(
+        [AddObservationStateFeatures(add_front_image=True, front_image_shape=(240, 320, 3))]
+    )
+    initial = {"front": (480, 640, 3)}  # should NOT override the step-provided (240, 320, 3)
+
+    out = aggregate_pipeline_dataset_features(
+        pipeline=rp,
+        initial_features={PipelineFeatureType.ACTION: {}, PipelineFeatureType.OBSERVATION: initial},
+        use_videos=True,
+        patterns=[f"{OBS_IMAGES}.front"],
+    )
+
+    key = f"{OBS_IMAGES}.front"
+    assert key in out
+    assert out[key]["shape"] == (240, 320, 3)  # from the step, not from initial
diff --git a/lerobot/tests/processor/test_pipeline_from_pretrained_helpers.py b/lerobot/tests/processor/test_pipeline_from_pretrained_helpers.py
new file mode 100644
index 0000000000000000000000000000000000000000..89d45cbaded95b31f42607af29c2a7a3ff200eea
--- /dev/null
+++ b/lerobot/tests/processor/test_pipeline_from_pretrained_helpers.py
@@ -0,0 +1,259 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Tests for DataProcessorPipeline.from_pretrained helper methods.
+
+These tests focus on the individual private methods that were extracted from
+the main from_pretrained method to improve modularity and testability.
+"""
+
+import json
+import tempfile
+from pathlib import Path
+
+import pytest
+
+from lerobot.processor.pipeline import DataProcessorPipeline, ProcessorMigrationError
+
+# Simplified Config Loading Tests
+
+
+def test_load_config_directory():
+    """Test loading config from directory."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create a config file
+        config_file = tmp_path / "processor.json"
+        test_config = {"name": "TestProcessor", "steps": []}
+        config_file.write_text(json.dumps(test_config))
+
+        # Load from directory
+        loaded_config, base_path = DataProcessorPipeline._load_config(str(tmp_path), "processor.json", {})
+
+        assert loaded_config == test_config
+        assert base_path == tmp_path
+
+
+def test_load_config_single_file():
+    """Test loading config from a single file path."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create a config file
+        config_file = tmp_path / "processor.json"
+        test_config = {"name": "TestProcessor", "steps": []}
+        config_file.write_text(json.dumps(test_config))
+
+        # Load using file path directly
+        loaded_config, base_path = DataProcessorPipeline._load_config(
+            str(config_file), "any_filename_ignored", {}
+        )
+
+        assert loaded_config == test_config
+        assert base_path == tmp_path
+
+
+def test_load_config_directory_file_not_found():
+    """Test directory loading when config file doesn't exist."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Directory exists but no processor.json
+        with pytest.raises(FileNotFoundError, match="not found in directory"):
+            DataProcessorPipeline._load_config(str(tmp_path), "processor.json", {})
+
+
+def test_load_config_directory_with_migration_detection():
+    """Test that missing config triggers migration detection."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create old-style config to trigger migration
+        (tmp_path / "config.json").write_text(json.dumps({"type": "act"}))
+
+        # Try to load processor.json (doesn't exist), should trigger migration
+        with pytest.raises(ProcessorMigrationError):
+            DataProcessorPipeline._load_config(str(tmp_path), "processor.json", {})
+
+
+def test_load_config_nonexistent_path_tries_hub():
+    """Test that nonexistent paths try Hub (simplified logic)."""
+    # This path doesn't exist locally, should try Hub
+    with pytest.raises(FileNotFoundError, match="on the HuggingFace Hub"):
+        DataProcessorPipeline._load_config("nonexistent/path", "processor.json", {})
+
+
+# Config Validation Tests
+
+
+def test_validate_loaded_config_valid_config():
+    """Test validation with valid processor config."""
+    valid_config = {"name": "TestProcessor", "steps": []}
+
+    # Should not raise any exception
+    DataProcessorPipeline._validate_loaded_config("any-path", valid_config, "processor.json")
+
+
+def test_validate_loaded_config_invalid_config():
+    """Test validation with invalid processor config."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Create non-processor config to trigger migration
+        (tmp_path / "config.json").write_text(json.dumps({"type": "act"}))
+
+        invalid_config = {"type": "act", "hidden_dim": 256}
+
+        with pytest.raises(ProcessorMigrationError):
+            DataProcessorPipeline._validate_loaded_config(str(tmp_path), invalid_config, "config.json")
+
+
+def test_validate_loaded_config_invalid_config_no_migration():
+    """Test validation with invalid config when no migration is detected."""
+    # Non-directory path (Hub repo) - no migration detection
+    invalid_config = {"type": "act", "hidden_dim": 256}
+
+    with pytest.raises(ValueError, match="not a valid processor configuration"):
+        DataProcessorPipeline._validate_loaded_config("user/repo", invalid_config, "config.json")
+
+
+# Step Class Resolution Tests
+
+
+def test_resolve_step_class_registry_name():
+    """Test resolution using registry name."""
+    from lerobot.processor.pipeline import ProcessorStep, ProcessorStepRegistry
+
+    # Register a test step
+    @ProcessorStepRegistry.register("test_step")
+    class TestStep(ProcessorStep):
+        def __call__(self, transition):
+            return transition
+
+        def transform_features(self, features):
+            return features
+
+    try:
+        step_entry = {"registry_name": "test_step"}
+        step_class, step_key = DataProcessorPipeline._resolve_step_class(step_entry)
+
+        assert step_class is TestStep
+        assert step_key == "test_step"
+    finally:
+        ProcessorStepRegistry.unregister("test_step")
+
+
+def test_resolve_step_class_registry_name_not_found():
+    """Test resolution with non-existent registry name."""
+    step_entry = {"registry_name": "nonexistent_step"}
+
+    with pytest.raises(ImportError, match="Failed to load processor step from registry"):
+        DataProcessorPipeline._resolve_step_class(step_entry)
+
+
+def test_resolve_step_class_import_path():
+    """Test resolution using full import path."""
+    # Use a valid existing class (this should work)
+    step_entry = {"class": "lerobot.processor.pipeline.ProcessorStep"}
+
+    # This should succeed - ProcessorStep can be imported, just not instantiated
+    step_class, step_key = DataProcessorPipeline._resolve_step_class(step_entry)
+
+    from lerobot.processor.pipeline import ProcessorStep
+
+    assert step_class is ProcessorStep
+    assert step_key == "ProcessorStep"
+
+
+def test_resolve_step_class_invalid_import_path():
+    """Test resolution with invalid import path."""
+    step_entry = {"class": "nonexistent.module.ClassName"}
+
+    with pytest.raises(ImportError, match="Failed to load processor step"):
+        DataProcessorPipeline._resolve_step_class(step_entry)
+
+
+# Override Validation Tests
+
+
+def test_validate_overrides_used_all_used():
+    """Test validation when all overrides are used."""
+    # Empty set means all overrides were used
+    remaining_overrides = set()
+    config = {"steps": [{"class": "SomeStep"}]}
+
+    # Should not raise
+    DataProcessorPipeline._validate_overrides_used(remaining_overrides, config)
+
+
+def test_validate_overrides_used_some_unused():
+    """Test validation when some overrides are unused."""
+    remaining_overrides = {"NonExistentStep", "AnotherMissingStep"}
+    config = {
+        "steps": [
+            {"registry_name": "normalize_step"},
+            {"class": "some.module.TransformStep"},
+        ]
+    }
+
+    with pytest.raises(KeyError, match="Override keys.*do not match any step"):
+        DataProcessorPipeline._validate_overrides_used(remaining_overrides, config)
+
+
+def test_validate_overrides_used_helpful_error_message():
+    """Test that error message includes available step keys."""
+    remaining_overrides = {"WrongStep"}
+    config = {
+        "steps": [
+            {"registry_name": "correct_step"},
+            {"class": "module.path.CorrectClass"},
+        ]
+    }
+
+    with pytest.raises(KeyError) as exc_info:
+        DataProcessorPipeline._validate_overrides_used(remaining_overrides, config)
+
+    error_msg = str(exc_info.value)
+    assert "Available step keys" in error_msg
+    assert "correct_step" in error_msg
+    assert "CorrectClass" in error_msg
+
+
+# Integration Tests for Simplified Logic
+
+
+def test_simplified_three_way_loading():
+    """Test that the simplified 3-way loading logic works correctly."""
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        tmp_path = Path(tmp_dir)
+
+        # Test 1: Directory loading
+        config_file = tmp_path / "processor.json"
+        test_config = {"name": "DirectoryTest", "steps": []}
+        config_file.write_text(json.dumps(test_config))
+
+        loaded_config, base_path = DataProcessorPipeline._load_config(str(tmp_path), "processor.json", {})
+        assert loaded_config["name"] == "DirectoryTest"
+        assert base_path == tmp_path
+
+        # Test 2: Single file loading
+        loaded_config, base_path = DataProcessorPipeline._load_config(
+            str(config_file), "ignored_filename", {}
+        )
+        assert loaded_config["name"] == "DirectoryTest"
+        assert base_path == tmp_path
diff --git a/lerobot/tests/processor/test_policy_robot_bridge.py b/lerobot/tests/processor/test_policy_robot_bridge.py
new file mode 100644
index 0000000000000000000000000000000000000000..6269c508f9deda0d5c2be34722a4007c63be445b
--- /dev/null
+++ b/lerobot/tests/processor/test_policy_robot_bridge.py
@@ -0,0 +1,526 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import tempfile
+from pathlib import Path
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType
+from lerobot.processor import (
+    DataProcessorPipeline,
+    PolicyActionToRobotActionProcessorStep,
+    ProcessorStepRegistry,
+    RobotActionToPolicyActionProcessorStep,
+)
+from lerobot.processor.converters import identity_transition
+from lerobot.utils.constants import ACTION
+from tests.conftest import assert_contract_is_typed
+
+
+def test_robot_to_policy_basic_action_conversion():
+    """Test basic robot action to policy action conversion."""
+    motor_names = ["joint1", "joint2", "joint3"]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+
+    robot_action = {
+        "joint1.pos": 1.0,
+        "joint2.pos": 2.0,
+        "joint3.pos": 3.0,
+    }
+
+    policy_action = processor.action(robot_action)
+
+    assert isinstance(policy_action, torch.Tensor)
+    assert policy_action.shape == (3,)
+    torch.testing.assert_close(policy_action, torch.tensor([1.0, 2.0, 3.0]))
+
+
+def test_robot_to_policy_action_conversion_preserves_order():
+    """Test that motor names order is preserved in conversion."""
+    motor_names = ["gripper", "arm", "wrist"]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+
+    robot_action = {
+        "arm.pos": 10.0,
+        "gripper.pos": 5.0,
+        "wrist.pos": 15.0,
+    }
+
+    policy_action = processor.action(robot_action)
+
+    expected = torch.tensor([5.0, 10.0, 15.0])
+    torch.testing.assert_close(policy_action, expected)
+
+
+def test_robot_to_policy_action_conversion_with_floats_and_tensors():
+    """Test conversion with mixed float and tensor values."""
+    motor_names = ["joint1", "joint2"]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+
+    robot_action = {
+        "joint1.pos": torch.tensor(1.5),
+        "joint2.pos": 2.5,  # Regular float
+    }
+
+    policy_action = processor.action(robot_action)
+
+    assert isinstance(policy_action, torch.Tensor)
+    torch.testing.assert_close(policy_action, torch.tensor([1.5, 2.5]))
+
+
+def test_robot_to_policy_action_length_mismatch_error():
+    """Test error when robot action length doesn't match motor names."""
+    motor_names = ["joint1", "joint2", "joint3"]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+
+    # Too few actions
+    robot_action = {"joint1.pos": 1.0, "joint2.pos": 2.0}
+
+    with pytest.raises(ValueError, match="Action must have 3 elements, got 2"):
+        processor.action(robot_action)
+
+    robot_action = {
+        "joint1.pos": 1.0,
+        "joint2.pos": 2.0,
+        "joint3.pos": 3.0,
+        "extra.pos": 4.0,
+    }
+
+    with pytest.raises(ValueError, match="Action must have 3 elements, got 4"):
+        processor.action(robot_action)
+
+
+def test_robot_to_policy_missing_motor_key_error():
+    """Test error when robot action is missing expected motor keys."""
+    motor_names = ["joint1", "joint2"]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+
+    robot_action = {
+        "joint1.pos": 1.0,
+        "wrong_key.pos": 2.0,
+    }
+
+    with pytest.raises(KeyError):
+        processor.action(robot_action)
+
+
+def test_robot_to_policy_transform_features():
+    """Test feature transformation for robot to policy action processor."""
+    motor_names = ["joint1", "joint2", "joint3"]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+
+    features = {
+        PipelineFeatureType.ACTION: {
+            "joint1.pos": {"type": FeatureType.ACTION, "shape": (1,)},
+            "joint2.pos": {"type": FeatureType.ACTION, "shape": (1,)},
+            "joint3.pos": {"type": FeatureType.ACTION, "shape": (1,)},
+            "other_data": {"type": FeatureType.ENV, "shape": (1,)},
+        }
+    }
+
+    transformed = processor.transform_features(features)
+
+    assert ACTION in transformed[PipelineFeatureType.ACTION]
+    action_feature = transformed[PipelineFeatureType.ACTION][ACTION]
+    assert action_feature.type == FeatureType.ACTION
+    assert action_feature.shape == (3,)
+
+    assert "joint1.pos" in transformed[PipelineFeatureType.ACTION]
+    assert "joint2.pos" in transformed[PipelineFeatureType.ACTION]
+    assert "joint3.pos" in transformed[PipelineFeatureType.ACTION]
+
+    assert "other_data" in transformed[PipelineFeatureType.ACTION]
+
+
+def test_robot_to_policy_get_config():
+    """Test configuration serialization."""
+    motor_names = ["motor1", "motor2"]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+
+    config = processor.get_config()
+    assert config == {"motor_names": motor_names}
+
+
+def test_robot_to_policy_state_dict():
+    """Test state dict operations."""
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=["joint1"])
+
+    state = processor.state_dict()
+    assert state == {}
+
+    processor.load_state_dict({})
+
+
+def test_robot_to_policy_single_motor():
+    """Test with single motor."""
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=["single_joint"])
+
+    robot_action = {"single_joint.pos": 42.0}
+    policy_action = processor.action(robot_action)
+
+    assert policy_action.shape == (1,)
+    torch.testing.assert_close(policy_action, torch.tensor([42.0]))
+
+
+def test_policy_to_robot_basic_action_conversion():
+    """Test basic policy action to robot action conversion."""
+    motor_names = ["joint1", "joint2", "joint3"]
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=motor_names)
+
+    policy_action = torch.tensor([1.0, 2.0, 3.0])
+    robot_action = processor.action(policy_action)
+
+    assert isinstance(robot_action, dict)
+    assert len(robot_action) == 3
+
+    expected = {
+        "joint1.pos": 1.0,
+        "joint2.pos": 2.0,
+        "joint3.pos": 3.0,
+    }
+
+    for key, expected_value in expected.items():
+        assert key in robot_action
+        actual_value = robot_action[key]
+        if isinstance(actual_value, torch.Tensor):
+            actual_value = actual_value.item()
+        assert actual_value == pytest.approx(expected_value)
+
+
+def test_policy_to_robot_action_conversion_preserves_order():
+    """Test that motor names order corresponds to tensor indices."""
+    motor_names = ["gripper", "arm", "wrist"]
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=motor_names)
+
+    policy_action = torch.tensor([5.0, 10.0, 15.0])
+    robot_action = processor.action(policy_action)
+
+    assert robot_action["gripper.pos"] == pytest.approx(5.0)
+    assert robot_action["arm.pos"] == pytest.approx(10.0)
+    assert robot_action["wrist.pos"] == pytest.approx(15.0)
+
+
+def test_policy_to_robot_action_conversion_with_numpy_input():
+    """Test conversion with numpy array input."""
+    import numpy as np
+
+    motor_names = ["joint1", "joint2"]
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=motor_names)
+
+    policy_action = np.array([1.5, 2.5])
+    robot_action = processor.action(policy_action)
+
+    assert robot_action["joint1.pos"] == pytest.approx(1.5)
+    assert robot_action["joint2.pos"] == pytest.approx(2.5)
+
+
+def test_policy_to_robot_action_length_mismatch_error():
+    """Test error when policy action length doesn't match motor names."""
+    motor_names = ["joint1", "joint2", "joint3"]
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=motor_names)
+
+    policy_action = torch.tensor([1.0, 2.0])
+
+    with pytest.raises(ValueError, match="Action must have 3 elements, got 2"):
+        processor.action(policy_action)
+
+    policy_action = torch.tensor([1.0, 2.0, 3.0, 4.0])
+
+    with pytest.raises(ValueError, match="Action must have 3 elements, got 4"):
+        processor.action(policy_action)
+
+
+def test_policy_to_robot_transform_features():
+    """Test feature transformation for policy to robot action processor."""
+    motor_names = ["joint1", "joint2"]
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=motor_names)
+
+    features = {
+        PipelineFeatureType.ACTION: {
+            ACTION: {"type": FeatureType.ACTION, "shape": (2,)},
+            "other_data": {"type": FeatureType.ENV, "shape": (1,)},
+        }
+    }
+
+    transformed = processor.transform_features(features)
+
+    assert "joint1.pos" in transformed[PipelineFeatureType.ACTION]
+    assert "joint2.pos" in transformed[PipelineFeatureType.ACTION]
+
+    for motor in motor_names:
+        motor_feature = transformed[PipelineFeatureType.ACTION][f"{motor}.pos"]
+        assert motor_feature.type == FeatureType.ACTION
+        assert motor_feature.shape == (1,)
+
+    assert ACTION in transformed[PipelineFeatureType.ACTION]
+
+    assert "other_data" in transformed[PipelineFeatureType.ACTION]
+
+
+def test_policy_to_robot_get_config():
+    """Test configuration serialization."""
+    motor_names = ["motor1", "motor2"]
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=motor_names)
+
+    config = processor.get_config()
+    assert config == {"motor_names": motor_names}
+
+
+def test_policy_to_robot_state_dict():
+    """Test state dict operations."""
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=["joint1"])
+
+    state = processor.state_dict()
+    assert state == {}
+
+    processor.load_state_dict({})
+
+
+def test_policy_to_robot_single_motor():
+    """Test with single motor."""
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=["single_joint"])
+
+    policy_action = torch.tensor([42.0])
+    robot_action = processor.action(policy_action)
+
+    assert len(robot_action) == 1
+    assert robot_action["single_joint.pos"] == pytest.approx(42.0)
+
+
+def test_robot_to_policy_registry():
+    """Test RobotActionToPolicyActionProcessorStep registry."""
+    assert "robot_action_to_policy_action_processor" in ProcessorStepRegistry.list()
+
+    retrieved_class = ProcessorStepRegistry.get("robot_action_to_policy_action_processor")
+    assert retrieved_class is RobotActionToPolicyActionProcessorStep
+
+    instance = retrieved_class(motor_names=["test"])
+    assert isinstance(instance, RobotActionToPolicyActionProcessorStep)
+    assert instance.motor_names == ["test"]
+
+
+def test_policy_to_robot_registry():
+    """Test PolicyActionToRobotActionProcessorStep registry."""
+    assert "policy_action_to_robot_action_processor" in ProcessorStepRegistry.list()
+
+    retrieved_class = ProcessorStepRegistry.get("policy_action_to_robot_action_processor")
+    assert retrieved_class is PolicyActionToRobotActionProcessorStep
+
+    instance = retrieved_class(motor_names=["test"])
+    assert isinstance(instance, PolicyActionToRobotActionProcessorStep)
+    assert instance.motor_names == ["test"]
+
+
+def test_save_and_load_robot_to_policy():
+    """Test saving and loading RobotActionToPolicyActionProcessorStep."""
+    motor_names = ["joint1", "joint2", "joint3"]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+    pipeline = DataProcessorPipeline([processor], name="TestRobotToPolicy")
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save pipeline
+        pipeline.save_pretrained(tmp_dir)
+
+        # Check config file exists
+        config_path = Path(tmp_dir) / "testrobottopolicy.json"
+        assert config_path.exists()
+
+        # Load pipeline
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            "testrobottopolicy.json",
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        assert loaded_pipeline.name == "TestRobotToPolicy"
+        assert len(loaded_pipeline) == 1
+
+        # Check loaded processor
+        loaded_processor = loaded_pipeline.steps[0]
+        assert isinstance(loaded_processor, RobotActionToPolicyActionProcessorStep)
+        assert loaded_processor.motor_names == motor_names
+
+        # Test functionality after loading
+        robot_action = {"joint1.pos": 1.0, "joint2.pos": 2.0, "joint3.pos": 3.0}
+        policy_action = loaded_processor.action(robot_action)
+        torch.testing.assert_close(policy_action, torch.tensor([1.0, 2.0, 3.0]))
+
+
+def test_save_and_load_policy_to_robot():
+    """Test saving and loading PolicyActionToRobotActionProcessorStep."""
+    motor_names = ["motor_a", "motor_b"]
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=motor_names)
+    pipeline = DataProcessorPipeline([processor], name="TestPolicyToRobot")
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save pipeline
+        pipeline.save_pretrained(tmp_dir)
+
+        # Load pipeline
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            "testpolicytorobot.json",
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        loaded_processor = loaded_pipeline.steps[0]
+        assert isinstance(loaded_processor, PolicyActionToRobotActionProcessorStep)
+        assert loaded_processor.motor_names == motor_names
+
+        policy_action = torch.tensor([10.0, 20.0])
+        robot_action = loaded_processor.action(policy_action)
+        assert robot_action["motor_a.pos"] == pytest.approx(10.0)
+        assert robot_action["motor_b.pos"] == pytest.approx(20.0)
+
+
+# Integration and chaining tests
+
+
+def test_round_trip_conversion():
+    """Test that robot->policy->robot conversion preserves values."""
+    motor_names = ["joint1", "joint2", "joint3"]
+    robot_to_policy = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+    policy_to_robot = PolicyActionToRobotActionProcessorStep(motor_names=motor_names)
+
+    original_robot_action = {
+        "joint1.pos": 1.5,
+        "joint2.pos": -2.3,
+        "joint3.pos": 0.7,
+    }
+
+    policy_action = robot_to_policy.action(original_robot_action)
+    final_robot_action = policy_to_robot.action(policy_action)
+
+    for key in original_robot_action:
+        original_val = original_robot_action[key]
+        final_val = final_robot_action[key]
+        if isinstance(final_val, torch.Tensor):
+            final_val = final_val.item()
+        assert final_val == pytest.approx(original_val, abs=1e-6)
+
+
+def test_chained_processors_in_pipeline():
+    """Test both processors chained in a pipeline."""
+    motor_names = ["joint1", "joint2"]
+    robot_to_policy = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+    policy_to_robot = PolicyActionToRobotActionProcessorStep(motor_names=motor_names)
+
+    pipeline = DataProcessorPipeline(
+        [robot_to_policy, policy_to_robot],
+        to_transition=identity_transition,
+        to_output=identity_transition,
+    )
+
+    assert len(pipeline.steps) == 2
+    assert isinstance(pipeline.steps[0], RobotActionToPolicyActionProcessorStep)
+    assert isinstance(pipeline.steps[1], PolicyActionToRobotActionProcessorStep)
+
+
+def test_robot_to_policy_features_contract(policy_feature_factory):
+    """Test feature transformation maintains proper typing contract."""
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=["j1", "j2"])
+    features = {
+        PipelineFeatureType.ACTION: {
+            "j1.pos": policy_feature_factory(FeatureType.ACTION, (1,)),
+            "j2.pos": policy_feature_factory(FeatureType.ACTION, (1,)),
+            "other": policy_feature_factory(FeatureType.ENV, (3,)),
+        }
+    }
+
+    out = processor.transform_features(features.copy())
+
+    assert_contract_is_typed(out)
+
+    assert ACTION in out[PipelineFeatureType.ACTION]
+    action_feature = out[PipelineFeatureType.ACTION][ACTION]
+    assert action_feature.type == FeatureType.ACTION
+    assert action_feature.shape == (2,)
+
+
+def test_policy_to_robot_features_contract(policy_feature_factory):
+    """Test feature transformation maintains proper typing contract."""
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=["m1", "m2", "m3"])
+    features = {
+        PipelineFeatureType.ACTION: {
+            ACTION: policy_feature_factory(FeatureType.ACTION, (3,)),
+            "other": policy_feature_factory(FeatureType.ENV, (1,)),
+        }
+    }
+
+    out = processor.transform_features(features.copy())
+
+    assert_contract_is_typed(out)
+
+    for motor in ["m1", "m2", "m3"]:
+        key = f"{motor}.pos"
+        assert key in out[PipelineFeatureType.ACTION]
+        motor_feature = out[PipelineFeatureType.ACTION][key]
+        assert motor_feature.type == FeatureType.ACTION
+        assert motor_feature.shape == (1,)
+
+
+def test_empty_motor_names_list():
+    """Test behavior with empty motor names list."""
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=[])
+
+    robot_action = {}
+    policy_action = processor.action(robot_action)
+
+    assert isinstance(policy_action, torch.Tensor)
+    assert policy_action.shape == (0,)
+
+
+def test_empty_motor_names_list_policy_to_robot():
+    """Test PolicyActionToRobotActionProcessorStep with empty motor names."""
+    processor = PolicyActionToRobotActionProcessorStep(motor_names=[])
+
+    policy_action = torch.tensor([])
+    robot_action = processor.action(policy_action)
+
+    assert isinstance(robot_action, dict)
+    assert len(robot_action) == 0
+
+
+def test_very_long_motor_names():
+    """Test with many motor names."""
+    motor_names = [f"joint_{i}" for i in range(100)]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+
+    robot_action = {f"joint_{i}.pos": float(i) for i in range(100)}
+    policy_action = processor.action(robot_action)
+
+    assert policy_action.shape == (100,)
+    expected = torch.tensor([float(i) for i in range(100)])
+    torch.testing.assert_close(policy_action, expected)
+
+
+def test_special_characters_in_motor_names():
+    """Test with special characters in motor names."""
+    motor_names = ["motor-1", "motor_2", "motor.3"]
+    processor = RobotActionToPolicyActionProcessorStep(motor_names=motor_names)
+
+    robot_action = {
+        "motor-1.pos": 1.0,
+        "motor_2.pos": 2.0,
+        "motor.3.pos": 3.0,
+    }
+
+    policy_action = processor.action(robot_action)
+    torch.testing.assert_close(policy_action, torch.tensor([1.0, 2.0, 3.0]))
diff --git a/lerobot/tests/processor/test_rename_processor.py b/lerobot/tests/processor/test_rename_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..efb9f93289377f988507225a27d21262921c26bd
--- /dev/null
+++ b/lerobot/tests/processor/test_rename_processor.py
@@ -0,0 +1,498 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import tempfile
+from pathlib import Path
+
+import numpy as np
+import torch
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType
+from lerobot.processor import (
+    DataProcessorPipeline,
+    ProcessorStepRegistry,
+    RenameObservationsProcessorStep,
+    TransitionKey,
+)
+from lerobot.processor.converters import create_transition, identity_transition
+from lerobot.processor.rename_processor import rename_stats
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_IMAGES, OBS_STATE
+from tests.conftest import assert_contract_is_typed
+
+
+def test_basic_renaming():
+    """Test basic key renaming functionality."""
+    rename_map = {
+        "old_key1": "new_key1",
+        "old_key2": "new_key2",
+    }
+    processor = RenameObservationsProcessorStep(rename_map=rename_map)
+
+    observation = {
+        "old_key1": torch.tensor([1.0, 2.0]),
+        "old_key2": np.array([3.0, 4.0]),
+        "unchanged_key": "keep_me",
+    }
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check renamed keys
+    assert "new_key1" in processed_obs
+    assert "new_key2" in processed_obs
+    assert "old_key1" not in processed_obs
+    assert "old_key2" not in processed_obs
+
+    # Check values are preserved
+    torch.testing.assert_close(processed_obs["new_key1"], torch.tensor([1.0, 2.0]))
+    np.testing.assert_array_equal(processed_obs["new_key2"], np.array([3.0, 4.0]))
+
+    # Check unchanged key is preserved
+    assert processed_obs["unchanged_key"] == "keep_me"
+
+
+def test_empty_rename_map():
+    """Test processor with empty rename map (should pass through unchanged)."""
+    processor = RenameObservationsProcessorStep(rename_map={})
+
+    observation = {
+        "key1": torch.tensor([1.0]),
+        "key2": "value2",
+    }
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # All keys should be unchanged
+    assert processed_obs.keys() == observation.keys()
+    torch.testing.assert_close(processed_obs["key1"], observation["key1"])
+    assert processed_obs["key2"] == observation["key2"]
+
+
+def test_none_observation():
+    """Test processor with None observation."""
+    processor = RenameObservationsProcessorStep(rename_map={"old": "new"})
+
+    transition = create_transition(observation={})
+    result = processor(transition)
+
+    # Should return transition unchanged
+    assert result == transition
+
+
+def test_overlapping_rename():
+    """Test renaming when new names might conflict."""
+    rename_map = {
+        "a": "b",
+        "b": "c",  # This creates a potential conflict
+    }
+    processor = RenameObservationsProcessorStep(rename_map=rename_map)
+
+    observation = {
+        "a": 1,
+        "b": 2,
+        "x": 3,
+    }
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that renaming happens correctly
+    assert "a" not in processed_obs
+    assert processed_obs["b"] == 1  # 'a' renamed to 'b'
+    assert processed_obs["c"] == 2  # original 'b' renamed to 'c'
+    assert processed_obs["x"] == 3
+
+
+def test_partial_rename():
+    """Test renaming only some keys."""
+    rename_map = {
+        OBS_STATE: "observation.proprio_state",
+        "pixels": OBS_IMAGE,
+    }
+    processor = RenameObservationsProcessorStep(rename_map=rename_map)
+
+    observation = {
+        OBS_STATE: torch.randn(10),
+        "pixels": np.random.randint(0, 256, (64, 64, 3), dtype=np.uint8),
+        "reward": 1.0,
+        "info": {"episode": 1},
+    }
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check renamed keys
+    assert "observation.proprio_state" in processed_obs
+    assert OBS_IMAGE in processed_obs
+    assert OBS_STATE not in processed_obs
+    assert "pixels" not in processed_obs
+
+    # Check unchanged keys
+    assert processed_obs["reward"] == 1.0
+    assert processed_obs["info"] == {"episode": 1}
+
+
+def test_get_config():
+    """Test configuration serialization."""
+    rename_map = {
+        "old1": "new1",
+        "old2": "new2",
+    }
+    processor = RenameObservationsProcessorStep(rename_map=rename_map)
+
+    config = processor.get_config()
+    assert config == {"rename_map": rename_map}
+
+
+def test_state_dict():
+    """Test state dict (should be empty for RenameProcessorStep)."""
+    processor = RenameObservationsProcessorStep(rename_map={"old": "new"})
+
+    state = processor.state_dict()
+    assert state == {}
+
+    # Load state dict should work even with empty dict
+    processor.load_state_dict({})
+
+
+def test_integration_with_robot_processor():
+    """Test integration with RobotProcessor pipeline."""
+    rename_map = {
+        "agent_pos": OBS_STATE,
+        "pixels": OBS_IMAGE,
+    }
+    rename_processor = RenameObservationsProcessorStep(rename_map=rename_map)
+
+    pipeline = DataProcessorPipeline(
+        [rename_processor], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    observation = {
+        "agent_pos": np.array([1.0, 2.0, 3.0]),
+        "pixels": np.zeros((32, 32, 3), dtype=np.uint8),
+        "other_data": "preserve_me",
+    }
+    transition = create_transition(
+        observation=observation, reward=0.5, done=False, truncated=False, info={}, complementary_data={}
+    )
+
+    result = pipeline(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check renaming worked through pipeline
+    assert OBS_STATE in processed_obs
+    assert OBS_IMAGE in processed_obs
+    assert "agent_pos" not in processed_obs
+    assert "pixels" not in processed_obs
+    assert processed_obs["other_data"] == "preserve_me"
+
+    # Check other transition elements unchanged
+    assert result[TransitionKey.REWARD] == 0.5
+    assert result[TransitionKey.DONE] is False
+
+
+def test_save_and_load_pretrained():
+    """Test saving and loading processor with RobotProcessor."""
+    rename_map = {
+        "old_state": OBS_STATE,
+        "old_image": OBS_IMAGE,
+    }
+    processor = RenameObservationsProcessorStep(rename_map=rename_map)
+    pipeline = DataProcessorPipeline([processor], name="TestRenameProcessorStep")
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save pipeline
+        pipeline.save_pretrained(tmp_dir)
+
+        # Check files were created
+        config_path = (
+            Path(tmp_dir) / "testrenameprocessorstep.json"
+        )  # Based on name="TestRenameProcessorStep"
+        assert config_path.exists()
+
+        # No state files should be created for RenameProcessorStep
+        state_files = list(Path(tmp_dir).glob("*.safetensors"))
+        assert len(state_files) == 0
+
+        # Load pipeline
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir,
+            config_filename="testrenameprocessorstep.json",
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        assert loaded_pipeline.name == "TestRenameProcessorStep"
+        assert len(loaded_pipeline) == 1
+
+        # Check that loaded processor works correctly
+        loaded_processor = loaded_pipeline.steps[0]
+        assert isinstance(loaded_processor, RenameObservationsProcessorStep)
+        assert loaded_processor.rename_map == rename_map
+
+        # Test functionality after loading
+        observation = {"old_state": [1, 2, 3], "old_image": "image_data"}
+        transition = create_transition(observation=observation)
+
+        result = loaded_pipeline(transition)
+        processed_obs = result[TransitionKey.OBSERVATION]
+
+        assert OBS_STATE in processed_obs
+        assert OBS_IMAGE in processed_obs
+        assert processed_obs[OBS_STATE] == [1, 2, 3]
+        assert processed_obs[OBS_IMAGE] == "image_data"
+
+
+def test_registry_functionality():
+    """Test that RenameProcessorStep is properly registered."""
+    # Check that it's registered
+    assert "rename_observations_processor" in ProcessorStepRegistry.list()
+
+    # Get from registry
+    retrieved_class = ProcessorStepRegistry.get("rename_observations_processor")
+    assert retrieved_class is RenameObservationsProcessorStep
+
+    # Create instance from registry
+    instance = retrieved_class(rename_map={"old": "new"})
+    assert isinstance(instance, RenameObservationsProcessorStep)
+    assert instance.rename_map == {"old": "new"}
+
+
+def test_registry_based_save_load():
+    """Test save/load using registry name instead of module path."""
+    processor = RenameObservationsProcessorStep(rename_map={"key1": "renamed_key1"})
+    pipeline = DataProcessorPipeline(
+        [processor], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    with tempfile.TemporaryDirectory() as tmp_dir:
+        # Save and load
+        pipeline.save_pretrained(tmp_dir)
+
+        # Verify config uses registry name
+        import json
+
+        with open(Path(tmp_dir) / "dataprocessorpipeline.json") as f:  # Default name is "RobotProcessor"
+            config = json.load(f)
+
+        assert "registry_name" in config["steps"][0]
+        assert config["steps"][0]["registry_name"] == "rename_observations_processor"
+        assert "class" not in config["steps"][0]  # Should use registry, not module path
+
+        # Load should work
+        loaded_pipeline = DataProcessorPipeline.from_pretrained(
+            tmp_dir, config_filename="dataprocessorpipeline.json"
+        )
+        loaded_processor = loaded_pipeline.steps[0]
+        assert isinstance(loaded_processor, RenameObservationsProcessorStep)
+        assert loaded_processor.rename_map == {"key1": "renamed_key1"}
+
+
+def test_chained_rename_processors():
+    """Test multiple RenameProcessorSteps in a pipeline."""
+    # First processor: rename raw keys to intermediate format
+    processor1 = RenameObservationsProcessorStep(
+        rename_map={
+            "pos": "agent_position",
+            "img": "camera_image",
+        }
+    )
+
+    # Second processor: rename to final format
+    processor2 = RenameObservationsProcessorStep(
+        rename_map={
+            "agent_position": OBS_STATE,
+            "camera_image": OBS_IMAGE,
+        }
+    )
+
+    pipeline = DataProcessorPipeline(
+        [processor1, processor2], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    observation = {
+        "pos": np.array([1.0, 2.0]),
+        "img": "image_data",
+        "extra": "keep_me",
+    }
+    transition = create_transition(observation=observation)
+
+    # Step through to see intermediate results
+    results = list(pipeline.step_through(transition))
+
+    # After first processor
+    assert "agent_position" in results[1][TransitionKey.OBSERVATION]
+    assert "camera_image" in results[1][TransitionKey.OBSERVATION]
+
+    # After second processor
+    final_obs = results[2][TransitionKey.OBSERVATION]
+    assert OBS_STATE in final_obs
+    assert OBS_IMAGE in final_obs
+    assert final_obs["extra"] == "keep_me"
+
+    # Original keys should be gone
+    assert "pos" not in final_obs
+    assert "img" not in final_obs
+    assert "agent_position" not in final_obs
+    assert "camera_image" not in final_obs
+
+
+def test_nested_observation_rename():
+    """Test renaming with nested observation structures."""
+    rename_map = {
+        f"{OBS_IMAGES}.left": "observation.camera.left_view",
+        f"{OBS_IMAGES}.right": "observation.camera.right_view",
+        "observation.proprio": "observation.proprioception",
+    }
+    processor = RenameObservationsProcessorStep(rename_map=rename_map)
+
+    observation = {
+        f"{OBS_IMAGES}.left": torch.randn(3, 64, 64),
+        f"{OBS_IMAGES}.right": torch.randn(3, 64, 64),
+        "observation.proprio": torch.randn(7),
+        "observation.gripper": torch.tensor([0.0]),  # Not renamed
+    }
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check renames
+    assert "observation.camera.left_view" in processed_obs
+    assert "observation.camera.right_view" in processed_obs
+    assert "observation.proprioception" in processed_obs
+
+    # Check unchanged key
+    assert "observation.gripper" in processed_obs
+
+    # Check old keys removed
+    assert f"{OBS_IMAGES}.left" not in processed_obs
+    assert f"{OBS_IMAGES}.right" not in processed_obs
+    assert "observation.proprio" not in processed_obs
+
+
+def test_value_types_preserved():
+    """Test that various value types are preserved during renaming."""
+    rename_map = {"old_tensor": "new_tensor", "old_array": "new_array", "old_scalar": "new_scalar"}
+    processor = RenameObservationsProcessorStep(rename_map=rename_map)
+
+    tensor_value = torch.randn(3, 3)
+    array_value = np.random.rand(2, 2)
+
+    observation = {
+        "old_tensor": tensor_value,
+        "old_array": array_value,
+        "old_scalar": 42,
+        "old_string": "hello",
+        "old_dict": {"nested": "value"},
+        "old_list": [1, 2, 3],
+    }
+    transition = create_transition(observation=observation)
+
+    result = processor(transition)
+    processed_obs = result[TransitionKey.OBSERVATION]
+
+    # Check that values and types are preserved
+    assert torch.equal(processed_obs["new_tensor"], tensor_value)
+    assert np.array_equal(processed_obs["new_array"], array_value)
+    assert processed_obs["new_scalar"] == 42
+    assert processed_obs["old_string"] == "hello"
+    assert processed_obs["old_dict"] == {"nested": "value"}
+    assert processed_obs["old_list"] == [1, 2, 3]
+
+
+def test_features_basic_renaming(policy_feature_factory):
+    processor = RenameObservationsProcessorStep(rename_map={"a": "x", "b": "y"})
+    features = {
+        PipelineFeatureType.OBSERVATION: {
+            "a": policy_feature_factory(FeatureType.VISUAL, (2,)),
+            "b": policy_feature_factory(FeatureType.VISUAL, (3,)),
+            "c": policy_feature_factory(FeatureType.VISUAL, (1,)),
+        },
+    }
+
+    out = processor.transform_features(features.copy())
+
+    # Values preserved and typed
+    assert out[PipelineFeatureType.OBSERVATION]["x"] == features[PipelineFeatureType.OBSERVATION]["a"]
+    assert out[PipelineFeatureType.OBSERVATION]["y"] == features[PipelineFeatureType.OBSERVATION]["b"]
+    assert out[PipelineFeatureType.OBSERVATION]["c"] == features[PipelineFeatureType.OBSERVATION]["c"]
+
+    assert_contract_is_typed(out)
+    # Input not mutated
+    assert set(features[PipelineFeatureType.OBSERVATION]) == {"a", "b", "c"}
+
+
+def test_features_overlapping_keys(policy_feature_factory):
+    # Overlapping renames: both 'a' and 'b' exist. 'a'->'b', 'b'->'c'
+    processor = RenameObservationsProcessorStep(rename_map={"a": "b", "b": "c"})
+    features = {
+        PipelineFeatureType.OBSERVATION: {
+            "a": policy_feature_factory(FeatureType.VISUAL, (1,)),
+            "b": policy_feature_factory(FeatureType.VISUAL, (2,)),
+        },
+    }
+    out = processor.transform_features(features)
+
+    assert set(out[PipelineFeatureType.OBSERVATION]) == {"b", "c"}
+    assert (
+        out[PipelineFeatureType.OBSERVATION]["b"] == features[PipelineFeatureType.OBSERVATION]["a"]
+    )  # 'a' renamed to'b'
+    assert (
+        out[PipelineFeatureType.OBSERVATION]["c"] == features[PipelineFeatureType.OBSERVATION]["b"]
+    )  # 'b' renamed to 'c'
+    assert_contract_is_typed(out)
+
+
+def test_features_chained_processors(policy_feature_factory):
+    # Chain two rename processors at the contract level
+    processor1 = RenameObservationsProcessorStep(rename_map={"pos": "agent_position", "img": "camera_image"})
+    processor2 = RenameObservationsProcessorStep(
+        rename_map={"agent_position": OBS_STATE, "camera_image": OBS_IMAGE}
+    )
+    pipeline = DataProcessorPipeline([processor1, processor2])
+
+    spec = {
+        PipelineFeatureType.OBSERVATION: {
+            "pos": policy_feature_factory(FeatureType.VISUAL, (7,)),
+            "img": policy_feature_factory(FeatureType.VISUAL, (3, 64, 64)),
+            "extra": policy_feature_factory(FeatureType.VISUAL, (1,)),
+        },
+    }
+    out = pipeline.transform_features(initial_features=spec)
+
+    assert set(out[PipelineFeatureType.OBSERVATION]) == {OBS_STATE, OBS_IMAGE, "extra"}
+    assert out[PipelineFeatureType.OBSERVATION][OBS_STATE] == spec[PipelineFeatureType.OBSERVATION]["pos"]
+    assert out[PipelineFeatureType.OBSERVATION][OBS_IMAGE] == spec[PipelineFeatureType.OBSERVATION]["img"]
+    assert out[PipelineFeatureType.OBSERVATION]["extra"] == spec[PipelineFeatureType.OBSERVATION]["extra"]
+    assert_contract_is_typed(out)
+
+
+def test_rename_stats_basic():
+    orig = {
+        OBS_STATE: {"mean": np.array([0.0]), "std": np.array([1.0])},
+        ACTION: {"mean": np.array([0.0])},
+    }
+    mapping = {OBS_STATE: "observation.robot_state"}
+    renamed = rename_stats(orig, mapping)
+    assert "observation.robot_state" in renamed and OBS_STATE not in renamed
+    # Ensure deep copy: mutate original and verify renamed unaffected
+    orig[OBS_STATE]["mean"][0] = 42.0
+    assert renamed["observation.robot_state"]["mean"][0] != 42.0
diff --git a/lerobot/tests/processor/test_sac_processor.py b/lerobot/tests/processor/test_sac_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..a1a4b285d5e3f54e2f38ffa4cb4f5f0af4b02117
--- /dev/null
+++ b/lerobot/tests/processor/test_sac_processor.py
@@ -0,0 +1,414 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tests for SAC policy processor."""
+
+import tempfile
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.policies.sac.configuration_sac import SACConfig
+from lerobot.policies.sac.processor_sac import make_sac_pre_post_processors
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DataProcessorPipeline,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    RenameObservationsProcessorStep,
+    TransitionKey,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import create_transition, transition_to_batch
+from lerobot.utils.constants import ACTION, OBS_STATE
+
+
+def create_default_config():
+    """Create a default SAC configuration for testing."""
+    config = SACConfig()
+    config.input_features = {
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(10,)),
+    }
+    config.output_features = {
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(5,)),
+    }
+    config.normalization_mapping = {
+        FeatureType.STATE: NormalizationMode.MEAN_STD,
+        FeatureType.ACTION: NormalizationMode.MIN_MAX,
+    }
+    config.device = "cpu"
+    return config
+
+
+def create_default_stats():
+    """Create default dataset statistics for testing."""
+    return {
+        OBS_STATE: {"mean": torch.zeros(10), "std": torch.ones(10)},
+        ACTION: {"min": torch.full((5,), -1.0), "max": torch.ones(5)},
+    }
+
+
+def test_make_sac_processor_basic():
+    """Test basic creation of SAC processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Check processor names
+    assert preprocessor.name == "policy_preprocessor"
+    assert postprocessor.name == "policy_postprocessor"
+
+    # Check steps in preprocessor
+    assert len(preprocessor.steps) == 4
+    assert isinstance(preprocessor.steps[0], RenameObservationsProcessorStep)
+    assert isinstance(preprocessor.steps[1], AddBatchDimensionProcessorStep)
+    assert isinstance(preprocessor.steps[2], DeviceProcessorStep)
+    assert isinstance(preprocessor.steps[3], NormalizerProcessorStep)
+
+    # Check steps in postprocessor
+    assert len(postprocessor.steps) == 2
+    assert isinstance(postprocessor.steps[0], UnnormalizerProcessorStep)
+    assert isinstance(postprocessor.steps[1], DeviceProcessorStep)
+
+
+def test_sac_processor_normalization_modes():
+    """Test that SAC processor correctly handles different normalization modes."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create test data
+    observation = {OBS_STATE: torch.randn(10) * 2}  # Larger values to test normalization
+    action = torch.rand(5) * 2 - 1  # Range [-1, 1]
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data is normalized and batched
+    # State should be mean-std normalized
+    # Action should be min-max normalized to [-1, 1]
+    assert processed[OBS_STATE].shape == (1, 10)
+    assert processed[TransitionKey.ACTION.value].shape == (1, 5)
+
+    # Process action through postprocessor
+    postprocessed = postprocessor(processed[TransitionKey.ACTION.value])
+
+    # Check that action is unnormalized (but still batched)
+    assert postprocessed.shape == (1, 5)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_sac_processor_cuda():
+    """Test SAC processor with CUDA device."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create CPU data
+    observation = {OBS_STATE: torch.randn(10)}
+    action = torch.randn(5)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data is on CUDA
+    assert processed[OBS_STATE].device.type == "cuda"
+    assert processed[TransitionKey.ACTION.value].device.type == "cuda"
+
+    # Process through postprocessor
+    postprocessed = postprocessor(processed[TransitionKey.ACTION.value])
+
+    # Check that action is back on CPU
+    assert postprocessed.device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_sac_processor_accelerate_scenario():
+    """Test SAC processor in simulated Accelerate scenario."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Simulate Accelerate: data already on GPU
+    device = torch.device("cuda:0")
+    observation = {OBS_STATE: torch.randn(10).to(device)}
+    action = torch.randn(5).to(device)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data stays on same GPU
+    assert processed[OBS_STATE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_sac_processor_multi_gpu():
+    """Test SAC processor with multi-GPU setup."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Simulate data on different GPU
+    device = torch.device("cuda:1")
+    observation = {OBS_STATE: torch.randn(10).to(device)}
+    action = torch.randn(5).to(device)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data stays on cuda:1
+    assert processed[OBS_STATE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+def test_sac_processor_without_stats():
+    """Test SAC processor creation without dataset statistics."""
+    config = create_default_config()
+
+    preprocessor, postprocessor = make_sac_pre_post_processors(config, dataset_stats=None)
+
+    # Should still create processors
+    assert preprocessor is not None
+    assert postprocessor is not None
+
+    # Process should still work
+    observation = {OBS_STATE: torch.randn(10)}
+    action = torch.randn(5)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed is not None
+
+
+def test_sac_processor_save_and_load():
+    """Test saving and loading SAC processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    with tempfile.TemporaryDirectory() as tmpdir:
+        # Save preprocessor
+        preprocessor.save_pretrained(tmpdir)
+
+        # Load preprocessor
+        loaded_preprocessor = DataProcessorPipeline.from_pretrained(
+            tmpdir, config_filename="policy_preprocessor.json"
+        )
+
+        # Test that loaded processor works
+        observation = {OBS_STATE: torch.randn(10)}
+        action = torch.randn(5)
+        transition = create_transition(observation, action)
+        batch = transition_to_batch(transition)
+
+        processed = loaded_preprocessor(batch)
+        assert processed[OBS_STATE].shape == (1, 10)
+        assert processed[TransitionKey.ACTION.value].shape == (1, 5)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_sac_processor_mixed_precision():
+    """Test SAC processor with mixed precision."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    # Create processor
+    preprocessor, postprocessor = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Replace DeviceProcessorStep with one that uses float16
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="float16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Update normalizer to use the same device as the device processor
+            norm_step = step  # Now type checker knows this is NormalizerProcessorStep
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=norm_step.features,
+                    norm_map=norm_step.norm_map,
+                    stats=norm_step.stats,
+                    device=config.device,
+                    dtype=torch.float16,  # Match the float16 dtype
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Create test data
+    observation = {OBS_STATE: torch.randn(10, dtype=torch.float32)}
+    action = torch.randn(5, dtype=torch.float32)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that data is converted to float16
+    assert processed[OBS_STATE].dtype == torch.float16
+    assert processed[TransitionKey.ACTION.value].dtype == torch.float16
+
+
+def test_sac_processor_batch_data():
+    """Test SAC processor with batched data."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Test with batched data
+    batch_size = 32
+    observation = {OBS_STATE: torch.randn(batch_size, 10)}
+    action = torch.randn(batch_size, 5)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+    processed = preprocessor(batch)
+
+    # Check that batch dimension is preserved
+    assert processed[OBS_STATE].shape == (batch_size, 10)
+    assert processed[TransitionKey.ACTION.value].shape == (batch_size, 5)
+
+
+def test_sac_processor_edge_cases():
+    """Test SAC processor with edge cases."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Test with observation that has no state key but still exists
+    observation = {"observation.dummy": torch.randn(1)}  # Some dummy observation to pass validation
+    action = torch.randn(5)
+    batch = {TransitionKey.ACTION.value: action, **observation}
+    processed = preprocessor(batch)
+    # observation.state wasn't in original, so it won't be in processed
+    assert OBS_STATE not in processed
+    assert processed[TransitionKey.ACTION.value].shape == (1, 5)
+
+    # Test with zero action (representing "null" action)
+    transition = create_transition(observation={OBS_STATE: torch.randn(10)}, action=torch.zeros(5))
+    batch = transition_to_batch(transition)
+    processed = preprocessor(batch)
+    assert processed[OBS_STATE].shape == (1, 10)
+    # Action should be present and batched, even if it's zeros
+    assert processed[TransitionKey.ACTION.value].shape == (1, 5)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_sac_processor_bfloat16_device_float32_normalizer():
+    """Test: DeviceProcessor(bfloat16) + NormalizerProcessor(float32) → output bfloat16 via automatic adaptation"""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, _ = make_sac_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Modify the pipeline to use bfloat16 device processor with float32 normalizer
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            # Device processor converts to bfloat16
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="bfloat16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Normalizer stays configured as float32 (will auto-adapt to bfloat16)
+            norm_step = step  # Now type checker knows this is NormalizerProcessorStep
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=norm_step.features,
+                    norm_map=norm_step.norm_map,
+                    stats=norm_step.stats,
+                    device=config.device,
+                    dtype=torch.float32,  # Deliberately configured as float32
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Verify initial normalizer configuration
+    normalizer_step = preprocessor.steps[3]  # NormalizerProcessorStep
+    assert normalizer_step.dtype == torch.float32
+
+    # Create test data
+    observation = {OBS_STATE: torch.randn(10, dtype=torch.float32)}  # Start with float32
+    action = torch.randn(5, dtype=torch.float32)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    # Process through full pipeline
+    processed = preprocessor(batch)
+
+    # Verify: DeviceProcessor → bfloat16, NormalizerProcessor adapts → final output is bfloat16
+    assert processed[OBS_STATE].dtype == torch.bfloat16
+    assert processed[TransitionKey.ACTION.value].dtype == torch.bfloat16
+
+    # Verify normalizer automatically adapted its internal state
+    assert normalizer_step.dtype == torch.bfloat16
+    for stat_tensor in normalizer_step._tensor_stats[OBS_STATE].values():
+        assert stat_tensor.dtype == torch.bfloat16
diff --git a/lerobot/tests/processor/test_smolvla_processor.py b/lerobot/tests/processor/test_smolvla_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..227b1dc359890b9f2ae7b0e933d88db5f85f701e
--- /dev/null
+++ b/lerobot/tests/processor/test_smolvla_processor.py
@@ -0,0 +1,459 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tests for SmolVLA policy processor."""
+
+from unittest.mock import patch
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PipelineFeatureType, PolicyFeature
+from lerobot.policies.smolvla.configuration_smolvla import SmolVLAConfig
+from lerobot.policies.smolvla.processor_smolvla import (
+    SmolVLANewLineProcessor,
+    make_smolvla_pre_post_processors,
+)
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DeviceProcessorStep,
+    EnvTransition,
+    NormalizerProcessorStep,
+    ProcessorStep,
+    RenameObservationsProcessorStep,
+    TransitionKey,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import create_transition, transition_to_batch
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_STATE
+
+
+class MockTokenizerProcessorStep(ProcessorStep):
+    """Mock tokenizer processor step for testing."""
+
+    def __init__(self, *args, **kwargs):
+        # Accept any arguments to mimic the real TokenizerProcessorStep interface
+        pass
+
+    def __call__(self, transition: EnvTransition) -> EnvTransition:
+        # Pass through transition unchanged
+        return transition
+
+    def transform_features(self, features):
+        # Pass through features unchanged
+        return features
+
+
+def create_default_config():
+    """Create a default SmolVLA configuration for testing."""
+    config = SmolVLAConfig()
+    config.input_features = {
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(8,)),
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+    config.normalization_mapping = {
+        FeatureType.STATE: NormalizationMode.MEAN_STD,
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,
+        FeatureType.ACTION: NormalizationMode.MIN_MAX,
+    }
+    config.device = "cpu"
+    config.vlm_model_name = "HuggingFaceTB/SmolVLM-Instruct"
+    config.pad_language_to = "max_length"
+    config.tokenizer_max_length = 100
+    return config
+
+
+def create_default_stats():
+    """Create default dataset statistics for testing."""
+    return {
+        OBS_STATE: {"mean": torch.zeros(8), "std": torch.ones(8)},
+        OBS_IMAGE: {},  # No normalization for images
+        ACTION: {"min": torch.full((7,), -1.0), "max": torch.ones(7)},
+    }
+
+
+def test_make_smolvla_processor_basic():
+    """Test basic creation of SmolVLA processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    with patch(
+        "lerobot.policies.smolvla.processor_smolvla.TokenizerProcessorStep", MockTokenizerProcessorStep
+    ):
+        preprocessor, postprocessor = make_smolvla_pre_post_processors(
+            config,
+            stats,
+        )
+
+    # Check processor names
+    assert preprocessor.name == "policy_preprocessor"
+    assert postprocessor.name == "policy_postprocessor"
+
+    # Check steps in preprocessor
+    assert len(preprocessor.steps) == 6
+    assert isinstance(preprocessor.steps[0], RenameObservationsProcessorStep)
+    assert isinstance(preprocessor.steps[1], AddBatchDimensionProcessorStep)
+    assert isinstance(preprocessor.steps[2], SmolVLANewLineProcessor)
+    # Step 3 would be TokenizerProcessorStep but it's mocked
+    assert isinstance(preprocessor.steps[4], DeviceProcessorStep)
+    assert isinstance(preprocessor.steps[5], NormalizerProcessorStep)
+
+    # Check steps in postprocessor
+    assert len(postprocessor.steps) == 2
+    assert isinstance(postprocessor.steps[0], UnnormalizerProcessorStep)
+    assert isinstance(postprocessor.steps[1], DeviceProcessorStep)
+
+
+def test_smolvla_newline_processor_single_task():
+    """Test SmolVLANewLineProcessor with single task string."""
+    processor = SmolVLANewLineProcessor()
+
+    # Test with task that doesn't have newline
+    transition = create_transition(complementary_data={"task": "test task"})
+    result = processor(transition)
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["task"] == "test task\n"
+
+    # Test with task that already has newline
+    transition = create_transition(complementary_data={"task": "test task\n"})
+    result = processor(transition)
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["task"] == "test task\n"
+
+
+def test_smolvla_newline_processor_list_of_tasks():
+    """Test SmolVLANewLineProcessor with list of task strings."""
+    processor = SmolVLANewLineProcessor()
+
+    # Test with list of tasks
+    tasks = ["task1", "task2\n", "task3"]
+    transition = create_transition(complementary_data={"task": tasks})
+    result = processor(transition)
+    expected = ["task1\n", "task2\n", "task3\n"]
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["task"] == expected
+
+
+def test_smolvla_newline_processor_empty_transition():
+    """Test SmolVLANewLineProcessor with empty transition."""
+    processor = SmolVLANewLineProcessor()
+
+    # Test with no complementary_data
+    transition = create_transition()
+    result = processor(transition)
+    assert result == transition
+
+    # Test with complementary_data but no task
+    transition = create_transition(complementary_data={"other": "data"})
+    result = processor(transition)
+    assert result == transition
+
+    # Test with None task
+    transition = create_transition(complementary_data={"task": None})
+    result = processor(transition)
+    assert result == transition
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_smolvla_processor_cuda():
+    """Test SmolVLA processor with CUDA device."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    # Mock the tokenizer processor to act as pass-through
+    class MockTokenizerProcessorStep(ProcessorStep):
+        def __init__(self, *args, **kwargs):
+            pass
+
+        def __call__(self, transition):
+            return transition
+
+        def state_dict(self):
+            return {}
+
+        def load_state_dict(self, state):
+            pass
+
+        def reset(self):
+            pass
+
+        def get_config(self):
+            return {"tokenizer_name": "HuggingFaceTB/SmolVLM-Instruct"}
+
+        def transform_features(self, features):
+            return features
+
+    with patch(
+        "lerobot.policies.smolvla.processor_smolvla.TokenizerProcessorStep", MockTokenizerProcessorStep
+    ):
+        preprocessor, postprocessor = make_smolvla_pre_post_processors(
+            config,
+            stats,
+        )
+
+    # Create CPU data
+    observation = {
+        OBS_STATE: torch.randn(8),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(7)
+    transition = create_transition(observation, action, complementary_data={"task": "test task"})
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is on CUDA
+    assert processed[OBS_STATE].device.type == "cuda"
+    assert processed[OBS_IMAGE].device.type == "cuda"
+    assert processed[TransitionKey.ACTION.value].device.type == "cuda"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_smolvla_processor_accelerate_scenario():
+    """Test SmolVLA processor in simulated Accelerate scenario."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    # Mock the tokenizer processor to act as pass-through
+    class MockTokenizerProcessorStep(ProcessorStep):
+        def __init__(self, *args, **kwargs):
+            pass
+
+        def __call__(self, transition):
+            return transition
+
+        def state_dict(self):
+            return {}
+
+        def load_state_dict(self, state):
+            pass
+
+        def reset(self):
+            pass
+
+        def get_config(self):
+            return {"tokenizer_name": "HuggingFaceTB/SmolVLM-Instruct"}
+
+        def transform_features(self, features):
+            return features
+
+    with patch(
+        "lerobot.policies.smolvla.processor_smolvla.TokenizerProcessorStep", MockTokenizerProcessorStep
+    ):
+        preprocessor, postprocessor = make_smolvla_pre_post_processors(
+            config,
+            stats,
+        )
+
+    # Simulate Accelerate: data already on GPU and batched
+    device = torch.device("cuda:0")
+    observation = {
+        OBS_STATE: torch.randn(1, 8).to(device),
+        OBS_IMAGE: torch.randn(1, 3, 224, 224).to(device),
+    }
+    action = torch.randn(1, 7).to(device)
+    transition = create_transition(observation, action, complementary_data={"task": ["test task"]})
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on same GPU
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_smolvla_processor_multi_gpu():
+    """Test SmolVLA processor with multi-GPU setup."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    # Mock the tokenizer processor to act as pass-through
+    class MockTokenizerProcessorStep(ProcessorStep):
+        def __init__(self, *args, **kwargs):
+            pass
+
+        def __call__(self, transition):
+            return transition
+
+        def state_dict(self):
+            return {}
+
+        def load_state_dict(self, state):
+            pass
+
+        def reset(self):
+            pass
+
+        def get_config(self):
+            return {"tokenizer_name": "HuggingFaceTB/SmolVLM-Instruct"}
+
+        def transform_features(self, features):
+            return features
+
+    with patch(
+        "lerobot.policies.smolvla.processor_smolvla.TokenizerProcessorStep", MockTokenizerProcessorStep
+    ):
+        preprocessor, postprocessor = make_smolvla_pre_post_processors(
+            config,
+            stats,
+        )
+
+    # Simulate data on different GPU
+    device = torch.device("cuda:1")
+    observation = {
+        OBS_STATE: torch.randn(1, 8).to(device),
+        OBS_IMAGE: torch.randn(1, 3, 224, 224).to(device),
+    }
+    action = torch.randn(1, 7).to(device)
+    transition = create_transition(observation, action, complementary_data={"task": ["test task"]})
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on cuda:1
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+def test_smolvla_processor_without_stats():
+    """Test SmolVLA processor creation without dataset statistics."""
+    config = create_default_config()
+
+    # Mock the tokenizer processor
+    with patch(
+        "lerobot.policies.smolvla.processor_smolvla.TokenizerProcessorStep", MockTokenizerProcessorStep
+    ):
+        preprocessor, postprocessor = make_smolvla_pre_post_processors(
+            config,
+            dataset_stats=None,
+        )
+
+    # Should still create processors
+    assert preprocessor is not None
+    assert postprocessor is not None
+
+
+def test_smolvla_newline_processor_state_dict():
+    """Test SmolVLANewLineProcessor state dict methods."""
+    processor = SmolVLANewLineProcessor()
+
+    # Test state_dict (should be empty)
+    state = processor.state_dict()
+    assert state == {}
+
+    # Test load_state_dict (should do nothing)
+    processor.load_state_dict({})
+
+    # Test reset (should do nothing)
+    processor.reset()
+
+    # Test get_config
+    config = processor.get_config()
+    assert config == {}
+
+
+def test_smolvla_newline_processor_transform_features():
+    """Test SmolVLANewLineProcessor transform_features method."""
+    processor = SmolVLANewLineProcessor()
+
+    # Test transform_features
+    features = {
+        PipelineFeatureType.OBSERVATION: {OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(10,))},
+    }
+    result = processor.transform_features(features)
+    assert result == features  # Should return unchanged
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_smolvla_processor_bfloat16_device_float32_normalizer():
+    """Test: DeviceProcessor(bfloat16) + NormalizerProcessor(float32) → output bfloat16 via automatic adaptation"""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    with patch(
+        "lerobot.policies.smolvla.processor_smolvla.TokenizerProcessorStep", MockTokenizerProcessorStep
+    ):
+        preprocessor, _ = make_smolvla_pre_post_processors(
+            config,
+            stats,
+        )
+
+    # Modify the pipeline to use bfloat16 device processor with float32 normalizer
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            # Device processor converts to bfloat16
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="bfloat16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Normalizer stays configured as float32 (will auto-adapt to bfloat16)
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=step.features,
+                    norm_map=step.norm_map,
+                    stats=step.stats,
+                    device=config.device,
+                    dtype=torch.float32,  # Deliberately configured as float32
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Verify initial normalizer configuration (SmolVLA has NormalizerProcessorStep at index 5)
+    normalizer_step = preprocessor.steps[5]  # NormalizerProcessorStep
+    assert normalizer_step.dtype == torch.float32
+
+    # Create test data with both state and visual observations
+    observation = {
+        OBS_STATE: torch.randn(8, dtype=torch.float32),
+        OBS_IMAGE: torch.randn(3, 224, 224, dtype=torch.float32),
+    }
+    action = torch.randn(7, dtype=torch.float32)
+    transition = create_transition(
+        observation, action, complementary_data={"task": "test bfloat16 adaptation"}
+    )
+
+    batch = transition_to_batch(transition)
+
+    # Process through full pipeline
+    processed = preprocessor(batch)
+
+    # Verify: DeviceProcessor → bfloat16, NormalizerProcessor adapts → final output is bfloat16
+    assert processed[OBS_STATE].dtype == torch.bfloat16
+    assert processed[OBS_IMAGE].dtype == torch.bfloat16  # IDENTITY normalization still gets dtype conversion
+    assert processed[TransitionKey.ACTION.value].dtype == torch.bfloat16
+
+    # Verify normalizer automatically adapted its internal state
+    assert normalizer_step.dtype == torch.bfloat16
+    # Check state stats (has normalization)
+    for stat_tensor in normalizer_step._tensor_stats[OBS_STATE].values():
+        assert stat_tensor.dtype == torch.bfloat16
+    # OBS_IMAGE uses IDENTITY normalization, so no stats to check
diff --git a/lerobot/tests/processor/test_tdmpc_processor.py b/lerobot/tests/processor/test_tdmpc_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..edbc25ae3e1d09f49139bb5b079de3589124238a
--- /dev/null
+++ b/lerobot/tests/processor/test_tdmpc_processor.py
@@ -0,0 +1,467 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tests for TDMPC policy processor."""
+
+import tempfile
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.policies.tdmpc.configuration_tdmpc import TDMPCConfig
+from lerobot.policies.tdmpc.processor_tdmpc import make_tdmpc_pre_post_processors
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DataProcessorPipeline,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    RenameObservationsProcessorStep,
+    TransitionKey,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import create_transition, transition_to_batch
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_STATE
+
+
+def create_default_config():
+    """Create a default TDMPC configuration for testing."""
+    config = TDMPCConfig()
+    config.input_features = {
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(12,)),
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(6,)),
+    }
+    config.normalization_mapping = {
+        FeatureType.STATE: NormalizationMode.MEAN_STD,
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,
+        FeatureType.ACTION: NormalizationMode.MIN_MAX,
+    }
+    config.device = "cpu"
+    return config
+
+
+def create_default_stats():
+    """Create default dataset statistics for testing."""
+    return {
+        OBS_STATE: {"mean": torch.zeros(12), "std": torch.ones(12)},
+        OBS_IMAGE: {},  # No normalization for images
+        ACTION: {"min": torch.full((6,), -1.0), "max": torch.ones(6)},
+    }
+
+
+def test_make_tdmpc_processor_basic():
+    """Test basic creation of TDMPC processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Check processor names
+    assert preprocessor.name == "policy_preprocessor"
+    assert postprocessor.name == "policy_postprocessor"
+
+    # Check steps in preprocessor
+    assert len(preprocessor.steps) == 4
+    assert isinstance(preprocessor.steps[0], RenameObservationsProcessorStep)
+    assert isinstance(preprocessor.steps[1], AddBatchDimensionProcessorStep)
+    assert isinstance(preprocessor.steps[2], DeviceProcessorStep)
+    assert isinstance(preprocessor.steps[3], NormalizerProcessorStep)
+
+    # Check steps in postprocessor
+    assert len(postprocessor.steps) == 2
+    assert isinstance(postprocessor.steps[0], UnnormalizerProcessorStep)
+    assert isinstance(postprocessor.steps[1], DeviceProcessorStep)
+
+
+def test_tdmpc_processor_normalization():
+    """Test that TDMPC processor correctly normalizes and unnormalizes data."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create test data
+    observation = {
+        OBS_STATE: torch.randn(12),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(6)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is processed and batched
+    assert processed[OBS_STATE].shape == (1, 12)
+    assert processed[OBS_IMAGE].shape == (1, 3, 224, 224)
+    assert processed[TransitionKey.ACTION.value].shape == (1, 6)
+
+    # Process action through postprocessor
+    postprocessed = postprocessor(processed[TransitionKey.ACTION.value])
+
+    # Check that action is unnormalized (but still batched)
+    assert postprocessed.shape == (1, 6)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_tdmpc_processor_cuda():
+    """Test TDMPC processor with CUDA device."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create CPU data
+    observation = {
+        OBS_STATE: torch.randn(12),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(6)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is on CUDA
+    assert processed[OBS_STATE].device.type == "cuda"
+    assert processed[OBS_IMAGE].device.type == "cuda"
+    assert processed[TransitionKey.ACTION.value].device.type == "cuda"
+
+    # Process through postprocessor
+    postprocessed = postprocessor(processed[TransitionKey.ACTION.value])
+
+    # Check that action is back on CPU
+    assert postprocessed.device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_tdmpc_processor_accelerate_scenario():
+    """Test TDMPC processor in simulated Accelerate scenario."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Simulate Accelerate: data already on GPU
+    device = torch.device("cuda:0")
+    observation = {
+        OBS_STATE: torch.randn(12).to(device),
+        OBS_IMAGE: torch.randn(3, 224, 224).to(device),
+    }
+    action = torch.randn(6).to(device)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on same GPU
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_tdmpc_processor_multi_gpu():
+    """Test TDMPC processor with multi-GPU setup."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Simulate data on different GPU
+    device = torch.device("cuda:1")
+    observation = {
+        OBS_STATE: torch.randn(12).to(device),
+        OBS_IMAGE: torch.randn(3, 224, 224).to(device),
+    }
+    action = torch.randn(6).to(device)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on cuda:1
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+def test_tdmpc_processor_without_stats():
+    """Test TDMPC processor creation without dataset statistics."""
+    config = create_default_config()
+
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(config, dataset_stats=None)
+
+    # Should still create processors
+    assert preprocessor is not None
+    assert postprocessor is not None
+
+    # Process should still work
+    observation = {
+        OBS_STATE: torch.randn(12),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(6)
+    transition = create_transition(observation, action)
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed is not None
+
+
+def test_tdmpc_processor_save_and_load():
+    """Test saving and loading TDMPC processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    with tempfile.TemporaryDirectory() as tmpdir:
+        # Save preprocessor
+        preprocessor.save_pretrained(tmpdir)
+
+        # Load preprocessor
+        loaded_preprocessor = DataProcessorPipeline.from_pretrained(
+            tmpdir, config_filename="policy_preprocessor.json"
+        )
+
+        # Test that loaded processor works
+        observation = {
+            OBS_STATE: torch.randn(12),
+            OBS_IMAGE: torch.randn(3, 224, 224),
+        }
+        action = torch.randn(6)
+        transition = create_transition(observation, action)
+
+        batch = transition_to_batch(transition)
+        processed = loaded_preprocessor(batch)
+        assert processed[OBS_STATE].shape == (1, 12)
+        assert processed[OBS_IMAGE].shape == (1, 3, 224, 224)
+        assert processed[TransitionKey.ACTION.value].shape == (1, 6)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_tdmpc_processor_mixed_precision():
+    """Test TDMPC processor with mixed precision."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    # Create processor
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Replace DeviceProcessorStep with one that uses float16
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="float16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Update normalizer to use the same device as the device processor
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=step.features,
+                    norm_map=step.norm_map,
+                    stats=step.stats,
+                    device=config.device,
+                    dtype=torch.float16,  # Match the float16 dtype
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Create test data
+    observation = {
+        OBS_STATE: torch.randn(12, dtype=torch.float32),
+        OBS_IMAGE: torch.randn(3, 224, 224, dtype=torch.float32),
+    }
+    action = torch.randn(6, dtype=torch.float32)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is converted to float16
+    assert processed[OBS_STATE].dtype == torch.float16
+    assert processed[OBS_IMAGE].dtype == torch.float16
+    assert processed[TransitionKey.ACTION.value].dtype == torch.float16
+
+
+def test_tdmpc_processor_batch_data():
+    """Test TDMPC processor with batched data."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Test with batched data
+    batch_size = 64
+    observation = {
+        OBS_STATE: torch.randn(batch_size, 12),
+        OBS_IMAGE: torch.randn(batch_size, 3, 224, 224),
+    }
+    action = torch.randn(batch_size, 6)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that batch dimension is preserved
+    assert processed[OBS_STATE].shape == (batch_size, 12)
+    assert processed[OBS_IMAGE].shape == (batch_size, 3, 224, 224)
+    assert processed[TransitionKey.ACTION.value].shape == (batch_size, 6)
+
+
+def test_tdmpc_processor_edge_cases():
+    """Test TDMPC processor with edge cases."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Test with only state observation (no image)
+    observation = {OBS_STATE: torch.randn(12)}
+    action = torch.randn(6)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed[OBS_STATE].shape == (1, 12)
+    assert OBS_IMAGE not in processed
+
+    # Test with only image observation (no state)
+    observation = {OBS_IMAGE: torch.randn(3, 224, 224)}
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed[OBS_IMAGE].shape == (1, 3, 224, 224)
+    assert OBS_STATE not in processed
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_tdmpc_processor_bfloat16_device_float32_normalizer():
+    """Test: DeviceProcessor(bfloat16) + NormalizerProcessor(float32) → output bfloat16 via automatic adaptation"""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, _ = make_tdmpc_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Modify the pipeline to use bfloat16 device processor with float32 normalizer
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            # Device processor converts to bfloat16
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="bfloat16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Normalizer stays configured as float32 (will auto-adapt to bfloat16)
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=step.features,
+                    norm_map=step.norm_map,
+                    stats=step.stats,
+                    device=config.device,
+                    dtype=torch.float32,  # Deliberately configured as float32
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Verify initial normalizer configuration
+    normalizer_step = preprocessor.steps[3]  # NormalizerProcessorStep
+    assert normalizer_step.dtype == torch.float32
+
+    # Create test data with both state and visual observations
+    observation = {
+        OBS_STATE: torch.randn(12, dtype=torch.float32),
+        OBS_IMAGE: torch.randn(3, 224, 224, dtype=torch.float32),
+    }
+    action = torch.randn(6, dtype=torch.float32)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through full pipeline
+    processed = preprocessor(batch)
+
+    # Verify: DeviceProcessor → bfloat16, NormalizerProcessor adapts → final output is bfloat16
+    assert processed[OBS_STATE].dtype == torch.bfloat16
+    assert processed[OBS_IMAGE].dtype == torch.bfloat16  # IDENTITY normalization still gets dtype conversion
+    assert processed[TransitionKey.ACTION.value].dtype == torch.bfloat16
+
+    # Verify normalizer automatically adapted its internal state
+    assert normalizer_step.dtype == torch.bfloat16
+    # Check state stats (has normalization)
+    for stat_tensor in normalizer_step._tensor_stats[OBS_STATE].values():
+        assert stat_tensor.dtype == torch.bfloat16
+    # OBS_IMAGE uses IDENTITY normalization, so no stats to check
diff --git a/lerobot/tests/processor/test_tokenizer_processor.py b/lerobot/tests/processor/test_tokenizer_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..2f1c4cc9cce29030841265127d5c64b1a8c37f81
--- /dev/null
+++ b/lerobot/tests/processor/test_tokenizer_processor.py
@@ -0,0 +1,1504 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Tests for the TokenizerProcessorStep class.
+"""
+
+import tempfile
+from unittest.mock import patch
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, PipelineFeatureType, PolicyFeature
+from lerobot.processor import DataProcessorPipeline, TokenizerProcessorStep
+from lerobot.processor.converters import create_transition, identity_transition
+from lerobot.types import TransitionKey
+from lerobot.utils.constants import (
+    ACTION,
+    OBS_IMAGE,
+    OBS_LANGUAGE,
+    OBS_LANGUAGE_SUBTASK_ATTENTION_MASK,
+    OBS_LANGUAGE_SUBTASK_TOKENS,
+    OBS_STATE,
+)
+from tests.utils import require_package
+
+
+class MockTokenizer:
+    """Mock tokenizer for testing that mimics transformers tokenizer interface."""
+
+    def __init__(self, vocab_size: int = 1000):
+        self.vocab_size = vocab_size
+
+    def __call__(
+        self,
+        text: str | list[str],
+        max_length: int = 512,
+        truncation: bool = True,
+        padding: str = "max_length",
+        padding_side: str = "right",
+        return_tensors: str = "pt",
+        **kwargs,
+    ) -> dict[str, torch.Tensor]:
+        """Mock tokenization that returns deterministic tokens based on text."""
+        texts = [text] if isinstance(text, str) else text
+
+        batch_size = len(texts)
+
+        # Create mock input_ids and attention_mask
+        input_ids = torch.zeros(batch_size, max_length, dtype=torch.long)
+        attention_mask = torch.zeros(batch_size, max_length, dtype=torch.long)
+
+        for i, txt in enumerate(texts):
+            # Simple mock: use hash of text to generate deterministic tokens
+            text_hash = hash(txt) % self.vocab_size
+            seq_len = min(len(txt.split()), max_length)
+
+            # Fill input_ids with simple pattern based on text
+            for j in range(seq_len):
+                input_ids[i, j] = (text_hash + j) % self.vocab_size
+
+            # Set attention mask for non-padded positions
+            attention_mask[i, :seq_len] = 1
+
+        result = {
+            "input_ids": input_ids,
+            "attention_mask": attention_mask,
+        }
+
+        # Return single sequence for single input to match transformers behavior
+        if len(texts) == 1:
+            result = {k: v.squeeze(0) for k, v in result.items()}
+
+        return result
+
+
+@pytest.fixture
+def mock_tokenizer():
+    """Provide a mock tokenizer for testing."""
+    return MockTokenizer(vocab_size=100)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_basic_tokenization(mock_auto_tokenizer):
+    """Test basic string tokenization functionality."""
+    # Mock AutoTokenizer.from_pretrained to return our mock tokenizer
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer", max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "pick up the red cube"},
+    )
+
+    result = processor(transition)
+
+    # Check that original task is preserved
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["task"] == "pick up the red cube"
+
+    # Check that tokens were added to observation
+    observation = result[TransitionKey.OBSERVATION]
+    assert f"{OBS_LANGUAGE}.tokens" in observation
+    assert f"{OBS_LANGUAGE}.attention_mask" in observation
+
+    # Check token structure
+    tokens = observation[f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = observation[f"{OBS_LANGUAGE}.attention_mask"]
+    assert isinstance(tokens, torch.Tensor)
+    assert isinstance(attention_mask, torch.Tensor)
+    assert tokens.shape == (10,)
+    assert attention_mask.shape == (10,)
+
+
+@require_package("transformers")
+def test_basic_tokenization_with_tokenizer_object():
+    """Test basic string tokenization functionality using tokenizer object directly."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "pick up the red cube"},
+    )
+
+    result = processor(transition)
+
+    # Check that original task is preserved
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["task"] == "pick up the red cube"
+
+    # Check that tokens were added to observation
+    observation = result[TransitionKey.OBSERVATION]
+    assert f"{OBS_LANGUAGE}.tokens" in observation
+    assert f"{OBS_LANGUAGE}.attention_mask" in observation
+
+    # Check token structure
+    tokens = observation[f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = observation[f"{OBS_LANGUAGE}.attention_mask"]
+    assert isinstance(tokens, torch.Tensor)
+    assert isinstance(attention_mask, torch.Tensor)
+    assert tokens.shape == (10,)
+    assert attention_mask.shape == (10,)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_list_of_strings_tokenization(mock_auto_tokenizer):
+    """Test tokenization of a list of strings."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer", max_length=8)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": ["pick up cube", "place on table"]},
+    )
+
+    result = processor(transition)
+
+    # Check that original task is preserved
+    assert result[TransitionKey.COMPLEMENTARY_DATA]["task"] == ["pick up cube", "place on table"]
+
+    # Check that tokens were added to observation
+    observation = result[TransitionKey.OBSERVATION]
+    tokens = observation[f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = observation[f"{OBS_LANGUAGE}.attention_mask"]
+    assert tokens.shape == (2, 8)  # batch_size=2, seq_len=8
+    assert attention_mask.shape == (2, 8)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_custom_keys(mock_auto_tokenizer):
+    """Test using custom task_key."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer", task_key="instruction", max_length=5)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"instruction": "move forward"},
+    )
+
+    result = processor(transition)
+
+    # Check that tokens are stored in observation regardless of task_key
+    observation = result[TransitionKey.OBSERVATION]
+    assert f"{OBS_LANGUAGE}.tokens" in observation
+    assert f"{OBS_LANGUAGE}.attention_mask" in observation
+
+    tokens = observation[f"{OBS_LANGUAGE}.tokens"]
+    assert tokens.shape == (5,)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_none_complementary_data(mock_auto_tokenizer):
+    """Test handling of None complementary_data."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer")
+
+    transition = create_transition(observation={}, complementary_data=None)
+
+    # create_transition converts None complementary_data to empty dict, so task key is missing
+    with pytest.raises(KeyError, match="task"):
+        processor(transition)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_missing_task_key(mock_auto_tokenizer):
+    """Test handling when task key is missing."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer")
+
+    transition = create_transition(observation={}, complementary_data={"other_field": "some value"})
+
+    with pytest.raises(KeyError, match="task"):
+        processor(transition)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_none_task_value(mock_auto_tokenizer):
+    """Test handling when task value is None."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer")
+
+    transition = create_transition(observation={}, complementary_data={"task": None})
+
+    with pytest.raises(ValueError, match="Task extracted from Complementary data is None"):
+        processor(transition)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_unsupported_task_type(mock_auto_tokenizer):
+    """Test handling of unsupported task types."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer")
+
+    # Test with integer task - get_task returns None, observation raises ValueError
+    transition = create_transition(observation={}, complementary_data={"task": 123})
+
+    with pytest.raises(ValueError, match="Task cannot be None"):
+        processor(transition)
+
+    # Test with mixed list - get_task returns None, observation raises ValueError
+    transition = create_transition(observation={}, complementary_data={"task": ["text", 123, "more text"]})
+
+    with pytest.raises(ValueError, match="Task cannot be None"):
+        processor(transition)
+
+
+@require_package("transformers")
+def test_no_tokenizer_error():
+    """Test that ValueError is raised when neither tokenizer nor tokenizer_name is provided."""
+    with pytest.raises(ValueError, match="Either 'tokenizer' or 'tokenizer_name' must be provided"):
+        TokenizerProcessorStep()
+
+
+@require_package("transformers")
+def test_invalid_tokenizer_name_error():
+    """Test that error is raised when invalid tokenizer_name is provided."""
+    with patch("lerobot.processor.tokenizer_processor.AutoTokenizer") as mock_auto_tokenizer:
+        # Mock import error
+        mock_auto_tokenizer.from_pretrained.side_effect = Exception("Model not found")
+
+        with pytest.raises(Exception, match="Model not found"):
+            TokenizerProcessorStep(tokenizer_name="invalid-tokenizer")
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_get_config_with_tokenizer_name(mock_auto_tokenizer):
+    """Test configuration serialization when using tokenizer_name."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(
+        tokenizer_name="test-tokenizer",
+        max_length=256,
+        task_key="instruction",
+        padding="longest",
+        truncation=False,
+    )
+
+    config = processor.get_config()
+
+    expected = {
+        "tokenizer_name": "test-tokenizer",
+        "max_length": 256,
+        "task_key": "instruction",
+        "padding_side": "right",
+        "padding": "longest",
+        "truncation": False,
+    }
+
+    assert config == expected
+
+
+@require_package("transformers")
+def test_get_config_with_tokenizer_object():
+    """Test configuration serialization when using tokenizer object."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+
+    processor = TokenizerProcessorStep(
+        tokenizer=mock_tokenizer,
+        max_length=256,
+        task_key="instruction",
+        padding="longest",
+        truncation=False,
+    )
+
+    config = processor.get_config()
+
+    # tokenizer_name should not be in config when tokenizer object is used
+    expected = {
+        "max_length": 256,
+        "task_key": "instruction",
+        "padding_side": "right",
+        "padding": "longest",
+        "truncation": False,
+    }
+
+    assert config == expected
+    assert "tokenizer_name" not in config
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_state_dict_methods(mock_auto_tokenizer):
+    """Test state_dict and load_state_dict methods."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer")
+
+    # Should return empty dict
+    state = processor.state_dict()
+    assert state == {}
+
+    # load_state_dict should not raise error
+    processor.load_state_dict({})
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_reset_method(mock_auto_tokenizer):
+    """Test reset method."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer")
+
+    # Should not raise error
+    processor.reset()
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_integration_with_robot_processor(mock_auto_tokenizer):
+    """Test integration with RobotProcessor."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    tokenizer_processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer", max_length=6)
+    robot_processor = DataProcessorPipeline(
+        [tokenizer_processor], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "test task"},
+    )
+
+    result = robot_processor(transition)
+
+    # Check that observation exists and tokenization was applied
+    assert TransitionKey.OBSERVATION in result
+    observation = result[TransitionKey.OBSERVATION]
+    assert f"{OBS_LANGUAGE}.tokens" in observation
+    assert f"{OBS_LANGUAGE}.attention_mask" in observation
+    tokens = observation[f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = observation[f"{OBS_LANGUAGE}.attention_mask"]
+    assert tokens.shape == (6,)
+    assert attention_mask.shape == (6,)
+
+    # Check that other data is preserved
+    assert torch.equal(
+        result[TransitionKey.OBSERVATION]["state"], transition[TransitionKey.OBSERVATION]["state"]
+    )
+    assert torch.equal(result[TransitionKey.ACTION], transition[TransitionKey.ACTION])
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_save_and_load_pretrained_with_tokenizer_name(mock_auto_tokenizer):
+    """Test saving and loading processor with tokenizer_name."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    original_processor = TokenizerProcessorStep(
+        tokenizer_name="test-tokenizer", max_length=32, task_key="instruction"
+    )
+
+    robot_processor = DataProcessorPipeline(
+        [original_processor], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    with tempfile.TemporaryDirectory() as temp_dir:
+        # Save processor
+        robot_processor.save_pretrained(temp_dir)
+
+        # Load processor - tokenizer will be recreated from saved config
+        loaded_processor = DataProcessorPipeline.from_pretrained(
+            temp_dir,
+            config_filename="dataprocessorpipeline.json",
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        # Test that loaded processor works
+        transition = create_transition(
+            observation={"state": torch.tensor([1.0, 2.0])},
+            action=torch.tensor([0.1, 0.2]),
+            complementary_data={"instruction": "test instruction"},
+        )
+
+        result = loaded_processor(transition)
+        assert TransitionKey.OBSERVATION in result
+        assert f"{OBS_LANGUAGE}.tokens" in result[TransitionKey.OBSERVATION]
+        assert f"{OBS_LANGUAGE}.attention_mask" in result[TransitionKey.OBSERVATION]
+
+
+@require_package("transformers")
+def test_save_and_load_pretrained_with_tokenizer_object():
+    """Test saving and loading processor with tokenizer object using overrides."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+
+    original_processor = TokenizerProcessorStep(
+        tokenizer=mock_tokenizer, max_length=32, task_key="instruction"
+    )
+
+    robot_processor = DataProcessorPipeline(
+        [original_processor], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    with tempfile.TemporaryDirectory() as temp_dir:
+        # Save processor
+        robot_processor.save_pretrained(temp_dir)
+
+        # Load processor with tokenizer override (since tokenizer object wasn't saved)
+        loaded_processor = DataProcessorPipeline.from_pretrained(
+            temp_dir,
+            config_filename="dataprocessorpipeline.json",
+            overrides={"tokenizer_processor": {"tokenizer": mock_tokenizer}},
+            to_transition=identity_transition,
+            to_output=identity_transition,
+        )
+
+        # Test that loaded processor works
+        transition = create_transition(
+            observation={"state": torch.tensor([1.0, 2.0])},
+            action=torch.tensor([0.1, 0.2]),
+            complementary_data={"instruction": "test instruction"},
+        )
+
+        result = loaded_processor(transition)
+        assert TransitionKey.OBSERVATION in result
+        assert f"{OBS_LANGUAGE}.tokens" in result[TransitionKey.OBSERVATION]
+        assert f"{OBS_LANGUAGE}.attention_mask" in result[TransitionKey.OBSERVATION]
+
+
+@require_package("transformers")
+def test_registry_functionality():
+    """Test that the processor is properly registered."""
+    from lerobot.processor import ProcessorStepRegistry
+
+    # Check that the processor is registered
+    assert "tokenizer_processor" in ProcessorStepRegistry.list()
+
+    # Check that we can retrieve it
+    retrieved_class = ProcessorStepRegistry.get("tokenizer_processor")
+    assert retrieved_class is TokenizerProcessorStep
+
+
+@require_package("transformers")
+def test_features_basic():
+    """Test basic feature contract functionality."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=128)
+
+    input_features = {
+        PipelineFeatureType.OBSERVATION: {OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(10,))},
+        PipelineFeatureType.ACTION: {ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(5,))},
+    }
+
+    output_features = processor.transform_features(input_features)
+
+    # Check that original features are preserved
+    assert OBS_STATE in output_features[PipelineFeatureType.OBSERVATION]
+    assert ACTION in output_features[PipelineFeatureType.ACTION]
+
+    # Check that tokenized features are added
+    assert f"{OBS_LANGUAGE}.tokens" in output_features[PipelineFeatureType.OBSERVATION]
+    assert f"{OBS_LANGUAGE}.attention_mask" in output_features[PipelineFeatureType.OBSERVATION]
+
+    # Check feature properties
+    tokens_feature = output_features[PipelineFeatureType.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask_feature = output_features[PipelineFeatureType.OBSERVATION][
+        f"{OBS_LANGUAGE}.attention_mask"
+    ]
+
+    assert tokens_feature.type == FeatureType.LANGUAGE
+    assert tokens_feature.shape == (128,)
+    assert attention_mask_feature.type == FeatureType.LANGUAGE
+    assert attention_mask_feature.shape == (128,)
+
+
+@require_package("transformers")
+def test_features_with_custom_max_length():
+    """Test feature contract with custom max_length."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=64)
+
+    input_features = {PipelineFeatureType.OBSERVATION: {}}
+    output_features = processor.transform_features(input_features)
+
+    # Check that features use correct max_length
+    assert f"{OBS_LANGUAGE}.tokens" in output_features[PipelineFeatureType.OBSERVATION]
+    assert f"{OBS_LANGUAGE}.attention_mask" in output_features[PipelineFeatureType.OBSERVATION]
+
+    tokens_feature = output_features[PipelineFeatureType.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask_feature = output_features[PipelineFeatureType.OBSERVATION][
+        f"{OBS_LANGUAGE}.attention_mask"
+    ]
+
+    assert tokens_feature.shape == (64,)
+    assert attention_mask_feature.shape == (64,)
+
+
+@require_package("transformers")
+def test_features_existing_features():
+    """Test feature contract when tokenized features already exist."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=256)
+
+    input_features = {
+        PipelineFeatureType.OBSERVATION: {
+            f"{OBS_LANGUAGE}.tokens": PolicyFeature(type=FeatureType.LANGUAGE, shape=(100,)),
+            f"{OBS_LANGUAGE}.attention_mask": PolicyFeature(type=FeatureType.LANGUAGE, shape=(100,)),
+        }
+    }
+
+    output_features = processor.transform_features(input_features)
+
+    # Should not overwrite existing features
+    assert output_features[PipelineFeatureType.OBSERVATION][f"{OBS_LANGUAGE}.tokens"].shape == (
+        100,
+    )  # Original shape preserved
+    assert output_features[PipelineFeatureType.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"].shape == (100,)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_tokenization_parameters(mock_auto_tokenizer):
+    """Test that tokenization parameters are correctly passed to tokenizer."""
+
+    # Create a custom mock that tracks calls
+    class TrackingMockTokenizer:
+        def __init__(self):
+            self.last_call_args = None
+            self.last_call_kwargs = None
+
+        def __call__(self, *args, **kwargs):
+            self.last_call_args = args
+            self.last_call_kwargs = kwargs
+            # Return minimal valid output
+            return {
+                "input_ids": torch.zeros(16, dtype=torch.long),
+                "attention_mask": torch.ones(16, dtype=torch.long),
+            }
+
+    tracking_tokenizer = TrackingMockTokenizer()
+    mock_auto_tokenizer.from_pretrained.return_value = tracking_tokenizer
+
+    processor = TokenizerProcessorStep(
+        tokenizer_name="test-tokenizer",
+        max_length=16,
+        padding="longest",
+        truncation=False,
+        padding_side="left",
+    )
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "test task"},
+    )
+
+    processor(transition)
+
+    # Check that parameters were passed correctly (task is converted to list)
+    assert tracking_tokenizer.last_call_args == (["test task"],)
+    assert tracking_tokenizer.last_call_kwargs["max_length"] == 16
+    assert tracking_tokenizer.last_call_kwargs["padding"] == "longest"
+    assert tracking_tokenizer.last_call_kwargs["padding_side"] == "left"
+    assert tracking_tokenizer.last_call_kwargs["truncation"] is False
+    assert tracking_tokenizer.last_call_kwargs["return_tensors"] == "pt"
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_preserves_other_complementary_data(mock_auto_tokenizer):
+    """Test that other complementary data fields are preserved."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer")
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={
+            "task": "test task",
+            "episode_id": 123,
+            "timestamp": 456.789,
+            "other_field": {"nested": "data"},
+        },
+    )
+
+    result = processor(transition)
+    comp_data = result[TransitionKey.COMPLEMENTARY_DATA]
+
+    # Check that all original fields are preserved
+    assert comp_data["task"] == "test task"
+    assert comp_data["episode_id"] == 123
+    assert comp_data["timestamp"] == 456.789
+    assert comp_data["other_field"] == {"nested": "data"}
+
+    # Check that tokens were added to observation
+    observation = result[TransitionKey.OBSERVATION]
+    assert f"{OBS_LANGUAGE}.tokens" in observation
+    assert f"{OBS_LANGUAGE}.attention_mask" in observation
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_deterministic_tokenization(mock_auto_tokenizer):
+    """Test that tokenization is deterministic for the same input."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer", max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "consistent test"},
+    )
+
+    result1 = processor(transition)
+    result2 = processor(transition)
+
+    tokens1 = result1[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask1 = result1[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+    tokens2 = result2[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask2 = result2[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+
+    # Results should be identical
+    assert torch.equal(tokens1, tokens2)
+    assert torch.equal(attention_mask1, attention_mask2)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_empty_string_task(mock_auto_tokenizer):
+    """Test handling of empty string task."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer", max_length=8)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": ""},
+    )
+
+    result = processor(transition)
+
+    # Should still tokenize (mock tokenizer handles empty strings)
+    observation = result[TransitionKey.OBSERVATION]
+    assert f"{OBS_LANGUAGE}.tokens" in observation
+    tokens = observation[f"{OBS_LANGUAGE}.tokens"]
+    assert tokens.shape == (8,)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_very_long_task(mock_auto_tokenizer):
+    """Test handling of very long task strings."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer", max_length=5, truncation=True)
+
+    long_task = " ".join(["word"] * 100)  # Very long task
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": long_task},
+    )
+
+    result = processor(transition)
+
+    # Should be truncated to max_length
+    observation = result[TransitionKey.OBSERVATION]
+    tokens = observation[f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = observation[f"{OBS_LANGUAGE}.attention_mask"]
+    assert tokens.shape == (5,)
+    assert attention_mask.shape == (5,)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_custom_padding_side(mock_auto_tokenizer):
+    """Test using custom padding_side parameter."""
+
+    # Create a mock tokenizer that tracks padding_side calls
+    class PaddingSideTrackingTokenizer:
+        def __init__(self):
+            self.padding_side_calls = []
+
+        def __call__(
+            self,
+            text,
+            max_length=512,
+            truncation=True,
+            padding="max_length",
+            padding_side="right",
+            return_tensors="pt",
+            **kwargs,
+        ):
+            self.padding_side_calls.append(padding_side)
+            # Return minimal valid output
+            return {
+                "input_ids": torch.zeros(max_length, dtype=torch.long),
+                "attention_mask": torch.ones(max_length, dtype=torch.long),
+            }
+
+    tracking_tokenizer = PaddingSideTrackingTokenizer()
+    mock_auto_tokenizer.from_pretrained.return_value = tracking_tokenizer
+
+    # Test left padding
+    processor_left = TokenizerProcessorStep(
+        tokenizer_name="test-tokenizer", max_length=10, padding_side="left"
+    )
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "test task"},
+    )
+    processor_left(transition)
+
+    assert tracking_tokenizer.padding_side_calls[-1] == "left"
+
+    # Test right padding (default)
+    processor_right = TokenizerProcessorStep(
+        tokenizer_name="test-tokenizer", max_length=10, padding_side="right"
+    )
+
+    processor_right(transition)
+
+    assert tracking_tokenizer.padding_side_calls[-1] == "right"
+
+
+@require_package("transformers")
+def test_device_detection_cpu():
+    """Test that tokenized tensors stay on CPU when other tensors are on CPU."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    # Create transition with CPU tensors
+    observation = {OBS_STATE: torch.randn(10)}  # CPU tensor
+    action = torch.randn(5)  # CPU tensor
+    transition = create_transition(
+        observation=observation, action=action, complementary_data={"task": "test task"}
+    )
+
+    result = processor(transition)
+
+    # Check that tokenized tensors are on CPU
+    tokens = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+
+    assert tokens.device.type == "cpu"
+    assert attention_mask.device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+@require_package("transformers")
+def test_device_detection_cuda():
+    """Test that tokenized tensors are moved to CUDA when other tensors are on CUDA."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    # Create transition with CUDA tensors
+    observation = {OBS_STATE: torch.randn(10).cuda()}  # CUDA tensor
+    action = torch.randn(5).cuda()  # CUDA tensor
+    transition = create_transition(
+        observation=observation, action=action, complementary_data={"task": "test task"}
+    )
+
+    result = processor(transition)
+
+    # Check that tokenized tensors are on CUDA
+    tokens = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+
+    assert tokens.device.type == "cuda"
+    assert attention_mask.device.type == "cuda"
+    assert tokens.device.index == 0  # Should be on same device as input
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+@require_package("transformers")
+def test_device_detection_multi_gpu():
+    """Test that tokenized tensors match device in multi-GPU setup."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    # Test with tensors on cuda:1
+    device = torch.device("cuda:1")
+    observation = {OBS_STATE: torch.randn(10).to(device)}
+    action = torch.randn(5).to(device)
+    transition = create_transition(
+        observation=observation, action=action, complementary_data={"task": "multi gpu test"}
+    )
+
+    result = processor(transition)
+
+    # Check that tokenized tensors are on cuda:1
+    tokens = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+
+    assert tokens.device == device
+    assert attention_mask.device == device
+
+
+@require_package("transformers")
+def test_device_detection_no_tensors():
+    """Test that tokenized tensors stay on CPU when no other tensors exist."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    # Create transition with no tensors
+    transition = create_transition(
+        observation={"metadata": {"key": "value"}},  # No tensors
+        complementary_data={"task": "no tensor test"},
+    )
+
+    result = processor(transition)
+
+    # Check that tokenized tensors are on CPU (default)
+    tokens = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+
+    assert tokens.device.type == "cpu"
+    assert attention_mask.device.type == "cpu"
+
+
+@require_package("transformers")
+def test_device_detection_mixed_devices():
+    """Test device detection when tensors are on different devices (uses first found)."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    if torch.cuda.is_available():
+        # Create transition with mixed devices
+        observation = {
+            "observation.cpu": torch.randn(10),  # CPU
+            "observation.cuda": torch.randn(10).cuda(),  # CUDA
+        }
+        transition = create_transition(
+            observation=observation, complementary_data={"task": "mixed device test"}
+        )
+
+        result = processor(transition)
+
+        # The device detection should use the first tensor found
+        # (iteration order depends on dict, but result should be consistent)
+        tokens = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+        attention_mask = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+
+        # Both should be on the same device
+        assert tokens.device == attention_mask.device
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+@require_package("transformers")
+def test_device_detection_from_action():
+    """Test that device is detected from action tensor when no observation tensors exist."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    # Create transition with action on CUDA but no observation tensors
+    observation = {"metadata": {"key": "value"}}  # No tensors in observation
+    action = torch.randn(5).cuda()
+    transition = create_transition(
+        observation=observation, action=action, complementary_data={"task": "action device test"}
+    )
+
+    result = processor(transition)
+
+    # Check that tokenized tensors match action's device
+    tokens = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+
+    assert tokens.device.type == "cuda"
+    assert attention_mask.device.type == "cuda"
+
+
+@require_package("transformers")
+def test_device_detection_preserves_dtype():
+    """Test that device detection doesn't affect dtype of tokenized tensors."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    # Create transition with float tensor (to test dtype isn't affected)
+    observation = {OBS_STATE: torch.randn(10, dtype=torch.float16)}
+    transition = create_transition(observation=observation, complementary_data={"task": "dtype test"})
+
+    result = processor(transition)
+
+    # Check that tokenized tensors have correct dtypes (not affected by input dtype)
+    tokens = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+
+    assert tokens.dtype == torch.long  # Should remain long
+    assert attention_mask.dtype == torch.bool  # Should be bool (converted in processor)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_integration_with_device_processor(mock_auto_tokenizer):
+    """Test that TokenizerProcessorStep works correctly with DeviceProcessorStep in pipeline."""
+    from lerobot.processor import DeviceProcessorStep
+
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    # Create pipeline with TokenizerProcessorStep then DeviceProcessorStep
+    tokenizer_processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer", max_length=6)
+    device_processor = DeviceProcessorStep(device="cuda:0")
+    robot_processor = DataProcessorPipeline(
+        [tokenizer_processor, device_processor],
+        to_transition=identity_transition,
+        to_output=identity_transition,
+    )
+
+    # Start with CPU tensors
+    transition = create_transition(
+        observation={OBS_STATE: torch.randn(10)},  # CPU
+        action=torch.randn(5),  # CPU
+        complementary_data={"task": "pipeline test"},
+    )
+
+    result = robot_processor(transition)
+
+    # All tensors should end up on CUDA (moved by DeviceProcessorStep)
+    assert result[TransitionKey.OBSERVATION][OBS_STATE].device.type == "cuda"
+    assert result[TransitionKey.ACTION].device.type == "cuda"
+
+    # Tokenized tensors should also be on CUDA
+    tokens = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+    assert tokens.device.type == "cuda"
+    assert attention_mask.device.type == "cuda"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+@require_package("transformers")
+def test_simulated_accelerate_scenario():
+    """Test scenario simulating Accelerate with data already on GPU."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    # Simulate Accelerate scenario: batch already on GPU
+    device = torch.device("cuda:0")
+    observation = {
+        OBS_STATE: torch.randn(1, 10).to(device),  # Batched, on GPU
+        OBS_IMAGE: torch.randn(1, 3, 224, 224).to(device),  # Batched, on GPU
+    }
+    action = torch.randn(1, 5).to(device)  # Batched, on GPU
+
+    transition = create_transition(
+        observation=observation,
+        action=action,
+        complementary_data={"task": ["accelerate test"]},  # List for batched task
+    )
+
+    result = processor(transition)
+
+    # Tokenized tensors should match GPU placement
+    tokens = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.tokens"]
+    attention_mask = result[TransitionKey.OBSERVATION][f"{OBS_LANGUAGE}.attention_mask"]
+
+    assert tokens.device == device
+    assert attention_mask.device == device
+    # MockTokenizer squeezes single-item batches, so shape is (max_length,) not (1, max_length)
+    assert tokens.shape == (10,)  # MockTokenizer behavior for single string in list
+    assert attention_mask.shape == (10,)
+
+
+# =============================================================================
+# Tests for get_subtask method
+# =============================================================================
+
+
+@require_package("transformers")
+def test_get_subtask_missing_key():
+    """Test get_subtask returns None when subtask key is missing from complementary_data."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task"},  # No "subtask" key
+    )
+
+    result = processor.get_subtask(transition)
+    assert result is None
+
+
+@require_package("transformers")
+def test_get_subtask_none_value():
+    """Test get_subtask returns None when subtask value is None."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": None},
+    )
+
+    result = processor.get_subtask(transition)
+    assert result is None
+
+
+@require_package("transformers")
+def test_get_subtask_none_complementary_data():
+    """Test get_subtask returns None when complementary_data is None."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data=None,  # No complementary data
+    )
+
+    result = processor.get_subtask(transition)
+    assert result is None
+
+
+@require_package("transformers")
+def test_get_subtask_string():
+    """Test get_subtask returns list with single string when subtask is a string."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": "pick up the cube"},
+    )
+
+    result = processor.get_subtask(transition)
+    assert result == ["pick up the cube"]
+    assert isinstance(result, list)
+    assert len(result) == 1
+
+
+@require_package("transformers")
+def test_get_subtask_list_of_strings():
+    """Test get_subtask returns the list when subtask is already a list of strings."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    subtask_list = ["pick up", "move to target", "place down"]
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": subtask_list},
+    )
+
+    result = processor.get_subtask(transition)
+    assert result == subtask_list
+    assert isinstance(result, list)
+    assert len(result) == 3
+
+
+@require_package("transformers")
+def test_get_subtask_unsupported_type_integer():
+    """Test get_subtask returns None when subtask is an unsupported type (integer)."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": 123},
+    )
+
+    result = processor.get_subtask(transition)
+    assert result is None
+
+
+@require_package("transformers")
+def test_get_subtask_unsupported_type_mixed_list():
+    """Test get_subtask returns None when subtask is a list with mixed types."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": ["valid string", 123, "another string"]},
+    )
+
+    result = processor.get_subtask(transition)
+    assert result is None
+
+
+@require_package("transformers")
+def test_get_subtask_unsupported_type_dict():
+    """Test get_subtask returns None when subtask is a dictionary."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": {"key": "value"}},
+    )
+
+    result = processor.get_subtask(transition)
+    assert result is None
+
+
+@require_package("transformers")
+def test_get_subtask_empty_string():
+    """Test get_subtask with empty string returns list with empty string."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": ""},
+    )
+
+    result = processor.get_subtask(transition)
+    assert result == [""]
+
+
+@require_package("transformers")
+def test_get_subtask_empty_list():
+    """Test get_subtask with empty list returns empty list."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": []},
+    )
+
+    result = processor.get_subtask(transition)
+    assert result == []
+
+
+# =============================================================================
+# Tests for subtask tokenization in observation method
+# =============================================================================
+
+
+@require_package("transformers")
+def test_subtask_tokenization_when_present():
+    """Test that subtask is tokenized and added to observation when present."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=8)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": "pick up the red cube"},
+    )
+
+    result = processor(transition)
+
+    # Check that subtask tokens were added to observation
+    observation = result[TransitionKey.OBSERVATION]
+    assert OBS_LANGUAGE_SUBTASK_TOKENS in observation
+    assert OBS_LANGUAGE_SUBTASK_ATTENTION_MASK in observation
+
+    # Check token structure
+    subtask_tokens = observation[OBS_LANGUAGE_SUBTASK_TOKENS]
+    subtask_attention_mask = observation[OBS_LANGUAGE_SUBTASK_ATTENTION_MASK]
+    assert isinstance(subtask_tokens, torch.Tensor)
+    assert isinstance(subtask_attention_mask, torch.Tensor)
+    assert subtask_tokens.shape == (8,)
+    assert subtask_attention_mask.shape == (8,)
+    assert subtask_attention_mask.dtype == torch.bool
+
+
+@require_package("transformers")
+def test_subtask_tokenization_not_added_when_none():
+    """Test that subtask tokens are NOT added to observation when subtask is None."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=8)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task"},  # No subtask
+    )
+
+    result = processor(transition)
+
+    # Check that subtask tokens were NOT added to observation
+    observation = result[TransitionKey.OBSERVATION]
+    assert OBS_LANGUAGE_SUBTASK_TOKENS not in observation
+    assert OBS_LANGUAGE_SUBTASK_ATTENTION_MASK not in observation
+
+    # But main task tokens should still be present
+    assert f"{OBS_LANGUAGE}.tokens" in observation
+    assert f"{OBS_LANGUAGE}.attention_mask" in observation
+
+
+@require_package("transformers")
+def test_subtask_tokenization_not_added_when_subtask_value_is_none():
+    """Test that subtask tokens are NOT added when subtask value is explicitly None."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=8)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": None},
+    )
+
+    result = processor(transition)
+
+    # Check that subtask tokens were NOT added to observation
+    observation = result[TransitionKey.OBSERVATION]
+    assert OBS_LANGUAGE_SUBTASK_TOKENS not in observation
+    assert OBS_LANGUAGE_SUBTASK_ATTENTION_MASK not in observation
+
+
+@require_package("transformers")
+def test_subtask_tokenization_list_of_strings():
+    """Test subtask tokenization with list of strings."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=8)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": ["pick up", "place down"]},
+    )
+
+    result = processor(transition)
+
+    # Check that subtask tokens were added to observation
+    observation = result[TransitionKey.OBSERVATION]
+    assert OBS_LANGUAGE_SUBTASK_TOKENS in observation
+    assert OBS_LANGUAGE_SUBTASK_ATTENTION_MASK in observation
+
+    # Check token structure for batch
+    subtask_tokens = observation[OBS_LANGUAGE_SUBTASK_TOKENS]
+    subtask_attention_mask = observation[OBS_LANGUAGE_SUBTASK_ATTENTION_MASK]
+    assert subtask_tokens.shape == (2, 8)  # batch_size=2, seq_len=8
+    assert subtask_attention_mask.shape == (2, 8)
+
+
+@require_package("transformers")
+def test_subtask_tokenization_device_cpu():
+    """Test that subtask tokens are on CPU when other tensors are on CPU."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    # Create transition with CPU tensors
+    observation = {OBS_STATE: torch.randn(10)}  # CPU tensor
+    action = torch.randn(5)  # CPU tensor
+    transition = create_transition(
+        observation=observation,
+        action=action,
+        complementary_data={"task": "main task", "subtask": "pick up cube"},
+    )
+
+    result = processor(transition)
+
+    # Check that subtask tokens are on CPU
+    subtask_tokens = result[TransitionKey.OBSERVATION][OBS_LANGUAGE_SUBTASK_TOKENS]
+    subtask_attention_mask = result[TransitionKey.OBSERVATION][OBS_LANGUAGE_SUBTASK_ATTENTION_MASK]
+
+    assert subtask_tokens.device.type == "cpu"
+    assert subtask_attention_mask.device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+@require_package("transformers")
+def test_subtask_tokenization_device_cuda():
+    """Test that subtask tokens are moved to CUDA when other tensors are on CUDA."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    # Create transition with CUDA tensors
+    observation = {OBS_STATE: torch.randn(10).cuda()}  # CUDA tensor
+    action = torch.randn(5).cuda()  # CUDA tensor
+    transition = create_transition(
+        observation=observation,
+        action=action,
+        complementary_data={"task": "main task", "subtask": "pick up cube"},
+    )
+
+    result = processor(transition)
+
+    # Check that subtask tokens are on CUDA
+    subtask_tokens = result[TransitionKey.OBSERVATION][OBS_LANGUAGE_SUBTASK_TOKENS]
+    subtask_attention_mask = result[TransitionKey.OBSERVATION][OBS_LANGUAGE_SUBTASK_ATTENTION_MASK]
+
+    assert subtask_tokens.device.type == "cuda"
+    assert subtask_attention_mask.device.type == "cuda"
+
+
+@require_package("transformers")
+def test_subtask_tokenization_preserves_other_observation_data():
+    """Test that subtask tokenization preserves other observation data."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    original_state = torch.tensor([1.0, 2.0, 3.0])
+    transition = create_transition(
+        observation={"state": original_state.clone()},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": "pick up cube"},
+    )
+
+    result = processor(transition)
+    observation = result[TransitionKey.OBSERVATION]
+
+    # Check that original observation data is preserved
+    assert torch.equal(observation["state"], original_state)
+
+    # Check that both task and subtask tokens are present
+    assert f"{OBS_LANGUAGE}.tokens" in observation
+    assert f"{OBS_LANGUAGE}.attention_mask" in observation
+    assert OBS_LANGUAGE_SUBTASK_TOKENS in observation
+    assert OBS_LANGUAGE_SUBTASK_ATTENTION_MASK in observation
+
+
+@require_package("transformers")
+def test_subtask_attention_mask_dtype():
+    """Test that subtask attention mask has correct dtype (bool)."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": "pick up cube"},
+    )
+
+    result = processor(transition)
+    observation = result[TransitionKey.OBSERVATION]
+
+    subtask_attention_mask = observation[OBS_LANGUAGE_SUBTASK_ATTENTION_MASK]
+    assert subtask_attention_mask.dtype == torch.bool
+
+
+@require_package("transformers")
+def test_subtask_tokenization_deterministic():
+    """Test that subtask tokenization is deterministic for the same input."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=10)
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": "consistent subtask"},
+    )
+
+    result1 = processor(transition)
+    result2 = processor(transition)
+
+    subtask_tokens1 = result1[TransitionKey.OBSERVATION][OBS_LANGUAGE_SUBTASK_TOKENS]
+    subtask_tokens2 = result2[TransitionKey.OBSERVATION][OBS_LANGUAGE_SUBTASK_TOKENS]
+    subtask_mask1 = result1[TransitionKey.OBSERVATION][OBS_LANGUAGE_SUBTASK_ATTENTION_MASK]
+    subtask_mask2 = result2[TransitionKey.OBSERVATION][OBS_LANGUAGE_SUBTASK_ATTENTION_MASK]
+
+    # Results should be identical
+    assert torch.equal(subtask_tokens1, subtask_tokens2)
+    assert torch.equal(subtask_mask1, subtask_mask2)
+
+
+@require_package("transformers")
+@patch("lerobot.processor.tokenizer_processor.AutoTokenizer")
+def test_subtask_tokenization_integration_with_pipeline(mock_auto_tokenizer):
+    """Test subtask tokenization works correctly with DataProcessorPipeline."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    mock_auto_tokenizer.from_pretrained.return_value = mock_tokenizer
+
+    tokenizer_processor = TokenizerProcessorStep(tokenizer_name="test-tokenizer", max_length=6)
+    robot_processor = DataProcessorPipeline(
+        [tokenizer_processor], to_transition=identity_transition, to_output=identity_transition
+    )
+
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": "subtask instruction"},
+    )
+
+    result = robot_processor(transition)
+
+    # Check that observation exists and both tokenizations were applied
+    assert TransitionKey.OBSERVATION in result
+    observation = result[TransitionKey.OBSERVATION]
+
+    # Check task tokens
+    assert f"{OBS_LANGUAGE}.tokens" in observation
+    assert f"{OBS_LANGUAGE}.attention_mask" in observation
+
+    # Check subtask tokens
+    assert OBS_LANGUAGE_SUBTASK_TOKENS in observation
+    assert OBS_LANGUAGE_SUBTASK_ATTENTION_MASK in observation
+
+    # Check shapes
+    assert observation[f"{OBS_LANGUAGE}.tokens"].shape == (6,)
+    assert observation[OBS_LANGUAGE_SUBTASK_TOKENS].shape == (6,)
+
+
+@require_package("transformers")
+def test_subtask_not_added_for_unsupported_types():
+    """Test that subtask tokens are not added when subtask has unsupported type."""
+    mock_tokenizer = MockTokenizer(vocab_size=100)
+    processor = TokenizerProcessorStep(tokenizer=mock_tokenizer, max_length=8)
+
+    # Test with integer subtask
+    transition = create_transition(
+        observation={"state": torch.tensor([1.0, 2.0])},
+        action=torch.tensor([0.1, 0.2]),
+        complementary_data={"task": "main task", "subtask": 123},
+    )
+
+    result = processor(transition)
+    observation = result[TransitionKey.OBSERVATION]
+
+    # Subtask tokens should NOT be added for unsupported types
+    assert OBS_LANGUAGE_SUBTASK_TOKENS not in observation
+    assert OBS_LANGUAGE_SUBTASK_ATTENTION_MASK not in observation
+
+    # But main task tokens should still be present
+    assert f"{OBS_LANGUAGE}.tokens" in observation
diff --git a/lerobot/tests/processor/test_vqbet_processor.py b/lerobot/tests/processor/test_vqbet_processor.py
new file mode 100644
index 0000000000000000000000000000000000000000..47e41dff4797dc9d729e1fcdb72c8755e2b80491
--- /dev/null
+++ b/lerobot/tests/processor/test_vqbet_processor.py
@@ -0,0 +1,462 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tests for VQBeT policy processor."""
+
+import tempfile
+
+import pytest
+import torch
+
+from lerobot.configs.types import FeatureType, NormalizationMode, PolicyFeature
+from lerobot.policies.vqbet.configuration_vqbet import VQBeTConfig
+from lerobot.policies.vqbet.processor_vqbet import make_vqbet_pre_post_processors
+from lerobot.processor import (
+    AddBatchDimensionProcessorStep,
+    DataProcessorPipeline,
+    DeviceProcessorStep,
+    NormalizerProcessorStep,
+    RenameObservationsProcessorStep,
+    TransitionKey,
+    UnnormalizerProcessorStep,
+)
+from lerobot.processor.converters import create_transition, transition_to_batch
+from lerobot.utils.constants import ACTION, OBS_IMAGE, OBS_STATE
+
+
+def create_default_config():
+    """Create a default VQBeT configuration for testing."""
+    config = VQBeTConfig()
+    config.input_features = {
+        OBS_STATE: PolicyFeature(type=FeatureType.STATE, shape=(8,)),
+        OBS_IMAGE: PolicyFeature(type=FeatureType.VISUAL, shape=(3, 224, 224)),
+    }
+    config.output_features = {
+        ACTION: PolicyFeature(type=FeatureType.ACTION, shape=(7,)),
+    }
+    config.normalization_mapping = {
+        FeatureType.STATE: NormalizationMode.MEAN_STD,
+        FeatureType.VISUAL: NormalizationMode.IDENTITY,
+        FeatureType.ACTION: NormalizationMode.MIN_MAX,
+    }
+    config.device = "cpu"
+    return config
+
+
+def create_default_stats():
+    """Create default dataset statistics for testing."""
+    return {
+        OBS_STATE: {"mean": torch.zeros(8), "std": torch.ones(8)},
+        OBS_IMAGE: {},  # No normalization for images
+        ACTION: {"min": torch.full((7,), -1.0), "max": torch.ones(7)},
+    }
+
+
+def test_make_vqbet_processor_basic():
+    """Test basic creation of VQBeT processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Check processor names
+    assert preprocessor.name == "policy_preprocessor"
+    assert postprocessor.name == "policy_postprocessor"
+
+    # Check steps in preprocessor
+    assert len(preprocessor.steps) == 4
+    assert isinstance(preprocessor.steps[0], RenameObservationsProcessorStep)
+    assert isinstance(preprocessor.steps[1], AddBatchDimensionProcessorStep)
+    assert isinstance(preprocessor.steps[2], DeviceProcessorStep)
+    assert isinstance(preprocessor.steps[3], NormalizerProcessorStep)
+
+    # Check steps in postprocessor
+    assert len(postprocessor.steps) == 2
+    assert isinstance(postprocessor.steps[0], UnnormalizerProcessorStep)
+    assert isinstance(postprocessor.steps[1], DeviceProcessorStep)
+
+
+def test_vqbet_processor_with_images():
+    """Test VQBeT processor with image and state observations."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create test data with images and states
+    observation = {
+        OBS_STATE: torch.randn(8),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(7)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is batched
+    assert processed[OBS_STATE].shape == (1, 8)
+    assert processed[OBS_IMAGE].shape == (1, 3, 224, 224)
+    assert processed[TransitionKey.ACTION.value].shape == (1, 7)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_vqbet_processor_cuda():
+    """Test VQBeT processor with CUDA device."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Create CPU data
+    observation = {
+        OBS_STATE: torch.randn(8),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(7)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is on CUDA
+    assert processed[OBS_STATE].device.type == "cuda"
+    assert processed[OBS_IMAGE].device.type == "cuda"
+    assert processed[TransitionKey.ACTION.value].device.type == "cuda"
+
+    # Process through postprocessor
+    postprocessed = postprocessor(processed[TransitionKey.ACTION.value])
+
+    # Check that action is back on CPU
+    assert postprocessed.device.type == "cpu"
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_vqbet_processor_accelerate_scenario():
+    """Test VQBeT processor in simulated Accelerate scenario."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Simulate Accelerate: data already on GPU and batched
+    device = torch.device("cuda:0")
+    observation = {
+        OBS_STATE: torch.randn(1, 8).to(device),
+        OBS_IMAGE: torch.randn(1, 3, 224, 224).to(device),
+    }
+    action = torch.randn(1, 7).to(device)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on same GPU
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+@pytest.mark.skipif(torch.cuda.device_count() < 2, reason="Requires at least 2 GPUs")
+def test_vqbet_processor_multi_gpu():
+    """Test VQBeT processor with multi-GPU setup."""
+    config = create_default_config()
+    config.device = "cuda:0"
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Simulate data on different GPU
+    device = torch.device("cuda:1")
+    observation = {
+        OBS_STATE: torch.randn(1, 8).to(device),
+        OBS_IMAGE: torch.randn(1, 3, 224, 224).to(device),
+    }
+    action = torch.randn(1, 7).to(device)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data stays on cuda:1
+    assert processed[OBS_STATE].device == device
+    assert processed[OBS_IMAGE].device == device
+    assert processed[TransitionKey.ACTION.value].device == device
+
+
+def test_vqbet_processor_without_stats():
+    """Test VQBeT processor creation without dataset statistics."""
+    config = create_default_config()
+
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(config, dataset_stats=None)
+
+    # Should still create processors
+    assert preprocessor is not None
+    assert postprocessor is not None
+
+    # Process should still work
+    observation = {
+        OBS_STATE: torch.randn(8),
+        OBS_IMAGE: torch.randn(3, 224, 224),
+    }
+    action = torch.randn(7)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    processed = preprocessor(batch)
+    assert processed is not None
+
+
+def test_vqbet_processor_save_and_load():
+    """Test saving and loading VQBeT processor."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    with tempfile.TemporaryDirectory() as tmpdir:
+        # Save preprocessor
+        preprocessor.save_pretrained(tmpdir)
+
+        # Load preprocessor
+        loaded_preprocessor = DataProcessorPipeline.from_pretrained(
+            tmpdir, config_filename="policy_preprocessor.json"
+        )
+
+        # Test that loaded processor works
+        observation = {
+            OBS_STATE: torch.randn(8),
+            OBS_IMAGE: torch.randn(3, 224, 224),
+        }
+        action = torch.randn(7)
+        transition = create_transition(observation, action)
+
+        batch = transition_to_batch(transition)
+        processed = loaded_preprocessor(batch)
+        assert processed[OBS_STATE].shape == (1, 8)
+        assert processed[OBS_IMAGE].shape == (1, 3, 224, 224)
+        assert processed[TransitionKey.ACTION.value].shape == (1, 7)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_vqbet_processor_mixed_precision():
+    """Test VQBeT processor with mixed precision."""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    # Create processor
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Replace DeviceProcessorStep with one that uses float16
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="float16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Update normalizer to use the same device as the device processor
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=step.features,
+                    norm_map=step.norm_map,
+                    stats=step.stats,
+                    device=config.device,
+                    dtype=torch.float16,  # Match the float16 dtype
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Create test data
+    observation = {
+        OBS_STATE: torch.randn(8, dtype=torch.float32),
+        OBS_IMAGE: torch.randn(3, 224, 224, dtype=torch.float32),
+    }
+    action = torch.randn(7, dtype=torch.float32)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that data is converted to float16
+    assert processed[OBS_STATE].dtype == torch.float16
+    assert processed[OBS_IMAGE].dtype == torch.float16
+    assert processed[TransitionKey.ACTION.value].dtype == torch.float16
+
+
+def test_vqbet_processor_large_batch():
+    """Test VQBeT processor with large batch sizes."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Test with large batch
+    batch_size = 128
+    observation = {
+        OBS_STATE: torch.randn(batch_size, 8),
+        OBS_IMAGE: torch.randn(batch_size, 3, 224, 224),
+    }
+    action = torch.randn(batch_size, 7)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through preprocessor
+
+    processed = preprocessor(batch)
+
+    # Check that batch dimension is preserved
+    assert processed[OBS_STATE].shape == (batch_size, 8)
+    assert processed[OBS_IMAGE].shape == (batch_size, 3, 224, 224)
+    assert processed[TransitionKey.ACTION.value].shape == (batch_size, 7)
+
+
+def test_vqbet_processor_sequential_processing():
+    """Test VQBeT processor with sequential data processing."""
+    config = create_default_config()
+    stats = create_default_stats()
+
+    preprocessor, postprocessor = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Process multiple samples sequentially
+    results = []
+    for _ in range(5):
+        observation = {
+            OBS_STATE: torch.randn(8),
+            OBS_IMAGE: torch.randn(3, 224, 224),
+        }
+        action = torch.randn(7)
+        transition = create_transition(observation, action)
+
+        batch = transition_to_batch(transition)
+
+        processed = preprocessor(batch)
+        results.append(processed)
+
+    # Check that all results are consistent
+    for result in results:
+        assert result[OBS_STATE].shape == (1, 8)
+        assert result[OBS_IMAGE].shape == (1, 3, 224, 224)
+        assert result[TransitionKey.ACTION.value].shape == (1, 7)
+
+
+@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA not available")
+def test_vqbet_processor_bfloat16_device_float32_normalizer():
+    """Test: DeviceProcessor(bfloat16) + NormalizerProcessor(float32) → output bfloat16 via automatic adaptation"""
+    config = create_default_config()
+    config.device = "cuda"
+    stats = create_default_stats()
+
+    preprocessor, _ = make_vqbet_pre_post_processors(
+        config,
+        stats,
+    )
+
+    # Modify the pipeline to use bfloat16 device processor with float32 normalizer
+    modified_steps = []
+    for step in preprocessor.steps:
+        if isinstance(step, DeviceProcessorStep):
+            # Device processor converts to bfloat16
+            modified_steps.append(DeviceProcessorStep(device=config.device, float_dtype="bfloat16"))
+        elif isinstance(step, NormalizerProcessorStep):
+            # Normalizer stays configured as float32 (will auto-adapt to bfloat16)
+            modified_steps.append(
+                NormalizerProcessorStep(
+                    features=step.features,
+                    norm_map=step.norm_map,
+                    stats=step.stats,
+                    device=config.device,
+                    dtype=torch.float32,  # Deliberately configured as float32
+                )
+            )
+        else:
+            modified_steps.append(step)
+    preprocessor.steps = modified_steps
+
+    # Verify initial normalizer configuration
+    normalizer_step = preprocessor.steps[3]  # NormalizerProcessorStep
+    assert normalizer_step.dtype == torch.float32
+
+    # Create test data with both state and visual observations
+    observation = {
+        OBS_STATE: torch.randn(8, dtype=torch.float32),
+        OBS_IMAGE: torch.randn(3, 224, 224, dtype=torch.float32),
+    }
+    action = torch.randn(7, dtype=torch.float32)
+    transition = create_transition(observation, action)
+
+    batch = transition_to_batch(transition)
+
+    # Process through full pipeline
+    processed = preprocessor(batch)
+
+    # Verify: DeviceProcessor → bfloat16, NormalizerProcessor adapts → final output is bfloat16
+    assert processed[OBS_STATE].dtype == torch.bfloat16
+    assert processed[OBS_IMAGE].dtype == torch.bfloat16  # IDENTITY normalization still gets dtype conversion
+    assert processed[TransitionKey.ACTION.value].dtype == torch.bfloat16
+
+    # Verify normalizer automatically adapted its internal state
+    assert normalizer_step.dtype == torch.bfloat16
+    # Check state stats (has normalization)
+    for stat_tensor in normalizer_step._tensor_stats[OBS_STATE].values():
+        assert stat_tensor.dtype == torch.bfloat16
+    # OBS_IMAGE uses IDENTITY normalization, so no stats to check
diff --git a/lerobot/tests/rl/test_actor.py b/lerobot/tests/rl/test_actor.py
new file mode 100644
index 0000000000000000000000000000000000000000..54e4d28700aa59e8b3d91eb426aac8a451085e68
--- /dev/null
+++ b/lerobot/tests/rl/test_actor.py
@@ -0,0 +1,209 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from concurrent import futures
+from unittest.mock import patch
+
+import pytest
+import torch
+from torch.multiprocessing import Event, Queue
+
+from lerobot.utils.constants import OBS_STR
+from lerobot.utils.transition import Transition
+from tests.utils import require_package
+
+
+def create_learner_service_stub():
+    import grpc
+
+    from lerobot.transport import services_pb2, services_pb2_grpc
+
+    class MockLearnerService(services_pb2_grpc.LearnerServiceServicer):
+        def __init__(self):
+            self.ready_call_count = 0
+            self.should_fail = False
+
+        def Ready(self, request, context):  # noqa: N802
+            self.ready_call_count += 1
+            if self.should_fail:
+                context.set_code(grpc.StatusCode.UNAVAILABLE)
+                context.set_details("Service unavailable")
+                raise grpc.RpcError("Service unavailable")
+            return services_pb2.Empty()
+
+    """Fixture to start a LearnerService gRPC server and provide a connected stub."""
+
+    servicer = MockLearnerService()
+
+    # Create a gRPC server and add our servicer to it.
+    server = grpc.server(futures.ThreadPoolExecutor(max_workers=4))
+    services_pb2_grpc.add_LearnerServiceServicer_to_server(servicer, server)
+    port = server.add_insecure_port("[::]:0")  # bind to a free port chosen by OS
+    server.start()  # start the server (non-blocking call):contentReference[oaicite:1]{index=1}
+
+    # Create a client channel and stub connected to the server's port.
+    channel = grpc.insecure_channel(f"localhost:{port}")
+    return services_pb2_grpc.LearnerServiceStub(channel), servicer, channel, server
+
+
+def close_service_stub(channel, server):
+    channel.close()
+    server.stop(None)
+
+
+@require_package("grpcio", "grpc")
+def test_establish_learner_connection_success():
+    from lerobot.rl.actor import establish_learner_connection
+
+    """Test successful connection establishment."""
+    stub, _servicer, channel, server = create_learner_service_stub()
+
+    shutdown_event = Event()
+
+    # Test successful connection
+    result = establish_learner_connection(stub, shutdown_event, attempts=5)
+
+    assert result is True
+
+    close_service_stub(channel, server)
+
+
+@require_package("grpcio", "grpc")
+def test_establish_learner_connection_failure():
+    from lerobot.rl.actor import establish_learner_connection
+
+    """Test connection failure."""
+    stub, servicer, channel, server = create_learner_service_stub()
+    servicer.should_fail = True
+
+    shutdown_event = Event()
+
+    # Test failed connection
+    with patch("time.sleep"):  # Speed up the test
+        result = establish_learner_connection(stub, shutdown_event, attempts=2)
+
+    assert result is False
+
+    close_service_stub(channel, server)
+
+
+@require_package("grpcio", "grpc")
+def test_push_transitions_to_transport_queue():
+    from lerobot.rl.actor import push_transitions_to_transport_queue
+    from lerobot.transport.utils import bytes_to_transitions
+    from tests.transport.test_transport_utils import assert_transitions_equal
+
+    """Test pushing transitions to transport queue."""
+    # Create mock transitions
+    transitions = []
+    for i in range(3):
+        transition = Transition(
+            state={OBS_STR: torch.randn(3, 64, 64), "state": torch.randn(10)},
+            action=torch.randn(5),
+            reward=torch.tensor(1.0 + i),
+            done=torch.tensor(False),
+            truncated=torch.tensor(False),
+            next_state={OBS_STR: torch.randn(3, 64, 64), "state": torch.randn(10)},
+            complementary_info={"step": torch.tensor(i)},
+        )
+        transitions.append(transition)
+
+    transitions_queue = Queue()
+
+    # Test pushing transitions
+    push_transitions_to_transport_queue(transitions, transitions_queue)
+
+    # Verify the data can be retrieved
+    serialized_data = transitions_queue.get()
+    assert isinstance(serialized_data, bytes)
+    deserialized_transitions = bytes_to_transitions(serialized_data)
+    assert len(deserialized_transitions) == len(transitions)
+    for i, deserialized_transition in enumerate(deserialized_transitions):
+        assert_transitions_equal(deserialized_transition, transitions[i])
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(3)  # force cross-platform watchdog
+def test_transitions_stream():
+    from lerobot.rl.actor import transitions_stream
+
+    """Test transitions stream functionality."""
+    shutdown_event = Event()
+    transitions_queue = Queue()
+
+    # Add test data to queue
+    test_data = [b"transition_data_1", b"transition_data_2", b"transition_data_3"]
+    for data in test_data:
+        transitions_queue.put(data)
+
+    # Collect streamed data
+    streamed_data = []
+    stream_generator = transitions_stream(shutdown_event, transitions_queue, 0.1)
+
+    # Process a few items
+    for i, message in enumerate(stream_generator):
+        streamed_data.append(message)
+        if i >= len(test_data) - 1:
+            shutdown_event.set()
+            break
+
+    # Verify we got messages
+    assert len(streamed_data) == len(test_data)
+    assert streamed_data[0].data == b"transition_data_1"
+    assert streamed_data[1].data == b"transition_data_2"
+    assert streamed_data[2].data == b"transition_data_3"
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(3)  # force cross-platform watchdog
+def test_interactions_stream():
+    from lerobot.rl.actor import interactions_stream
+    from lerobot.transport.utils import bytes_to_python_object, python_object_to_bytes
+
+    """Test interactions stream functionality."""
+    shutdown_event = Event()
+    interactions_queue = Queue()
+
+    # Create test interaction data (similar structure to what would be sent)
+    test_interactions = [
+        {"episode_reward": 10.5, "step": 1, "policy_fps": 30.2},
+        {"episode_reward": 15.2, "step": 2, "policy_fps": 28.7},
+        {"episode_reward": 8.7, "step": 3, "policy_fps": 29.1},
+    ]
+
+    # Serialize the interaction data as it would be in practice
+    test_data = [
+        interactions_queue.put(python_object_to_bytes(interaction)) for interaction in test_interactions
+    ]
+
+    # Collect streamed data
+    streamed_data = []
+    stream_generator = interactions_stream(shutdown_event, interactions_queue, 0.1)
+
+    # Process the items
+    for i, message in enumerate(stream_generator):
+        streamed_data.append(message)
+        if i >= len(test_data) - 1:
+            shutdown_event.set()
+            break
+
+    # Verify we got messages
+    assert len(streamed_data) == len(test_data)
+
+    # Verify the messages can be deserialized back to original data
+    for i, message in enumerate(streamed_data):
+        deserialized_interaction = bytes_to_python_object(message.data)
+        assert deserialized_interaction == test_interactions[i]
diff --git a/lerobot/tests/rl/test_actor_learner.py b/lerobot/tests/rl/test_actor_learner.py
new file mode 100644
index 0000000000000000000000000000000000000000..e13862d82e37d910da733079a532ae3c8e212c7d
--- /dev/null
+++ b/lerobot/tests/rl/test_actor_learner.py
@@ -0,0 +1,298 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import socket
+import threading
+import time
+
+import pytest
+import torch
+from torch.multiprocessing import Event, Queue
+
+from lerobot.configs.train import TrainRLServerPipelineConfig
+from lerobot.policies.sac.configuration_sac import SACConfig
+from lerobot.utils.constants import OBS_STR
+from lerobot.utils.transition import Transition
+from tests.utils import require_package
+
+
+def create_test_transitions(count: int = 3) -> list[Transition]:
+    """Create test transitions for integration testing."""
+    transitions = []
+    for i in range(count):
+        transition = Transition(
+            state={OBS_STR: torch.randn(3, 64, 64), "state": torch.randn(10)},
+            action=torch.randn(5),
+            reward=torch.tensor(1.0 + i),
+            done=torch.tensor(i == count - 1),  # Last transition is done
+            truncated=torch.tensor(False),
+            next_state={OBS_STR: torch.randn(3, 64, 64), "state": torch.randn(10)},
+            complementary_info={"step": torch.tensor(i), "episode_id": i // 2},
+        )
+        transitions.append(transition)
+    return transitions
+
+
+def create_test_interactions(count: int = 3) -> list[dict]:
+    """Create test interactions for integration testing."""
+    interactions = []
+    for i in range(count):
+        interaction = {
+            "episode_reward": 10.0 + i * 5,
+            "step": i * 100,
+            "policy_fps": 30.0 + i,
+            "intervention_rate": 0.1 * i,
+            "episode_length": 200 + i * 50,
+        }
+        interactions.append(interaction)
+    return interactions
+
+
+def find_free_port():
+    """Finds a free port on the local machine."""
+    with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
+        s.bind(("", 0))  # Bind to port 0 to let the OS choose a free port
+        s.listen(1)
+        port = s.getsockname()[1]
+        return port
+
+
+@pytest.fixture
+def cfg():
+    cfg = TrainRLServerPipelineConfig()
+
+    port = find_free_port()
+
+    policy_cfg = SACConfig()
+    policy_cfg.actor_learner_config.learner_host = "127.0.0.1"
+    policy_cfg.actor_learner_config.learner_port = port
+    policy_cfg.concurrency.actor = "threads"
+    policy_cfg.concurrency.learner = "threads"
+    policy_cfg.actor_learner_config.queue_get_timeout = 0.1
+
+    cfg.policy = policy_cfg
+
+    return cfg
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(10)  # force cross-platform watchdog
+def test_end_to_end_transitions_flow(cfg):
+    from lerobot.rl.actor import (
+        establish_learner_connection,
+        learner_service_client,
+        push_transitions_to_transport_queue,
+        send_transitions,
+    )
+    from lerobot.rl.learner import start_learner
+    from lerobot.transport.utils import bytes_to_transitions
+    from tests.transport.test_transport_utils import assert_transitions_equal
+
+    """Test complete transitions flow from actor to learner."""
+    transitions_actor_queue = Queue()
+    transitions_learner_queue = Queue()
+
+    interactions_queue = Queue()
+    parameters_queue = Queue()
+    shutdown_event = Event()
+
+    learner_thread = threading.Thread(
+        target=start_learner,
+        args=(parameters_queue, transitions_learner_queue, interactions_queue, shutdown_event, cfg),
+    )
+    learner_thread.start()
+
+    policy_cfg = cfg.policy
+    learner_client, channel = learner_service_client(
+        host=policy_cfg.actor_learner_config.learner_host, port=policy_cfg.actor_learner_config.learner_port
+    )
+
+    assert establish_learner_connection(learner_client, shutdown_event, attempts=5)
+
+    send_transitions_thread = threading.Thread(
+        target=send_transitions, args=(cfg, transitions_actor_queue, shutdown_event, learner_client, channel)
+    )
+    send_transitions_thread.start()
+
+    input_transitions = create_test_transitions(count=5)
+
+    push_transitions_to_transport_queue(input_transitions, transitions_actor_queue)
+
+    # Wait for learner to start
+    time.sleep(0.1)
+
+    shutdown_event.set()
+
+    # Wait for learner to receive transitions
+    learner_thread.join()
+    send_transitions_thread.join()
+    channel.close()
+
+    received_transitions = []
+    while not transitions_learner_queue.empty():
+        received_transitions.extend(bytes_to_transitions(transitions_learner_queue.get()))
+
+    assert len(received_transitions) == len(input_transitions)
+    for i, transition in enumerate(received_transitions):
+        assert_transitions_equal(transition, input_transitions[i])
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(10)
+def test_end_to_end_interactions_flow(cfg):
+    from lerobot.rl.actor import (
+        establish_learner_connection,
+        learner_service_client,
+        send_interactions,
+    )
+    from lerobot.rl.learner import start_learner
+    from lerobot.transport.utils import bytes_to_python_object, python_object_to_bytes
+
+    """Test complete interactions flow from actor to learner."""
+    # Queues for actor-learner communication
+    interactions_actor_queue = Queue()
+    interactions_learner_queue = Queue()
+
+    # Other queues required by the learner
+    parameters_queue = Queue()
+    transitions_learner_queue = Queue()
+
+    shutdown_event = Event()
+
+    # Start the learner in a separate thread
+    learner_thread = threading.Thread(
+        target=start_learner,
+        args=(parameters_queue, transitions_learner_queue, interactions_learner_queue, shutdown_event, cfg),
+    )
+    learner_thread.start()
+
+    # Establish connection from actor to learner
+    policy_cfg = cfg.policy
+    learner_client, channel = learner_service_client(
+        host=policy_cfg.actor_learner_config.learner_host, port=policy_cfg.actor_learner_config.learner_port
+    )
+
+    assert establish_learner_connection(learner_client, shutdown_event, attempts=5)
+
+    # Start the actor's interaction sending process in a separate thread
+    send_interactions_thread = threading.Thread(
+        target=send_interactions,
+        args=(cfg, interactions_actor_queue, shutdown_event, learner_client, channel),
+    )
+    send_interactions_thread.start()
+
+    # Create and push test interactions to the actor's queue
+    input_interactions = create_test_interactions(count=5)
+    for interaction in input_interactions:
+        interactions_actor_queue.put(python_object_to_bytes(interaction))
+
+    # Wait for the communication to happen
+    time.sleep(0.1)
+
+    # Signal shutdown and wait for threads to complete
+    shutdown_event.set()
+    learner_thread.join()
+    send_interactions_thread.join()
+    channel.close()
+
+    # Verify that the learner received the interactions
+    received_interactions = []
+    while not interactions_learner_queue.empty():
+        received_interactions.append(bytes_to_python_object(interactions_learner_queue.get()))
+
+    assert len(received_interactions) == len(input_interactions)
+
+    # Sort by a unique key to handle potential reordering in queues
+    received_interactions.sort(key=lambda x: x["step"])
+    input_interactions.sort(key=lambda x: x["step"])
+
+    for received, expected in zip(received_interactions, input_interactions, strict=False):
+        assert received == expected
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.parametrize("data_size", ["small", "large"])
+@pytest.mark.timeout(10)
+def test_end_to_end_parameters_flow(cfg, data_size):
+    from lerobot.rl.actor import establish_learner_connection, learner_service_client, receive_policy
+    from lerobot.rl.learner import start_learner
+    from lerobot.transport.utils import bytes_to_state_dict, state_to_bytes
+
+    """Test complete parameter flow from learner to actor, with small and large data."""
+    # Actor's local queue to receive params
+    parameters_actor_queue = Queue()
+    # Learner's queue to send params from
+    parameters_learner_queue = Queue()
+
+    # Other queues required by the learner
+    transitions_learner_queue = Queue()
+    interactions_learner_queue = Queue()
+
+    shutdown_event = Event()
+
+    # Start the learner in a separate thread
+    learner_thread = threading.Thread(
+        target=start_learner,
+        args=(
+            parameters_learner_queue,
+            transitions_learner_queue,
+            interactions_learner_queue,
+            shutdown_event,
+            cfg,
+        ),
+    )
+    learner_thread.start()
+
+    # Establish connection from actor to learner
+    policy_cfg = cfg.policy
+    learner_client, channel = learner_service_client(
+        host=policy_cfg.actor_learner_config.learner_host, port=policy_cfg.actor_learner_config.learner_port
+    )
+
+    assert establish_learner_connection(learner_client, shutdown_event, attempts=5)
+
+    # Start the actor's parameter receiving process in a separate thread
+    receive_params_thread = threading.Thread(
+        target=receive_policy,
+        args=(cfg, parameters_actor_queue, shutdown_event, learner_client, channel),
+    )
+    receive_params_thread.start()
+
+    # Create test parameters based on parametrization
+    if data_size == "small":
+        input_params = {"layer.weight": torch.randn(128, 64)}
+    else:  # "large"
+        # CHUNK_SIZE is 2MB, so this tensor (4MB) will force chunking
+        input_params = {"large_layer.weight": torch.randn(1024, 1024)}
+
+    # Simulate learner having new parameters to send
+    parameters_learner_queue.put(state_to_bytes(input_params))
+
+    # Wait for the actor to receive the parameters
+    time.sleep(0.1)
+
+    # Signal shutdown and wait for threads to complete
+    shutdown_event.set()
+    learner_thread.join()
+    receive_params_thread.join()
+    channel.close()
+
+    # Verify that the actor received the parameters correctly
+    received_params = bytes_to_state_dict(parameters_actor_queue.get())
+
+    assert received_params.keys() == input_params.keys()
+    for key in input_params:
+        assert torch.allclose(received_params[key], input_params[key])
diff --git a/lerobot/tests/rl/test_learner_service.py b/lerobot/tests/rl/test_learner_service.py
new file mode 100644
index 0000000000000000000000000000000000000000..d967388f07c2e82bddcab21566dcedfb4f7add13
--- /dev/null
+++ b/lerobot/tests/rl/test_learner_service.py
@@ -0,0 +1,374 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import threading
+import time
+from concurrent import futures
+from multiprocessing import Event, Queue
+
+import pytest
+
+from tests.utils import require_package  # our gRPC servicer class
+
+
+@pytest.fixture(scope="function")
+def learner_service_stub():
+    shutdown_event = Event()
+    parameters_queue = Queue()
+    transitions_queue = Queue()
+    interactions_queue = Queue()
+    seconds_between_pushes = 1
+    client, channel, server = create_learner_service_stub(
+        shutdown_event, parameters_queue, transitions_queue, interactions_queue, seconds_between_pushes
+    )
+
+    yield client  # provide the stub to the test function
+
+    close_learner_service_stub(channel, server)
+
+
+@require_package("grpcio", "grpc")
+def create_learner_service_stub(
+    shutdown_event: Event,
+    parameters_queue: Queue,
+    transitions_queue: Queue,
+    interactions_queue: Queue,
+    seconds_between_pushes: int,
+    queue_get_timeout: float = 0.1,
+):
+    import grpc
+
+    from lerobot.rl.learner_service import LearnerService
+    from lerobot.transport import services_pb2_grpc  # generated from .proto
+
+    """Fixture to start a LearnerService gRPC server and provide a connected stub."""
+
+    servicer = LearnerService(
+        shutdown_event=shutdown_event,
+        parameters_queue=parameters_queue,
+        seconds_between_pushes=seconds_between_pushes,
+        transition_queue=transitions_queue,
+        interaction_message_queue=interactions_queue,
+        queue_get_timeout=queue_get_timeout,
+    )
+
+    # Create a gRPC server and add our servicer to it.
+    server = grpc.server(futures.ThreadPoolExecutor(max_workers=4))
+    services_pb2_grpc.add_LearnerServiceServicer_to_server(servicer, server)
+    port = server.add_insecure_port("[::]:0")  # bind to a free port chosen by OS
+    server.start()  # start the server (non-blocking call):contentReference[oaicite:1]{index=1}
+
+    # Create a client channel and stub connected to the server's port.
+    channel = grpc.insecure_channel(f"localhost:{port}")
+    return services_pb2_grpc.LearnerServiceStub(channel), channel, server
+
+
+@require_package("grpcio", "grpc")
+def close_learner_service_stub(channel, server):
+    channel.close()
+    server.stop(None)
+
+
+@pytest.mark.timeout(3)  # force cross-platform watchdog
+def test_ready_method(learner_service_stub):
+    from lerobot.transport import services_pb2
+
+    """Test the ready method of the UserService."""
+    request = services_pb2.Empty()
+    response = learner_service_stub.Ready(request)
+    assert response == services_pb2.Empty()
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(3)  # force cross-platform watchdog
+def test_send_interactions():
+    from lerobot.transport import services_pb2
+
+    shutdown_event = Event()
+
+    parameters_queue = Queue()
+    transitions_queue = Queue()
+    interactions_queue = Queue()
+    seconds_between_pushes = 1
+    client, channel, server = create_learner_service_stub(
+        shutdown_event, parameters_queue, transitions_queue, interactions_queue, seconds_between_pushes
+    )
+
+    list_of_interaction_messages = [
+        services_pb2.InteractionMessage(transfer_state=services_pb2.TransferState.TRANSFER_BEGIN, data=b"1"),
+        services_pb2.InteractionMessage(transfer_state=services_pb2.TransferState.TRANSFER_MIDDLE, data=b"2"),
+        services_pb2.InteractionMessage(transfer_state=services_pb2.TransferState.TRANSFER_END, data=b"3"),
+        services_pb2.InteractionMessage(transfer_state=services_pb2.TransferState.TRANSFER_END, data=b"4"),
+        services_pb2.InteractionMessage(transfer_state=services_pb2.TransferState.TRANSFER_END, data=b"5"),
+        services_pb2.InteractionMessage(transfer_state=services_pb2.TransferState.TRANSFER_BEGIN, data=b"6"),
+        services_pb2.InteractionMessage(transfer_state=services_pb2.TransferState.TRANSFER_MIDDLE, data=b"7"),
+        services_pb2.InteractionMessage(transfer_state=services_pb2.TransferState.TRANSFER_END, data=b"8"),
+    ]
+
+    def mock_interactions_stream():
+        yield from list_of_interaction_messages
+
+        return services_pb2.Empty()
+
+    response = client.SendInteractions(mock_interactions_stream())
+    assert response == services_pb2.Empty()
+
+    close_learner_service_stub(channel, server)
+
+    # Extract the data from the interactions queue
+    interactions = []
+    while not interactions_queue.empty():
+        interactions.append(interactions_queue.get())
+
+    assert interactions == [b"123", b"4", b"5", b"678"]
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(3)  # force cross-platform watchdog
+def test_send_transitions():
+    from lerobot.transport import services_pb2
+
+    """Test the SendTransitions method with various transition data."""
+    shutdown_event = Event()
+    parameters_queue = Queue()
+    transitions_queue = Queue()
+    interactions_queue = Queue()
+    seconds_between_pushes = 1
+
+    client, channel, server = create_learner_service_stub(
+        shutdown_event, parameters_queue, transitions_queue, interactions_queue, seconds_between_pushes
+    )
+
+    # Create test transition messages
+    list_of_transition_messages = [
+        services_pb2.Transition(
+            transfer_state=services_pb2.TransferState.TRANSFER_BEGIN, data=b"transition_1"
+        ),
+        services_pb2.Transition(
+            transfer_state=services_pb2.TransferState.TRANSFER_MIDDLE, data=b"transition_2"
+        ),
+        services_pb2.Transition(transfer_state=services_pb2.TransferState.TRANSFER_END, data=b"transition_3"),
+        services_pb2.Transition(transfer_state=services_pb2.TransferState.TRANSFER_BEGIN, data=b"batch_1"),
+        services_pb2.Transition(transfer_state=services_pb2.TransferState.TRANSFER_END, data=b"batch_2"),
+    ]
+
+    def mock_transitions_stream():
+        yield from list_of_transition_messages
+
+    response = client.SendTransitions(mock_transitions_stream())
+    assert response == services_pb2.Empty()
+
+    close_learner_service_stub(channel, server)
+
+    # Extract the data from the transitions queue
+    transitions = []
+    while not transitions_queue.empty():
+        transitions.append(transitions_queue.get())
+
+    # Should have assembled the chunked data
+    assert transitions == [b"transition_1transition_2transition_3", b"batch_1batch_2"]
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(3)  # force cross-platform watchdog
+def test_send_transitions_empty_stream():
+    from lerobot.transport import services_pb2
+
+    """Test SendTransitions with empty stream."""
+    shutdown_event = Event()
+    parameters_queue = Queue()
+    transitions_queue = Queue()
+    interactions_queue = Queue()
+    seconds_between_pushes = 1
+
+    client, channel, server = create_learner_service_stub(
+        shutdown_event, parameters_queue, transitions_queue, interactions_queue, seconds_between_pushes
+    )
+
+    def empty_stream():
+        return iter([])
+
+    response = client.SendTransitions(empty_stream())
+    assert response == services_pb2.Empty()
+
+    close_learner_service_stub(channel, server)
+
+    # Queue should remain empty
+    assert transitions_queue.empty()
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(10)  # force cross-platform watchdog
+def test_stream_parameters():
+    import time
+
+    from lerobot.transport import services_pb2
+
+    """Test the StreamParameters method."""
+    shutdown_event = Event()
+    parameters_queue = Queue()
+    transitions_queue = Queue()
+    interactions_queue = Queue()
+    seconds_between_pushes = 0.2  # Short delay for testing
+
+    client, channel, server = create_learner_service_stub(
+        shutdown_event, parameters_queue, transitions_queue, interactions_queue, seconds_between_pushes
+    )
+
+    # Add test parameters to the queue
+    test_params = [b"param_batch_1", b"param_batch_2"]
+    for param in test_params:
+        parameters_queue.put(param)
+
+    # Start streaming parameters
+    request = services_pb2.Empty()
+    stream = client.StreamParameters(request)
+
+    # Collect streamed parameters and timestamps
+    received_params = []
+    timestamps = []
+
+    for response in stream:
+        received_params.append(response.data)
+        timestamps.append(time.time())
+
+        # We should receive one last item
+        break
+
+    parameters_queue.put(b"param_batch_3")
+
+    for response in stream:
+        received_params.append(response.data)
+        timestamps.append(time.time())
+
+        # We should receive only one item
+        break
+
+    shutdown_event.set()
+    close_learner_service_stub(channel, server)
+
+    assert received_params == [b"param_batch_2", b"param_batch_3"]
+
+    # Check the time difference between the two sends
+    time_diff = timestamps[1] - timestamps[0]
+    # Check if the time difference is close to the expected push frequency
+    assert time_diff == pytest.approx(seconds_between_pushes, abs=0.1)
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(3)  # force cross-platform watchdog
+def test_stream_parameters_with_shutdown():
+    from lerobot.transport import services_pb2
+
+    """Test StreamParameters handles shutdown gracefully."""
+    shutdown_event = Event()
+    parameters_queue = Queue()
+    transitions_queue = Queue()
+    interactions_queue = Queue()
+    seconds_between_pushes = 0.1
+    queue_get_timeout = 0.001
+
+    client, channel, server = create_learner_service_stub(
+        shutdown_event,
+        parameters_queue,
+        transitions_queue,
+        interactions_queue,
+        seconds_between_pushes,
+        queue_get_timeout=queue_get_timeout,
+    )
+
+    test_params = [b"param_batch_1", b"stop", b"param_batch_3", b"param_batch_4"]
+
+    # create a thread that will put the parameters in the queue
+    def producer():
+        for param in test_params:
+            parameters_queue.put(param)
+            time.sleep(0.1)
+
+    producer_thread = threading.Thread(target=producer)
+    producer_thread.start()
+
+    # Start streaming
+    request = services_pb2.Empty()
+    stream = client.StreamParameters(request)
+
+    # Collect streamed parameters
+    received_params = []
+
+    for response in stream:
+        received_params.append(response.data)
+
+        if response.data == b"stop":
+            shutdown_event.set()
+
+    producer_thread.join()
+    close_learner_service_stub(channel, server)
+
+    assert received_params == [b"param_batch_1", b"stop"]
+
+
+@require_package("grpcio", "grpc")
+@pytest.mark.timeout(3)  # force cross-platform watchdog
+def test_stream_parameters_waits_and_retries_on_empty_queue():
+    import threading
+    import time
+
+    from lerobot.transport import services_pb2
+
+    """Test that StreamParameters waits and retries when the queue is empty."""
+    shutdown_event = Event()
+    parameters_queue = Queue()
+    transitions_queue = Queue()
+    interactions_queue = Queue()
+    seconds_between_pushes = 0.05
+    queue_get_timeout = 0.01
+
+    client, channel, server = create_learner_service_stub(
+        shutdown_event,
+        parameters_queue,
+        transitions_queue,
+        interactions_queue,
+        seconds_between_pushes,
+        queue_get_timeout=queue_get_timeout,
+    )
+
+    request = services_pb2.Empty()
+    stream = client.StreamParameters(request)
+
+    received_params = []
+
+    def producer():
+        # Let the consumer start and find an empty queue.
+        # It will wait `seconds_between_pushes` (0.05s), then `get` will timeout after `queue_get_timeout` (0.01s).
+        # Total time for the first empty loop is > 0.06s. We wait a bit longer to be safe.
+        time.sleep(0.06)
+        parameters_queue.put(b"param_after_wait")
+        time.sleep(0.05)
+        parameters_queue.put(b"param_after_wait_2")
+
+    producer_thread = threading.Thread(target=producer)
+    producer_thread.start()
+
+    # The consumer will block here until the producer sends an item.
+    for response in stream:
+        received_params.append(response.data)
+        if response.data == b"param_after_wait_2":
+            break  # We only need one item for this test.
+
+    shutdown_event.set()
+    producer_thread.join()
+    close_learner_service_stub(channel, server)
+
+    assert received_params == [b"param_after_wait", b"param_after_wait_2"]
diff --git a/lerobot/tests/rl/test_queue.py b/lerobot/tests/rl/test_queue.py
new file mode 100644
index 0000000000000000000000000000000000000000..b6716fbd642465f96a5e5561f46bf7c7d48d459d
--- /dev/null
+++ b/lerobot/tests/rl/test_queue.py
@@ -0,0 +1,166 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import threading
+import time
+from queue import Queue
+
+from torch.multiprocessing import Queue as TorchMPQueue
+
+from lerobot.rl.queue import get_last_item_from_queue
+
+
+def test_get_last_item_single_item():
+    """Test getting the last item when queue has only one item."""
+    queue = Queue()
+    queue.put("single_item")
+
+    result = get_last_item_from_queue(queue)
+
+    assert result == "single_item"
+    assert queue.empty()
+
+
+def test_get_last_item_multiple_items():
+    """Test getting the last item when queue has multiple items."""
+    queue = Queue()
+    items = ["first", "second", "third", "fourth", "last"]
+
+    for item in items:
+        queue.put(item)
+
+    result = get_last_item_from_queue(queue)
+
+    assert result == "last"
+    assert queue.empty()
+
+
+def test_get_last_item_multiple_items_with_torch_queue():
+    """Test getting the last item when queue has multiple items."""
+    queue = TorchMPQueue()
+    items = ["first", "second", "third", "fourth", "last"]
+
+    for item in items:
+        queue.put(item)
+
+    result = get_last_item_from_queue(queue)
+
+    assert result == "last"
+    assert queue.empty()
+
+
+def test_get_last_item_different_types():
+    """Test with different data types in the queue."""
+    queue = Queue()
+    items = [1, 2.5, "string", {"key": "value"}, [1, 2, 3], ("tuple", "data")]
+
+    for item in items:
+        queue.put(item)
+
+    result = get_last_item_from_queue(queue)
+
+    assert result == ("tuple", "data")
+    assert queue.empty()
+
+
+def test_get_last_item_maxsize_queue():
+    """Test with a queue that has a maximum size."""
+    queue = Queue(maxsize=5)
+
+    # Fill the queue
+    for i in range(5):
+        queue.put(i)
+
+    # Give the queue time to fill
+    time.sleep(0.1)
+
+    result = get_last_item_from_queue(queue)
+
+    assert result == 4
+    assert queue.empty()
+
+
+def test_get_last_item_with_none_values():
+    """Test with None values in the queue."""
+    queue = Queue()
+    items = [1, None, 2, None, 3]
+
+    for item in items:
+        queue.put(item)
+
+    # Give the queue time to fill
+    time.sleep(0.1)
+
+    result = get_last_item_from_queue(queue)
+
+    assert result == 3
+    assert queue.empty()
+
+
+def test_get_last_item_blocking_timeout():
+    """Test get_last_item_from_queue returns None on timeout."""
+    queue = Queue()
+    result = get_last_item_from_queue(queue, block=True, timeout=0.1)
+    assert result is None
+
+
+def test_get_last_item_non_blocking_empty():
+    """Test get_last_item_from_queue with block=False on an empty queue returns None."""
+    queue = Queue()
+    result = get_last_item_from_queue(queue, block=False)
+    assert result is None
+
+
+def test_get_last_item_non_blocking_success():
+    """Test get_last_item_from_queue with block=False on a non-empty queue."""
+    queue = Queue()
+    items = ["first", "second", "last"]
+    for item in items:
+        queue.put(item)
+
+    # Give the queue time to fill
+    time.sleep(0.1)
+
+    result = get_last_item_from_queue(queue, block=False)
+    assert result == "last"
+    assert queue.empty()
+
+
+def test_get_last_item_blocking_waits_for_item():
+    """Test that get_last_item_from_queue waits for an item if block=True."""
+    queue = Queue()
+    result = []
+
+    def producer():
+        queue.put("item1")
+        queue.put("item2")
+
+    def consumer():
+        # This will block until the producer puts the first item
+        item = get_last_item_from_queue(queue, block=True, timeout=0.2)
+        result.append(item)
+
+    producer_thread = threading.Thread(target=producer)
+    consumer_thread = threading.Thread(target=consumer)
+
+    producer_thread.start()
+    consumer_thread.start()
+
+    producer_thread.join()
+    consumer_thread.join()
+
+    assert result == ["item2"]
+    assert queue.empty()
diff --git a/lerobot/tests/robots/test_reachy2.py b/lerobot/tests/robots/test_reachy2.py
new file mode 100644
index 0000000000000000000000000000000000000000..d3f32b1c269366bdd0a38d348b099590ebec3a2f
--- /dev/null
+++ b/lerobot/tests/robots/test_reachy2.py
@@ -0,0 +1,329 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from unittest.mock import MagicMock, patch
+
+import numpy as np
+import pytest
+
+pytest.importorskip("reachy2_sdk")
+
+from lerobot.robots.reachy2 import (
+    REACHY2_ANTENNAS_JOINTS,
+    REACHY2_L_ARM_JOINTS,
+    REACHY2_NECK_JOINTS,
+    REACHY2_R_ARM_JOINTS,
+    REACHY2_VEL,
+    Reachy2Robot,
+    Reachy2RobotConfig,
+)
+
+# {lerobot_keys: reachy2_sdk_keys}
+REACHY2_JOINTS = {
+    **REACHY2_NECK_JOINTS,
+    **REACHY2_ANTENNAS_JOINTS,
+    **REACHY2_R_ARM_JOINTS,
+    **REACHY2_L_ARM_JOINTS,
+}
+
+PARAMS = [
+    {},  # default config
+    {"with_mobile_base": False},
+    {"with_mobile_base": False, "with_l_arm": False, "with_antennas": False},
+    {"with_r_arm": False, "with_neck": False, "with_antennas": False},
+    {"use_external_commands": True, "disable_torque_on_disconnect": True},
+    {"use_external_commands": True, "with_mobile_base": False, "with_neck": False},
+    {"disable_torque_on_disconnect": False},
+    {"max_relative_target": 5},
+    {"with_right_teleop_camera": False},
+    {"with_left_teleop_camera": False, "with_right_teleop_camera": False},
+    {"with_left_teleop_camera": False, "with_torso_camera": True},
+]
+
+
+def _make_reachy2_sdk_mock():
+    class JointSpy:
+        __slots__ = (
+            "present_position",
+            "_goal_position",
+            "_on_set",
+        )
+
+        def __init__(self, present_position=0.0, on_set=None):
+            self.present_position = present_position
+            self._goal_position = present_position
+            self._on_set = on_set
+
+        @property
+        def goal_position(self):
+            return self._goal_position
+
+        @goal_position.setter
+        def goal_position(self, v):
+            self._goal_position = v
+            if self._on_set:
+                self._on_set()
+
+    r = MagicMock(name="ReachySDKMock")
+    r.is_connected.return_value = True
+
+    def _connect():
+        r.is_connected.return_value = True
+
+    def _disconnect():
+        r.is_connected.return_value = False
+
+    # Global counter of goal_position sets
+    r._goal_position_set_total = 0
+
+    def _on_any_goal_set():
+        r._goal_position_set_total += 1
+
+    # Mock joints with some dummy positions
+    joints = {
+        k: JointSpy(
+            present_position=float(i),
+            on_set=_on_any_goal_set,
+        )
+        for i, k in enumerate(REACHY2_JOINTS.values())
+    }
+    r.joints = joints
+
+    # Mock mobile base with some dummy odometry
+    r.mobile_base = MagicMock()
+    r.mobile_base.odometry = {
+        "x": 0.1,
+        "y": -0.2,
+        "theta": 21.3,
+        "vx": 0.001,
+        "vy": 0.002,
+        "vtheta": 0.0,
+    }
+
+    r.connect = MagicMock(side_effect=_connect)
+    r.disconnect = MagicMock(side_effect=_disconnect)
+
+    # Mock methods
+    r.turn_on = MagicMock()
+    r.reset_default_limits = MagicMock()
+    r.send_goal_positions = MagicMock()
+    r.turn_off_smoothly = MagicMock()
+    r.mobile_base.set_goal_speed = MagicMock()
+    r.mobile_base.send_speed_command = MagicMock()
+
+    return r
+
+
+def _make_reachy2_camera_mock(*args, **kwargs):
+    cfg = args[0] if args else kwargs.get("config")
+    name = getattr(cfg, "name", kwargs.get("name", "cam"))
+    image_type = getattr(cfg, "image_type", kwargs.get("image_type", "cam"))
+    width = getattr(cfg, "width", kwargs.get("width", 640))
+    height = getattr(cfg, "height", kwargs.get("height", 480))
+
+    cam = MagicMock(name=f"Reachy2CameraMock:{name}")
+    cam.name = name
+    cam.image_type = image_type
+    cam.width = width
+    cam.height = height
+    cam.connect = MagicMock()
+    cam.disconnect = MagicMock()
+    cam.async_read = MagicMock(side_effect=lambda: np.zeros((height, width, 3), dtype=np.uint8))
+    cam.read_latest = MagicMock(side_effect=lambda: np.zeros((height, width, 3), dtype=np.uint8))
+    return cam
+
+
+@pytest.fixture(params=PARAMS, ids=lambda p: "default" if not p else ",".join(p.keys()))
+def reachy2(request):
+    with (
+        patch(
+            "lerobot.robots.reachy2.robot_reachy2.ReachySDK",
+            side_effect=lambda *a, **k: _make_reachy2_sdk_mock(),
+        ),
+        patch(
+            "lerobot.cameras.reachy2_camera.reachy2_camera.Reachy2Camera",
+            side_effect=_make_reachy2_camera_mock,
+        ),
+    ):
+        overrides = request.param
+        cfg = Reachy2RobotConfig(ip_address="192.168.0.200", **overrides)
+        robot = Reachy2Robot(cfg)
+        yield robot
+        if robot.is_connected:
+            robot.disconnect()
+
+
+def test_connect_disconnect(reachy2):
+    assert not reachy2.is_connected
+
+    reachy2.connect()
+    assert reachy2.is_connected
+
+    reachy2.reachy.turn_on.assert_called_once()
+    reachy2.reachy.reset_default_limits.assert_called_once()
+
+    reachy2.disconnect()
+    assert not reachy2.is_connected
+
+    if reachy2.config.disable_torque_on_disconnect:
+        reachy2.reachy.turn_off_smoothly.assert_called_once()
+    else:
+        reachy2.reachy.turn_off_smoothly.assert_not_called()
+    reachy2.reachy.disconnect.assert_called_once()
+
+
+def test_get_joints_dict(reachy2):
+    reachy2.connect()
+
+    if reachy2.config.with_neck:
+        assert "neck_yaw.pos" in reachy2.joints_dict
+        assert "neck_pitch.pos" in reachy2.joints_dict
+        assert "neck_roll.pos" in reachy2.joints_dict
+    else:
+        assert "neck_yaw.pos" not in reachy2.joints_dict
+        assert "neck_pitch.pos" not in reachy2.joints_dict
+        assert "neck_roll.pos" not in reachy2.joints_dict
+
+    if reachy2.config.with_antennas:
+        assert "l_antenna.pos" in reachy2.joints_dict
+        assert "r_antenna.pos" in reachy2.joints_dict
+    else:
+        assert "l_antenna.pos" not in reachy2.joints_dict
+        assert "r_antenna.pos" not in reachy2.joints_dict
+
+    if reachy2.config.with_r_arm:
+        assert "r_shoulder_pitch.pos" in reachy2.joints_dict
+        assert "r_shoulder_roll.pos" in reachy2.joints_dict
+        assert "r_elbow_yaw.pos" in reachy2.joints_dict
+        assert "r_elbow_pitch.pos" in reachy2.joints_dict
+        assert "r_wrist_roll.pos" in reachy2.joints_dict
+        assert "r_wrist_pitch.pos" in reachy2.joints_dict
+        assert "r_wrist_yaw.pos" in reachy2.joints_dict
+        assert "r_gripper.pos" in reachy2.joints_dict
+    else:
+        assert "r_shoulder_pitch.pos" not in reachy2.joints_dict
+        assert "r_shoulder_roll.pos" not in reachy2.joints_dict
+        assert "r_elbow_yaw.pos" not in reachy2.joints_dict
+        assert "r_elbow_pitch.pos" not in reachy2.joints_dict
+        assert "r_wrist_roll.pos" not in reachy2.joints_dict
+        assert "r_wrist_pitch.pos" not in reachy2.joints_dict
+        assert "r_wrist_yaw.pos" not in reachy2.joints_dict
+        assert "r_gripper.pos" not in reachy2.joints_dict
+
+    if reachy2.config.with_l_arm:
+        assert "l_shoulder_pitch.pos" in reachy2.joints_dict
+        assert "l_shoulder_roll.pos" in reachy2.joints_dict
+        assert "l_elbow_yaw.pos" in reachy2.joints_dict
+        assert "l_elbow_pitch.pos" in reachy2.joints_dict
+        assert "l_wrist_roll.pos" in reachy2.joints_dict
+        assert "l_wrist_pitch.pos" in reachy2.joints_dict
+        assert "l_wrist_yaw.pos" in reachy2.joints_dict
+        assert "l_gripper.pos" in reachy2.joints_dict
+    else:
+        assert "l_shoulder_pitch.pos" not in reachy2.joints_dict
+        assert "l_shoulder_roll.pos" not in reachy2.joints_dict
+        assert "l_elbow_yaw.pos" not in reachy2.joints_dict
+        assert "l_elbow_pitch.pos" not in reachy2.joints_dict
+        assert "l_wrist_roll.pos" not in reachy2.joints_dict
+        assert "l_wrist_pitch.pos" not in reachy2.joints_dict
+        assert "l_wrist_yaw.pos" not in reachy2.joints_dict
+        assert "l_gripper.pos" not in reachy2.joints_dict
+
+
+def test_get_observation(reachy2):
+    reachy2.connect()
+    obs = reachy2.get_observation()
+
+    expected_keys = set(reachy2.joints_dict)
+    expected_keys.update(f"{v}" for v in REACHY2_VEL if reachy2.config.with_mobile_base)
+    expected_keys.update(reachy2.cameras.keys())
+    assert set(obs.keys()) == expected_keys
+
+    for motor in reachy2.joints_dict:
+        assert obs[motor] == reachy2.reachy.joints[REACHY2_JOINTS[motor]].present_position
+    if reachy2.config.with_mobile_base:
+        for vel in REACHY2_VEL:
+            assert obs[vel] == reachy2.reachy.mobile_base.odometry[REACHY2_VEL[vel]]
+    if reachy2.config.with_left_teleop_camera:
+        assert obs["teleop_left"].shape == (
+            reachy2.config.cameras["teleop_left"].height,
+            reachy2.config.cameras["teleop_left"].width,
+            3,
+        )
+    if reachy2.config.with_right_teleop_camera:
+        assert obs["teleop_right"].shape == (
+            reachy2.config.cameras["teleop_right"].height,
+            reachy2.config.cameras["teleop_right"].width,
+            3,
+        )
+    if reachy2.config.with_torso_camera:
+        assert obs["torso_rgb"].shape == (
+            reachy2.config.cameras["torso_rgb"].height,
+            reachy2.config.cameras["torso_rgb"].width,
+            3,
+        )
+
+
+def test_send_action(reachy2):
+    reachy2.connect()
+
+    action = {k: i * 10.0 for i, k in enumerate(reachy2.joints_dict.keys(), start=1)}
+    if reachy2.config.with_mobile_base:
+        action.update({k: i * 0.1 for i, k in enumerate(REACHY2_VEL.keys(), start=1)})
+
+    previous_present_position = {
+        k: reachy2.reachy.joints[REACHY2_JOINTS[k]].present_position for k in reachy2.joints_dict
+    }
+    returned = reachy2.send_action(action)
+
+    if reachy2.config.max_relative_target is None:
+        assert returned == action
+
+    assert reachy2.reachy._goal_position_set_total == len(reachy2.joints_dict)
+    for motor in reachy2.joints_dict:
+        expected_pos = action[motor]
+        real_pos = reachy2.reachy.joints[REACHY2_JOINTS[motor]].goal_position
+        if reachy2.config.max_relative_target is None:
+            assert real_pos == expected_pos
+        else:
+            assert real_pos == previous_present_position[motor] + np.sign(expected_pos) * min(
+                abs(expected_pos - real_pos), reachy2.config.max_relative_target
+            )
+
+    if reachy2.config.with_mobile_base:
+        goal_speed = [i * 0.1 for i, _ in enumerate(REACHY2_VEL.keys(), start=1)]
+        reachy2.reachy.mobile_base.set_goal_speed.assert_called_once_with(*goal_speed)
+
+    if reachy2.config.use_external_commands:
+        reachy2.reachy.send_goal_positions.assert_not_called()
+        if reachy2.config.with_mobile_base:
+            reachy2.reachy.mobile_base.send_speed_command.assert_not_called()
+    else:
+        reachy2.reachy.send_goal_positions.assert_called_once()
+        if reachy2.config.with_mobile_base:
+            reachy2.reachy.mobile_base.send_speed_command.assert_called_once()
+
+
+def test_no_part_declared():
+    with pytest.raises(ValueError):
+        _ = Reachy2RobotConfig(
+            ip_address="192.168.0.200",
+            with_mobile_base=False,
+            with_l_arm=False,
+            with_r_arm=False,
+            with_neck=False,
+            with_antennas=False,
+        )
diff --git a/lerobot/tests/robots/test_so100_follower.py b/lerobot/tests/robots/test_so100_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..b61d0ca0147ad546c8e114b40f781004d784c399
--- /dev/null
+++ b/lerobot/tests/robots/test_so100_follower.py
@@ -0,0 +1,111 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from contextlib import contextmanager
+from unittest.mock import MagicMock, patch
+
+import pytest
+
+from lerobot.robots.so_follower import (
+    SO100Follower,
+    SO100FollowerConfig,
+)
+
+
+def _make_bus_mock() -> MagicMock:
+    """Return a bus mock with just the attributes used by the robot."""
+    bus = MagicMock(name="FeetechBusMock")
+    bus.is_connected = False
+
+    def _connect():
+        bus.is_connected = True
+
+    def _disconnect(_disable=True):
+        bus.is_connected = False
+
+    bus.connect.side_effect = _connect
+    bus.disconnect.side_effect = _disconnect
+
+    @contextmanager
+    def _dummy_cm():
+        yield
+
+    bus.torque_disabled.side_effect = _dummy_cm
+
+    return bus
+
+
+@pytest.fixture
+def follower():
+    bus_mock = _make_bus_mock()
+
+    def _bus_side_effect(*_args, **kwargs):
+        bus_mock.motors = kwargs["motors"]
+        motors_order: list[str] = list(bus_mock.motors)
+
+        bus_mock.sync_read.return_value = {motor: idx for idx, motor in enumerate(motors_order, 1)}
+        bus_mock.sync_write.return_value = None
+        bus_mock.write.return_value = None
+        bus_mock.disable_torque.return_value = None
+        bus_mock.enable_torque.return_value = None
+        bus_mock.is_calibrated = True
+        return bus_mock
+
+    with (
+        patch(
+            "lerobot.robots.so_follower.so_follower.FeetechMotorsBus",
+            side_effect=_bus_side_effect,
+        ),
+        patch.object(SO100Follower, "configure", lambda self: None),
+    ):
+        cfg = SO100FollowerConfig(port="/dev/null")
+        robot = SO100Follower(cfg)
+        yield robot
+        if robot.is_connected:
+            robot.disconnect()
+
+
+def test_connect_disconnect(follower):
+    assert not follower.is_connected
+
+    follower.connect()
+    assert follower.is_connected
+
+    follower.disconnect()
+    assert not follower.is_connected
+
+
+def test_get_observation(follower):
+    follower.connect()
+    obs = follower.get_observation()
+
+    expected_keys = {f"{m}.pos" for m in follower.bus.motors}
+    assert set(obs.keys()) == expected_keys
+
+    for idx, motor in enumerate(follower.bus.motors, 1):
+        assert obs[f"{motor}.pos"] == idx
+
+
+def test_send_action(follower):
+    follower.connect()
+
+    action = {f"{m}.pos": i * 10 for i, m in enumerate(follower.bus.motors, 1)}
+    returned = follower.send_action(action)
+
+    assert returned == action
+
+    goal_pos = {m: (i + 1) * 10 for i, m in enumerate(follower.bus.motors)}
+    follower.bus.sync_write.assert_called_once_with("Goal_Position", goal_pos)
diff --git a/lerobot/tests/robots/test_unitree_g1.py b/lerobot/tests/robots/test_unitree_g1.py
new file mode 100644
index 0000000000000000000000000000000000000000..8cc85b5723d6eaddee59e41ba87f3511c828d000
--- /dev/null
+++ b/lerobot/tests/robots/test_unitree_g1.py
@@ -0,0 +1,267 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Tests for Unitree G1 robot. Meant to be run in an environment where the Unitree SDK is installed."""
+
+from unittest.mock import MagicMock, patch
+
+import numpy as np
+import pytest
+
+from lerobot.utils.import_utils import _unitree_sdk_available
+
+if not _unitree_sdk_available:
+    pytest.skip("Unitree SDK not available", allow_module_level=True)
+
+from lerobot.robots.unitree_g1.config_unitree_g1 import UnitreeG1Config
+from lerobot.robots.unitree_g1.g1_utils import (
+    NUM_MOTORS,
+    REMOTE_AXES,
+    REMOTE_BUTTONS,
+    REMOTE_KEYS,
+    G1_29_JointArmIndex,
+    G1_29_JointIndex,
+    default_remote_input,
+    get_gravity_orientation,
+)
+
+# ---------------------------------------------------------------------------
+# Unit tests for g1_utils (no SDK needed)
+# ---------------------------------------------------------------------------
+
+
+class TestG1Utils:
+    def test_num_motors(self):
+        assert NUM_MOTORS == 29
+
+    def test_joint_index_count(self):
+        assert len(G1_29_JointIndex) == 29
+
+    def test_joint_arm_index_count(self):
+        assert len(G1_29_JointArmIndex) == 14
+
+    def test_arm_indices_are_subset_of_full(self):
+        full_values = {j.value for j in G1_29_JointIndex}
+        arm_values = {j.value for j in G1_29_JointArmIndex}
+        assert arm_values.issubset(full_values)
+
+    def test_arm_indices_start_at_15(self):
+        assert min(j.value for j in G1_29_JointArmIndex) == 15
+        assert max(j.value for j in G1_29_JointArmIndex) == 28
+
+    def test_enum_naming_consistency(self):
+        """Verify all wrist joints use consistent PascalCase naming."""
+        wrist_joints = [j for j in G1_29_JointIndex if "Wrist" in j.name]
+        for j in wrist_joints:
+            # Should be "WristYaw", "WristPitch", "WristRoll" — no lowercase after "Wrist"
+            after_wrist = j.name.split("Wrist")[1]
+            assert after_wrist[0].isupper(), f"{j.name} has inconsistent casing after 'Wrist'"
+
+    def test_remote_keys_structure(self):
+        assert len(REMOTE_AXES) == 4
+        assert len(REMOTE_BUTTONS) == 16
+        assert len(REMOTE_KEYS) == 20
+        assert REMOTE_KEYS == REMOTE_AXES + REMOTE_BUTTONS
+
+    def test_default_remote_input(self):
+        d = default_remote_input()
+        assert len(d) == 20
+        assert all(v == 0.0 for v in d.values())
+        assert set(d.keys()) == set(REMOTE_KEYS)
+
+    def test_gravity_orientation_identity(self):
+        """Quaternion [1, 0, 0, 0] (no rotation) should give gravity along -z."""
+        g = get_gravity_orientation([1.0, 0.0, 0.0, 0.0])
+        assert g.shape == (3,)
+        assert g.dtype == np.float32
+        np.testing.assert_allclose(g, [0.0, 0.0, -1.0], atol=1e-6)
+
+    def test_gravity_orientation_dtype(self):
+        g = get_gravity_orientation(np.array([1.0, 0.0, 0.0, 0.0]))
+        assert g.dtype == np.float32
+
+
+# ---------------------------------------------------------------------------
+# Unit tests for UnitreeG1Config (no SDK needed)
+# ---------------------------------------------------------------------------
+
+
+class TestUnitreeG1Config:
+    def test_default_config(self):
+        cfg = UnitreeG1Config()
+        assert len(cfg.kp) == 29
+        assert len(cfg.kd) == 29
+        assert len(cfg.default_positions) == 29
+        assert cfg.is_simulation is True
+        assert cfg.controller is None
+        assert cfg.gravity_compensation is False
+
+    def test_gains_are_positive(self):
+        cfg = UnitreeG1Config()
+        assert all(v > 0 for v in cfg.kp)
+        assert all(v > 0 for v in cfg.kd)
+
+    def test_config_copies_gains(self):
+        """Each config instance should have its own copy of gains."""
+        cfg1 = UnitreeG1Config()
+        cfg2 = UnitreeG1Config()
+        cfg1.kp[0] = 999.0
+        assert cfg2.kp[0] != 999.0
+
+
+# ---------------------------------------------------------------------------
+# Robot mock and integration tests
+# ---------------------------------------------------------------------------
+
+
+def _make_lowstate_msg_mock():
+    """Create a mock that mimics the SDK LowState_ message."""
+    msg = MagicMock()
+    for i in range(29):
+        motor = MagicMock()
+        motor.q = float(i) * 0.1
+        motor.dq = float(i) * 0.01
+        motor.tau_est = float(i) * 0.001
+        motor.temperature = 30.0 + i
+        msg.motor_state.__getitem__ = lambda self, idx, _motors={}: _motors.setdefault(
+            idx, MagicMock(q=idx * 0.1, dq=idx * 0.01, tau_est=idx * 0.001, temperature=30.0 + idx)
+        )
+
+    msg.imu_state.quaternion = [1.0, 0.0, 0.0, 0.0]
+    msg.imu_state.gyroscope = [0.1, 0.2, 0.3]
+    msg.imu_state.accelerometer = [0.0, 0.0, 9.81]
+    msg.imu_state.rpy = [0.0, 0.0, 0.0]
+    msg.imu_state.temperature = 25.0
+    msg.wireless_remote = b"\x00" * 40
+    msg.mode_machine = 0
+    return msg
+
+
+def _make_sdk_mocks():
+    """Create mocks for the Unitree SDK modules used by UnitreeG1."""
+    lowcmd_default = MagicMock()
+    lowcmd_default.mode_pr = 0
+    lowcmd_default.motor_cmd = [MagicMock() for _ in range(35)]
+
+    crc_mock = MagicMock()
+    crc_mock.Crc.return_value = 0
+
+    lowstate_msg = _make_lowstate_msg_mock()
+
+    subscriber_mock = MagicMock()
+    subscriber_mock.Read.return_value = lowstate_msg
+
+    publisher_mock = MagicMock()
+
+    return {
+        "lowcmd_default": lowcmd_default,
+        "crc_mock": crc_mock,
+        "subscriber_mock": subscriber_mock,
+        "publisher_mock": publisher_mock,
+        "lowstate_msg": lowstate_msg,
+    }
+
+
+@pytest.fixture
+def unitree_g1():
+    """Create a UnitreeG1 robot with all SDK dependencies mocked."""
+    mocks = _make_sdk_mocks()
+
+    mock_channel_init = MagicMock()
+    mock_channel_pub = MagicMock(return_value=mocks["publisher_mock"])
+    mock_channel_sub = MagicMock(return_value=mocks["subscriber_mock"])
+
+    with (
+        patch(
+            "lerobot.robots.unitree_g1.unitree_g1.make_cameras_from_configs",
+            return_value={},
+        ),
+        patch(
+            "lerobot.robots.unitree_g1.unitree_g1.G1_29_ArmIK",
+            return_value=MagicMock(),
+        ),
+        patch(
+            "lerobot.robots.unitree_g1.unitree_g1._SDKChannelFactoryInitialize",
+            mock_channel_init,
+        ),
+        patch(
+            "lerobot.robots.unitree_g1.unitree_g1._SDKChannelPublisher",
+            mock_channel_pub,
+        ),
+        patch(
+            "lerobot.robots.unitree_g1.unitree_g1._SDKChannelSubscriber",
+            mock_channel_sub,
+        ),
+        patch(
+            "lerobot.robots.unitree_g1.unitree_g1.unitree_hg_msg_dds__LowCmd_",
+            MagicMock(return_value=mocks["lowcmd_default"]),
+        ),
+        patch(
+            "lerobot.robots.unitree_g1.unitree_g1.hg_LowCmd",
+            MagicMock,
+        ),
+        patch(
+            "lerobot.robots.unitree_g1.unitree_g1.hg_LowState",
+            MagicMock,
+        ),
+        patch(
+            "lerobot.robots.unitree_g1.unitree_g1.CRC",
+            MagicMock(return_value=mocks["crc_mock"]),
+        ),
+    ):
+        from lerobot.robots.unitree_g1.unitree_g1 import UnitreeG1
+
+        cfg = UnitreeG1Config(is_simulation=True, gravity_compensation=False)
+        robot = UnitreeG1(cfg)
+        yield robot, mocks
+        if robot.is_connected:
+            robot.disconnect()
+
+
+def test_init_state(unitree_g1):
+    robot, _ = unitree_g1
+    assert not robot.is_connected
+    assert robot.controller is None
+
+
+def test_observation_features(unitree_g1):
+    robot, _ = unitree_g1
+    features = robot.observation_features
+    # Should have .q for all 29 joints (no cameras configured)
+    assert len(features) == 29
+    for joint in G1_29_JointIndex:
+        assert f"{joint.name}.q" in features
+
+
+def test_action_features_no_controller(unitree_g1):
+    robot, _ = unitree_g1
+    features = robot.action_features
+    # Without controller: all 29 joints
+    assert len(features) == 29
+    for joint in G1_29_JointIndex:
+        assert f"{joint.name}.q" in features
+
+
+def test_get_observation_before_connect(unitree_g1):
+    robot, _ = unitree_g1
+    obs = robot.get_observation()
+    assert obs == {}
+
+
+def test_disconnect_idempotent(unitree_g1):
+    robot, _ = unitree_g1
+    # Should not raise even when not connected
+    robot.disconnect()
diff --git a/lerobot/tests/scripts/test_edit_dataset_parsing.py b/lerobot/tests/scripts/test_edit_dataset_parsing.py
new file mode 100644
index 0000000000000000000000000000000000000000..4d758ae35cd00cd176812e681efb32eea3c08461
--- /dev/null
+++ b/lerobot/tests/scripts/test_edit_dataset_parsing.py
@@ -0,0 +1,89 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import draccus
+import pytest
+
+from lerobot.scripts.lerobot_edit_dataset import (
+    ConvertImageToVideoConfig,
+    DeleteEpisodesConfig,
+    EditDatasetConfig,
+    InfoConfig,
+    MergeConfig,
+    ModifyTasksConfig,
+    OperationConfig,
+    RemoveFeatureConfig,
+    SplitConfig,
+    _validate_config,
+)
+
+
+def parse_cfg(cli_args: list[str]) -> EditDatasetConfig:
+    """Helper to parse CLI args into an EditDatasetConfig via draccus."""
+    return draccus.parse(EditDatasetConfig, args=cli_args)
+
+
+class TestOperationTypeParsing:
+    """Test that --operation.type correctly selects the right config subclass."""
+
+    @pytest.mark.parametrize(
+        "type_name, expected_cls",
+        [
+            ("delete_episodes", DeleteEpisodesConfig),
+            ("split", SplitConfig),
+            ("merge", MergeConfig),
+            ("remove_feature", RemoveFeatureConfig),
+            ("modify_tasks", ModifyTasksConfig),
+            ("convert_image_to_video", ConvertImageToVideoConfig),
+            ("info", InfoConfig),
+        ],
+    )
+    def test_operation_type_resolves_correct_class(self, type_name, expected_cls):
+        cfg = parse_cfg(
+            ["--repo_id", "test/repo", "--new_repo_id", "test/merged", "--operation.type", type_name]
+        )
+        assert isinstance(cfg.operation, expected_cls), (
+            f"Expected {expected_cls.__name__}, got {type(cfg.operation).__name__}"
+        )
+
+    def test_merge_requires_new_repo_id(self):
+        cfg = parse_cfg(["--operation.type", "merge"])
+        with pytest.raises(ValueError, match="--new_repo_id is required for merge"):
+            _validate_config(cfg)
+
+    def test_non_merge_requires_repo_id(self):
+        cfg = parse_cfg(["--operation.type", "delete_episodes"])
+        with pytest.raises(ValueError, match="--repo_id is required for delete_episodes"):
+            _validate_config(cfg)
+
+    @pytest.mark.parametrize(
+        "type_name, expected_cls",
+        [
+            ("delete_episodes", DeleteEpisodesConfig),
+            ("split", SplitConfig),
+            ("merge", MergeConfig),
+            ("remove_feature", RemoveFeatureConfig),
+            ("modify_tasks", ModifyTasksConfig),
+            ("convert_image_to_video", ConvertImageToVideoConfig),
+            ("info", InfoConfig),
+        ],
+    )
+    def test_get_choice_name_roundtrips(self, type_name, expected_cls):
+        cfg = parse_cfg(
+            ["--repo_id", "test/repo", "--new_repo_id", "test/merged", "--operation.type", type_name]
+        )
+        resolved_name = OperationConfig.get_choice_name(type(cfg.operation))
+        assert resolved_name == type_name
diff --git a/lerobot/tests/teleoperators/test_reachy2_teleoperator.py b/lerobot/tests/teleoperators/test_reachy2_teleoperator.py
new file mode 100644
index 0000000000000000000000000000000000000000..dd8c5904c336cedbc5c01c2aed996354e4a0bad0
--- /dev/null
+++ b/lerobot/tests/teleoperators/test_reachy2_teleoperator.py
@@ -0,0 +1,150 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from unittest.mock import MagicMock, patch
+
+import pytest
+
+from lerobot.teleoperators.reachy2_teleoperator import (
+    REACHY2_ANTENNAS_JOINTS,
+    REACHY2_L_ARM_JOINTS,
+    REACHY2_NECK_JOINTS,
+    REACHY2_R_ARM_JOINTS,
+    REACHY2_VEL,
+    Reachy2Teleoperator,
+    Reachy2TeleoperatorConfig,
+)
+
+# {lerobot_keys: reachy2_sdk_keys}
+REACHY2_JOINTS = {
+    **REACHY2_NECK_JOINTS,
+    **REACHY2_ANTENNAS_JOINTS,
+    **REACHY2_R_ARM_JOINTS,
+    **REACHY2_L_ARM_JOINTS,
+}
+
+PARAMS = [
+    {},  # default config
+    {"with_mobile_base": False},
+    {"with_mobile_base": False, "with_l_arm": False, "with_antennas": False},
+    {"with_r_arm": False, "with_neck": False, "with_antennas": False},
+    {"with_mobile_base": False, "with_neck": False},
+    {"use_present_position": True},
+]
+
+
+def _make_reachy2_sdk_mock():
+    r = MagicMock(name="ReachySDKMock")
+    r.is_connected.return_value = True
+
+    def _connect():
+        r.is_connected.return_value = True
+
+    def _disconnect():
+        r.is_connected.return_value = False
+
+    # Mock joints with some dummy positions
+    joints = {
+        k: MagicMock(
+            present_position=float(i),
+            goal_position=float(i) + 0.5,
+        )
+        for i, k in enumerate(REACHY2_JOINTS.values())
+    }
+    r.joints = joints
+
+    # Mock mobile base with some dummy odometry
+    r.mobile_base = MagicMock()
+    r.mobile_base.last_cmd_vel = {
+        "vx": -0.2,
+        "vy": 0.2,
+        "vtheta": 11.0,
+    }
+    r.mobile_base.odometry = {
+        "x": 1.0,
+        "y": 2.0,
+        "theta": 20.0,
+        "vx": 0.1,
+        "vy": -0.1,
+        "vtheta": 8.0,
+    }
+
+    r.connect = MagicMock(side_effect=_connect)
+    r.disconnect = MagicMock(side_effect=_disconnect)
+
+    return r
+
+
+@pytest.fixture(params=PARAMS, ids=lambda p: "default" if not p else ",".join(p.keys()))
+def reachy2(request):
+    with (
+        patch(
+            "lerobot.teleoperators.reachy2_teleoperator.reachy2_teleoperator.ReachySDK",
+            side_effect=lambda *a, **k: _make_reachy2_sdk_mock(),
+        ),
+    ):
+        overrides = request.param
+        cfg = Reachy2TeleoperatorConfig(ip_address="192.168.0.200", **overrides)
+        robot = Reachy2Teleoperator(cfg)
+        yield robot
+        if robot.is_connected:
+            robot.disconnect()
+
+
+def test_connect_disconnect(reachy2):
+    assert not reachy2.is_connected
+
+    reachy2.connect()
+    assert reachy2.is_connected
+
+    reachy2.disconnect()
+    assert not reachy2.is_connected
+
+    reachy2.reachy.disconnect.assert_called_once()
+
+
+def test_get_action(reachy2):
+    reachy2.connect()
+    action = reachy2.get_action()
+
+    expected_keys = set(reachy2.joints_dict)
+    expected_keys.update(f"{v}" for v in REACHY2_VEL if reachy2.config.with_mobile_base)
+    assert set(action.keys()) == expected_keys
+
+    for motor in reachy2.joints_dict:
+        if reachy2.config.use_present_position:
+            assert action[motor] == reachy2.reachy.joints[REACHY2_JOINTS[motor]].present_position
+        else:
+            assert action[motor] == reachy2.reachy.joints[REACHY2_JOINTS[motor]].goal_position
+    if reachy2.config.with_mobile_base:
+        if reachy2.config.use_present_position:
+            for vel in REACHY2_VEL:
+                assert action[vel] == reachy2.reachy.mobile_base.odometry[REACHY2_VEL[vel]]
+        else:
+            for vel in REACHY2_VEL:
+                assert action[vel] == reachy2.reachy.mobile_base.last_cmd_vel[REACHY2_VEL[vel]]
+
+
+def test_no_part_declared():
+    with pytest.raises(ValueError):
+        _ = Reachy2TeleoperatorConfig(
+            ip_address="192.168.0.200",
+            with_mobile_base=False,
+            with_l_arm=False,
+            with_r_arm=False,
+            with_neck=False,
+            with_antennas=False,
+        )
diff --git a/lerobot/tests/teleoperators/test_unitree_g1_teleoperator.py b/lerobot/tests/teleoperators/test_unitree_g1_teleoperator.py
new file mode 100644
index 0000000000000000000000000000000000000000..52f4a8482a5310cc49e72d061efeca57accaa451
--- /dev/null
+++ b/lerobot/tests/teleoperators/test_unitree_g1_teleoperator.py
@@ -0,0 +1,309 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Tests for Unitree G1 teleoperator. Meant to be run in an environment where the Unitree SDK is installed."""
+
+from unittest.mock import MagicMock
+
+import pytest
+
+from lerobot.utils.import_utils import _unitree_sdk_available
+
+if not _unitree_sdk_available:
+    pytest.skip("Unitree SDK not available", allow_module_level=True)
+
+from lerobot.robots.unitree_g1.g1_utils import REMOTE_AXES
+from lerobot.teleoperators.unitree_g1.config_unitree_g1 import (
+    ExoskeletonArmPortConfig,
+    UnitreeG1TeleoperatorConfig,
+)
+from lerobot.teleoperators.unitree_g1.unitree_g1 import RemoteController, UnitreeG1Teleoperator
+
+# ---------------------------------------------------------------------------
+# Tests for RemoteController
+# ---------------------------------------------------------------------------
+
+
+def _make_joystick_mock():
+    """Create a mock Joystick class matching the SDK interface."""
+    joystick = MagicMock()
+    # Axes are Axis objects with .data attribute
+    joystick.lx = MagicMock(data=0.0, smooth=0.03, deadzone=0.01)
+    joystick.ly = MagicMock(data=0.0, smooth=0.03, deadzone=0.01)
+    joystick.rx = MagicMock(data=0.0, smooth=0.03, deadzone=0.01)
+    joystick.ry = MagicMock(data=0.0, smooth=0.03, deadzone=0.01)
+    # Buttons are Button objects with .data attribute
+    for name in ["RB", "LB", "start", "back", "RT", "LT", "A", "B", "X", "Y", "up", "right", "down", "left"]:
+        setattr(joystick, name, MagicMock(data=0))
+    return joystick
+
+
+@pytest.fixture
+def remote_controller():
+    """Create a RemoteController with a mocked Joystick."""
+    mock_joystick = _make_joystick_mock()
+
+    rc = RemoteController()
+    rc._joystick = mock_joystick
+    yield rc, mock_joystick
+
+
+def test_remote_controller_init(remote_controller):
+    rc, _ = remote_controller
+    assert rc.lx == 0.0
+    assert rc.ly == 0.0
+    assert rc.rx == 0.0
+    assert rc.ry == 0.0
+    assert len(rc.button) == 16
+    assert all(b == 0 for b in rc.button)
+
+
+def test_sync_remote_action(remote_controller):
+    rc, _ = remote_controller
+    rc.lx = 0.5
+    rc.ly = -0.3
+    rc.rx = 0.1
+    rc.ry = 0.0
+    rc._sync_remote_action()
+
+    assert rc.remote_action["remote.lx"] == 0.5
+    assert rc.remote_action["remote.ly"] == -0.3
+    assert rc.remote_action["remote.rx"] == 0.1
+    assert rc.remote_action["remote.ry"] == 0.0
+
+
+def test_set_from_wireless_calls_extract(remote_controller):
+    rc, mock_joystick = remote_controller
+    # Set up the mock to populate data after extract
+    mock_joystick.lx.data = 0.5
+    mock_joystick.ly.data = -0.3
+    mock_joystick.rx.data = 0.1
+    mock_joystick.ry.data = 0.0
+
+    wireless_data = b"\x00" * 40
+    rc.set_from_wireless(wireless_data)
+
+    mock_joystick.extract.assert_called_once_with(wireless_data)
+    assert rc.lx == 0.5
+    assert rc.ly == -0.3
+
+
+def test_set_from_wireless_short_data(remote_controller):
+    rc, mock_joystick = remote_controller
+    rc.set_from_wireless(b"\x00" * 10)  # Too short
+    mock_joystick.extract.assert_not_called()
+
+
+def test_set_from_wireless_buttons(remote_controller):
+    rc, mock_joystick = remote_controller
+    # Simulate RB pressed
+    mock_joystick.RB.data = 1
+    mock_joystick.lx.data = 0.0
+    mock_joystick.ly.data = 0.0
+    mock_joystick.rx.data = 0.0
+    mock_joystick.ry.data = 0.0
+
+    rc.set_from_wireless(b"\x00" * 40)
+    assert rc.button[0] == 1  # RB maps to button[0]
+
+
+def test_set_from_exo_left(remote_controller):
+    rc, _ = remote_controller
+    rc.use_left_exo_joystick = True
+    rc.left_center_x = 2048
+    rc.left_center_y = 2048
+
+    raw16 = [0] * 16
+    raw16[11] = 3048  # X axis: (3048 - 2048) / 2047.5 ≈ 0.488
+    raw16[13] = 1048  # Y axis: (1048 - 2048) / 2047.5 ≈ -0.488
+    raw16[12] = 0  # Button pressed (below ADC_HALF)
+
+    rc.set_from_exo(raw16, "left")
+    assert rc.lx == pytest.approx((3048 - 2048) / 2047.5, abs=1e-3)
+    assert rc.ly == pytest.approx((1048 - 2048) / 2047.5, abs=1e-3)
+    assert rc.button[4] == 1  # Left button maps to button[4]
+
+
+def test_set_from_exo_clears_button(remote_controller):
+    rc, _ = remote_controller
+    rc.use_left_exo_joystick = True
+    rc.button[4] = 1  # Pre-set
+
+    raw16 = [0] * 16
+    raw16[12] = 4000  # Button NOT pressed (above ADC_HALF)
+
+    rc.set_from_exo(raw16, "left")
+    assert rc.button[4] == 0  # Should be cleared
+
+
+def test_set_from_exo_ignored_when_not_enabled(remote_controller):
+    rc, _ = remote_controller
+    rc.use_left_exo_joystick = False
+    raw16 = [0] * 16
+    raw16[11] = 3000
+
+    rc.set_from_exo(raw16, "left")
+    assert rc.lx == 0.0  # Unchanged
+
+
+# ---------------------------------------------------------------------------
+# Tests for UnitreeG1TeleoperatorConfig (no SDK needed)
+# ---------------------------------------------------------------------------
+
+
+class TestTeleoperatorConfig:
+    def test_default_config(self):
+        cfg = UnitreeG1TeleoperatorConfig()
+        assert cfg.left_arm_config.port == ""
+        assert cfg.right_arm_config.port == ""
+        assert cfg.frozen_joints == ""
+
+    def test_config_with_ports(self):
+        cfg = UnitreeG1TeleoperatorConfig(
+            left_arm_config=ExoskeletonArmPortConfig(port="/dev/ttyACM0"),
+            right_arm_config=ExoskeletonArmPortConfig(port="/dev/ttyACM1"),
+        )
+        assert cfg.left_arm_config.port == "/dev/ttyACM0"
+        assert cfg.right_arm_config.port == "/dev/ttyACM1"
+
+
+# ---------------------------------------------------------------------------
+# Tests for UnitreeG1Teleoperator
+# ---------------------------------------------------------------------------
+
+
+@pytest.fixture
+def teleop_remote_only():
+    """Create a UnitreeG1Teleoperator in remote-only mode (no exo arms)."""
+    cfg = UnitreeG1TeleoperatorConfig()  # No ports = remote-only mode
+    teleop = UnitreeG1Teleoperator(cfg)
+    yield teleop
+
+
+def test_remote_only_connect(teleop_remote_only):
+    """Remote-only mode should connect immediately without serial ports."""
+    teleop = teleop_remote_only
+    teleop.connect()
+    assert teleop.is_connected
+    assert not teleop._arm_control_enabled
+
+
+def test_remote_only_action_features(teleop_remote_only):
+    teleop = teleop_remote_only
+    features = teleop.action_features
+    # Remote-only: just the 4 remote axes
+    assert set(features.keys()) == set(REMOTE_AXES)
+
+
+def test_feedback_features(teleop_remote_only):
+    teleop = teleop_remote_only
+    features = teleop.feedback_features
+    assert "wireless_remote" in features
+    assert features["wireless_remote"] is bytes
+
+
+def test_remote_only_get_action(teleop_remote_only):
+    teleop = teleop_remote_only
+    teleop.connect()
+    action = teleop.get_action()
+    assert set(action.keys()) == set(REMOTE_AXES)
+    assert all(isinstance(v, float) for v in action.values())
+
+
+def test_send_feedback(teleop_remote_only):
+    teleop = teleop_remote_only
+    teleop.connect()
+    # Should not raise
+    teleop.send_feedback({"wireless_remote": b"\x00" * 40})
+
+
+def test_send_feedback_missing_key(teleop_remote_only):
+    teleop = teleop_remote_only
+    teleop.connect()
+    # Should not raise even with missing key
+    teleop.send_feedback({"other_key": 42})
+
+
+def test_asymmetric_exo_ports_raises():
+    """Configuring only one exo port should raise ValueError."""
+    cfg = UnitreeG1TeleoperatorConfig(
+        left_arm_config=ExoskeletonArmPortConfig(port="/dev/ttyACM0"),
+        # right_arm_config left empty
+    )
+    with pytest.raises(ValueError, match="set both left/right"):
+        UnitreeG1Teleoperator(cfg)
+
+
+# ---------------------------------------------------------------------------
+# Tests for ExoskeletonArm (needs serial mock)
+# ---------------------------------------------------------------------------
+
+
+class TestExoskeletonArm:
+    def test_parse_raw16_valid(self):
+        from lerobot.teleoperators.unitree_g1.exo_serial import parse_raw16
+
+        line = b"100 200 300 400 500 600 700 800 900 1000 1100 1200 1300 1400 1500 1600\n"
+        result = parse_raw16(line)
+        assert result is not None
+        assert len(result) == 16
+        assert result[0] == 100
+        assert result[15] == 1600
+
+    def test_parse_raw16_too_short(self):
+        from lerobot.teleoperators.unitree_g1.exo_serial import parse_raw16
+
+        line = b"100 200 300\n"
+        assert parse_raw16(line) is None
+
+    def test_parse_raw16_garbage(self):
+        from lerobot.teleoperators.unitree_g1.exo_serial import parse_raw16
+
+        assert parse_raw16(b"not numbers at all\n") is None
+        assert parse_raw16(b"\xff\xfe\xfd\n") is None
+        assert parse_raw16(b"") is None
+
+    def test_calibrate_requires_connection(self):
+        from lerobot.teleoperators.unitree_g1.exo_serial import ExoskeletonArm
+
+        arm = ExoskeletonArm(
+            port="/dev/null",
+            calibration_fpath=MagicMock(is_file=MagicMock(return_value=False)),
+            side="left",
+        )
+        with pytest.raises(RuntimeError, match="not connected"):
+            arm.calibrate()
+
+    def test_is_connected_false_by_default(self):
+        from lerobot.teleoperators.unitree_g1.exo_serial import ExoskeletonArm
+
+        arm = ExoskeletonArm(
+            port="/dev/null",
+            calibration_fpath=MagicMock(is_file=MagicMock(return_value=False)),
+            side="left",
+        )
+        assert not arm.is_connected
+        assert not arm.is_calibrated
+
+    def test_read_raw_when_disconnected(self):
+        from lerobot.teleoperators.unitree_g1.exo_serial import ExoskeletonArm
+
+        arm = ExoskeletonArm(
+            port="/dev/null",
+            calibration_fpath=MagicMock(is_file=MagicMock(return_value=False)),
+            side="left",
+        )
+        assert arm.read_raw() is None
diff --git a/lerobot/tests/test_available.py b/lerobot/tests/test_available.py
new file mode 100644
index 0000000000000000000000000000000000000000..19e39b2b6dc83dc5d7f65c9442a90e3b9aa8dc69
--- /dev/null
+++ b/lerobot/tests/test_available.py
@@ -0,0 +1,60 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import importlib
+
+import gymnasium as gym
+import pytest
+
+import lerobot
+from lerobot.policies.act.modeling_act import ACTPolicy
+from lerobot.policies.diffusion.modeling_diffusion import DiffusionPolicy
+from lerobot.policies.tdmpc.modeling_tdmpc import TDMPCPolicy
+from lerobot.policies.vqbet.modeling_vqbet import VQBeTPolicy
+from tests.utils import require_env
+
+
+@pytest.mark.parametrize("env_name, task_name", lerobot.env_task_pairs)
+@require_env
+def test_available_env_task(env_name: str, task_name: list):
+    """
+    This test verifies that all environments listed in `lerobot/__init__.py` can
+    be successfully imported — if they're installed — and that their
+    `available_tasks_per_env` are valid.
+    """
+    package_name = f"gym_{env_name}"
+    importlib.import_module(package_name)
+    gym_handle = f"{package_name}/{task_name}"
+    assert gym_handle in gym.envs.registry, gym_handle
+
+
+def test_available_policies():
+    """
+    This test verifies that the class attribute `name` for all policies is
+    consistent with those listed in `lerobot/__init__.py`.
+    """
+    policy_classes = [ACTPolicy, DiffusionPolicy, TDMPCPolicy, VQBeTPolicy]
+    policies = [pol_cls.name for pol_cls in policy_classes]
+    assert set(policies) == set(lerobot.available_policies), policies
+
+
+def test_print():
+    print(lerobot.available_envs)
+    print(lerobot.available_tasks_per_env)
+    print(lerobot.available_datasets)
+    print(lerobot.available_datasets_per_env)
+    print(lerobot.available_real_world_datasets)
+    print(lerobot.available_policies)
+    print(lerobot.available_policies_per_env)
diff --git a/lerobot/tests/test_cli_peft.py b/lerobot/tests/test_cli_peft.py
new file mode 100644
index 0000000000000000000000000000000000000000..42fef47416dd3d461bd01e60989d3ae144bc4b47
--- /dev/null
+++ b/lerobot/tests/test_cli_peft.py
@@ -0,0 +1,235 @@
+import importlib
+import os
+from unittest.mock import MagicMock, patch
+
+import pytest
+from safetensors.torch import load_file
+
+from .utils import require_package
+
+# Skip this entire module in CI
+pytestmark = pytest.mark.skipif(
+    os.environ.get("CI") == "true" or os.environ.get("GITHUB_ACTIONS") == "true",
+    reason="This test requires peft and is very slow, not meant for CI",
+)
+
+
+def run_command(cmd, module, args):
+    module = importlib.import_module(f"lerobot.scripts.{module}")
+    with patch("sys.argv", [cmd] + args):
+        module.main()
+
+
+def lerobot_train(args):
+    return run_command(cmd="lerobot-train", module="lerobot_train", args=args)
+
+
+def lerobot_record(args):
+    return run_command(cmd="lerobot-record", module="lerobot_record", args=args)
+
+
+def resolve_model_id_for_peft_training(policy_type):
+    """PEFT training needs pretrained models, this finds the pretrained model of a policy type for PEFT training."""
+    if policy_type == "smolvla":
+        return "lerobot/smolvla_base"
+
+    raise ValueError(f"No pretrained model known for {policy_type}. PEFT training will not work.")
+
+
+@pytest.mark.parametrize("policy_type", ["smolvla"])
+@require_package("peft")
+def test_peft_training_push_to_hub_works(policy_type, tmp_path):
+    """Ensure that push to hub stores PEFT only the adapter, not the full model weights."""
+    output_dir = tmp_path / f"output_{policy_type}"
+    upload_folder_contents = set()
+
+    model_id = resolve_model_id_for_peft_training(policy_type)
+
+    def mock_upload_folder(*args, **kwargs):
+        folder_path = kwargs["folder_path"]
+        # we include more than is actually uploaded since we ignore {allow,ignore}_patterns of upload_folders()
+        upload_folder_contents.update(os.listdir(folder_path))
+        return MagicMock()
+
+    with (
+        patch("huggingface_hub.HfApi.create_repo"),
+        patch("huggingface_hub.HfApi.upload_folder", mock_upload_folder),
+    ):
+        lerobot_train(
+            [
+                f"--policy.path={model_id}",
+                "--policy.push_to_hub=true",
+                "--policy.repo_id=foo/bar",
+                "--policy.input_features=null",
+                "--policy.output_features=null",
+                "--peft.method=LORA",
+                "--dataset.repo_id=lerobot/pusht",
+                "--dataset.episodes=[0, 1]",
+                "--steps=1",
+                f"--output_dir={output_dir}",
+            ]
+        )
+
+        assert "adapter_model.safetensors" in upload_folder_contents
+        assert "config.json" in upload_folder_contents
+        assert "adapter_config.json" in upload_folder_contents
+
+
+@pytest.mark.parametrize("policy_type", ["smolvla"])
+@require_package("peft")
+def test_peft_training_works(policy_type, tmp_path):
+    """Check whether the standard case of fine-tuning a (partially) pre-trained policy with PEFT works."""
+    output_dir = tmp_path / f"output_{policy_type}"
+    model_id = resolve_model_id_for_peft_training(policy_type)
+
+    lerobot_train(
+        [
+            f"--policy.path={model_id}",
+            "--policy.push_to_hub=false",
+            "--policy.input_features=null",
+            "--policy.output_features=null",
+            "--peft.method=LORA",
+            "--dataset.repo_id=lerobot/pusht",
+            "--dataset.episodes=[0, 1]",
+            "--steps=1",
+            f"--output_dir={output_dir}",
+        ]
+    )
+
+    policy_dir = output_dir / "checkpoints" / "last" / "pretrained_model"
+
+    for file in ["adapter_config.json", "adapter_model.safetensors", "config.json"]:
+        assert (policy_dir / file).exists()
+
+    # This is the default case where we train a pre-trained policy from scratch with new data.
+    # We assume that we target policy-specific modules but fully fine-tune action and state projections
+    # so these must be part of the trained state dict.
+    state_dict = load_file(policy_dir / "adapter_model.safetensors")
+
+    adapted_keys = [
+        "state_proj",
+        "action_in_proj",
+        "action_out_proj",
+        "action_time_mlp_in",
+        "action_time_mlp_out",
+    ]
+
+    found_keys = [
+        module_key
+        for module_key in adapted_keys
+        for state_dict_key in state_dict
+        if f".{module_key}." in state_dict_key
+    ]
+
+    assert set(found_keys) == set(adapted_keys)
+
+
+@pytest.mark.parametrize("policy_type", ["smolvla"])
+@require_package("peft")
+def test_peft_training_params_are_fewer(policy_type, tmp_path):
+    """Check whether the standard case of fine-tuning a (partially) pre-trained policy with PEFT works."""
+    output_dir = tmp_path / f"output_{policy_type}"
+    model_id = resolve_model_id_for_peft_training(policy_type)
+
+    def dummy_update_policy(
+        train_metrics, policy, batch, optimizer, grad_clip_norm: float, accelerator, **kwargs
+    ):
+        params_total = sum(p.numel() for p in policy.parameters())
+        params_trainable = sum(p.numel() for p in policy.parameters() if p.requires_grad)
+
+        assert params_total > params_trainable
+
+        return train_metrics, {}
+
+    with patch("lerobot.scripts.lerobot_train.update_policy", dummy_update_policy):
+        lerobot_train(
+            [
+                f"--policy.path={model_id}",
+                "--policy.push_to_hub=false",
+                "--policy.input_features=null",
+                "--policy.output_features=null",
+                "--peft.method=LORA",
+                "--dataset.repo_id=lerobot/pusht",
+                "--dataset.episodes=[0, 1]",
+                "--steps=1",
+                f"--output_dir={output_dir}",
+            ]
+        )
+
+
+class DummyRobot:
+    name = "dummy"
+    cameras = []
+    action_features = {"foo": 1.0, "bar": 2.0}
+    observation_features = {"obs1": 1.0, "obs2": 2.0}
+    is_connected = True
+
+    def connect(self, *args):
+        pass
+
+    def disconnect(self):
+        pass
+
+
+def dummy_make_robot_from_config(*args, **kwargs):
+    return DummyRobot()
+
+
+@pytest.mark.parametrize("policy_type", ["smolvla"])
+@require_package("peft")
+def test_peft_record_loads_policy(policy_type, tmp_path):
+    """Train a policy with PEFT and attempt to load it with `lerobot-record`."""
+    from peft import PeftModel
+
+    output_dir = tmp_path / f"output_{policy_type}"
+    model_id = resolve_model_id_for_peft_training(policy_type)
+
+    lerobot_train(
+        [
+            f"--policy.path={model_id}",
+            "--policy.push_to_hub=false",
+            "--policy.input_features=null",
+            "--policy.output_features=null",
+            "--peft.method=LORA",
+            "--dataset.repo_id=lerobot/pusht",
+            "--dataset.episodes=[0, 1]",
+            "--steps=1",
+            f"--output_dir={output_dir}",
+        ]
+    )
+
+    policy_dir = output_dir / "checkpoints" / "last" / "pretrained_model"
+    dataset_dir = tmp_path / "eval_pusht"
+    single_task = "move the table"
+    loaded_policy = None
+
+    def dummy_record_loop(*args, **kwargs):
+        nonlocal loaded_policy
+
+        if "dataset" not in kwargs:
+            return
+
+        dataset = kwargs["dataset"]
+        dataset.add_frame({"task": single_task})
+        loaded_policy = kwargs["policy"]
+
+    with (
+        patch("lerobot.scripts.lerobot_record.make_robot_from_config", dummy_make_robot_from_config),
+        # disable record loop since we're only interested in successful loading of the policy.
+        patch("lerobot.scripts.lerobot_record.record_loop", dummy_record_loop),
+        # disable speech output
+        patch("lerobot.utils.utils.say"),
+    ):
+        lerobot_record(
+            [
+                f"--policy.path={policy_dir}",
+                "--robot.type=so101_follower",
+                "--robot.port=/dev/null",
+                "--dataset.repo_id=lerobot/eval_pusht",
+                f'--dataset.single_task="{single_task}"',
+                f"--dataset.root={dataset_dir}",
+                "--dataset.push_to_hub=false",
+            ]
+        )
+
+        assert isinstance(loaded_policy, PeftModel)
diff --git a/lerobot/tests/test_control_robot.py b/lerobot/tests/test_control_robot.py
new file mode 100644
index 0000000000000000000000000000000000000000..7725884677575f8e84b7cbf3aa8975ef3209ea0b
--- /dev/null
+++ b/lerobot/tests/test_control_robot.py
@@ -0,0 +1,123 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from unittest.mock import patch
+
+from lerobot.scripts.lerobot_calibrate import CalibrateConfig, calibrate
+from lerobot.scripts.lerobot_record import DatasetRecordConfig, RecordConfig, record
+from lerobot.scripts.lerobot_replay import DatasetReplayConfig, ReplayConfig, replay
+from lerobot.scripts.lerobot_teleoperate import TeleoperateConfig, teleoperate
+from tests.fixtures.constants import DUMMY_REPO_ID
+from tests.mocks.mock_robot import MockRobotConfig
+from tests.mocks.mock_teleop import MockTeleopConfig
+
+
+def test_calibrate():
+    robot_cfg = MockRobotConfig()
+    cfg = CalibrateConfig(robot=robot_cfg)
+    calibrate(cfg)
+
+
+def test_teleoperate():
+    robot_cfg = MockRobotConfig()
+    teleop_cfg = MockTeleopConfig()
+    cfg = TeleoperateConfig(
+        robot=robot_cfg,
+        teleop=teleop_cfg,
+        teleop_time_s=0.1,
+    )
+    teleoperate(cfg)
+
+
+def test_record_and_resume(tmp_path):
+    robot_cfg = MockRobotConfig()
+    teleop_cfg = MockTeleopConfig()
+    dataset_cfg = DatasetRecordConfig(
+        repo_id=DUMMY_REPO_ID,
+        single_task="Dummy task",
+        root=tmp_path / "record",
+        num_episodes=1,
+        episode_time_s=0.1,
+        reset_time_s=0,
+        push_to_hub=False,
+    )
+    cfg = RecordConfig(
+        robot=robot_cfg,
+        dataset=dataset_cfg,
+        teleop=teleop_cfg,
+        play_sounds=False,
+    )
+
+    dataset = record(cfg)
+
+    assert dataset.fps == 30
+    assert dataset.meta.total_episodes == dataset.num_episodes == 1
+    assert dataset.meta.total_frames == dataset.num_frames == 3
+    assert dataset.meta.total_tasks == 1
+
+    cfg.resume = True
+    # Mock the revision to prevent Hub calls during resume
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "record")
+        dataset = record(cfg)
+
+    assert dataset.meta.total_episodes == dataset.num_episodes == 2
+    assert dataset.meta.total_frames == dataset.num_frames == 6
+    assert dataset.meta.total_tasks == 1
+
+
+def test_record_and_replay(tmp_path):
+    robot_cfg = MockRobotConfig()
+    teleop_cfg = MockTeleopConfig()
+    record_dataset_cfg = DatasetRecordConfig(
+        repo_id=DUMMY_REPO_ID,
+        single_task="Dummy task",
+        root=tmp_path / "record_and_replay",
+        num_episodes=1,
+        episode_time_s=0.1,
+        push_to_hub=False,
+    )
+    record_cfg = RecordConfig(
+        robot=robot_cfg,
+        dataset=record_dataset_cfg,
+        teleop=teleop_cfg,
+        play_sounds=False,
+    )
+    replay_dataset_cfg = DatasetReplayConfig(
+        repo_id=DUMMY_REPO_ID,
+        episode=0,
+        root=tmp_path / "record_and_replay",
+    )
+    replay_cfg = ReplayConfig(
+        robot=robot_cfg,
+        dataset=replay_dataset_cfg,
+        play_sounds=False,
+    )
+
+    record(record_cfg)
+
+    # Mock the revision to prevent Hub calls during replay
+    with (
+        patch("lerobot.datasets.dataset_metadata.get_safe_version") as mock_get_safe_version,
+        patch("lerobot.datasets.dataset_metadata.snapshot_download") as mock_snapshot_download,
+    ):
+        mock_get_safe_version.return_value = "v3.0"
+        mock_snapshot_download.return_value = str(tmp_path / "record_and_replay")
+        replay(replay_cfg)
diff --git a/lerobot/tests/training/test_multi_gpu.py b/lerobot/tests/training/test_multi_gpu.py
new file mode 100644
index 0000000000000000000000000000000000000000..bb234e2e7d22807ad5615a79abfd309a2b7be8df
--- /dev/null
+++ b/lerobot/tests/training/test_multi_gpu.py
@@ -0,0 +1,211 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Multi-GPU Training Tests
+
+This module tests multi-GPU training functionality with accelerate.
+These tests are designed to run on machines with 2+ GPUs and are executed
+in the nightly CI workflow.
+
+The tests automatically generate accelerate configs and launch training
+with subprocess to properly test the distributed training environment.
+"""
+
+import os
+import subprocess
+import tempfile
+from pathlib import Path
+
+import pytest
+import torch
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+
+
+def get_num_available_gpus():
+    """Returns the number of available GPUs."""
+    if not torch.cuda.is_available():
+        return 0
+    return torch.cuda.device_count()
+
+
+def download_dataset(repo_id, episodes):
+    """
+    Pre-download dataset to avoid race conditions in multi-GPU training.
+
+    Args:
+        repo_id: HuggingFace dataset repository ID
+        episodes: List of episode indices to download
+    """
+    # Simply instantiating the dataset will download it
+    _ = LeRobotDataset(repo_id, episodes=episodes)
+    print(f"Dataset {repo_id} downloaded successfully")
+
+
+def run_accelerate_training(config_args, num_processes=4, temp_dir=None):
+    """
+    Helper function to run training with accelerate launch.
+
+    Args:
+        config_args: List of config arguments to pass to lerobot_train.py
+        num_processes: Number of processes (GPUs) to use
+        temp_dir: Temporary directory for outputs
+
+    Returns:
+        subprocess.CompletedProcess result
+    """
+
+    config_path = Path(temp_dir) / "accelerate_config.yaml"
+
+    # Write YAML config
+    with open(config_path, "w") as f:
+        f.write("compute_environment: LOCAL_MACHINE\n")
+        f.write("distributed_type: MULTI_GPU\n")
+        f.write("mixed_precision: 'no'\n")
+        f.write(f"num_processes: {num_processes}\n")
+        f.write("use_cpu: false\n")
+        f.write("gpu_ids: all\n")
+        f.write("downcast_bf16: 'no'\n")
+        f.write("machine_rank: 0\n")
+        f.write("main_training_function: main\n")
+        f.write("num_machines: 1\n")
+        f.write("rdzv_backend: static\n")
+        f.write("same_network: true\n")
+
+    cmd = [
+        "accelerate",
+        "launch",
+        "--config_file",
+        str(config_path),
+        "-m",
+        "lerobot.scripts.lerobot_train",
+    ] + config_args
+
+    result = subprocess.run(
+        cmd,
+        capture_output=True,
+        text=True,
+        env={**os.environ, "CUDA_VISIBLE_DEVICES": ",".join(map(str, range(num_processes)))},
+    )
+
+    return result
+
+
+@pytest.mark.skipif(
+    get_num_available_gpus() < 2,
+    reason="Multi-GPU tests require at least 2 GPUs",
+)
+class TestMultiGPUTraining:
+    """Test suite for multi-GPU training functionality."""
+
+    def test_basic_multi_gpu_training(self):
+        """
+        Test that basic multi-GPU training runs successfully.
+        Verifies that the training completes without errors.
+        """
+        # Pre-download dataset to avoid race conditions
+        download_dataset("lerobot/pusht", episodes=[0])
+
+        with tempfile.TemporaryDirectory() as temp_dir:
+            output_dir = Path(temp_dir) / "outputs"
+
+            config_args = [
+                "--dataset.repo_id=lerobot/pusht",
+                "--dataset.episodes=[0]",
+                "--policy.type=act",
+                "--policy.device=cuda",
+                "--policy.push_to_hub=false",
+                f"--output_dir={output_dir}",
+                "--batch_size=4",
+                "--steps=10",
+                "--eval_freq=-1",
+                "--log_freq=5",
+                "--save_freq=10",
+                "--seed=42",
+                "--num_workers=0",
+            ]
+
+            result = run_accelerate_training(config_args, num_processes=4, temp_dir=temp_dir)
+
+            # Check that training completed successfully
+            assert result.returncode == 0, (
+                f"Multi-GPU training failed with return code {result.returncode}\n"
+                f"STDOUT:\n{result.stdout}\n"
+                f"STDERR:\n{result.stderr}"
+            )
+
+            # Verify checkpoint was saved
+            checkpoints_dir = output_dir / "checkpoints"
+            assert checkpoints_dir.exists(), "Checkpoints directory was not created"
+
+            # Verify that training completed
+            assert "End of training" in result.stdout or "End of training" in result.stderr
+
+    def test_checkpoint_saving_multi_gpu(self):
+        """
+        Test that checkpoints are correctly saved during multi-GPU training.
+        Only the main process (rank 0) should save checkpoints.
+        """
+        # Pre-download dataset to avoid race conditions
+        download_dataset("lerobot/pusht", episodes=[0])
+
+        with tempfile.TemporaryDirectory() as temp_dir:
+            output_dir = Path(temp_dir) / "outputs"
+
+            config_args = [
+                "--dataset.repo_id=lerobot/pusht",
+                "--dataset.episodes=[0]",
+                "--policy.type=act",
+                "--policy.device=cuda",
+                "--policy.push_to_hub=false",
+                f"--output_dir={output_dir}",
+                "--batch_size=4",
+                "--steps=20",
+                "--eval_freq=-1",
+                "--log_freq=5",
+                "--save_freq=10",
+                "--seed=42",
+                "--num_workers=0",
+            ]
+
+            result = run_accelerate_training(config_args, num_processes=2, temp_dir=temp_dir)
+
+            assert result.returncode == 0, (
+                f"Training failed:\nSTDOUT:\n{result.stdout}\n\nSTDERR:\n{result.stderr}"
+            )
+
+            # Verify checkpoint directory exists
+            checkpoints_dir = output_dir / "checkpoints"
+            assert checkpoints_dir.exists(), "Checkpoints directory not created"
+
+            # Count checkpoint directories (should have checkpoint at step 10 and 20)
+            checkpoint_dirs = [d for d in checkpoints_dir.iterdir() if d.is_dir()]
+            assert len(checkpoint_dirs) >= 1, f"Expected at least 1 checkpoint, found {len(checkpoint_dirs)}"
+
+            # Verify checkpoint contents
+            for checkpoint_dir in checkpoint_dirs:
+                # Check for model files
+                model_files = list(checkpoint_dir.rglob("*.safetensors"))
+                assert len(model_files) > 0, f"No model files in checkpoint {checkpoint_dir}"
+
+                # Check for training state
+                training_state_dir = checkpoint_dir / "training_state"
+                assert training_state_dir.exists(), f"No training state in checkpoint {checkpoint_dir}"
+
+                # Verify optimizer state exists
+                optimizer_state = training_state_dir / "optimizer_state.safetensors"
+                assert optimizer_state.exists(), f"No optimizer state in checkpoint {checkpoint_dir}"
diff --git a/lerobot/tests/training/test_visual_validation.py b/lerobot/tests/training/test_visual_validation.py
new file mode 100644
index 0000000000000000000000000000000000000000..89351e3c292356e670ce0b8907af67e75f95fd98
--- /dev/null
+++ b/lerobot/tests/training/test_visual_validation.py
@@ -0,0 +1,157 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Visual Feature Consistency Tests
+
+This module tests the `validate_visual_features_consistency` function,
+which ensures that visual features (camera observations) in a dataset/env
+match the expectations defined in a policy configuration.
+
+The purpose of this check is to prevent mismatches between what a policy expects
+(e.g., `observation.images.camera1`, `camera2`, `camera3`) and what a dataset or
+environment actually provides (e.g., `observation.images.top`, `side`, or fewer cameras).
+"""
+
+from pathlib import Path
+
+import numpy as np
+import pytest
+
+from lerobot.configs.default import DatasetConfig
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.policies.factory import make_policy_config
+from lerobot.scripts.lerobot_train import train
+from lerobot.utils.device_utils import auto_select_torch_device
+
+pytest.importorskip("transformers")
+
+DUMMY_REPO_ID = "dummy/repo"
+
+
+@pytest.fixture
+def temp_dir(tmp_path):
+    return tmp_path
+
+
+DUMMY_STATE_DIM = 6
+DUMMY_ACTION_DIM = 6
+IMAGE_SIZE = 8
+DEVICE = auto_select_torch_device()
+
+
+def make_dummy_dataset(camera_keys, tmp_path):
+    """Creates a minimal dummy dataset for testing rename_mapping logic."""
+    features = {
+        "action": {"dtype": "float32", "shape": (DUMMY_ACTION_DIM,), "names": None},
+        "observation.state": {"dtype": "float32", "shape": (DUMMY_STATE_DIM,), "names": None},
+    }
+    for cam in camera_keys:
+        features[f"observation.images.{cam}"] = {
+            "dtype": "image",
+            "shape": (IMAGE_SIZE, IMAGE_SIZE, 3),
+            "names": ["height", "width", "channel"],
+        }
+    dataset = LeRobotDataset.create(
+        repo_id=DUMMY_REPO_ID,
+        fps=30,
+        features=features,
+        root=tmp_path / "_dataset",
+    )
+    root = tmp_path / "_dataset"
+    for ep_idx in range(2):
+        for _ in range(3):
+            frame = {
+                "action": np.random.randn(DUMMY_ACTION_DIM).astype(np.float32),
+                "observation.state": np.random.randn(DUMMY_STATE_DIM).astype(np.float32),
+            }
+            for cam in camera_keys:
+                frame[f"observation.images.{cam}"] = np.random.randint(
+                    0, 255, size=(IMAGE_SIZE, IMAGE_SIZE, 3), dtype=np.uint8
+                )
+            frame["task"] = f"task_{ep_idx}"
+            dataset.add_frame(frame)
+        dataset.save_episode()
+
+    dataset.finalize()
+    return dataset, root
+
+
+def custom_validate(train_config: TrainPipelineConfig, policy_path: str, empty_cameras: int):
+    train_config.policy = PreTrainedConfig.from_pretrained(policy_path)
+    train_config.policy.pretrained_path = Path(policy_path)
+    # override empty_cameras and push_to_hub for testing
+    train_config.policy.empty_cameras = empty_cameras
+    train_config.policy.push_to_hub = False
+    if train_config.use_policy_training_preset:
+        train_config.optimizer = train_config.policy.get_optimizer_preset()
+        train_config.scheduler = train_config.policy.get_scheduler_preset()
+    return train_config
+
+
+@pytest.mark.skip(reason="Skipping this test as it results OOM")
+@pytest.mark.parametrize(
+    "camera_keys, empty_cameras, rename_map, expect_success",
+    [
+        # case 1: dataset has fewer cameras than policy (3 instead of 4), but we specify empty_cameras=1 for smolvla, pi0, pi05
+        (["camera1", "camera2", "camera3"], 1, {}, True),
+        # case 2: dataset has 2 cameras with different names, rename_mapping provided
+        (
+            ["top", "side"],
+            0,
+            {
+                "observation.images.top": "observation.images.camera1",
+                "observation.images.side": "observation.images.camera2",
+            },
+            True,
+        ),
+        # case 3: dataset has 2 cameras, policy expects 3, names do not match, no empty_cameras
+        (["top", "side"], 0, {}, False),
+        # TODO: case 4: dataset has 2 cameras, policy expects 3, no rename_map, no empty_cameras, should raise for smolvla
+        # (["camera1", "camera2"], 0, {}, False),
+    ],
+)
+def test_train_with_camera_mismatch(camera_keys, empty_cameras, rename_map, expect_success, tmp_path):
+    """Tests that training works or fails depending on camera/feature alignment."""
+
+    _dataset, root = make_dummy_dataset(camera_keys, tmp_path)
+    pretrained_path = "lerobot/smolvla_base"
+    dataset_config = DatasetConfig(repo_id=DUMMY_REPO_ID, root=root)
+    policy_config = make_policy_config(
+        "smolvla",
+        optimizer_lr=0.01,
+        push_to_hub=False,
+        pretrained_path=pretrained_path,
+        device=DEVICE,
+    )
+    policy_config.empty_cameras = empty_cameras
+    train_config = TrainPipelineConfig(
+        dataset=dataset_config,
+        policy=policy_config,
+        rename_map=rename_map,
+        output_dir=tmp_path / "_output",
+        steps=1,
+    )
+    train_config = custom_validate(train_config, policy_path=pretrained_path, empty_cameras=empty_cameras)
+    # HACK: disable the internal CLI validation step for tests, we did it with custom_validate
+    train_config.validate = lambda: None
+    if expect_success:
+        train(train_config)
+    else:
+        with pytest.raises(ValueError):
+            train(train_config)
diff --git a/lerobot/tests/transport/test_transport_utils.py b/lerobot/tests/transport/test_transport_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..63632a8f417ff4fe82d417978f7c8a558575fc12
--- /dev/null
+++ b/lerobot/tests/transport/test_transport_utils.py
@@ -0,0 +1,572 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import io
+from multiprocessing import Event, Queue
+from pickle import UnpicklingError
+
+import pytest
+import torch
+
+from lerobot.utils.constants import ACTION
+from lerobot.utils.transition import Transition
+from tests.utils import require_cuda, require_package
+
+
+@require_package("grpcio", "grpc")
+def test_bytes_buffer_size_empty_buffer():
+    from lerobot.transport.utils import bytes_buffer_size
+
+    """Test with an empty buffer."""
+    buffer = io.BytesIO()
+    assert bytes_buffer_size(buffer) == 0
+    # Ensure position is reset to beginning
+    assert buffer.tell() == 0
+
+
+@require_package("grpcio", "grpc")
+def test_bytes_buffer_size_small_buffer():
+    from lerobot.transport.utils import bytes_buffer_size
+
+    """Test with a small buffer."""
+    buffer = io.BytesIO(b"Hello, World!")
+    assert bytes_buffer_size(buffer) == 13
+    assert buffer.tell() == 0
+
+
+@require_package("grpcio", "grpc")
+def test_bytes_buffer_size_large_buffer():
+    from lerobot.transport.utils import CHUNK_SIZE, bytes_buffer_size
+
+    """Test with a large buffer."""
+    data = b"x" * (CHUNK_SIZE * 2 + 1000)
+    buffer = io.BytesIO(data)
+    assert bytes_buffer_size(buffer) == len(data)
+    assert buffer.tell() == 0
+
+
+@require_package("grpcio", "grpc")
+def test_send_bytes_in_chunks_empty_data():
+    from lerobot.transport.utils import send_bytes_in_chunks, services_pb2
+
+    """Test sending empty data."""
+    message_class = services_pb2.InteractionMessage
+    chunks = list(send_bytes_in_chunks(b"", message_class))
+    assert len(chunks) == 0
+
+
+@require_package("grpcio", "grpc")
+def test_single_chunk_small_data():
+    from lerobot.transport.utils import send_bytes_in_chunks, services_pb2
+
+    """Test data that fits in a single chunk."""
+    data = b"Some data"
+    message_class = services_pb2.InteractionMessage
+    chunks = list(send_bytes_in_chunks(data, message_class))
+
+    assert len(chunks) == 1
+    assert chunks[0].data == b"Some data"
+    assert chunks[0].transfer_state == services_pb2.TransferState.TRANSFER_END
+
+
+@require_package("grpcio", "grpc")
+def test_not_silent_mode():
+    from lerobot.transport.utils import send_bytes_in_chunks, services_pb2
+
+    """Test not silent mode."""
+    data = b"Some data"
+    message_class = services_pb2.InteractionMessage
+    chunks = list(send_bytes_in_chunks(data, message_class, silent=False))
+    assert len(chunks) == 1
+    assert chunks[0].data == b"Some data"
+
+
+@require_package("grpcio", "grpc")
+def test_send_bytes_in_chunks_large_data():
+    from lerobot.transport.utils import CHUNK_SIZE, send_bytes_in_chunks, services_pb2
+
+    """Test sending large data."""
+    data = b"x" * (CHUNK_SIZE * 2 + 1000)
+    message_class = services_pb2.InteractionMessage
+    chunks = list(send_bytes_in_chunks(data, message_class))
+    assert len(chunks) == 3
+    assert chunks[0].data == b"x" * CHUNK_SIZE
+    assert chunks[0].transfer_state == services_pb2.TransferState.TRANSFER_BEGIN
+    assert chunks[1].data == b"x" * CHUNK_SIZE
+    assert chunks[1].transfer_state == services_pb2.TransferState.TRANSFER_MIDDLE
+    assert chunks[2].data == b"x" * 1000
+    assert chunks[2].transfer_state == services_pb2.TransferState.TRANSFER_END
+
+
+@require_package("grpcio", "grpc")
+def test_send_bytes_in_chunks_large_data_with_exact_chunk_size():
+    from lerobot.transport.utils import CHUNK_SIZE, send_bytes_in_chunks, services_pb2
+
+    """Test sending large data with exact chunk size."""
+    data = b"x" * CHUNK_SIZE
+    message_class = services_pb2.InteractionMessage
+    chunks = list(send_bytes_in_chunks(data, message_class))
+    assert len(chunks) == 1
+    assert chunks[0].data == data
+    assert chunks[0].transfer_state == services_pb2.TransferState.TRANSFER_END
+
+
+@require_package("grpcio", "grpc")
+def test_receive_bytes_in_chunks_empty_data():
+    from lerobot.transport.utils import receive_bytes_in_chunks
+
+    """Test receiving empty data."""
+    queue = Queue()
+    shutdown_event = Event()
+
+    # Empty iterator
+    receive_bytes_in_chunks(iter([]), queue, shutdown_event)
+
+    assert queue.empty()
+
+
+@require_package("grpcio", "grpc")
+def test_receive_bytes_in_chunks_single_chunk():
+    from lerobot.transport.utils import receive_bytes_in_chunks, services_pb2
+
+    """Test receiving a single chunk message."""
+    queue = Queue()
+    shutdown_event = Event()
+
+    data = b"Single chunk data"
+    chunks = [
+        services_pb2.InteractionMessage(data=data, transfer_state=services_pb2.TransferState.TRANSFER_END)
+    ]
+
+    receive_bytes_in_chunks(iter(chunks), queue, shutdown_event)
+
+    assert queue.get(timeout=0.01) == data
+    assert queue.empty()
+
+
+@require_package("grpcio", "grpc")
+def test_receive_bytes_in_chunks_single_not_end_chunk():
+    from lerobot.transport.utils import receive_bytes_in_chunks, services_pb2
+
+    """Test receiving a single chunk message."""
+    queue = Queue()
+    shutdown_event = Event()
+
+    data = b"Single chunk data"
+    chunks = [
+        services_pb2.InteractionMessage(data=data, transfer_state=services_pb2.TransferState.TRANSFER_MIDDLE)
+    ]
+
+    receive_bytes_in_chunks(iter(chunks), queue, shutdown_event)
+
+    assert queue.empty()
+
+
+@require_package("grpcio", "grpc")
+def test_receive_bytes_in_chunks_multiple_chunks():
+    from lerobot.transport.utils import receive_bytes_in_chunks, services_pb2
+
+    """Test receiving a multi-chunk message."""
+    queue = Queue()
+    shutdown_event = Event()
+
+    chunks = [
+        services_pb2.InteractionMessage(
+            data=b"First ", transfer_state=services_pb2.TransferState.TRANSFER_BEGIN
+        ),
+        services_pb2.InteractionMessage(
+            data=b"Middle ", transfer_state=services_pb2.TransferState.TRANSFER_MIDDLE
+        ),
+        services_pb2.InteractionMessage(data=b"Last", transfer_state=services_pb2.TransferState.TRANSFER_END),
+    ]
+
+    receive_bytes_in_chunks(iter(chunks), queue, shutdown_event)
+
+    assert queue.get(timeout=0.01) == b"First Middle Last"
+    assert queue.empty()
+
+
+@require_package("grpcio", "grpc")
+def test_receive_bytes_in_chunks_multiple_messages():
+    from lerobot.transport.utils import receive_bytes_in_chunks, services_pb2
+
+    """Test receiving multiple complete messages in sequence."""
+    queue = Queue()
+    shutdown_event = Event()
+
+    chunks = [
+        # First message - single chunk
+        services_pb2.InteractionMessage(
+            data=b"Message1", transfer_state=services_pb2.TransferState.TRANSFER_END
+        ),
+        # Second message - multi chunk
+        services_pb2.InteractionMessage(
+            data=b"Start2 ", transfer_state=services_pb2.TransferState.TRANSFER_BEGIN
+        ),
+        services_pb2.InteractionMessage(
+            data=b"Middle2 ", transfer_state=services_pb2.TransferState.TRANSFER_MIDDLE
+        ),
+        services_pb2.InteractionMessage(data=b"End2", transfer_state=services_pb2.TransferState.TRANSFER_END),
+        # Third message - single chunk
+        services_pb2.InteractionMessage(
+            data=b"Message3", transfer_state=services_pb2.TransferState.TRANSFER_END
+        ),
+    ]
+
+    receive_bytes_in_chunks(iter(chunks), queue, shutdown_event)
+
+    # Should have three messages in queue
+    assert queue.get(timeout=0.01) == b"Message1"
+    assert queue.get(timeout=0.01) == b"Start2 Middle2 End2"
+    assert queue.get(timeout=0.01) == b"Message3"
+    assert queue.empty()
+
+
+@require_package("grpcio", "grpc")
+def test_receive_bytes_in_chunks_shutdown_during_receive():
+    from lerobot.transport.utils import receive_bytes_in_chunks, services_pb2
+
+    """Test that shutdown event stops receiving mid-stream."""
+    queue = Queue()
+    shutdown_event = Event()
+    shutdown_event.set()
+
+    chunks = [
+        services_pb2.InteractionMessage(
+            data=b"First ", transfer_state=services_pb2.TransferState.TRANSFER_BEGIN
+        ),
+        services_pb2.InteractionMessage(
+            data=b"Middle ", transfer_state=services_pb2.TransferState.TRANSFER_MIDDLE
+        ),
+        services_pb2.InteractionMessage(data=b"Last", transfer_state=services_pb2.TransferState.TRANSFER_END),
+    ]
+
+    receive_bytes_in_chunks(iter(chunks), queue, shutdown_event)
+
+    assert queue.empty()
+
+
+@require_package("grpcio", "grpc")
+def test_receive_bytes_in_chunks_only_begin_chunk():
+    from lerobot.transport.utils import receive_bytes_in_chunks, services_pb2
+
+    """Test receiving only a BEGIN chunk without END."""
+    queue = Queue()
+    shutdown_event = Event()
+
+    chunks = [
+        services_pb2.InteractionMessage(
+            data=b"Start", transfer_state=services_pb2.TransferState.TRANSFER_BEGIN
+        ),
+        # No END chunk
+    ]
+
+    receive_bytes_in_chunks(iter(chunks), queue, shutdown_event)
+
+    assert queue.empty()
+
+
+@require_package("grpcio", "grpc")
+def test_receive_bytes_in_chunks_missing_begin():
+    from lerobot.transport.utils import receive_bytes_in_chunks, services_pb2
+
+    """Test receiving chunks starting with MIDDLE instead of BEGIN."""
+    queue = Queue()
+    shutdown_event = Event()
+
+    chunks = [
+        # Missing BEGIN
+        services_pb2.InteractionMessage(
+            data=b"Middle", transfer_state=services_pb2.TransferState.TRANSFER_MIDDLE
+        ),
+        services_pb2.InteractionMessage(data=b"End", transfer_state=services_pb2.TransferState.TRANSFER_END),
+    ]
+
+    receive_bytes_in_chunks(iter(chunks), queue, shutdown_event)
+
+    # The implementation continues from where it is, so we should get partial data
+    assert queue.get(timeout=0.01) == b"MiddleEnd"
+    assert queue.empty()
+
+
+# Tests for state_to_bytes and bytes_to_state_dict
+@require_package("grpcio", "grpc")
+def test_state_to_bytes_empty_dict():
+    from lerobot.transport.utils import bytes_to_state_dict, state_to_bytes
+
+    """Test converting empty state dict to bytes."""
+    state_dict = {}
+    data = state_to_bytes(state_dict)
+    reconstructed = bytes_to_state_dict(data)
+    assert reconstructed == state_dict
+
+
+@require_package("grpcio", "grpc")
+def test_bytes_to_state_dict_empty_data():
+    from lerobot.transport.utils import bytes_to_state_dict
+
+    """Test converting empty data to state dict."""
+    with pytest.raises(EOFError):
+        bytes_to_state_dict(b"")
+
+
+@require_package("grpcio", "grpc")
+def test_state_to_bytes_simple_dict():
+    from lerobot.transport.utils import bytes_to_state_dict, state_to_bytes
+
+    """Test converting simple state dict to bytes."""
+    state_dict = {
+        "layer1.weight": torch.randn(10, 5),
+        "layer1.bias": torch.randn(10),
+        "layer2.weight": torch.randn(1, 10),
+        "layer2.bias": torch.randn(1),
+    }
+
+    data = state_to_bytes(state_dict)
+    assert isinstance(data, bytes)
+    assert len(data) > 0
+
+    reconstructed = bytes_to_state_dict(data)
+
+    assert len(reconstructed) == len(state_dict)
+    for key in state_dict:
+        assert key in reconstructed
+        assert torch.allclose(state_dict[key], reconstructed[key])
+
+
+@require_package("grpcio", "grpc")
+def test_state_to_bytes_various_dtypes():
+    from lerobot.transport.utils import bytes_to_state_dict, state_to_bytes
+
+    """Test converting state dict with various tensor dtypes."""
+    state_dict = {
+        "float32": torch.randn(5, 5),
+        "float64": torch.randn(3, 3).double(),
+        "int32": torch.randint(0, 100, (4, 4), dtype=torch.int32),
+        "int64": torch.randint(0, 100, (2, 2), dtype=torch.int64),
+        "bool": torch.tensor([True, False, True]),
+        "uint8": torch.randint(0, 255, (3, 3), dtype=torch.uint8),
+    }
+
+    data = state_to_bytes(state_dict)
+    reconstructed = bytes_to_state_dict(data)
+
+    for key in state_dict:
+        assert reconstructed[key].dtype == state_dict[key].dtype
+        if state_dict[key].dtype == torch.bool:
+            assert torch.equal(state_dict[key], reconstructed[key])
+        else:
+            assert torch.allclose(state_dict[key], reconstructed[key])
+
+
+@require_package("grpcio", "grpc")
+def test_bytes_to_state_dict_invalid_data():
+    from lerobot.transport.utils import bytes_to_state_dict
+
+    """Test bytes_to_state_dict with invalid data."""
+    with pytest.raises(UnpicklingError):
+        bytes_to_state_dict(b"This is not a valid torch save file")
+
+
+@require_cuda
+@require_package("grpcio", "grpc")
+def test_state_to_bytes_various_dtypes_cuda():
+    from lerobot.transport.utils import bytes_to_state_dict, state_to_bytes
+
+    """Test converting state dict with various tensor dtypes."""
+    state_dict = {
+        "float32": torch.randn(5, 5).cuda(),
+        "float64": torch.randn(3, 3).double().cuda(),
+        "int32": torch.randint(0, 100, (4, 4), dtype=torch.int32).cuda(),
+        "int64": torch.randint(0, 100, (2, 2), dtype=torch.int64).cuda(),
+        "bool": torch.tensor([True, False, True]),
+        "uint8": torch.randint(0, 255, (3, 3), dtype=torch.uint8),
+    }
+
+    data = state_to_bytes(state_dict)
+    reconstructed = bytes_to_state_dict(data)
+
+    for key in state_dict:
+        assert reconstructed[key].dtype == state_dict[key].dtype
+        if state_dict[key].dtype == torch.bool:
+            assert torch.equal(state_dict[key], reconstructed[key])
+        else:
+            assert torch.allclose(state_dict[key], reconstructed[key])
+
+
+@require_package("grpcio", "grpc")
+def test_python_object_to_bytes_none():
+    from lerobot.transport.utils import bytes_to_python_object, python_object_to_bytes
+
+    """Test converting None to bytes."""
+    obj = None
+    data = python_object_to_bytes(obj)
+    reconstructed = bytes_to_python_object(data)
+    assert reconstructed is None
+
+
+@pytest.mark.parametrize(
+    "obj",
+    [
+        42,
+        -123,
+        3.14159,
+        -2.71828,
+        "Hello, World!",
+        "Unicode: 你好世界 🌍",
+        True,
+        False,
+        b"byte string",
+        [],
+        [1, 2, 3],
+        [1, "two", 3.0, True, None],
+        {},
+        {"key": "value", "number": 123, "nested": {"a": 1}},
+        (),
+        (1, 2, 3),
+    ],
+)
+@require_package("grpcio", "grpc")
+def test_python_object_to_bytes_simple_types(obj):
+    from lerobot.transport.utils import bytes_to_python_object, python_object_to_bytes
+
+    """Test converting simple Python types."""
+    data = python_object_to_bytes(obj)
+    reconstructed = bytes_to_python_object(data)
+    assert reconstructed == obj
+    assert type(reconstructed) is type(obj)
+
+
+@require_package("grpcio", "grpc")
+def test_python_object_to_bytes_with_tensors():
+    from lerobot.transport.utils import bytes_to_python_object, python_object_to_bytes
+
+    """Test converting objects containing PyTorch tensors."""
+    obj = {
+        "tensor": torch.randn(5, 5),
+        "list_with_tensor": [1, 2, torch.randn(3, 3), "string"],
+        "nested": {
+            "tensor1": torch.randn(2, 2),
+            "tensor2": torch.tensor([1, 2, 3]),
+        },
+    }
+
+    data = python_object_to_bytes(obj)
+    reconstructed = bytes_to_python_object(data)
+
+    assert torch.allclose(obj["tensor"], reconstructed["tensor"])
+    assert reconstructed["list_with_tensor"][0] == 1
+    assert reconstructed["list_with_tensor"][3] == "string"
+    assert torch.allclose(obj["list_with_tensor"][2], reconstructed["list_with_tensor"][2])
+    assert torch.allclose(obj["nested"]["tensor1"], reconstructed["nested"]["tensor1"])
+    assert torch.equal(obj["nested"]["tensor2"], reconstructed["nested"]["tensor2"])
+
+
+@require_package("grpcio", "grpc")
+def test_transitions_to_bytes_empty_list():
+    from lerobot.transport.utils import bytes_to_transitions, transitions_to_bytes
+
+    """Test converting empty transitions list."""
+    transitions = []
+    data = transitions_to_bytes(transitions)
+    reconstructed = bytes_to_transitions(data)
+    assert reconstructed == transitions
+    assert isinstance(reconstructed, list)
+
+
+@require_package("grpcio", "grpc")
+def test_transitions_to_bytes_single_transition():
+    from lerobot.transport.utils import bytes_to_transitions, transitions_to_bytes
+
+    """Test converting a single transition."""
+    transition = Transition(
+        state={"image": torch.randn(3, 64, 64), "state": torch.randn(10)},
+        action=torch.randn(5),
+        reward=torch.tensor(1.5),
+        done=torch.tensor(False),
+        next_state={"image": torch.randn(3, 64, 64), "state": torch.randn(10)},
+    )
+
+    transitions = [transition]
+    data = transitions_to_bytes(transitions)
+    reconstructed = bytes_to_transitions(data)
+
+    assert len(reconstructed) == 1
+
+    assert_transitions_equal(transitions[0], reconstructed[0])
+
+
+@require_package("grpcio", "grpc")
+def assert_transitions_equal(t1: Transition, t2: Transition):
+    """Helper to assert two transitions are equal."""
+    assert_observation_equal(t1["state"], t2["state"])
+    assert torch.allclose(t1[ACTION], t2[ACTION])
+    assert torch.allclose(t1["reward"], t2["reward"])
+    assert torch.equal(t1["done"], t2["done"])
+    assert_observation_equal(t1["next_state"], t2["next_state"])
+
+
+@require_package("grpcio", "grpc")
+def assert_observation_equal(o1: dict, o2: dict):
+    """Helper to assert two observations are equal."""
+    assert set(o1.keys()) == set(o2.keys())
+    for key in o1:
+        assert torch.allclose(o1[key], o2[key])
+
+
+@require_package("grpcio", "grpc")
+def test_transitions_to_bytes_multiple_transitions():
+    from lerobot.transport.utils import bytes_to_transitions, transitions_to_bytes
+
+    """Test converting multiple transitions."""
+    transitions = []
+    for i in range(5):
+        transition = Transition(
+            state={"data": torch.randn(10)},
+            action=torch.randn(3),
+            reward=torch.tensor(float(i)),
+            done=torch.tensor(i == 4),
+            next_state={"data": torch.randn(10)},
+        )
+        transitions.append(transition)
+
+    data = transitions_to_bytes(transitions)
+    reconstructed = bytes_to_transitions(data)
+
+    assert len(reconstructed) == len(transitions)
+    for original, reconstructed_item in zip(transitions, reconstructed, strict=False):
+        assert_transitions_equal(original, reconstructed_item)
+
+
+@require_package("grpcio", "grpc")
+def test_receive_bytes_in_chunks_unknown_state():
+    from lerobot.transport.utils import receive_bytes_in_chunks
+
+    """Test receive_bytes_in_chunks with an unknown transfer state."""
+
+    # Mock the gRPC message object, which has `transfer_state` and `data` attributes.
+    class MockMessage:
+        def __init__(self, transfer_state, data):
+            self.transfer_state = transfer_state
+            self.data = data
+
+    # 10 is not a valid TransferState enum value
+    bad_iterator = [MockMessage(transfer_state=10, data=b"bad_data")]
+    output_queue = Queue()
+    shutdown_event = Event()
+
+    with pytest.raises(ValueError, match="Received unknown transfer state"):
+        receive_bytes_in_chunks(bad_iterator, output_queue, shutdown_event)
diff --git a/lerobot/tests/utils.py b/lerobot/tests/utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..33c5548042af61ecf984897f4eba6c4d07e47d91
--- /dev/null
+++ b/lerobot/tests/utils.py
@@ -0,0 +1,201 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import os
+import platform
+from functools import wraps
+
+import pytest
+import torch
+
+from lerobot import available_cameras, available_motors, available_robots
+from lerobot.utils.device_utils import auto_select_torch_device
+from lerobot.utils.import_utils import is_package_available
+
+DEVICE = os.environ.get("LEROBOT_TEST_DEVICE", str(auto_select_torch_device()))
+
+TEST_ROBOT_TYPES = []
+for robot_type in available_robots:
+    TEST_ROBOT_TYPES += [(robot_type, True), (robot_type, False)]
+
+TEST_CAMERA_TYPES = []
+for camera_type in available_cameras:
+    TEST_CAMERA_TYPES += [(camera_type, True), (camera_type, False)]
+
+TEST_MOTOR_TYPES = []
+for motor_type in available_motors:
+    TEST_MOTOR_TYPES += [(motor_type, True), (motor_type, False)]
+
+# Camera indices used for connecting physical cameras
+OPENCV_CAMERA_INDEX = int(os.environ.get("LEROBOT_TEST_OPENCV_CAMERA_INDEX", 0))
+INTELREALSENSE_SERIAL_NUMBER = int(os.environ.get("LEROBOT_TEST_INTELREALSENSE_SERIAL_NUMBER", 128422271614))
+
+DYNAMIXEL_PORT = os.environ.get("LEROBOT_TEST_DYNAMIXEL_PORT", "/dev/tty.usbmodem575E0032081")
+DYNAMIXEL_MOTORS = {
+    "shoulder_pan": [1, "xl430-w250"],
+    "shoulder_lift": [2, "xl430-w250"],
+    "elbow_flex": [3, "xl330-m288"],
+    "wrist_flex": [4, "xl330-m288"],
+    "wrist_roll": [5, "xl330-m288"],
+    "gripper": [6, "xl330-m288"],
+}
+
+FEETECH_PORT = os.environ.get("LEROBOT_TEST_FEETECH_PORT", "/dev/tty.usbmodem585A0080971")
+FEETECH_MOTORS = {
+    "shoulder_pan": [1, "sts3215"],
+    "shoulder_lift": [2, "sts3215"],
+    "elbow_flex": [3, "sts3215"],
+    "wrist_flex": [4, "sts3215"],
+    "wrist_roll": [5, "sts3215"],
+    "gripper": [6, "sts3215"],
+}
+
+
+def require_x86_64_kernel(func):
+    """
+    Decorator that skips the test if plateform device is not an x86_64 cpu.
+    """
+    from functools import wraps
+
+    @wraps(func)
+    def wrapper(*args, **kwargs):
+        if platform.machine() != "x86_64":
+            pytest.skip("requires x86_64 plateform")
+        return func(*args, **kwargs)
+
+    return wrapper
+
+
+def require_cpu(func):
+    """
+    Decorator that skips the test if device is not cpu.
+    """
+    from functools import wraps
+
+    @wraps(func)
+    def wrapper(*args, **kwargs):
+        if DEVICE != "cpu":
+            pytest.skip("requires cpu")
+        return func(*args, **kwargs)
+
+    return wrapper
+
+
+def require_cuda(func):
+    """
+    Decorator that skips the test if cuda is not available.
+    """
+    from functools import wraps
+
+    @wraps(func)
+    def wrapper(*args, **kwargs):
+        if not torch.cuda.is_available():
+            pytest.skip("requires cuda")
+        return func(*args, **kwargs)
+
+    return wrapper
+
+
+def require_hf_token(func):
+    """
+    Decorator that skips the test if no Hugging Face Hub token is available.
+    """
+
+    @wraps(func)
+    def wrapper(*args, **kwargs):
+        from huggingface_hub import get_token
+
+        if get_token() is None:
+            pytest.skip("requires HF token for gated model access")
+        return func(*args, **kwargs)
+
+    return wrapper
+
+
+def require_env(func):
+    """
+    Decorator that skips the test if the required environment package is not installed.
+    As it need 'env_name' in args, it also checks whether it is provided as an argument.
+    If 'env_name' is None, this check is skipped.
+    """
+
+    @wraps(func)
+    def wrapper(*args, **kwargs):
+        # Determine if 'env_name' is provided and extract its value
+        arg_names = func.__code__.co_varnames[: func.__code__.co_argcount]
+        if "env_name" in arg_names:
+            # Get the index of 'env_name' and retrieve the value from args
+            index = arg_names.index("env_name")
+            env_name = args[index] if len(args) > index else kwargs.get("env_name")
+        else:
+            raise ValueError("Function does not have 'env_name' as an argument.")
+
+        # Perform the package check
+        package_name = f"gym_{env_name}"
+        if env_name is not None and not is_package_available(package_name):
+            pytest.skip(f"gym-{env_name} not installed")
+
+        return func(*args, **kwargs)
+
+    return wrapper
+
+
+def require_package_arg(func):
+    """
+    Decorator that skips the test if the required package is not installed.
+    This is similar to `require_env` but more general in that it can check any package (not just environments).
+    As it need 'required_packages' in args, it also checks whether it is provided as an argument.
+    If 'required_packages' is None, this check is skipped.
+    """
+
+    @wraps(func)
+    def wrapper(*args, **kwargs):
+        # Determine if 'required_packages' is provided and extract its value
+        arg_names = func.__code__.co_varnames[: func.__code__.co_argcount]
+        if "required_packages" in arg_names:
+            # Get the index of 'required_packages' and retrieve the value from args
+            index = arg_names.index("required_packages")
+            required_packages = args[index] if len(args) > index else kwargs.get("required_packages")
+        else:
+            raise ValueError("Function does not have 'required_packages' as an argument.")
+
+        if required_packages is None:
+            return func(*args, **kwargs)
+
+        # Perform the package check
+        for package in required_packages:
+            if not is_package_available(package):
+                pytest.skip(f"{package} not installed")
+
+        return func(*args, **kwargs)
+
+    return wrapper
+
+
+def require_package(package_name, import_name=None):
+    """
+    Decorator that skips the test if the specified package is not installed.
+    """
+
+    def decorator(func):
+        @wraps(func)
+        def wrapper(*args, **kwargs):
+            if not is_package_available(pkg_name=package_name, import_name=import_name):
+                pytest.skip(f"{package_name} not installed")
+            return func(*args, **kwargs)
+
+        return wrapper
+
+    return decorator
diff --git a/lerobot/tests/utils/test_encoding_utils.py b/lerobot/tests/utils/test_encoding_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..8a02312210209ace07115c60b2f7ea1ff17aefe6
--- /dev/null
+++ b/lerobot/tests/utils/test_encoding_utils.py
@@ -0,0 +1,171 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import pytest
+
+from lerobot.motors.encoding_utils import (
+    decode_sign_magnitude,
+    decode_twos_complement,
+    encode_sign_magnitude,
+    encode_twos_complement,
+)
+
+
+@pytest.mark.parametrize(
+    "value, sign_bit_index, expected",
+    [
+        (5, 4, 5),
+        (0, 4, 0),
+        (7, 3, 7),
+        (-1, 4, 17),
+        (-8, 4, 24),
+        (-3, 3, 11),
+    ],
+)
+def test_encode_sign_magnitude(value, sign_bit_index, expected):
+    assert encode_sign_magnitude(value, sign_bit_index) == expected
+
+
+@pytest.mark.parametrize(
+    "encoded, sign_bit_index, expected",
+    [
+        (5, 4, 5),
+        (0, 4, 0),
+        (7, 3, 7),
+        (17, 4, -1),
+        (24, 4, -8),
+        (11, 3, -3),
+    ],
+)
+def test_decode_sign_magnitude(encoded, sign_bit_index, expected):
+    assert decode_sign_magnitude(encoded, sign_bit_index) == expected
+
+
+@pytest.mark.parametrize(
+    "encoded, sign_bit_index",
+    [
+        (16, 4),
+        (-9, 3),
+    ],
+)
+def test_encode_raises_on_overflow(encoded, sign_bit_index):
+    with pytest.raises(ValueError):
+        encode_sign_magnitude(encoded, sign_bit_index)
+
+
+def test_encode_decode_sign_magnitude():
+    for sign_bit_index in range(2, 6):
+        max_val = (1 << sign_bit_index) - 1
+        for value in range(-max_val, max_val + 1):
+            encoded = encode_sign_magnitude(value, sign_bit_index)
+            decoded = decode_sign_magnitude(encoded, sign_bit_index)
+            assert decoded == value, f"Failed at value={value}, index={sign_bit_index}"
+
+
+@pytest.mark.parametrize(
+    "value, n_bytes, expected",
+    [
+        (0, 1, 0),
+        (5, 1, 5),
+        (-1, 1, 255),
+        (-128, 1, 128),
+        (-2, 1, 254),
+        (127, 1, 127),
+        (0, 2, 0),
+        (5, 2, 5),
+        (-1, 2, 65_535),
+        (-32_768, 2, 32_768),
+        (-2, 2, 65_534),
+        (32_767, 2, 32_767),
+        (0, 4, 0),
+        (5, 4, 5),
+        (-1, 4, 4_294_967_295),
+        (-2_147_483_648, 4, 2_147_483_648),
+        (-2, 4, 4_294_967_294),
+        (2_147_483_647, 4, 2_147_483_647),
+    ],
+)
+def test_encode_twos_complement(value, n_bytes, expected):
+    assert encode_twos_complement(value, n_bytes) == expected
+
+
+@pytest.mark.parametrize(
+    "value, n_bytes, expected",
+    [
+        (0, 1, 0),
+        (5, 1, 5),
+        (255, 1, -1),
+        (128, 1, -128),
+        (254, 1, -2),
+        (127, 1, 127),
+        (0, 2, 0),
+        (5, 2, 5),
+        (65_535, 2, -1),
+        (32_768, 2, -32_768),
+        (65_534, 2, -2),
+        (32_767, 2, 32_767),
+        (0, 4, 0),
+        (5, 4, 5),
+        (4_294_967_295, 4, -1),
+        (2_147_483_648, 4, -2_147_483_648),
+        (4_294_967_294, 4, -2),
+        (2_147_483_647, 4, 2_147_483_647),
+    ],
+)
+def test_decode_twos_complement(value, n_bytes, expected):
+    assert decode_twos_complement(value, n_bytes) == expected
+
+
+@pytest.mark.parametrize(
+    "value, n_bytes",
+    [
+        (-129, 1),
+        (128, 1),
+        (-32_769, 2),
+        (32_768, 2),
+        (-2_147_483_649, 4),
+        (2_147_483_648, 4),
+    ],
+)
+def test_encode_twos_complement_out_of_range(value, n_bytes):
+    with pytest.raises(ValueError):
+        encode_twos_complement(value, n_bytes)
+
+
+@pytest.mark.parametrize(
+    "value, n_bytes",
+    [
+        (-128, 1),
+        (-1, 1),
+        (0, 1),
+        (1, 1),
+        (127, 1),
+        (-32_768, 2),
+        (-1, 2),
+        (0, 2),
+        (1, 2),
+        (32_767, 2),
+        (-2_147_483_648, 4),
+        (-1, 4),
+        (0, 4),
+        (1, 4),
+        (2_147_483_647, 4),
+    ],
+)
+def test_encode_decode_twos_complement(value, n_bytes):
+    encoded = encode_twos_complement(value, n_bytes)
+    decoded = decode_twos_complement(encoded, n_bytes)
+    assert decoded == value, f"Failed at value={value}, n_bytes={n_bytes}"
diff --git a/lerobot/tests/utils/test_io_utils.py b/lerobot/tests/utils/test_io_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..0beea639d6a1f35d6e16c05342db610408c23ef4
--- /dev/null
+++ b/lerobot/tests/utils/test_io_utils.py
@@ -0,0 +1,90 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import json
+from pathlib import Path
+from typing import Any
+
+import pytest
+
+from lerobot.utils.io_utils import deserialize_json_into_object
+
+
+@pytest.fixture
+def tmp_json_file(tmp_path: Path):
+    """Writes `data` to a temporary JSON file and returns the file's path."""
+
+    def _write(data: Any) -> Path:
+        file_path = tmp_path / "data.json"
+        with file_path.open("w", encoding="utf-8") as f:
+            json.dump(data, f)
+        return file_path
+
+    return _write
+
+
+def test_simple_dict(tmp_json_file):
+    data = {"name": "Alice", "age": 30}
+    json_path = tmp_json_file(data)
+    obj = {"name": "", "age": 0}
+    assert deserialize_json_into_object(json_path, obj) == data
+
+
+def test_nested_structure(tmp_json_file):
+    data = {"items": [1, 2, 3], "info": {"active": True}}
+    json_path = tmp_json_file(data)
+    obj = {"items": [0, 0, 0], "info": {"active": False}}
+    assert deserialize_json_into_object(json_path, obj) == data
+
+
+def test_tuple_conversion(tmp_json_file):
+    data = {"coords": [10.5, 20.5]}
+    json_path = tmp_json_file(data)
+    obj = {"coords": (0.0, 0.0)}
+    result = deserialize_json_into_object(json_path, obj)
+    assert result["coords"] == (10.5, 20.5)
+
+
+def test_type_mismatch_raises(tmp_json_file):
+    data = {"numbers": {"bad": "structure"}}
+    json_path = tmp_json_file(data)
+    obj = {"numbers": [0, 0]}
+    with pytest.raises(TypeError):
+        deserialize_json_into_object(json_path, obj)
+
+
+def test_missing_key_raises(tmp_json_file):
+    data = {"one": 1}
+    json_path = tmp_json_file(data)
+    obj = {"one": 0, "two": 0}
+    with pytest.raises(ValueError):
+        deserialize_json_into_object(json_path, obj)
+
+
+def test_extra_key_raises(tmp_json_file):
+    data = {"one": 1, "two": 2}
+    json_path = tmp_json_file(data)
+    obj = {"one": 0}
+    with pytest.raises(ValueError):
+        deserialize_json_into_object(json_path, obj)
+
+
+def test_list_length_mismatch_raises(tmp_json_file):
+    data = {"nums": [1, 2, 3]}
+    json_path = tmp_json_file(data)
+    obj = {"nums": [0, 0]}
+    with pytest.raises(ValueError):
+        deserialize_json_into_object(json_path, obj)
diff --git a/lerobot/tests/utils/test_logging_utils.py b/lerobot/tests/utils/test_logging_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..1207534c0e5d87368256ddeed614d5b54f38e9c5
--- /dev/null
+++ b/lerobot/tests/utils/test_logging_utils.py
@@ -0,0 +1,159 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import pytest
+
+from lerobot.utils.logging_utils import AverageMeter, MetricsTracker
+
+
+@pytest.fixture
+def mock_metrics():
+    return {"loss": AverageMeter("loss", ":.3f"), "accuracy": AverageMeter("accuracy", ":.2f")}
+
+
+class MockAccelerator:
+    def __init__(self, num_processes: int):
+        self.num_processes = num_processes
+
+
+def test_average_meter_initialization():
+    meter = AverageMeter("loss", ":.2f")
+    assert meter.name == "loss"
+    assert meter.fmt == ":.2f"
+    assert meter.val == 0.0
+    assert meter.avg == 0.0
+    assert meter.sum == 0.0
+    assert meter.count == 0.0
+
+
+def test_average_meter_update():
+    meter = AverageMeter("accuracy")
+    meter.update(5, n=2)
+    assert meter.val == 5
+    assert meter.sum == 10
+    assert meter.count == 2
+    assert meter.avg == 5
+
+
+def test_average_meter_reset():
+    meter = AverageMeter("loss")
+    meter.update(3, 4)
+    meter.reset()
+    assert meter.val == 0.0
+    assert meter.avg == 0.0
+    assert meter.sum == 0.0
+    assert meter.count == 0.0
+
+
+def test_average_meter_str():
+    meter = AverageMeter("metric", ":.1f")
+    meter.update(4.567, 3)
+    assert str(meter) == "metric:4.6"
+
+
+def test_metrics_tracker_initialization(mock_metrics):
+    tracker = MetricsTracker(
+        batch_size=32, num_frames=1000, num_episodes=50, metrics=mock_metrics, initial_step=10
+    )
+    assert tracker.steps == 10
+    assert tracker.samples == 10 * 32
+    assert tracker.episodes == tracker.samples / (1000 / 50)
+    assert tracker.epochs == tracker.samples / 1000
+    assert "loss" in tracker.metrics
+    assert "accuracy" in tracker.metrics
+
+
+def test_metrics_tracker_step(mock_metrics):
+    tracker = MetricsTracker(
+        batch_size=32, num_frames=1000, num_episodes=50, metrics=mock_metrics, initial_step=5
+    )
+    tracker.step()
+    assert tracker.steps == 6
+    assert tracker.samples == 6 * 32
+    assert tracker.episodes == tracker.samples / (1000 / 50)
+    assert tracker.epochs == tracker.samples / 1000
+
+
+def test_metrics_tracker_initialization_with_accelerator(mock_metrics):
+    tracker = MetricsTracker(
+        batch_size=32,
+        num_frames=1000,
+        num_episodes=50,
+        metrics=mock_metrics,
+        initial_step=10,
+        accelerator=MockAccelerator(num_processes=2),
+    )
+    assert tracker.steps == 10
+    assert tracker.samples == 10 * 32 * 2
+    assert tracker.episodes == tracker.samples / (1000 / 50)
+    assert tracker.epochs == tracker.samples / 1000
+
+
+def test_metrics_tracker_step_with_accelerator(mock_metrics):
+    tracker = MetricsTracker(
+        batch_size=32,
+        num_frames=1000,
+        num_episodes=50,
+        metrics=mock_metrics,
+        initial_step=5,
+        accelerator=MockAccelerator(num_processes=2),
+    )
+    tracker.step()
+    assert tracker.steps == 6
+    assert tracker.samples == (5 * 32 * 2) + (32 * 2)
+    assert tracker.episodes == tracker.samples / (1000 / 50)
+    assert tracker.epochs == tracker.samples / 1000
+
+
+def test_metrics_tracker_getattr(mock_metrics):
+    tracker = MetricsTracker(batch_size=32, num_frames=1000, num_episodes=50, metrics=mock_metrics)
+    assert tracker.loss == mock_metrics["loss"]
+    assert tracker.accuracy == mock_metrics["accuracy"]
+    with pytest.raises(AttributeError):
+        _ = tracker.non_existent_metric
+
+
+def test_metrics_tracker_setattr(mock_metrics):
+    tracker = MetricsTracker(batch_size=32, num_frames=1000, num_episodes=50, metrics=mock_metrics)
+    tracker.loss = 2.0
+    assert tracker.loss.val == 2.0
+
+
+def test_metrics_tracker_str(mock_metrics):
+    tracker = MetricsTracker(batch_size=32, num_frames=1000, num_episodes=50, metrics=mock_metrics)
+    tracker.loss.update(3.456, 1)
+    tracker.accuracy.update(0.876, 1)
+    output = str(tracker)
+    assert "loss:3.456" in output
+    assert "accuracy:0.88" in output
+
+
+def test_metrics_tracker_to_dict(mock_metrics):
+    tracker = MetricsTracker(batch_size=32, num_frames=1000, num_episodes=50, metrics=mock_metrics)
+    tracker.loss.update(5, 2)
+    metrics_dict = tracker.to_dict()
+    assert isinstance(metrics_dict, dict)
+    assert metrics_dict["loss"] == 5  # average value
+    assert metrics_dict["steps"] == tracker.steps
+
+
+def test_metrics_tracker_reset_averages(mock_metrics):
+    tracker = MetricsTracker(batch_size=32, num_frames=1000, num_episodes=50, metrics=mock_metrics)
+    tracker.loss.update(10, 3)
+    tracker.accuracy.update(0.95, 5)
+    tracker.reset_averages()
+    assert tracker.loss.avg == 0.0
+    assert tracker.accuracy.avg == 0.0
diff --git a/lerobot/tests/utils/test_process.py b/lerobot/tests/utils/test_process.py
new file mode 100644
index 0000000000000000000000000000000000000000..e2b00cae9cb2afc3a760da1d53b77db9d424684d
--- /dev/null
+++ b/lerobot/tests/utils/test_process.py
@@ -0,0 +1,112 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import multiprocessing
+import os
+import signal
+import threading
+from unittest.mock import patch
+
+import pytest
+
+from lerobot.rl.process import ProcessSignalHandler
+
+
+# Fixture to reset shutdown_event_counter and original signal handlers before and after each test
+@pytest.fixture(autouse=True)
+def reset_globals_and_handlers():
+    # Store original signal handlers
+    original_handlers = {
+        sig: signal.getsignal(sig)
+        for sig in [signal.SIGINT, signal.SIGTERM, signal.SIGHUP, signal.SIGQUIT]
+        if hasattr(signal, sig.name)
+    }
+
+    yield
+
+    # Restore original signal handlers
+    for sig, handler in original_handlers.items():
+        signal.signal(sig, handler)
+
+
+def test_setup_process_handlers_event_with_threads():
+    """Test that setup_process_handlers returns the correct event type."""
+    handler = ProcessSignalHandler(use_threads=True)
+    shutdown_event = handler.shutdown_event
+    assert isinstance(shutdown_event, threading.Event), "Should be a threading.Event"
+    assert not shutdown_event.is_set(), "Event should initially be unset"
+
+
+def test_setup_process_handlers_event_with_processes():
+    """Test that setup_process_handlers returns the correct event type."""
+    handler = ProcessSignalHandler(use_threads=False)
+    shutdown_event = handler.shutdown_event
+    assert isinstance(shutdown_event, type(multiprocessing.Event())), "Should be a multiprocessing.Event"
+    assert not shutdown_event.is_set(), "Event should initially be unset"
+
+
+@pytest.mark.parametrize("use_threads", [True, False])
+@pytest.mark.parametrize(
+    "sig",
+    [
+        signal.SIGINT,
+        signal.SIGTERM,
+        # SIGHUP and SIGQUIT are not reliably available on all platforms (e.g. Windows)
+        pytest.param(
+            signal.SIGHUP,
+            marks=pytest.mark.skipif(not hasattr(signal, "SIGHUP"), reason="SIGHUP not available"),
+        ),
+        pytest.param(
+            signal.SIGQUIT,
+            marks=pytest.mark.skipif(not hasattr(signal, "SIGQUIT"), reason="SIGQUIT not available"),
+        ),
+    ],
+)
+def test_signal_handler_sets_event(use_threads, sig):
+    """Test that the signal handler sets the event on receiving a signal."""
+    handler = ProcessSignalHandler(use_threads=use_threads)
+    shutdown_event = handler.shutdown_event
+
+    assert handler.counter == 0
+
+    os.kill(os.getpid(), sig)
+
+    # In some environments, the signal might take a moment to be handled.
+    shutdown_event.wait(timeout=1.0)
+
+    assert shutdown_event.is_set(), f"Event should be set after receiving signal {sig}"
+
+    # Ensure the internal counter was incremented
+    assert handler.counter == 1
+
+
+@pytest.mark.parametrize("use_threads", [True, False])
+@patch("sys.exit")
+def test_force_shutdown_on_second_signal(mock_sys_exit, use_threads):
+    """Test that a second signal triggers a force shutdown."""
+    handler = ProcessSignalHandler(use_threads=use_threads)
+
+    os.kill(os.getpid(), signal.SIGINT)
+    # Give a moment for the first signal to be processed
+    import time
+
+    time.sleep(0.1)
+    os.kill(os.getpid(), signal.SIGINT)
+
+    time.sleep(0.1)
+
+    assert handler.counter == 2
+    mock_sys_exit.assert_called_once_with(1)
diff --git a/lerobot/tests/utils/test_random_utils.py b/lerobot/tests/utils/test_random_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..e3a5d420f2c1dd9d6be2a88173aad517c92033db
--- /dev/null
+++ b/lerobot/tests/utils/test_random_utils.py
@@ -0,0 +1,125 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import random
+
+import numpy as np
+import pytest
+import torch
+
+from lerobot.utils.random_utils import (
+    deserialize_numpy_rng_state,
+    deserialize_python_rng_state,
+    deserialize_rng_state,
+    deserialize_torch_rng_state,
+    get_rng_state,
+    seeded_context,
+    serialize_numpy_rng_state,
+    serialize_python_rng_state,
+    serialize_rng_state,
+    serialize_torch_rng_state,
+    set_rng_state,
+    set_seed,
+)
+
+
+@pytest.fixture
+def fixed_seed():
+    """Fixture to set a consistent initial seed for each test."""
+    set_seed(12345)
+    yield
+
+
+def test_serialize_deserialize_python_rng(fixed_seed):
+    # Save state after generating val1
+    _ = random.random()
+    st = serialize_python_rng_state()
+    # Next random is val2
+    val2 = random.random()
+    # Restore the state, so the next random should match val2
+    deserialize_python_rng_state(st)
+    val3 = random.random()
+    assert val2 == val3
+
+
+def test_serialize_deserialize_numpy_rng(fixed_seed):
+    _ = np.random.rand()
+    st = serialize_numpy_rng_state()
+    val2 = np.random.rand()
+    deserialize_numpy_rng_state(st)
+    val3 = np.random.rand()
+    assert val2 == val3
+
+
+def test_serialize_deserialize_torch_rng(fixed_seed):
+    _ = torch.rand(1).item()
+    st = serialize_torch_rng_state()
+    val2 = torch.rand(1).item()
+    deserialize_torch_rng_state(st)
+    val3 = torch.rand(1).item()
+    assert val2 == val3
+
+
+def test_serialize_deserialize_rng(fixed_seed):
+    # Generate one from each library
+    _ = random.random()
+    _ = np.random.rand()
+    _ = torch.rand(1).item()
+    # Serialize
+    st = serialize_rng_state()
+    # Generate second set
+    val_py2 = random.random()
+    val_np2 = np.random.rand()
+    val_th2 = torch.rand(1).item()
+    # Restore, so the next draws should match val_py2, val_np2, val_th2
+    deserialize_rng_state(st)
+    assert random.random() == val_py2
+    assert np.random.rand() == val_np2
+    assert torch.rand(1).item() == val_th2
+
+
+def test_get_set_rng_state(fixed_seed):
+    st = get_rng_state()
+    val1 = (random.random(), np.random.rand(), torch.rand(1).item())
+    # Change states
+    random.random()
+    np.random.rand()
+    torch.rand(1)
+    # Restore
+    set_rng_state(st)
+    val2 = (random.random(), np.random.rand(), torch.rand(1).item())
+    assert val1 == val2
+
+
+def test_set_seed():
+    set_seed(1337)
+    val1 = (random.random(), np.random.rand(), torch.rand(1).item())
+    set_seed(1337)
+    val2 = (random.random(), np.random.rand(), torch.rand(1).item())
+    assert val1 == val2
+
+
+def test_seeded_context(fixed_seed):
+    val1 = (random.random(), np.random.rand(), torch.rand(1).item())
+    with seeded_context(1337):
+        seeded_val1 = (random.random(), np.random.rand(), torch.rand(1).item())
+    val2 = (random.random(), np.random.rand(), torch.rand(1).item())
+    with seeded_context(1337):
+        seeded_val2 = (random.random(), np.random.rand(), torch.rand(1).item())
+
+    assert seeded_val1 == seeded_val2
+    assert all(a != b for a, b in zip(val1, seeded_val1, strict=True))  # changed inside the context
+    assert all(a != b for a, b in zip(val2, seeded_val2, strict=True))  # changed again after exiting
diff --git a/lerobot/tests/utils/test_replay_buffer.py b/lerobot/tests/utils/test_replay_buffer.py
new file mode 100644
index 0000000000000000000000000000000000000000..b9d3a1ac04631d6639fa08b2dafb840156a08ca7
--- /dev/null
+++ b/lerobot/tests/utils/test_replay_buffer.py
@@ -0,0 +1,681 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import sys
+from collections.abc import Callable
+
+import pytest
+import torch
+
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.rl.buffer import BatchTransition, ReplayBuffer, random_crop_vectorized
+from lerobot.utils.constants import ACTION, DONE, OBS_IMAGE, OBS_STATE, OBS_STR, REWARD
+from tests.fixtures.constants import DUMMY_REPO_ID
+
+
+def state_dims() -> list[str]:
+    return [OBS_IMAGE, OBS_STATE]
+
+
+@pytest.fixture
+def replay_buffer() -> ReplayBuffer:
+    return create_empty_replay_buffer()
+
+
+def clone_state(state: dict) -> dict:
+    return {k: v.clone() for k, v in state.items()}
+
+
+def create_empty_replay_buffer(
+    optimize_memory: bool = False,
+    use_drq: bool = False,
+    image_augmentation_function: Callable | None = None,
+) -> ReplayBuffer:
+    buffer_capacity = 10
+    device = "cpu"
+    return ReplayBuffer(
+        buffer_capacity,
+        device,
+        state_dims(),
+        optimize_memory=optimize_memory,
+        use_drq=use_drq,
+        image_augmentation_function=image_augmentation_function,
+    )
+
+
+def create_random_image() -> torch.Tensor:
+    return torch.rand(3, 84, 84)
+
+
+def create_dummy_transition() -> dict:
+    return {
+        OBS_IMAGE: create_random_image(),
+        ACTION: torch.randn(4),
+        "reward": torch.tensor(1.0),
+        OBS_STATE: torch.randn(
+            10,
+        ),
+        "done": torch.tensor(False),
+        "truncated": torch.tensor(False),
+        "complementary_info": {},
+    }
+
+
+def create_dataset_from_replay_buffer(tmp_path) -> tuple[LeRobotDataset, ReplayBuffer]:
+    dummy_state_1 = create_dummy_state()
+    dummy_action_1 = create_dummy_action()
+
+    dummy_state_2 = create_dummy_state()
+    dummy_action_2 = create_dummy_action()
+
+    dummy_state_3 = create_dummy_state()
+    dummy_action_3 = create_dummy_action()
+
+    dummy_state_4 = create_dummy_state()
+    dummy_action_4 = create_dummy_action()
+
+    replay_buffer = create_empty_replay_buffer()
+    replay_buffer.add(dummy_state_1, dummy_action_1, 1.0, dummy_state_1, False, False)
+    replay_buffer.add(dummy_state_2, dummy_action_2, 1.0, dummy_state_2, False, False)
+    replay_buffer.add(dummy_state_3, dummy_action_3, 1.0, dummy_state_3, True, True)
+    replay_buffer.add(dummy_state_4, dummy_action_4, 1.0, dummy_state_4, True, True)
+
+    root = tmp_path / "test"
+    return (replay_buffer.to_lerobot_dataset(DUMMY_REPO_ID, root=root), replay_buffer)
+
+
+def create_dummy_state() -> dict:
+    return {
+        OBS_IMAGE: create_random_image(),
+        OBS_STATE: torch.randn(
+            10,
+        ),
+    }
+
+
+def get_tensor_memory_consumption(tensor):
+    return tensor.nelement() * tensor.element_size()
+
+
+def get_tensors_memory_consumption(obj, visited_addresses):
+    total_size = 0
+
+    address = id(obj)
+    if address in visited_addresses:
+        return 0
+
+    visited_addresses.add(address)
+
+    if isinstance(obj, torch.Tensor):
+        return get_tensor_memory_consumption(obj)
+    elif isinstance(obj, (list | tuple)):
+        for item in obj:
+            total_size += get_tensors_memory_consumption(item, visited_addresses)
+    elif isinstance(obj, dict):
+        for value in obj.values():
+            total_size += get_tensors_memory_consumption(value, visited_addresses)
+    elif hasattr(obj, "__dict__"):
+        # It's an object, we need to get the size of the attributes
+        for _, attr in vars(obj).items():
+            total_size += get_tensors_memory_consumption(attr, visited_addresses)
+
+    return total_size
+
+
+def get_object_memory(obj):
+    # Track visited addresses to avoid infinite loops
+    # and cases when two properties point to the same object
+    visited_addresses = set()
+
+    # Get the size of the object in bytes
+    total_size = sys.getsizeof(obj)
+
+    # Get the size of the tensor attributes
+    total_size += get_tensors_memory_consumption(obj, visited_addresses)
+
+    return total_size
+
+
+def create_dummy_action() -> torch.Tensor:
+    return torch.randn(4)
+
+
+def dict_properties() -> list:
+    return ["state", "next_state"]
+
+
+@pytest.fixture
+def dummy_state() -> dict:
+    return create_dummy_state()
+
+
+@pytest.fixture
+def next_dummy_state() -> dict:
+    return create_dummy_state()
+
+
+@pytest.fixture
+def dummy_action() -> torch.Tensor:
+    return torch.randn(4)
+
+
+def test_empty_buffer_sample_raises_error(replay_buffer):
+    assert len(replay_buffer) == 0, "Replay buffer should be empty."
+    assert replay_buffer.capacity == 10, "Replay buffer capacity should be 10."
+    with pytest.raises(RuntimeError, match="Cannot sample from an empty buffer"):
+        replay_buffer.sample(1)
+
+
+def test_zero_capacity_buffer_raises_error():
+    with pytest.raises(ValueError, match="Capacity must be greater than 0."):
+        ReplayBuffer(0, "cpu", [OBS_STR, "next_observation"])
+
+
+def test_add_transition(replay_buffer, dummy_state, dummy_action):
+    replay_buffer.add(dummy_state, dummy_action, 1.0, dummy_state, False, False)
+    assert len(replay_buffer) == 1, "Replay buffer should have one transition after adding."
+    assert torch.equal(replay_buffer.actions[0], dummy_action), (
+        "Action should be equal to the first transition."
+    )
+    assert replay_buffer.rewards[0] == 1.0, "Reward should be equal to the first transition."
+    assert not replay_buffer.dones[0], "Done should be False for the first transition."
+    assert not replay_buffer.truncateds[0], "Truncated should be False for the first transition."
+
+    for dim in state_dims():
+        assert torch.equal(replay_buffer.states[dim][0], dummy_state[dim]), (
+            "Observation should be equal to the first transition."
+        )
+        assert torch.equal(replay_buffer.next_states[dim][0], dummy_state[dim]), (
+            "Next observation should be equal to the first transition."
+        )
+
+
+def test_add_over_capacity():
+    replay_buffer = ReplayBuffer(2, "cpu", [OBS_STR, "next_observation"])
+    dummy_state_1 = create_dummy_state()
+    dummy_action_1 = create_dummy_action()
+
+    dummy_state_2 = create_dummy_state()
+    dummy_action_2 = create_dummy_action()
+
+    dummy_state_3 = create_dummy_state()
+    dummy_action_3 = create_dummy_action()
+
+    replay_buffer.add(dummy_state_1, dummy_action_1, 1.0, dummy_state_1, False, False)
+    replay_buffer.add(dummy_state_2, dummy_action_2, 1.0, dummy_state_2, False, False)
+    replay_buffer.add(dummy_state_3, dummy_action_3, 1.0, dummy_state_3, True, True)
+
+    assert len(replay_buffer) == 2, "Replay buffer should have 2 transitions after adding 3."
+
+    for dim in state_dims():
+        assert torch.equal(replay_buffer.states[dim][0], dummy_state_3[dim]), (
+            "Observation should be equal to the first transition."
+        )
+        assert torch.equal(replay_buffer.next_states[dim][0], dummy_state_3[dim]), (
+            "Next observation should be equal to the first transition."
+        )
+
+    assert torch.equal(replay_buffer.actions[0], dummy_action_3), (
+        "Action should be equal to the last transition."
+    )
+    assert replay_buffer.rewards[0] == 1.0, "Reward should be equal to the last transition."
+    assert replay_buffer.dones[0], "Done should be True for the first transition."
+    assert replay_buffer.truncateds[0], "Truncated should be True for the first transition."
+
+
+def test_sample_from_empty_buffer(replay_buffer):
+    with pytest.raises(RuntimeError, match="Cannot sample from an empty buffer"):
+        replay_buffer.sample(1)
+
+
+def test_sample_with_1_transition(replay_buffer, dummy_state, next_dummy_state, dummy_action):
+    replay_buffer.add(dummy_state, dummy_action, 1.0, next_dummy_state, False, False)
+    got_batch_transition = replay_buffer.sample(1)
+
+    expected_batch_transition = BatchTransition(
+        state=clone_state(dummy_state),
+        action=dummy_action.clone(),
+        reward=1.0,
+        next_state=clone_state(next_dummy_state),
+        done=False,
+        truncated=False,
+    )
+
+    for buffer_property in dict_properties():
+        for k, v in expected_batch_transition[buffer_property].items():
+            got_state = got_batch_transition[buffer_property][k]
+
+            assert got_state.shape[0] == 1, f"{k} should have 1 transition."
+            assert got_state.device.type == "cpu", f"{k} should be on cpu."
+
+            assert torch.equal(got_state[0], v), f"{k} should be equal to the expected batch transition."
+
+    for key, _value in expected_batch_transition.items():
+        if key in dict_properties():
+            continue
+
+        got_value = got_batch_transition[key]
+
+        v_tensor = expected_batch_transition[key]
+        if not isinstance(v_tensor, torch.Tensor):
+            v_tensor = torch.tensor(v_tensor)
+
+        assert got_value.shape[0] == 1, f"{key} should have 1 transition."
+        assert got_value.device.type == "cpu", f"{key} should be on cpu."
+        assert torch.equal(got_value[0], v_tensor), f"{key} should be equal to the expected batch transition."
+
+
+def test_sample_with_batch_bigger_than_buffer_size(
+    replay_buffer, dummy_state, next_dummy_state, dummy_action
+):
+    replay_buffer.add(dummy_state, dummy_action, 1.0, next_dummy_state, False, False)
+    got_batch_transition = replay_buffer.sample(10)
+
+    expected_batch_transition = BatchTransition(
+        state=dummy_state,
+        action=dummy_action,
+        reward=1.0,
+        next_state=next_dummy_state,
+        done=False,
+        truncated=False,
+    )
+
+    for buffer_property in dict_properties():
+        for k in expected_batch_transition[buffer_property]:
+            got_state = got_batch_transition[buffer_property][k]
+
+            assert got_state.shape[0] == 1, f"{k} should have 1 transition."
+
+    for key in expected_batch_transition:
+        if key in dict_properties():
+            continue
+
+        got_value = got_batch_transition[key]
+        assert got_value.shape[0] == 1, f"{key} should have 1 transition."
+
+
+def test_sample_batch(replay_buffer):
+    dummy_state_1 = create_dummy_state()
+    dummy_action_1 = create_dummy_action()
+
+    dummy_state_2 = create_dummy_state()
+    dummy_action_2 = create_dummy_action()
+
+    dummy_state_3 = create_dummy_state()
+    dummy_action_3 = create_dummy_action()
+
+    dummy_state_4 = create_dummy_state()
+    dummy_action_4 = create_dummy_action()
+
+    replay_buffer.add(dummy_state_1, dummy_action_1, 1.0, dummy_state_1, False, False)
+    replay_buffer.add(dummy_state_2, dummy_action_2, 2.0, dummy_state_2, False, False)
+    replay_buffer.add(dummy_state_3, dummy_action_3, 3.0, dummy_state_3, True, True)
+    replay_buffer.add(dummy_state_4, dummy_action_4, 4.0, dummy_state_4, True, True)
+
+    dummy_states = [dummy_state_1, dummy_state_2, dummy_state_3, dummy_state_4]
+    dummy_actions = [dummy_action_1, dummy_action_2, dummy_action_3, dummy_action_4]
+
+    got_batch_transition = replay_buffer.sample(3)
+
+    for buffer_property in dict_properties():
+        for k in got_batch_transition[buffer_property]:
+            got_state = got_batch_transition[buffer_property][k]
+
+            assert got_state.shape[0] == 3, f"{k} should have 3 transition."
+
+            for got_state_item in got_state:
+                assert any(torch.equal(got_state_item, dummy_state[k]) for dummy_state in dummy_states), (
+                    f"{k} should be equal to one of the dummy states."
+                )
+
+    for got_action_item in got_batch_transition[ACTION]:
+        assert any(torch.equal(got_action_item, dummy_action) for dummy_action in dummy_actions), (
+            "Actions should be equal to the dummy actions."
+        )
+
+    for k in got_batch_transition:
+        if k in dict_properties() or k == "complementary_info":
+            continue
+
+        got_value = got_batch_transition[k]
+        assert got_value.shape[0] == 3, f"{k} should have 3 transition."
+
+
+def test_to_lerobot_dataset_with_empty_buffer(replay_buffer):
+    with pytest.raises(ValueError, match="The replay buffer is empty. Cannot convert to a dataset."):
+        replay_buffer.to_lerobot_dataset("dummy_repo")
+
+
+def test_to_lerobot_dataset(tmp_path):
+    ds, buffer = create_dataset_from_replay_buffer(tmp_path)
+
+    assert len(ds) == len(buffer), "Dataset should have the same size as the Replay Buffer"
+    assert ds.fps == 1, "FPS should be 1"
+    assert ds.repo_id == "dummy/repo", "The dataset should have `dummy/repo` repo id"
+
+    for dim in state_dims():
+        assert dim in ds.features
+        assert ds.features[dim]["shape"] == buffer.states[dim][0].shape
+
+    assert ds.num_episodes == 2
+    assert ds.num_frames == 4
+
+    for j, value in enumerate(ds):
+        print(torch.equal(value[OBS_IMAGE], buffer.next_states[OBS_IMAGE][j]))
+
+    for i in range(len(ds)):
+        for feature, value in ds[i].items():
+            if feature == ACTION:
+                assert torch.equal(value, buffer.actions[i])
+            elif feature == REWARD:
+                assert torch.equal(value, buffer.rewards[i])
+            elif feature == DONE:
+                assert torch.equal(value, buffer.dones[i])
+            elif feature == OBS_IMAGE:
+                # Tensor -> numpy is not precise, so we have some diff there
+                # TODO: Check and fix it
+                torch.testing.assert_close(value, buffer.states[OBS_IMAGE][i], rtol=0.3, atol=0.003)
+            elif feature == OBS_STATE:
+                assert torch.equal(value, buffer.states[OBS_STATE][i])
+
+
+def test_from_lerobot_dataset(tmp_path):
+    dummy_state_1 = create_dummy_state()
+    dummy_action_1 = create_dummy_action()
+
+    dummy_state_2 = create_dummy_state()
+    dummy_action_2 = create_dummy_action()
+
+    dummy_state_3 = create_dummy_state()
+    dummy_action_3 = create_dummy_action()
+
+    dummy_state_4 = create_dummy_state()
+    dummy_action_4 = create_dummy_action()
+
+    replay_buffer = create_empty_replay_buffer()
+    replay_buffer.add(dummy_state_1, dummy_action_1, 1.0, dummy_state_1, False, False)
+    replay_buffer.add(dummy_state_2, dummy_action_2, 1.0, dummy_state_2, False, False)
+    replay_buffer.add(dummy_state_3, dummy_action_3, 1.0, dummy_state_3, True, True)
+    replay_buffer.add(dummy_state_4, dummy_action_4, 1.0, dummy_state_4, True, True)
+
+    root = tmp_path / "test"
+    ds = replay_buffer.to_lerobot_dataset(DUMMY_REPO_ID, root=root)
+
+    reconverted_buffer = ReplayBuffer.from_lerobot_dataset(
+        ds, state_keys=list(state_dims()), device="cpu", capacity=replay_buffer.capacity, use_drq=False
+    )
+
+    # Check only the part of the buffer that's actually filled with data
+    assert torch.equal(
+        reconverted_buffer.actions[: len(replay_buffer)],
+        replay_buffer.actions[: len(replay_buffer)],
+    ), "Actions from converted buffer should be equal to the original replay buffer."
+    assert torch.equal(
+        reconverted_buffer.rewards[: len(replay_buffer)], replay_buffer.rewards[: len(replay_buffer)]
+    ), "Rewards from converted buffer should be equal to the original replay buffer."
+    assert torch.equal(
+        reconverted_buffer.dones[: len(replay_buffer)], replay_buffer.dones[: len(replay_buffer)]
+    ), "Dones from converted buffer should be equal to the original replay buffer."
+
+    # Lerobot DS haven't supported truncateds yet
+    expected_truncateds = torch.zeros(len(replay_buffer)).bool()
+    assert torch.equal(reconverted_buffer.truncateds[: len(replay_buffer)], expected_truncateds), (
+        "Truncateds from converted buffer should be equal False"
+    )
+
+    assert torch.equal(
+        replay_buffer.states[OBS_STATE][: len(replay_buffer)],
+        reconverted_buffer.states[OBS_STATE][: len(replay_buffer)],
+    ), "State should be the same after converting to dataset and return back"
+
+    for i in range(4):
+        torch.testing.assert_close(
+            replay_buffer.states[OBS_IMAGE][i],
+            reconverted_buffer.states[OBS_IMAGE][i],
+            rtol=0.4,
+            atol=0.004,
+        )
+
+    # The 2, 3 frames have done flag, so their values will be equal to the current state
+    for i in range(2):
+        # In the current implementation we take the next state from the `states` and ignore `next_states`
+        next_index = (i + 1) % 4
+
+        torch.testing.assert_close(
+            replay_buffer.states[OBS_IMAGE][next_index],
+            reconverted_buffer.next_states[OBS_IMAGE][i],
+            rtol=0.4,
+            atol=0.004,
+        )
+
+    for i in range(2, 4):
+        assert torch.equal(
+            replay_buffer.states[OBS_STATE][i],
+            reconverted_buffer.next_states[OBS_STATE][i],
+        )
+
+
+def test_buffer_sample_alignment():
+    # Initialize buffer
+    buffer = ReplayBuffer(capacity=100, device="cpu", state_keys=["state_value"], storage_device="cpu")
+
+    # Fill buffer with patterned data
+    for i in range(100):
+        signature = float(i) / 100.0
+        state = {"state_value": torch.tensor([[signature]]).float()}
+        action = torch.tensor([[2.0 * signature]]).float()
+        reward = 3.0 * signature
+
+        is_end = (i + 1) % 10 == 0
+        if is_end:
+            next_state = {"state_value": torch.tensor([[signature]]).float()}
+            done = True
+        else:
+            next_signature = float(i + 1) / 100.0
+            next_state = {"state_value": torch.tensor([[next_signature]]).float()}
+            done = False
+
+        buffer.add(state, action, reward, next_state, done, False)
+
+    # Sample and verify
+    batch = buffer.sample(50)
+
+    for i in range(50):
+        state_sig = batch["state"]["state_value"][i].item()
+        action_val = batch[ACTION][i].item()
+        reward_val = batch["reward"][i].item()
+        next_state_sig = batch["next_state"]["state_value"][i].item()
+        is_done = batch["done"][i].item() > 0.5
+
+        # Verify relationships
+        assert abs(action_val - 2.0 * state_sig) < 1e-4, (
+            f"Action {action_val} should be 2x state signature {state_sig}"
+        )
+
+        assert abs(reward_val - 3.0 * state_sig) < 1e-4, (
+            f"Reward {reward_val} should be 3x state signature {state_sig}"
+        )
+
+        if is_done:
+            assert abs(next_state_sig - state_sig) < 1e-4, (
+                f"For done states, next_state {next_state_sig} should equal state {state_sig}"
+            )
+        else:
+            # Either it's the next sequential state (+0.01) or same state (for episode boundaries)
+            valid_next = (
+                abs(next_state_sig - state_sig - 0.01) < 1e-4 or abs(next_state_sig - state_sig) < 1e-4
+            )
+            assert valid_next, (
+                f"Next state {next_state_sig} should be either state+0.01 or same as state {state_sig}"
+            )
+
+
+def test_memory_optimization():
+    dummy_state_1 = create_dummy_state()
+    dummy_action_1 = create_dummy_action()
+
+    dummy_state_2 = create_dummy_state()
+    dummy_action_2 = create_dummy_action()
+
+    dummy_state_3 = create_dummy_state()
+    dummy_action_3 = create_dummy_action()
+
+    dummy_state_4 = create_dummy_state()
+    dummy_action_4 = create_dummy_action()
+
+    replay_buffer = create_empty_replay_buffer()
+    replay_buffer.add(dummy_state_1, dummy_action_1, 1.0, dummy_state_2, False, False)
+    replay_buffer.add(dummy_state_2, dummy_action_2, 1.0, dummy_state_3, False, False)
+    replay_buffer.add(dummy_state_3, dummy_action_3, 1.0, dummy_state_4, False, False)
+    replay_buffer.add(dummy_state_4, dummy_action_4, 1.0, dummy_state_4, True, True)
+
+    optimized_replay_buffer = create_empty_replay_buffer(True)
+    optimized_replay_buffer.add(dummy_state_1, dummy_action_1, 1.0, dummy_state_2, False, False)
+    optimized_replay_buffer.add(dummy_state_2, dummy_action_2, 1.0, dummy_state_3, False, False)
+    optimized_replay_buffer.add(dummy_state_3, dummy_action_3, 1.0, dummy_state_4, False, False)
+    optimized_replay_buffer.add(dummy_state_4, dummy_action_4, 1.0, None, True, True)
+
+    assert get_object_memory(optimized_replay_buffer) < get_object_memory(replay_buffer), (
+        "Optimized replay buffer should be smaller than the original replay buffer"
+    )
+
+
+def test_check_image_augmentations_with_drq_and_dummy_image_augmentation_function(dummy_state, dummy_action):
+    def dummy_image_augmentation_function(x):
+        return torch.ones_like(x) * 10
+
+    replay_buffer = create_empty_replay_buffer(
+        use_drq=True, image_augmentation_function=dummy_image_augmentation_function
+    )
+
+    replay_buffer.add(dummy_state, dummy_action, 1.0, dummy_state, False, False)
+
+    sampled_transitions = replay_buffer.sample(1)
+    assert torch.all(sampled_transitions["state"][OBS_IMAGE] == 10), "Image augmentations should be applied"
+    assert torch.all(sampled_transitions["next_state"][OBS_IMAGE] == 10), (
+        "Image augmentations should be applied"
+    )
+
+
+def test_check_image_augmentations_with_drq_and_default_image_augmentation_function(
+    dummy_state, dummy_action
+):
+    replay_buffer = create_empty_replay_buffer(use_drq=True)
+
+    replay_buffer.add(dummy_state, dummy_action, 1.0, dummy_state, False, False)
+
+    # Let's check that it doesn't fail and shapes are correct
+    sampled_transitions = replay_buffer.sample(1)
+    assert sampled_transitions["state"][OBS_IMAGE].shape == (1, 3, 84, 84)
+    assert sampled_transitions["next_state"][OBS_IMAGE].shape == (1, 3, 84, 84)
+
+
+def test_random_crop_vectorized_basic():
+    # Create a batch of 2 images with known patterns
+    batch_size, channels, height, width = 2, 3, 10, 8
+    images = torch.zeros((batch_size, channels, height, width))
+
+    # Fill with unique values for testing
+    for b in range(batch_size):
+        images[b] = b + 1
+
+    crop_size = (6, 4)  # Smaller than original
+    cropped = random_crop_vectorized(images, crop_size)
+
+    # Check output shape
+    assert cropped.shape == (batch_size, channels, *crop_size)
+
+    # Check that values are preserved (should be either 1s or 2s for respective batches)
+    assert torch.all(cropped[0] == 1)
+    assert torch.all(cropped[1] == 2)
+
+
+def test_random_crop_vectorized_invalid_size():
+    images = torch.zeros((2, 3, 10, 8))
+
+    # Test crop size larger than image
+    with pytest.raises(ValueError, match="Requested crop size .* is bigger than the image size"):
+        random_crop_vectorized(images, (12, 8))
+
+    with pytest.raises(ValueError, match="Requested crop size .* is bigger than the image size"):
+        random_crop_vectorized(images, (10, 10))
+
+
+def _populate_buffer_for_async_test(capacity: int = 10) -> ReplayBuffer:
+    """Create a small buffer with deterministic 3×128×128 images and 11-D state."""
+    buffer = ReplayBuffer(
+        capacity=capacity,
+        device="cpu",
+        state_keys=[OBS_IMAGE, OBS_STATE],
+        storage_device="cpu",
+    )
+
+    for i in range(capacity):
+        img = torch.ones(3, 128, 128) * i
+        state_vec = torch.arange(11).float() + i
+        state = {
+            OBS_IMAGE: img,
+            OBS_STATE: state_vec,
+        }
+        buffer.add(
+            state=state,
+            action=torch.tensor([0.0]),
+            reward=0.0,
+            next_state=state,
+            done=False,
+            truncated=False,
+        )
+    return buffer
+
+
+def test_async_iterator_shapes_basic():
+    buffer = _populate_buffer_for_async_test()
+    batch_size = 2
+    iterator = buffer.get_iterator(batch_size=batch_size, async_prefetch=True, queue_size=1)
+    batch = next(iterator)
+
+    images = batch["state"][OBS_IMAGE]
+    states = batch["state"][OBS_STATE]
+
+    assert images.shape == (batch_size, 3, 128, 128)
+    assert states.shape == (batch_size, 11)
+
+    next_images = batch["next_state"][OBS_IMAGE]
+    next_states = batch["next_state"][OBS_STATE]
+
+    assert next_images.shape == (batch_size, 3, 128, 128)
+    assert next_states.shape == (batch_size, 11)
+
+
+def test_async_iterator_multiple_iterations():
+    buffer = _populate_buffer_for_async_test()
+    batch_size = 2
+    iterator = buffer.get_iterator(batch_size=batch_size, async_prefetch=True, queue_size=2)
+
+    for _ in range(5):
+        batch = next(iterator)
+        images = batch["state"][OBS_IMAGE]
+        states = batch["state"][OBS_STATE]
+        assert images.shape == (batch_size, 3, 128, 128)
+        assert states.shape == (batch_size, 11)
+
+        next_images = batch["next_state"][OBS_IMAGE]
+        next_states = batch["next_state"][OBS_STATE]
+        assert next_images.shape == (batch_size, 3, 128, 128)
+        assert next_states.shape == (batch_size, 11)
+
+    # Ensure iterator can be disposed without blocking
+    del iterator
diff --git a/lerobot/tests/utils/test_train_utils.py b/lerobot/tests/utils/test_train_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..4791caf588aabc2490fd3dfc3163b8fcafb96dac
--- /dev/null
+++ b/lerobot/tests/utils/test_train_utils.py
@@ -0,0 +1,114 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from pathlib import Path
+from unittest.mock import Mock, patch
+
+from lerobot.utils.constants import (
+    CHECKPOINTS_DIR,
+    LAST_CHECKPOINT_LINK,
+    OPTIMIZER_PARAM_GROUPS,
+    OPTIMIZER_STATE,
+    RNG_STATE,
+    SCHEDULER_STATE,
+    TRAINING_STATE_DIR,
+    TRAINING_STEP,
+)
+from lerobot.utils.train_utils import (
+    get_step_checkpoint_dir,
+    get_step_identifier,
+    load_training_state,
+    load_training_step,
+    save_checkpoint,
+    save_training_state,
+    save_training_step,
+    update_last_checkpoint,
+)
+
+
+def test_get_step_identifier():
+    assert get_step_identifier(5, 1000) == "000005"
+    assert get_step_identifier(123, 100_000) == "000123"
+    assert get_step_identifier(456789, 1_000_000) == "0456789"
+
+
+def test_get_step_checkpoint_dir():
+    output_dir = Path("/checkpoints")
+    step_dir = get_step_checkpoint_dir(output_dir, 1000, 5)
+    assert step_dir == output_dir / CHECKPOINTS_DIR / "000005"
+
+
+def test_save_load_training_step(tmp_path):
+    save_training_step(5000, tmp_path)
+    assert (tmp_path / TRAINING_STEP).is_file()
+
+
+def test_load_training_step(tmp_path):
+    step = 5000
+    save_training_step(step, tmp_path)
+    loaded_step = load_training_step(tmp_path)
+    assert loaded_step == step
+
+
+def test_update_last_checkpoint(tmp_path):
+    checkpoint = tmp_path / "0005"
+    checkpoint.mkdir()
+    update_last_checkpoint(checkpoint)
+    last_checkpoint = tmp_path / LAST_CHECKPOINT_LINK
+    assert last_checkpoint.is_symlink()
+    assert last_checkpoint.resolve() == checkpoint
+
+
+@patch("lerobot.utils.train_utils.save_training_state")
+def test_save_checkpoint(mock_save_training_state, tmp_path, optimizer):
+    policy = Mock()
+    cfg = Mock()
+    save_checkpoint(tmp_path, 10, cfg, policy, optimizer)
+    policy.save_pretrained.assert_called_once()
+    cfg.save_pretrained.assert_called_once()
+    mock_save_training_state.assert_called_once()
+
+
+@patch("lerobot.utils.train_utils.save_training_state")
+def test_save_checkpoint_peft(mock_save_training_state, tmp_path, optimizer):
+    policy = Mock()
+    policy.config = Mock()
+    policy.config.save_pretrained = Mock()
+    cfg = Mock()
+    cfg.use_peft = True
+    save_checkpoint(tmp_path, 10, cfg, policy, optimizer)
+    policy.save_pretrained.assert_called_once()
+    cfg.save_pretrained.assert_called_once()
+    policy.config.save_pretrained.assert_called_once()
+    mock_save_training_state.assert_called_once()
+
+
+def test_save_training_state(tmp_path, optimizer, scheduler):
+    save_training_state(tmp_path, 10, optimizer, scheduler)
+    assert (tmp_path / TRAINING_STATE_DIR).is_dir()
+    assert (tmp_path / TRAINING_STATE_DIR / TRAINING_STEP).is_file()
+    assert (tmp_path / TRAINING_STATE_DIR / RNG_STATE).is_file()
+    assert (tmp_path / TRAINING_STATE_DIR / OPTIMIZER_STATE).is_file()
+    assert (tmp_path / TRAINING_STATE_DIR / OPTIMIZER_PARAM_GROUPS).is_file()
+    assert (tmp_path / TRAINING_STATE_DIR / SCHEDULER_STATE).is_file()
+
+
+def test_save_load_training_state(tmp_path, optimizer, scheduler):
+    save_training_state(tmp_path, 10, optimizer, scheduler)
+    loaded_step, loaded_optimizer, loaded_scheduler = load_training_state(tmp_path, optimizer, scheduler)
+    assert loaded_step == 10
+    assert loaded_optimizer is optimizer
+    assert loaded_scheduler is scheduler
diff --git a/lerobot/tests/utils/test_visualization_utils.py b/lerobot/tests/utils/test_visualization_utils.py
new file mode 100644
index 0000000000000000000000000000000000000000..c8e5a92a86064397af41159f4249633342d2890a
--- /dev/null
+++ b/lerobot/tests/utils/test_visualization_utils.py
@@ -0,0 +1,229 @@
+#!/usr/bin/env python
+
+# Copyright 2025 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import importlib
+import sys
+from types import SimpleNamespace
+
+import numpy as np
+import pytest
+
+from lerobot.types import TransitionKey
+from lerobot.utils.constants import OBS_STATE
+
+
+@pytest.fixture
+def mock_rerun(monkeypatch):
+    """
+    Provide a mock `rerun` module so tests don't depend on the real library.
+    Also reload the module-under-test so it binds to this mock `rr`.
+    """
+    calls = []
+
+    class DummyScalar:
+        def __init__(self, value):
+            self.value = float(value)
+
+    class DummyImage:
+        def __init__(self, arr):
+            self.arr = arr
+
+    def dummy_log(key, obj=None, **kwargs):
+        # Accept either positional `obj` or keyword `entity` and record remaining kwargs.
+        if obj is None and "entity" in kwargs:
+            obj = kwargs.pop("entity")
+        calls.append((key, obj, kwargs))
+
+    dummy_rr = SimpleNamespace(
+        Scalars=DummyScalar,
+        Image=DummyImage,
+        log=dummy_log,
+        init=lambda *a, **k: None,
+        spawn=lambda *a, **k: None,
+    )
+
+    # Inject fake module into sys.modules
+    monkeypatch.setitem(sys.modules, "rerun", dummy_rr)
+
+    # Now import and reload the module under test, to bind to our rerun mock
+    import lerobot.utils.visualization_utils as vu
+
+    importlib.reload(vu)
+
+    # Expose both the reloaded module and the call recorder
+    yield vu, calls
+
+
+def _keys(calls):
+    """Helper to extract just the keys logged to rr.log"""
+    return [k for (k, _obj, _kw) in calls]
+
+
+def _obj_for(calls, key):
+    """Find the first object logged under a given key."""
+    for k, obj, _kw in calls:
+        if k == key:
+            return obj
+    raise KeyError(f"Key {key} not found in calls: {calls}")
+
+
+def _kwargs_for(calls, key):
+    for k, _obj, kw in calls:
+        if k == key:
+            return kw
+    raise KeyError(f"Key {key} not found in calls: {calls}")
+
+
+def test_log_rerun_data_envtransition_scalars_and_image(mock_rerun):
+    vu, calls = mock_rerun
+
+    # Build EnvTransition dict
+    obs = {
+        f"{OBS_STATE}.temperature": np.float32(25.0),
+        # CHW image should be converted to HWC for rr.Image
+        "observation.camera": np.zeros((3, 10, 20), dtype=np.uint8),
+    }
+    act = {
+        "action.throttle": 0.7,
+        # 1D array should log individual Scalars with suffix _i
+        "action.vector": np.array([1.0, 2.0], dtype=np.float32),
+    }
+    transition = {
+        TransitionKey.OBSERVATION: obs,
+        TransitionKey.ACTION: act,
+    }
+
+    # Extract observation and action data from transition like in the real call sites
+    obs_data = transition.get(TransitionKey.OBSERVATION, {})
+    action_data = transition.get(TransitionKey.ACTION, {})
+    vu.log_rerun_data(observation=obs_data, action=action_data)
+
+    # We expect:
+    # - observation.state.temperature -> Scalars
+    # - observation.camera -> Image (HWC) with static=True
+    # - action.throttle -> Scalars
+    # - action.vector_0, action.vector_1 -> Scalars
+    expected_keys = {
+        f"{OBS_STATE}.temperature",
+        "observation.camera",
+        "action.throttle",
+        "action.vector_0",
+        "action.vector_1",
+    }
+    assert set(_keys(calls)) == expected_keys
+
+    # Check scalar types and values
+    temp_obj = _obj_for(calls, f"{OBS_STATE}.temperature")
+    assert type(temp_obj).__name__ == "DummyScalar"
+    assert temp_obj.value == pytest.approx(25.0)
+
+    throttle_obj = _obj_for(calls, "action.throttle")
+    assert type(throttle_obj).__name__ == "DummyScalar"
+    assert throttle_obj.value == pytest.approx(0.7)
+
+    v0 = _obj_for(calls, "action.vector_0")
+    v1 = _obj_for(calls, "action.vector_1")
+    assert type(v0).__name__ == "DummyScalar"
+    assert type(v1).__name__ == "DummyScalar"
+    assert v0.value == pytest.approx(1.0)
+    assert v1.value == pytest.approx(2.0)
+
+    # Check image handling: CHW -> HWC
+    img_obj = _obj_for(calls, "observation.camera")
+    assert type(img_obj).__name__ == "DummyImage"
+    assert img_obj.arr.shape == (10, 20, 3)  # transposed
+    assert _kwargs_for(calls, "observation.camera").get("static", False) is True  # static=True for images
+
+
+def test_log_rerun_data_plain_list_ordering_and_prefixes(mock_rerun):
+    vu, calls = mock_rerun
+
+    # First dict without prefixes treated as observation
+    # Second dict without prefixes treated as action
+    obs_plain = {
+        "temp": 1.5,
+        # Already HWC image => should stay as-is
+        "img": np.zeros((5, 6, 3), dtype=np.uint8),
+        "none": None,  # should be skipped
+    }
+    act_plain = {
+        "throttle": 0.3,
+        "vec": np.array([9, 8, 7], dtype=np.float32),
+    }
+
+    # Extract observation and action data from list like the old function logic did
+    # First dict was treated as observation, second as action
+    vu.log_rerun_data(observation=obs_plain, action=act_plain)
+
+    # Expected keys with auto-prefixes
+    expected = {
+        "observation.temp",
+        "observation.img",
+        "action.throttle",
+        "action.vec_0",
+        "action.vec_1",
+        "action.vec_2",
+    }
+    logged = set(_keys(calls))
+    assert logged == expected
+
+    # Scalars
+    t = _obj_for(calls, "observation.temp")
+    assert type(t).__name__ == "DummyScalar"
+    assert t.value == pytest.approx(1.5)
+
+    throttle = _obj_for(calls, "action.throttle")
+    assert type(throttle).__name__ == "DummyScalar"
+    assert throttle.value == pytest.approx(0.3)
+
+    # Image stays HWC
+    img = _obj_for(calls, "observation.img")
+    assert type(img).__name__ == "DummyImage"
+    assert img.arr.shape == (5, 6, 3)
+    assert _kwargs_for(calls, "observation.img").get("static", False) is True
+
+    # Vectors
+    for i, val in enumerate([9, 8, 7]):
+        o = _obj_for(calls, f"action.vec_{i}")
+        assert type(o).__name__ == "DummyScalar"
+        assert o.value == pytest.approx(val)
+
+
+def test_log_rerun_data_kwargs_only(mock_rerun):
+    vu, calls = mock_rerun
+
+    vu.log_rerun_data(
+        observation={"observation.temp": 10.0, "observation.gray": np.zeros((8, 8, 1), dtype=np.uint8)},
+        action={"action.a": 1.0},
+    )
+
+    keys = set(_keys(calls))
+    assert "observation.temp" in keys
+    assert "observation.gray" in keys
+    assert "action.a" in keys
+
+    temp = _obj_for(calls, "observation.temp")
+    assert type(temp).__name__ == "DummyScalar"
+    assert temp.value == pytest.approx(10.0)
+
+    img = _obj_for(calls, "observation.gray")
+    assert type(img).__name__ == "DummyImage"
+    assert img.arr.shape == (8, 8, 1)  # remains HWC
+    assert _kwargs_for(calls, "observation.gray").get("static", False) is True
+
+    a = _obj_for(calls, "action.a")
+    assert type(a).__name__ == "DummyScalar"
+    assert a.value == pytest.approx(1.0)
diff --git a/lerobot_patches/apply.sh b/lerobot_patches/apply.sh
new file mode 100644
index 0000000000000000000000000000000000000000..9ab7203721da2dceb22c8860435b6ebb87588cea
--- /dev/null
+++ b/lerobot_patches/apply.sh
@@ -0,0 +1,22 @@
+#!/bin/bash
+set -e
+BASE="${1:?Usage: ./apply.sh /path/to/site-packages}"
+SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
+
+if [ -d "$BASE/lerobot" ]; then
+    TARGET="$BASE/lerobot"
+elif [ -d "$BASE/src/lerobot" ]; then
+    TARGET="$BASE/src/lerobot"
+else
+    echo "Error: Cannot find lerobot in $BASE"
+    exit 1
+fi
+
+echo "Patching $TARGET..."
+cp "$SCRIPT_DIR/src/lerobot/configs/train.py" "$TARGET/configs/train.py"
+cp "$SCRIPT_DIR/src/lerobot/datasets/factory.py" "$TARGET/datasets/factory.py"
+cp "$SCRIPT_DIR/src/lerobot/scripts/lerobot_train.py" "$TARGET/scripts/lerobot_train.py"
+cp "$SCRIPT_DIR/src/lerobot/scripts/lerobot_record.py" "$TARGET/scripts/lerobot_record.py"
+cp "$SCRIPT_DIR/src/lerobot/robots/so_follower/so_follower.py" "$TARGET/robots/so_follower/so_follower.py"
+cp "$SCRIPT_DIR/src/lerobot/policies/pi05/modeling_pi05.py" "$TARGET/policies/pi05/modeling_pi05.py"
+echo "Done"
diff --git a/lerobot_patches/src/lerobot/configs/train.py b/lerobot_patches/src/lerobot/configs/train.py
new file mode 100644
index 0000000000000000000000000000000000000000..8c6b0809a0f2fe56b269d0e161b9b51740b2ae09
--- /dev/null
+++ b/lerobot_patches/src/lerobot/configs/train.py
@@ -0,0 +1,228 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import builtins
+import datetime as dt
+import os
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import Any
+
+import draccus
+from huggingface_hub import hf_hub_download
+from huggingface_hub.errors import HfHubHTTPError
+
+from lerobot import envs
+from lerobot.configs import parser
+from lerobot.configs.default import DatasetConfig, EvalConfig, PeftConfig, WandBConfig
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.optim import OptimizerConfig
+from lerobot.optim.schedulers import LRSchedulerConfig
+from lerobot.utils.hub import HubMixin
+
+TRAIN_CONFIG_NAME = "train_config.json"
+
+
+@dataclass
+class TrainPipelineConfig(HubMixin):
+    dataset: DatasetConfig
+    env: envs.EnvConfig | None = None
+    policy: PreTrainedConfig | None = None
+    # Set `dir` to where you would like to save all of the run outputs. If you run another training session
+    # with the same value for `dir` its contents will be overwritten unless you set `resume` to true.
+    output_dir: Path | None = None
+    job_name: str | None = None
+    # Set `resume` to true to resume a previous run. In order for this to work, you will need to make sure
+    # `dir` is the directory of an existing run with at least one checkpoint in it.
+    # Note that when resuming a run, the default behavior is to use the configuration from the checkpoint,
+    # regardless of what's provided with the training command at the time of resumption.
+    resume: bool = False
+    # `seed` is used for training (eg: model initialization, dataset shuffling)
+    # AND for the evaluation environments.
+    seed: int | None = 1000
+    # Number of workers for the dataloader.
+    num_workers: int = 4
+    batch_size: int = 8
+    steps: int = 100_000
+    eval_freq: int = 20_000
+    log_freq: int = 200
+    tolerance_s: float = 1e-4
+    save_checkpoint: bool = True
+    # Checkpoint is saved every `save_freq` training iterations and after the last training step.
+    save_freq: int = 20_000
+    # Stop training early after this many steps (for local testing).
+    # The LR scheduler still uses `steps` for its schedule shape.
+    early_stop_steps: int | None = None
+    use_policy_training_preset: bool = True
+    optimizer: OptimizerConfig | None = None
+    scheduler: LRSchedulerConfig | None = None
+    eval: EvalConfig = field(default_factory=EvalConfig)
+    wandb: WandBConfig = field(default_factory=WandBConfig)
+    peft: PeftConfig | None = None
+
+    # RA-BC (Reward-Aligned Behavior Cloning) parameters
+    use_rabc: bool = False  # Enable reward-weighted training
+    rabc_progress_path: str | None = None  # Path to precomputed SARM progress parquet file
+    rabc_kappa: float = 0.01  # Hard threshold for high-quality samples
+    rabc_epsilon: float = 1e-6  # Small constant for numerical stability
+    rabc_head_mode: str | None = "sparse"  # For dual-head models: "sparse" or "dense"
+
+    # Rename map for the observation to override the image and state keys
+    rename_map: dict[str, str] = field(default_factory=dict)
+    checkpoint_path: Path | None = field(init=False, default=None)
+
+    def validate(self) -> None:
+        # HACK: We parse again the cli args here to get the pretrained paths if there was some.
+        policy_path = parser.get_path_arg("policy")
+        if policy_path:
+            # Only load the policy config
+            cli_overrides = parser.get_cli_overrides("policy")
+            self.policy = PreTrainedConfig.from_pretrained(policy_path, cli_overrides=cli_overrides)
+            self.policy.pretrained_path = Path(policy_path)
+        elif self.resume:
+            # The entire train config is already loaded, we just need to get the checkpoint dir
+            config_path = parser.parse_arg("config_path")
+            if not config_path:
+                raise ValueError(
+                    f"A config_path is expected when resuming a run. Please specify path to {TRAIN_CONFIG_NAME}"
+                )
+
+            if not Path(config_path).resolve().exists():
+                raise NotADirectoryError(
+                    f"{config_path=} is expected to be a local path. "
+                    "Resuming from the hub is not supported for now."
+                )
+
+            policy_dir = Path(config_path).parent
+            if self.policy is not None:
+                self.policy.pretrained_path = policy_dir
+            self.checkpoint_path = policy_dir.parent
+
+        if self.policy is None:
+            raise ValueError(
+                "Policy is not configured. Please specify a pretrained policy with `--policy.path`."
+            )
+
+        if not self.job_name:
+            if self.env is None:
+                self.job_name = f"{self.policy.type}"
+            else:
+                self.job_name = f"{self.env.type}_{self.policy.type}"
+
+        if not self.resume and isinstance(self.output_dir, Path) and self.output_dir.is_dir():
+            raise FileExistsError(
+                f"Output directory {self.output_dir} already exists and resume is {self.resume}. "
+                f"Please change your output directory so that {self.output_dir} is not overwritten."
+            )
+        elif not self.output_dir:
+            now = dt.datetime.now()
+            train_dir = f"{now:%Y-%m-%d}/{now:%H-%M-%S}_{self.job_name}"
+            self.output_dir = Path("outputs/train") / train_dir
+
+        if isinstance(self.dataset.repo_id, list):
+            raise NotImplementedError("LeRobotMultiDataset is not currently implemented.")
+
+        if not self.use_policy_training_preset and (self.optimizer is None or self.scheduler is None):
+            raise ValueError("Optimizer and Scheduler must be set when the policy presets are not used.")
+        elif self.use_policy_training_preset and not self.resume:
+            self.optimizer = self.policy.get_optimizer_preset()
+            self.scheduler = self.policy.get_scheduler_preset()
+
+        if self.policy.push_to_hub and not self.policy.repo_id:
+            raise ValueError(
+                "'policy.repo_id' argument missing. Please specify it to push the model to the hub."
+            )
+
+        if not self.policy.push_to_hub:
+            raise ValueError(
+                "'policy.push_to_hub' is not enabled. Checkpoints must be pushed to the hub. "
+                "Set --policy.push_to_hub=true and --policy.repo_id=<your_repo>."
+            )
+
+        if not self.wandb.enable:
+            raise ValueError(
+                "WandB logging is not enabled. Training must be monitored. "
+                "Set --wandb.enable=true --wandb.project=<your_project>."
+            )
+
+        if self.use_rabc and not self.rabc_progress_path:
+            # Auto-detect from dataset path
+            repo_id = self.dataset.repo_id
+            if self.dataset.root:
+                self.rabc_progress_path = str(Path(self.dataset.root) / "sarm_progress.parquet")
+            else:
+                self.rabc_progress_path = f"hf://datasets/{repo_id}/sarm_progress.parquet"
+
+    @classmethod
+    def __get_path_fields__(cls) -> list[str]:
+        """This enables the parser to load config from the policy using `--policy.path=local/dir`"""
+        return ["policy"]
+
+    def to_dict(self) -> dict[str, Any]:
+        return draccus.encode(self)  # type: ignore[no-any-return]  # because of the third-party library draccus uses Any as the return type
+
+    def _save_pretrained(self, save_directory: Path) -> None:
+        with open(save_directory / TRAIN_CONFIG_NAME, "w") as f, draccus.config_type("json"):
+            draccus.dump(self, f, indent=4)
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type["TrainPipelineConfig"],
+        pretrained_name_or_path: str | Path,
+        *,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict[Any, Any] | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        **kwargs: Any,
+    ) -> "TrainPipelineConfig":
+        model_id = str(pretrained_name_or_path)
+        config_file: str | None = None
+        if Path(model_id).is_dir():
+            if TRAIN_CONFIG_NAME in os.listdir(model_id):
+                config_file = os.path.join(model_id, TRAIN_CONFIG_NAME)
+            else:
+                print(f"{TRAIN_CONFIG_NAME} not found in {Path(model_id).resolve()}")
+        elif Path(model_id).is_file():
+            config_file = model_id
+        else:
+            try:
+                config_file = hf_hub_download(
+                    repo_id=model_id,
+                    filename=TRAIN_CONFIG_NAME,
+                    revision=revision,
+                    cache_dir=cache_dir,
+                    force_download=force_download,
+                    proxies=proxies,
+                    resume_download=resume_download,
+                    token=token,
+                    local_files_only=local_files_only,
+                )
+            except HfHubHTTPError as e:
+                raise FileNotFoundError(
+                    f"{TRAIN_CONFIG_NAME} not found on the HuggingFace Hub in {model_id}"
+                ) from e
+
+        cli_args = kwargs.pop("cli_args", [])
+        with draccus.config_type("json"):
+            return draccus.parse(cls, config_file, args=cli_args)
+
+
+@dataclass(kw_only=True)
+class TrainRLServerPipelineConfig(TrainPipelineConfig):
+    # NOTE: In RL, we don't need an offline dataset
+    # TODO: Make `TrainPipelineConfig.dataset` optional
+    dataset: DatasetConfig | None = None  # type: ignore[assignment] # because the parent class has made it's type non-optional
diff --git a/lerobot_patches/src/lerobot/datasets/factory.py b/lerobot_patches/src/lerobot/datasets/factory.py
new file mode 100644
index 0000000000000000000000000000000000000000..8114dfb5ba5aa9772927a15d5cdb075e9d59e69f
--- /dev/null
+++ b/lerobot_patches/src/lerobot/datasets/factory.py
@@ -0,0 +1,153 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import logging
+from pprint import pformat
+
+import torch
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.datasets.lerobot_dataset import (
+    LeRobotDataset,
+    LeRobotDatasetMetadata,
+    MultiLeRobotDataset,
+)
+from lerobot.datasets.streaming_dataset import StreamingLeRobotDataset
+from lerobot.datasets.transforms import ImageTransforms
+from lerobot.utils.constants import ACTION, OBS_PREFIX, REWARD
+
+IMAGENET_STATS = {
+    "mean": [[[0.485]], [[0.456]], [[0.406]]],  # (c,1,1)
+    "std": [[[0.229]], [[0.224]], [[0.225]]],  # (c,1,1)
+}
+
+
+def resolve_delta_timestamps(
+    cfg: PreTrainedConfig, ds_meta: LeRobotDatasetMetadata
+) -> dict[str, list] | None:
+    """Resolves delta_timestamps by reading from the 'delta_indices' properties of the PreTrainedConfig.
+
+    Args:
+        cfg (PreTrainedConfig): The PreTrainedConfig to read delta_indices from.
+        ds_meta (LeRobotDatasetMetadata): The dataset from which features and fps are used to build
+            delta_timestamps against.
+
+    Returns:
+        dict[str, list] | None: A dictionary of delta_timestamps, e.g.:
+            {
+                "observation.state": [-0.04, -0.02, 0]
+                "observation.action": [-0.02, 0, 0.02]
+            }
+            returns `None` if the resulting dict is empty.
+    """
+    delta_timestamps = {}
+    for key in ds_meta.features:
+        if key == REWARD and cfg.reward_delta_indices is not None:
+            delta_timestamps[key] = [i / ds_meta.fps for i in cfg.reward_delta_indices]
+        if key == ACTION and cfg.action_delta_indices is not None:
+            delta_timestamps[key] = [i / ds_meta.fps for i in cfg.action_delta_indices]
+        if key.startswith(OBS_PREFIX) and cfg.observation_delta_indices is not None:
+            delta_timestamps[key] = [i / ds_meta.fps for i in cfg.observation_delta_indices]
+
+    if len(delta_timestamps) == 0:
+        delta_timestamps = None
+
+    return delta_timestamps
+
+
+def make_dataset(cfg: TrainPipelineConfig) -> LeRobotDataset | MultiLeRobotDataset:
+    """Handles the logic of setting up delta timestamps and image transforms before creating a dataset.
+
+    Args:
+        cfg (TrainPipelineConfig): A TrainPipelineConfig config which contains a DatasetConfig and a PreTrainedConfig.
+
+    Raises:
+        NotImplementedError: The MultiLeRobotDataset is currently deactivated.
+
+    Returns:
+        LeRobotDataset | MultiLeRobotDataset
+    """
+    image_transforms = (
+        ImageTransforms(cfg.dataset.image_transforms) if cfg.dataset.image_transforms.enable else None
+    )
+
+    # Support SO100Dataset via repo_id starting with "so100:"
+    # Format: "so100:/path/to/data_root:/path/to/index.json:/path/to/stats.json"
+    if isinstance(cfg.dataset.repo_id, str) and cfg.dataset.repo_id.startswith("so100:"):
+        from so100_dataset import SO100Dataset
+        parts = cfg.dataset.repo_id.split(":")
+        data_root = parts[1]
+        index_path = parts[2] if len(parts) > 2 else None
+        stats_path = parts[3] if len(parts) > 3 else None
+        dataset = SO100Dataset(
+            data_root=data_root,
+            index_path=index_path,
+            stats_path=stats_path,
+            video_backend=cfg.dataset.video_backend or "pyav",
+            chunk_size=cfg.policy.chunk_size if hasattr(cfg.policy, "chunk_size") else 50,
+            image_transforms=image_transforms,
+        )
+        if cfg.dataset.use_imagenet_stats:
+            for key in dataset.meta.camera_keys:
+                for stats_type, stats in IMAGENET_STATS.items():
+                    dataset.meta.stats[key] = dataset.meta.stats.get(key, {})
+                    dataset.meta.stats[key][stats_type] = torch.tensor(stats, dtype=torch.float32)
+        return dataset
+
+    if isinstance(cfg.dataset.repo_id, str):
+        ds_meta = LeRobotDatasetMetadata(
+            cfg.dataset.repo_id, root=cfg.dataset.root, revision=cfg.dataset.revision
+        )
+        delta_timestamps = resolve_delta_timestamps(cfg.policy, ds_meta)
+        if not cfg.dataset.streaming:
+            dataset = LeRobotDataset(
+                cfg.dataset.repo_id,
+                root=cfg.dataset.root,
+                episodes=cfg.dataset.episodes,
+                delta_timestamps=delta_timestamps,
+                image_transforms=image_transforms,
+                revision=cfg.dataset.revision,
+                video_backend=cfg.dataset.video_backend,
+                tolerance_s=cfg.tolerance_s,
+            )
+        else:
+            dataset = StreamingLeRobotDataset(
+                cfg.dataset.repo_id,
+                root=cfg.dataset.root,
+                episodes=cfg.dataset.episodes,
+                delta_timestamps=delta_timestamps,
+                image_transforms=image_transforms,
+                revision=cfg.dataset.revision,
+                max_num_shards=cfg.num_workers,
+                tolerance_s=cfg.tolerance_s,
+            )
+    else:
+        dataset = MultiLeRobotDataset(
+            cfg.dataset.repo_id,
+            image_transforms=image_transforms,
+            video_backend=cfg.dataset.video_backend,
+        )
+        logging.info(
+            "Multiple datasets were provided. Applied the following index mapping to the provided datasets: "
+            f"{pformat(dataset.repo_id_to_index, indent=2)}"
+        )
+
+    if cfg.dataset.use_imagenet_stats:
+        for key in dataset.meta.camera_keys:
+            for stats_type, stats in IMAGENET_STATS.items():
+                dataset.meta.stats[key][stats_type] = torch.tensor(stats, dtype=torch.float32)
+
+    return dataset
diff --git a/lerobot_patches/src/lerobot/policies/pi05/modeling_pi05.py b/lerobot_patches/src/lerobot/policies/pi05/modeling_pi05.py
new file mode 100644
index 0000000000000000000000000000000000000000..1eb34b40b23439aabcc98e9f1f3256abc8993814
--- /dev/null
+++ b/lerobot_patches/src/lerobot/policies/pi05/modeling_pi05.py
@@ -0,0 +1,1277 @@
+#!/usr/bin/env python
+
+# Copyright 2025 Physical Intelligence and The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import builtins
+import logging
+import math
+from collections import deque
+from pathlib import Path
+from typing import TYPE_CHECKING, Literal, TypedDict
+
+import torch
+import torch.nn.functional as F  # noqa: N812
+from torch import Tensor, nn
+from typing_extensions import Unpack
+
+from lerobot.utils.import_utils import _transformers_available
+
+# Conditional import for type checking and lazy loading
+if TYPE_CHECKING or _transformers_available:
+    from transformers.models.auto import CONFIG_MAPPING
+    from transformers.models.gemma import modeling_gemma
+    from transformers.models.gemma.modeling_gemma import GemmaForCausalLM
+    from transformers.models.paligemma.modeling_paligemma import PaliGemmaForConditionalGeneration
+else:
+    CONFIG_MAPPING = None
+    modeling_gemma = None
+    GemmaForCausalLM = None
+    PaliGemmaForConditionalGeneration = None
+
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.policies.pi05.configuration_pi05 import DEFAULT_IMAGE_SIZE, PI05Config
+from lerobot.policies.pretrained import PreTrainedPolicy, T
+from lerobot.policies.rtc.modeling_rtc import RTCProcessor
+from lerobot.utils.constants import (
+    ACTION,
+    OBS_LANGUAGE_ATTENTION_MASK,
+    OBS_LANGUAGE_TOKENS,
+    OPENPI_ATTENTION_MASK_VALUE,
+)
+
+
+class ActionSelectKwargs(TypedDict, total=False):
+    inference_delay: int | None
+    prev_chunk_left_over: Tensor | None
+    execution_horizon: int | None
+
+
+def get_safe_dtype(target_dtype, device_type):
+    """Get a safe dtype for the given device type."""
+    if device_type == "mps" and target_dtype == torch.float64:
+        return torch.float32
+    if device_type == "cpu":
+        # CPU doesn't support bfloat16, use float32 instead
+        if target_dtype == torch.bfloat16:
+            return torch.float32
+        if target_dtype == torch.float64:
+            return torch.float64
+    return target_dtype
+
+
+def create_sinusoidal_pos_embedding(  # see openpi `create_sinusoidal_pos_embedding` (exact copy)
+    time: torch.Tensor, dimension: int, min_period: float, max_period: float, device="cpu"
+) -> Tensor:
+    """Computes sine-cosine positional embedding vectors for scalar positions."""
+    if dimension % 2 != 0:
+        raise ValueError(f"dimension ({dimension}) must be divisible by 2")
+
+    if time.ndim != 1:
+        raise ValueError("The time tensor is expected to be of shape `(batch_size, )`.")
+
+    dtype = get_safe_dtype(torch.float64, device.type)
+    fraction = torch.linspace(0.0, 1.0, dimension // 2, dtype=dtype, device=device)
+    period = min_period * (max_period / min_period) ** fraction
+
+    # Compute the outer product
+    scaling_factor = 1.0 / period * 2 * math.pi
+    sin_input = scaling_factor[None, :] * time[:, None]
+    return torch.cat([torch.sin(sin_input), torch.cos(sin_input)], dim=1)
+
+
+def sample_beta(alpha, beta, bsize, device):  # see openpi `sample_beta` (exact copy)
+    alpha_t = torch.as_tensor(alpha, dtype=torch.float32, device=device)
+    beta_t = torch.as_tensor(beta, dtype=torch.float32, device=device)
+    dist = torch.distributions.Beta(alpha_t, beta_t)
+    return dist.sample((bsize,))
+
+
+def make_att_2d_masks(pad_masks, att_masks):  # see openpi `make_att_2d_masks` (exact copy)
+    """Copied from big_vision.
+
+    Tokens can attend to valid inputs tokens which have a cumulative mask_ar
+    smaller or equal to theirs. This way `mask_ar` int[B, N] can be used to
+    setup several types of attention, for example:
+
+      [[1 1 1 1 1 1]]: pure causal attention.
+
+      [[0 0 0 1 1 1]]: prefix-lm attention. The first 3 tokens can attend between
+          themselves and the last 3 tokens have a causal attention. The first
+          entry could also be a 1 without changing behaviour.
+
+      [[1 0 1 0 1 0 0 1 0 0]]: causal attention between 4 blocks. Tokens of a
+          block can attend all previous blocks and all tokens on the same block.
+
+    Args:
+      input_mask: bool[B, N] true if its part of the input, false if padding.
+      mask_ar: int32[B, N] mask that's 1 where previous tokens cannot depend on
+        it and 0 where it shares the same attention mask as the previous token.
+    """
+    if att_masks.ndim != 2:
+        raise ValueError(att_masks.ndim)
+    if pad_masks.ndim != 2:
+        raise ValueError(pad_masks.ndim)
+
+    cumsum = torch.cumsum(att_masks, dim=1)
+    att_2d_masks = cumsum[:, None, :] <= cumsum[:, :, None]
+    pad_2d_masks = pad_masks[:, None, :] * pad_masks[:, :, None]
+    return att_2d_masks & pad_2d_masks
+
+
+def pad_vector(vector, new_dim):
+    """Pad the last dimension of a vector to new_dim with zeros.
+
+    Can be (batch_size x sequence_length x features_dimension)
+    or (batch_size x features_dimension)
+    """
+    if vector.shape[-1] >= new_dim:
+        return vector
+    return F.pad(vector, (0, new_dim - vector.shape[-1]))
+
+
+def resize_with_pad_torch(  # see openpi `resize_with_pad_torch` (exact copy)
+    images: torch.Tensor,
+    height: int,
+    width: int,
+    mode: str = "bilinear",
+) -> torch.Tensor:
+    """PyTorch version of resize_with_pad. Resizes an image to a target height and width without distortion
+    by padding with black. If the image is float32, it must be in the range [-1, 1].
+
+    Args:
+        images: Tensor of shape [*b, h, w, c] or [*b, c, h, w]
+        height: Target height
+        width: Target width
+        mode: Interpolation mode ('bilinear', 'nearest', etc.)
+
+    Returns:
+        Resized and padded tensor with same shape format as input
+    """
+    # Check if input is in channels-last format [*b, h, w, c] or channels-first [*b, c, h, w]
+    if images.shape[-1] <= 4:  # Assume channels-last format
+        channels_last = True
+        if images.dim() == 3:
+            images = images.unsqueeze(0)  # Add batch dimension
+        images = images.permute(0, 3, 1, 2)  # [b, h, w, c] -> [b, c, h, w]
+    else:
+        channels_last = False
+        if images.dim() == 3:
+            images = images.unsqueeze(0)  # Add batch dimension
+
+    batch_size, channels, cur_height, cur_width = images.shape
+
+    # Calculate resize ratio
+    ratio = max(cur_width / width, cur_height / height)
+    resized_height = int(cur_height / ratio)
+    resized_width = int(cur_width / ratio)
+
+    # Resize
+    resized_images = F.interpolate(
+        images,
+        size=(resized_height, resized_width),
+        mode=mode,
+        align_corners=False if mode == "bilinear" else None,
+    )
+
+    # Handle dtype-specific clipping
+    if images.dtype == torch.uint8:
+        resized_images = torch.round(resized_images).clamp(0, 255).to(torch.uint8)
+    elif images.dtype == torch.float32:
+        resized_images = resized_images.clamp(-1.0, 1.0)
+    else:
+        raise ValueError(f"Unsupported image dtype: {images.dtype}")
+
+    # Calculate padding
+    pad_h0, remainder_h = divmod(height - resized_height, 2)
+    pad_h1 = pad_h0 + remainder_h
+    pad_w0, remainder_w = divmod(width - resized_width, 2)
+    pad_w1 = pad_w0 + remainder_w
+
+    # Pad
+    constant_value = 0 if images.dtype == torch.uint8 else -1.0
+    padded_images = F.pad(
+        resized_images,
+        (pad_w0, pad_w1, pad_h0, pad_h1),  # left, right, top, bottom
+        mode="constant",
+        value=constant_value,
+    )
+
+    # Convert back to original format if needed
+    if channels_last:
+        padded_images = padded_images.permute(0, 2, 3, 1)  # [b, c, h, w] -> [b, h, w, c]
+
+    return padded_images
+
+
+# Define the complete layer computation function for gradient checkpointing
+def compute_layer_complete(
+    layer_idx, inputs_embeds, attention_mask, position_ids, adarms_cond, paligemma, gemma_expert
+):
+    models = [paligemma.language_model, gemma_expert.model]
+    query_states = []
+    key_states = []
+    value_states = []
+    gates = []
+    for i, hidden_states in enumerate(inputs_embeds):
+        layer = models[i].layers[layer_idx]
+        hidden_states, gate = layer.input_layernorm(hidden_states, cond=adarms_cond[i])  # noqa: PLW2901
+        gates.append(gate)
+        input_shape = hidden_states.shape[:-1]
+        hidden_shape = (*input_shape, -1, layer.self_attn.head_dim)
+        query_state = layer.self_attn.q_proj(hidden_states).view(hidden_shape).transpose(1, 2)
+        key_state = layer.self_attn.k_proj(hidden_states).view(hidden_shape).transpose(1, 2)
+        value_state = layer.self_attn.v_proj(hidden_states).view(hidden_shape).transpose(1, 2)
+        query_states.append(query_state)
+        key_states.append(key_state)
+        value_states.append(value_state)
+    # Concatenate and process attention
+    query_states = torch.cat(query_states, dim=2)
+    key_states = torch.cat(key_states, dim=2)
+    value_states = torch.cat(value_states, dim=2)
+    dummy_tensor = torch.zeros(
+        query_states.shape[0],
+        query_states.shape[2],
+        query_states.shape[-1],
+        device=query_states.device,
+        dtype=query_states.dtype,
+    )
+    cos, sin = paligemma.model.language_model.rotary_emb(dummy_tensor, position_ids)
+    query_states, key_states = modeling_gemma.apply_rotary_pos_emb(
+        query_states, key_states, cos, sin, unsqueeze_dim=1
+    )
+    batch_size = query_states.shape[0]
+    scaling = paligemma.language_model.layers[layer_idx].self_attn.scaling
+    # Attention computation
+    att_output, _ = modeling_gemma.eager_attention_forward(
+        paligemma.language_model.layers[layer_idx].self_attn,
+        query_states,
+        key_states,
+        value_states,
+        attention_mask,
+        scaling,
+    )
+    # Get head_dim from the current layer, not from the model
+    head_dim = paligemma.language_model.layers[layer_idx].self_attn.head_dim
+    att_output = att_output.reshape(batch_size, -1, 1 * 8 * head_dim)
+    # Process layer outputs
+    outputs_embeds = []
+    start_pos = 0
+    for i, hidden_states in enumerate(inputs_embeds):
+        layer = models[i].layers[layer_idx]
+        end_pos = start_pos + hidden_states.shape[1]
+        if att_output.dtype != layer.self_attn.o_proj.weight.dtype:
+            att_output = att_output.to(layer.self_attn.o_proj.weight.dtype)
+        out_emb = layer.self_attn.o_proj(att_output[:, start_pos:end_pos])
+        # first residual
+        out_emb = modeling_gemma._gated_residual(hidden_states, out_emb, gates[i])  # noqa: SLF001
+        after_first_residual = out_emb.clone()
+        out_emb, gate = layer.post_attention_layernorm(out_emb, cond=adarms_cond[i])
+        # Convert to bfloat16 if the next layer (mlp) uses bfloat16
+        if layer.mlp.up_proj.weight.dtype == torch.bfloat16:
+            out_emb = out_emb.to(dtype=torch.bfloat16)
+        out_emb = layer.mlp(out_emb)
+        # second residual
+        out_emb = modeling_gemma._gated_residual(after_first_residual, out_emb, gate)  # noqa: SLF001
+        outputs_embeds.append(out_emb)
+        start_pos = end_pos
+    return outputs_embeds
+
+
+class GemmaConfig:  # see openpi `gemma.py: Config`
+    """Configuration for Gemma model variants."""
+
+    def __init__(self, width, depth, mlp_dim, num_heads, num_kv_heads, head_dim):
+        self.width = width
+        self.depth = depth
+        self.mlp_dim = mlp_dim
+        self.num_heads = num_heads
+        self.num_kv_heads = num_kv_heads
+        self.head_dim = head_dim
+
+
+def get_gemma_config(variant: str) -> GemmaConfig:  # see openpi `gemma.py: get_config`
+    """Returns config for specified gemma variant."""
+    if variant == "gemma_300m":
+        return GemmaConfig(
+            width=1024,
+            depth=18,
+            mlp_dim=4096,
+            num_heads=8,
+            num_kv_heads=1,
+            head_dim=256,
+        )
+    elif variant == "gemma_2b":
+        return GemmaConfig(
+            width=2048,
+            depth=18,
+            mlp_dim=16_384,
+            num_heads=8,
+            num_kv_heads=1,
+            head_dim=256,
+        )
+    else:
+        raise ValueError(f"Unknown variant: {variant}")
+
+
+class PaliGemmaWithExpertModel(
+    nn.Module
+):  # see openpi `gemma_pytorch.py: PaliGemmaWithExpertModel` this class is almost a exact copy of PaliGemmaWithExpertModel in openpi
+    """PaliGemma model with action expert for PI05."""
+
+    def __init__(
+        self,
+        vlm_config,
+        action_expert_config,
+        use_adarms=None,
+        precision: Literal["bfloat16", "float32"] = "bfloat16",
+        image_size: int = DEFAULT_IMAGE_SIZE,
+        freeze_vision_encoder: bool = False,
+        train_expert_only: bool = False,
+    ):
+        if use_adarms is None:
+            use_adarms = [False, False]
+        super().__init__()
+        self.freeze_vision_encoder = freeze_vision_encoder
+        self.train_expert_only = train_expert_only
+
+        vlm_config_hf = CONFIG_MAPPING["paligemma"]()
+        vlm_config_hf._vocab_size = 257152  # noqa: SLF001
+        vlm_config_hf.image_token_index = 257152
+        vlm_config_hf.text_config.hidden_size = vlm_config.width
+        vlm_config_hf.text_config.intermediate_size = vlm_config.mlp_dim
+        vlm_config_hf.text_config.num_attention_heads = vlm_config.num_heads
+        vlm_config_hf.text_config.head_dim = vlm_config.head_dim
+        vlm_config_hf.text_config.num_hidden_layers = vlm_config.depth
+        vlm_config_hf.text_config.num_key_value_heads = vlm_config.num_kv_heads
+        vlm_config_hf.text_config.hidden_activation = "gelu_pytorch_tanh"
+        vlm_config_hf.text_config.torch_dtype = "float32"
+        vlm_config_hf.text_config.vocab_size = 257152
+        vlm_config_hf.text_config.use_adarms = use_adarms[0]
+        vlm_config_hf.text_config.adarms_cond_dim = vlm_config.width if use_adarms[0] else None
+        vlm_config_hf.vision_config.image_size = image_size
+        vlm_config_hf.vision_config.intermediate_size = 4304
+        vlm_config_hf.vision_config.projection_dim = 2048
+        vlm_config_hf.vision_config.projector_hidden_act = "gelu_fast"
+        vlm_config_hf.vision_config.torch_dtype = "float32"
+
+        action_expert_config_hf = CONFIG_MAPPING["gemma"](
+            head_dim=action_expert_config.head_dim,
+            hidden_size=action_expert_config.width,
+            intermediate_size=action_expert_config.mlp_dim,
+            num_attention_heads=action_expert_config.num_heads,
+            num_hidden_layers=action_expert_config.depth,
+            num_key_value_heads=action_expert_config.num_kv_heads,
+            vocab_size=257152,
+            hidden_activation="gelu_pytorch_tanh",
+            torch_dtype="float32",
+            use_adarms=use_adarms[1],
+            adarms_cond_dim=action_expert_config.width if use_adarms[1] else None,
+        )
+
+        self.paligemma = PaliGemmaForConditionalGeneration(config=vlm_config_hf)
+        self.gemma_expert = GemmaForCausalLM(config=action_expert_config_hf)
+        self.gemma_expert.model.embed_tokens = None
+
+        self.to_bfloat16_for_selected_params(precision)
+        self._set_requires_grad()
+
+    def to_bfloat16_for_selected_params(self, precision: Literal["bfloat16", "float32"] = "bfloat16"):
+        if precision == "bfloat16":
+            self.to(dtype=torch.bfloat16)
+        elif precision == "float32":
+            self.to(dtype=torch.float32)
+            return
+        else:
+            raise ValueError(f"Invalid precision: {precision}")
+
+        params_to_keep_float32 = [
+            "vision_tower.vision_model.embeddings.patch_embedding.weight",
+            "vision_tower.vision_model.embeddings.patch_embedding.bias",
+            "vision_tower.vision_model.embeddings.position_embedding.weight",
+            "input_layernorm",
+            "post_attention_layernorm",
+            "model.norm",
+        ]
+
+        for name, param in self.named_parameters():
+            if any(selector in name for selector in params_to_keep_float32):
+                param.data = param.data.to(dtype=torch.float32)
+
+    def _set_requires_grad(self):
+        if self.freeze_vision_encoder:
+            self.paligemma.vision_tower.eval()
+            for param in self.paligemma.vision_tower.parameters():
+                param.requires_grad = False
+        if self.train_expert_only:
+            self.paligemma.eval()
+            for param in self.paligemma.parameters():
+                param.requires_grad = False
+
+    def train(self, mode: bool = True):
+        super().train(mode)
+        if self.freeze_vision_encoder:
+            self.paligemma.vision_tower.eval()
+        if self.train_expert_only:
+            self.paligemma.eval()
+
+    def embed_image(self, image: torch.Tensor):
+        return self.paligemma.model.get_image_features(image)
+
+    def embed_language_tokens(self, tokens: torch.Tensor):
+        return self.paligemma.language_model.embed_tokens(tokens)
+
+    def forward(
+        self,
+        attention_mask: torch.Tensor | None = None,
+        position_ids: torch.LongTensor | None = None,
+        past_key_values: list[torch.FloatTensor] | None = None,
+        inputs_embeds: list[torch.FloatTensor] | None = None,
+        use_cache: bool | None = None,
+        adarms_cond: list[torch.Tensor] | None = None,
+    ):
+        if adarms_cond is None:
+            adarms_cond = [None, None]
+        if inputs_embeds[1] is None:
+            prefix_output = self.paligemma.language_model.forward(
+                inputs_embeds=inputs_embeds[0],
+                attention_mask=attention_mask,
+                position_ids=position_ids,
+                past_key_values=past_key_values,
+                use_cache=use_cache,
+                adarms_cond=adarms_cond[0] if adarms_cond is not None else None,
+            )
+            prefix_past_key_values = prefix_output.past_key_values
+            prefix_output = prefix_output.last_hidden_state
+            suffix_output = None
+        elif inputs_embeds[0] is None:
+            suffix_output = self.gemma_expert.model.forward(
+                inputs_embeds=inputs_embeds[1],
+                attention_mask=attention_mask,
+                position_ids=position_ids,
+                past_key_values=past_key_values,
+                use_cache=use_cache,
+                adarms_cond=adarms_cond[1] if adarms_cond is not None else None,
+            )
+            suffix_output = suffix_output.last_hidden_state
+            prefix_output = None
+            prefix_past_key_values = None
+        else:
+            models = [self.paligemma.language_model, self.gemma_expert.model]
+            num_layers = self.paligemma.config.text_config.num_hidden_layers
+
+            # Check if gradient checkpointing is enabled for any of the models
+            use_gradient_checkpointing = (
+                hasattr(self.gemma_expert.model, "gradient_checkpointing")
+                and self.gemma_expert.model.gradient_checkpointing
+                and self.training
+            ) or (hasattr(self, "gradient_checkpointing") and self.gradient_checkpointing and self.training)
+
+            # Process all layers with gradient checkpointing if enabled
+            for layer_idx in range(num_layers):
+                if use_gradient_checkpointing:
+                    inputs_embeds = torch.utils.checkpoint.checkpoint(
+                        compute_layer_complete,
+                        layer_idx,
+                        inputs_embeds,
+                        attention_mask,
+                        position_ids,
+                        adarms_cond,
+                        use_reentrant=False,
+                        preserve_rng_state=False,
+                        paligemma=self.paligemma,
+                        gemma_expert=self.gemma_expert,
+                    )
+                else:
+                    inputs_embeds = compute_layer_complete(
+                        layer_idx,
+                        inputs_embeds,
+                        attention_mask,
+                        position_ids,
+                        adarms_cond,
+                        paligemma=self.paligemma,
+                        gemma_expert=self.gemma_expert,
+                    )
+
+            # final norm
+            def compute_final_norms(inputs_embeds, adarms_cond):
+                outputs_embeds = []
+                for i, hidden_states in enumerate(inputs_embeds):
+                    out_emb, _ = models[i].norm(hidden_states, cond=adarms_cond[i])
+                    outputs_embeds.append(out_emb)
+                return outputs_embeds
+
+            # Apply gradient checkpointing to final norm if enabled
+            if use_gradient_checkpointing:
+                outputs_embeds = torch.utils.checkpoint.checkpoint(
+                    compute_final_norms,
+                    inputs_embeds,
+                    adarms_cond,
+                    use_reentrant=False,
+                    preserve_rng_state=False,
+                )
+            else:
+                outputs_embeds = compute_final_norms(inputs_embeds, adarms_cond)
+
+            prefix_output = outputs_embeds[0]
+            suffix_output = outputs_embeds[1]
+            prefix_past_key_values = None
+
+        return [prefix_output, suffix_output], prefix_past_key_values
+
+
+class PI05Pytorch(nn.Module):  # see openpi `PI0Pytorch`
+    """Core PI05 PyTorch model."""
+
+    def __init__(self, config: PI05Config, rtc_processor: RTCProcessor | None = None):
+        super().__init__()
+        self.config = config
+        self.rtc_processor = rtc_processor
+
+        paligemma_config = get_gemma_config(config.paligemma_variant)
+        action_expert_config = get_gemma_config(config.action_expert_variant)
+
+        if config.image_resolution[0] != config.image_resolution[1]:
+            raise ValueError(
+                f"PaliGemma expects square image resolution, invalid resolution: {config.image_resolution}"
+            )
+
+        self.paligemma_with_expert = PaliGemmaWithExpertModel(
+            paligemma_config,
+            action_expert_config,
+            use_adarms=[False, True],
+            precision=config.dtype,
+            image_size=config.image_resolution[0],
+            freeze_vision_encoder=config.freeze_vision_encoder,
+            train_expert_only=config.train_expert_only,
+        )
+
+        self.action_in_proj = nn.Linear(config.max_action_dim, action_expert_config.width)
+        self.action_out_proj = nn.Linear(action_expert_config.width, config.max_action_dim)
+
+        self.time_mlp_in = nn.Linear(action_expert_config.width, action_expert_config.width)
+        self.time_mlp_out = nn.Linear(action_expert_config.width, action_expert_config.width)
+
+        # Initialize gradient checkpointing flag
+        self.gradient_checkpointing_enabled = False
+
+        # Compile model if requested
+        if config.compile_model:
+            torch.set_float32_matmul_precision("high")
+            self.sample_actions = torch.compile(self.sample_actions, mode=config.compile_mode)
+            # Also compile the main forward pass used during training
+            self.forward = torch.compile(self.forward, mode=config.compile_mode)
+
+        msg = """An incorrect transformer version is used, please create an issue on https://github.com/huggingface/lerobot/issues"""
+
+        pass  # Version check disabled
+
+    def gradient_checkpointing_enable(self):
+        """Enable gradient checkpointing for memory optimization."""
+        self.gradient_checkpointing_enabled = True
+        self.paligemma_with_expert.paligemma.language_model.gradient_checkpointing = True
+        self.paligemma_with_expert.paligemma.vision_tower.gradient_checkpointing = True
+        self.paligemma_with_expert.gemma_expert.model.gradient_checkpointing = True
+        logging.info("Enabled gradient checkpointing for PI05Pytorch model")
+
+    def gradient_checkpointing_disable(self):
+        """Disable gradient checkpointing."""
+        self.gradient_checkpointing_enabled = False
+        self.paligemma_with_expert.paligemma.language_model.gradient_checkpointing = False
+        self.paligemma_with_expert.paligemma.vision_tower.gradient_checkpointing = False
+        self.paligemma_with_expert.gemma_expert.model.gradient_checkpointing = False
+        logging.info("Disabled gradient checkpointing for PI05Pytorch model")
+
+    def _rtc_enabled(self):
+        return self.config.rtc_config is not None and self.config.rtc_config.enabled
+
+    def _apply_checkpoint(self, func, *args, **kwargs):
+        """Helper method to apply gradient checkpointing if enabled."""
+        if self.gradient_checkpointing_enabled and self.training:
+            return torch.utils.checkpoint.checkpoint(
+                func, *args, use_reentrant=False, preserve_rng_state=False, **kwargs
+            )
+        return func(*args, **kwargs)
+
+    def _prepare_attention_masks_4d(self, att_2d_masks):
+        """Helper method to prepare 4D attention masks for transformer."""
+        att_2d_masks_4d = att_2d_masks[:, None, :, :]
+        return torch.where(att_2d_masks_4d, 0.0, OPENPI_ATTENTION_MASK_VALUE)
+
+    def sample_noise(self, shape, device):
+        return torch.normal(
+            mean=0.0,
+            std=1.0,
+            size=shape,
+            dtype=torch.float32,
+            device=device,
+        )
+
+    def sample_time(self, bsize, device):
+        time_beta = sample_beta(
+            self.config.time_sampling_beta_alpha, self.config.time_sampling_beta_beta, bsize, device
+        )
+        time = time_beta * self.config.time_sampling_scale + self.config.time_sampling_offset
+        return time.to(dtype=torch.float32, device=device)
+
+    def embed_prefix(
+        self, images, img_masks, tokens, masks
+    ) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor]:
+        """Embed images with SigLIP and language tokens with embedding layer."""
+        embs = []
+        pad_masks = []
+        att_masks = []
+
+        # Process images
+        for img, img_mask in zip(images, img_masks, strict=True):
+
+            def image_embed_func(img):
+                return self.paligemma_with_expert.embed_image(img)
+
+            img_emb = self._apply_checkpoint(image_embed_func, img)
+            bsize, num_img_embs = img_emb.shape[:2]
+
+            embs.append(img_emb)
+            pad_masks.append(img_mask[:, None].expand(bsize, num_img_embs))
+            att_masks += [0] * num_img_embs
+
+        # Process language tokens
+        def lang_embed_func(tokens):
+            lang_emb = self.paligemma_with_expert.embed_language_tokens(tokens)
+            lang_emb_dim = lang_emb.shape[-1]
+            return lang_emb * math.sqrt(lang_emb_dim)
+
+        lang_emb = self._apply_checkpoint(lang_embed_func, tokens)
+        embs.append(lang_emb)
+        pad_masks.append(masks)
+
+        num_lang_embs = lang_emb.shape[1]
+        att_masks += [0] * num_lang_embs
+
+        embs = torch.cat(embs, dim=1)
+        pad_masks = torch.cat(pad_masks, dim=1)
+        att_masks = torch.tensor(att_masks, dtype=torch.bool, device=pad_masks.device)
+
+        bsize = pad_masks.shape[0]
+        att_masks = att_masks[None, :].expand(bsize, len(att_masks))
+
+        return embs, pad_masks, att_masks
+
+    def embed_suffix(self, noisy_actions, timestep):
+        """Embed noisy_actions, timestep to prepare for Expert Gemma processing."""
+        embs = []
+        pad_masks = []
+        att_masks = []
+
+        # Embed timestep using sine-cosine positional encoding
+        time_emb = create_sinusoidal_pos_embedding(
+            timestep,
+            self.action_in_proj.out_features,
+            min_period=self.config.min_period,
+            max_period=self.config.max_period,
+            device=timestep.device,
+        )
+        time_emb = time_emb.type(dtype=timestep.dtype)
+
+        # Fuse timestep + action information using an MLP
+        def action_proj_func(noisy_actions):
+            return self.action_in_proj(noisy_actions)
+
+        action_emb = self._apply_checkpoint(action_proj_func, noisy_actions)
+
+        def time_mlp_func(time_emb):
+            x = self.time_mlp_in(time_emb)
+            x = F.silu(x)
+            x = self.time_mlp_out(x)
+            return F.silu(x)
+
+        time_emb = self._apply_checkpoint(time_mlp_func, time_emb)
+        action_time_emb = action_emb
+        adarms_cond = time_emb
+
+        embs.append(action_time_emb)
+        bsize, action_time_dim = action_time_emb.shape[:2]
+        action_time_mask = torch.ones(bsize, action_time_dim, dtype=torch.bool, device=timestep.device)
+        pad_masks.append(action_time_mask)
+
+        # Set attention masks so that image, language and state inputs do not attend to action tokens
+        att_masks += [1] + ([0] * (self.config.chunk_size - 1))
+
+        embs = torch.cat(embs, dim=1)
+        pad_masks = torch.cat(pad_masks, dim=1)
+        att_masks = torch.tensor(att_masks, dtype=embs.dtype, device=embs.device)
+        att_masks = att_masks[None, :].expand(bsize, len(att_masks))
+
+        return embs, pad_masks, att_masks, adarms_cond
+
+    def forward(self, images, img_masks, tokens, masks, actions, noise=None, time=None) -> Tensor:
+        """Do a full training forward pass and compute the loss."""
+        if noise is None:
+            noise = self.sample_noise(actions.shape, actions.device)
+
+        if time is None:
+            time = self.sample_time(actions.shape[0], actions.device)
+
+        time_expanded = time[:, None, None]
+        x_t = time_expanded * noise + (1 - time_expanded) * actions
+        u_t = noise - actions
+
+        prefix_embs, prefix_pad_masks, prefix_att_masks = self.embed_prefix(images, img_masks, tokens, masks)
+        suffix_embs, suffix_pad_masks, suffix_att_masks, adarms_cond = self.embed_suffix(x_t, time)
+
+        if (
+            self.paligemma_with_expert.paligemma.language_model.layers[0].self_attn.q_proj.weight.dtype
+            == torch.bfloat16
+        ):
+            suffix_embs = suffix_embs.to(dtype=torch.bfloat16)
+            prefix_embs = prefix_embs.to(dtype=torch.bfloat16)
+
+        pad_masks = torch.cat([prefix_pad_masks, suffix_pad_masks], dim=1)
+        att_masks = torch.cat([prefix_att_masks, suffix_att_masks], dim=1)
+
+        att_2d_masks = make_att_2d_masks(pad_masks, att_masks)
+        position_ids = torch.cumsum(pad_masks, dim=1) - 1
+
+        att_2d_masks_4d = self._prepare_attention_masks_4d(att_2d_masks)
+
+        def forward_func(prefix_embs, suffix_embs, att_2d_masks_4d, position_ids, adarms_cond):
+            (_, suffix_out), _ = self.paligemma_with_expert.forward(
+                attention_mask=att_2d_masks_4d,
+                position_ids=position_ids,
+                past_key_values=None,
+                inputs_embeds=[prefix_embs, suffix_embs],
+                use_cache=False,
+                adarms_cond=[None, adarms_cond],
+            )
+            return suffix_out
+
+        suffix_out = self._apply_checkpoint(
+            forward_func, prefix_embs, suffix_embs, att_2d_masks_4d, position_ids, adarms_cond
+        )
+
+        suffix_out = suffix_out[:, -self.config.chunk_size :]
+        suffix_out = suffix_out.to(dtype=torch.float32)
+
+        def action_out_proj_func(suffix_out):
+            return self.action_out_proj(suffix_out)
+
+        v_t = self._apply_checkpoint(action_out_proj_func, suffix_out)
+
+        return F.mse_loss(u_t, v_t, reduction="none")
+
+    @torch.no_grad()  # see openpi `sample_actions` (slightly adapted)
+    def sample_actions(
+        self,
+        images,
+        img_masks,
+        tokens,
+        masks,
+        noise=None,
+        num_steps=None,
+        **kwargs: Unpack[ActionSelectKwargs],
+    ) -> Tensor:
+        """Do a full inference forward and compute the action."""
+        if num_steps is None:
+            num_steps = self.config.num_inference_steps
+
+        bsize = tokens.shape[0]
+        device = tokens.device
+
+        if noise is None:
+            # Sample noise with padded dimension as expected by action_in_proj
+            actions_shape = (
+                bsize,
+                self.config.chunk_size,
+                self.config.max_action_dim,
+            )  # Use config max_action_dim for internal processing
+            noise = self.sample_noise(actions_shape, device)
+
+        prefix_embs, prefix_pad_masks, prefix_att_masks = self.embed_prefix(images, img_masks, tokens, masks)
+        prefix_att_2d_masks = make_att_2d_masks(prefix_pad_masks, prefix_att_masks)
+        prefix_position_ids = torch.cumsum(prefix_pad_masks, dim=1) - 1
+
+        prefix_att_2d_masks_4d = self._prepare_attention_masks_4d(prefix_att_2d_masks)
+        self.paligemma_with_expert.paligemma.language_model.config._attn_implementation = "eager"  # noqa: SLF001
+
+        _, past_key_values = self.paligemma_with_expert.forward(
+            attention_mask=prefix_att_2d_masks_4d,
+            position_ids=prefix_position_ids,
+            past_key_values=None,
+            inputs_embeds=[prefix_embs, None],
+            use_cache=True,
+        )
+
+        dt = -1.0 / num_steps
+
+        x_t = noise
+        for step in range(num_steps):
+            time = 1.0 + step * dt
+            time_tensor = torch.tensor(time, dtype=torch.float32, device=device).expand(bsize)
+
+            def denoise_step_partial_call(input_x_t, current_timestep=time_tensor):
+                return self.denoise_step(
+                    prefix_pad_masks=prefix_pad_masks,
+                    past_key_values=past_key_values,
+                    x_t=input_x_t,
+                    timestep=current_timestep,
+                )
+
+            if self._rtc_enabled():
+                inference_delay = kwargs.get("inference_delay")
+                prev_chunk_left_over = kwargs.get("prev_chunk_left_over")
+                execution_horizon = kwargs.get("execution_horizon")
+
+                v_t = self.rtc_processor.denoise_step(
+                    x_t=x_t,
+                    prev_chunk_left_over=prev_chunk_left_over,
+                    inference_delay=inference_delay,
+                    time=time,
+                    original_denoise_step_partial=denoise_step_partial_call,
+                    execution_horizon=execution_horizon,
+                )
+            else:
+                v_t = denoise_step_partial_call(x_t)
+
+            x_t = x_t + dt * v_t
+
+            if self.rtc_processor is not None and self.rtc_processor.is_debug_enabled():
+                self.rtc_processor.track(time=time, x_t=x_t, v_t=v_t)
+
+        return x_t
+
+    def denoise_step(
+        self,
+        prefix_pad_masks,
+        past_key_values,
+        x_t,
+        timestep,
+    ):
+        """Apply one denoising step of the noise `x_t` at a given timestep."""
+        suffix_embs, suffix_pad_masks, suffix_att_masks, adarms_cond = self.embed_suffix(x_t, timestep)
+
+        suffix_len = suffix_pad_masks.shape[1]
+        batch_size = prefix_pad_masks.shape[0]
+        prefix_len = prefix_pad_masks.shape[1]
+
+        prefix_pad_2d_masks = prefix_pad_masks[:, None, :].expand(batch_size, suffix_len, prefix_len)
+        suffix_att_2d_masks = make_att_2d_masks(suffix_pad_masks, suffix_att_masks)
+        full_att_2d_masks = torch.cat([prefix_pad_2d_masks, suffix_att_2d_masks], dim=2)
+
+        prefix_offsets = torch.sum(prefix_pad_masks, dim=-1)[:, None]
+        position_ids = prefix_offsets + torch.cumsum(suffix_pad_masks, dim=1) - 1
+
+        full_att_2d_masks_4d = self._prepare_attention_masks_4d(full_att_2d_masks)
+        self.paligemma_with_expert.gemma_expert.model.config._attn_implementation = "eager"  # noqa: SLF001
+
+        outputs_embeds, _ = self.paligemma_with_expert.forward(
+            attention_mask=full_att_2d_masks_4d,
+            position_ids=position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=[None, suffix_embs],
+            use_cache=False,
+            adarms_cond=[None, adarms_cond],
+        )
+
+        suffix_out = outputs_embeds[1]
+        suffix_out = suffix_out[:, -self.config.chunk_size :]
+        suffix_out = suffix_out.to(dtype=torch.float32)
+        return self.action_out_proj(suffix_out)
+
+
+class PI05Policy(PreTrainedPolicy):
+    """PI05 Policy for LeRobot."""
+
+    config_class = PI05Config
+    name = "pi05"
+
+    def __init__(
+        self,
+        config: PI05Config,
+        **kwargs,
+    ):
+        """
+        Args:
+            config: Policy configuration class instance.
+        """
+        super().__init__(config)
+        config.validate_features()
+        self.config = config
+
+        # Initialize the core PI05 model
+        self.init_rtc_processor()
+        self.model = PI05Pytorch(config, rtc_processor=self.rtc_processor)
+
+        # Enable gradient checkpointing if requested
+        if config.gradient_checkpointing:
+            self.model.gradient_checkpointing_enable()
+
+        self.model.to(config.device)
+
+        self.reset()
+
+    @classmethod
+    def from_pretrained(
+        cls: builtins.type[T],
+        pretrained_name_or_path: str | Path,
+        *,
+        config: PreTrainedConfig | None = None,
+        force_download: bool = False,
+        resume_download: bool | None = None,
+        proxies: dict | None = None,
+        token: str | bool | None = None,
+        cache_dir: str | Path | None = None,
+        local_files_only: bool = False,
+        revision: str | None = None,
+        strict: bool = True,
+        **kwargs,
+    ) -> T:
+        """Override the from_pretrained method to handle key remapping and display important disclaimer."""
+        print(
+            "The PI05 model is a direct port of the OpenPI implementation. \n"
+            "This implementation follows the original OpenPI structure for compatibility. \n"
+            "Original implementation: https://github.com/Physical-Intelligence/openpi"
+        )
+        if pretrained_name_or_path is None:
+            raise ValueError("pretrained_name_or_path is required")
+
+        # Use provided config if available, otherwise create default config
+        if config is None:
+            config = PreTrainedConfig.from_pretrained(
+                pretrained_name_or_path=pretrained_name_or_path,
+                force_download=force_download,
+                resume_download=resume_download,
+                proxies=proxies,
+                token=token,
+                cache_dir=cache_dir,
+                local_files_only=local_files_only,
+                revision=revision,
+                **kwargs,
+            )
+
+        # Initialize model without loading weights
+        # Check if dataset_stats were provided in kwargs
+        model = cls(config, **kwargs)
+
+        # Now manually load and remap the state dict
+        try:
+            # Try to load the pytorch_model.bin or model.safetensors file
+            print(f"Loading model from: {pretrained_name_or_path}")
+            try:
+                from transformers.utils import cached_file
+
+                # Try safetensors first
+                resolved_file = cached_file(
+                    pretrained_name_or_path,
+                    "model.safetensors",
+                    cache_dir=kwargs.get("cache_dir"),
+                    force_download=kwargs.get("force_download", False),
+                    resume_download=kwargs.get("resume_download"),
+                    proxies=kwargs.get("proxies"),
+                    use_auth_token=kwargs.get("use_auth_token"),
+                    revision=kwargs.get("revision"),
+                    local_files_only=kwargs.get("local_files_only", False),
+                )
+                from safetensors.torch import load_file
+
+                original_state_dict = load_file(resolved_file)
+                print("✓ Loaded state dict from model.safetensors")
+            except Exception as e:
+                print(f"Could not load state dict from remote files: {e}")
+                print("Returning model without loading pretrained weights")
+                return model
+
+            # First, fix any key differences # see openpi `model.py, _fix_pytorch_state_dict_keys`
+            fixed_state_dict = model._fix_pytorch_state_dict_keys(original_state_dict, model.config)
+
+            # Then add "model." prefix for all keys that don't already have it
+            remapped_state_dict = {}
+            remap_count = 0
+
+            for key, value in fixed_state_dict.items():
+                if not key.startswith("model."):
+                    new_key = f"model.{key}"
+                    remapped_state_dict[new_key] = value
+                    remap_count += 1
+                    if remap_count <= 10:  # Only print first 10 to avoid spam
+                        print(f"Remapped: {key} -> {new_key}")
+                else:
+                    remapped_state_dict[key] = value
+
+            if remap_count > 0:
+                print(f"Remapped {remap_count} state dict keys")
+
+            # Load the remapped state dict into the model
+            missing_keys, unexpected_keys = model.load_state_dict(remapped_state_dict, strict=strict)
+
+            if missing_keys:
+                print(f"Missing keys when loading state dict: {len(missing_keys)} keys")
+                if len(missing_keys) <= 5:
+                    for key in missing_keys:
+                        print(f"  - {key}")
+                else:
+                    for key in missing_keys[:5]:
+                        print(f"  - {key}")
+                    print(f"  ... and {len(missing_keys) - 5} more")
+
+            if unexpected_keys:
+                print(f"Unexpected keys when loading state dict: {len(unexpected_keys)} keys")
+                if len(unexpected_keys) <= 5:
+                    for key in unexpected_keys:
+                        print(f"  - {key}")
+                else:
+                    for key in unexpected_keys[:5]:
+                        print(f"  - {key}")
+                    print(f"  ... and {len(unexpected_keys) - 5} more")
+
+            if not missing_keys and not unexpected_keys:
+                print("All keys loaded successfully!")
+
+        except Exception as e:
+            print(f"Warning: Could not remap state dict keys: {e}")
+
+        return model
+
+    def _fix_pytorch_state_dict_keys(
+        self, state_dict, model_config
+    ):  # see openpi `BaseModelConfig, _fix_pytorch_state_dict_keys`
+        """Fix state dict keys to match current model architecture."""
+        import re
+
+        fixed_state_dict = {}
+
+        for key, value in state_dict.items():
+            new_key = key
+
+            # Handle layer norm structure changes: .weight -> .dense.weight + .dense.bias
+            # For gemma expert layers
+            if re.match(
+                r"paligemma_with_expert\.gemma_expert\.model\.layers\.\d+\.(input_layernorm|post_attention_layernorm)\.weight",
+                key,
+            ):
+                # Check if the model actually has adaRMS enabled for the expert
+                expert_uses_adarms = getattr(
+                    self.model.paligemma_with_expert.gemma_expert.config, "use_adarms", False
+                )
+                if expert_uses_adarms:
+                    logging.warning(f"Skipping layer norm key (adaRMS mismatch): {key}")
+                    continue
+
+            if re.match(r"paligemma_with_expert\.gemma_expert\.model\.norm\.weight", key):
+                # Check if the model actually has adaRMS enabled for the expert
+                expert_uses_adarms = getattr(
+                    self.model.paligemma_with_expert.gemma_expert.config, "use_adarms", False
+                )
+                if expert_uses_adarms:
+                    logging.warning(f"Skipping norm key (adaRMS mismatch): {key}")
+                    continue
+
+            # Handle MLP naming changes for pi05
+            # pi05 model expects time_mlp_*, but checkpoint might have action_time_mlp_*
+            if key.startswith("action_time_mlp_in."):
+                new_key = key.replace("action_time_mlp_in.", "time_mlp_in.")
+            elif key.startswith("action_time_mlp_out."):
+                new_key = key.replace("action_time_mlp_out.", "time_mlp_out.")
+            # Also handle state_proj which shouldn't exist in pi05
+            if key.startswith("state_proj."):
+                logging.warning(f"Skipping state_proj key in pi05 mode: {key}")
+                continue
+
+            # Handle vision tower embedding layer potential differences
+            if "patch_embedding" in key:
+                # Some checkpoints might have this, but current model expects different structure
+                logging.warning(f"Vision embedding key might need handling: {key}")
+
+            fixed_state_dict[new_key] = value
+
+        return fixed_state_dict
+
+    def get_optim_params(self) -> dict:
+        return self.parameters()
+
+    def reset(self):
+        """Reset internal state - called when environment resets."""
+        self._action_queue = deque(maxlen=self.config.n_action_steps)
+        self._queues = {
+            ACTION: deque(maxlen=self.config.n_action_steps),
+        }
+
+    def init_rtc_processor(self):
+        """Initialize RTC processor if RTC is enabled in config."""
+        self.rtc_processor = None
+
+        # Create processor if config provided
+        # If RTC is not enabled - we can still track the denoising data
+        if self.config.rtc_config is not None:
+            self.rtc_processor = RTCProcessor(self.config.rtc_config)
+
+            model_value = getattr(self, "model", None)
+            if model_value is not None:
+                model_value.rtc_processor = self.rtc_processor
+
+    def _rtc_enabled(self) -> bool:
+        return self.config.rtc_config is not None and self.config.rtc_config.enabled
+
+    def _preprocess_images(self, batch: dict[str, Tensor]) -> tuple[list[Tensor], list[Tensor]]:
+        """Preprocess images for the model.
+
+        Images from LeRobot are typically in [B, C, H, W] format and normalized to [0, 1].
+        PaliGemma expects images in [B, C, H, W] format and normalized to [-1, 1].
+        """
+        images = []
+        img_masks = []
+
+        # Get device from model parameters
+        device = next(self.parameters()).device
+
+        present_img_keys = [key for key in self.config.image_features if key in batch]
+        missing_img_keys = [key for key in self.config.image_features if key not in batch]
+
+        if len(present_img_keys) == 0:
+            raise ValueError(
+                f"All image features are missing from the batch. At least one expected. "
+                f"(batch: {batch.keys()}) (image_features: {self.config.image_features})"
+            )
+
+        # Preprocess image features present in the batch
+        for key in present_img_keys:
+            img = batch[key]
+
+            # Ensure tensor is on the same device as the model
+            if img.device != device:
+                img = img.to(device)
+
+            # Ensure float32 dtype for consistency
+            if img.dtype != torch.float32:
+                img = img.to(torch.float32)
+
+            # from openpi preprocess_observation_pytorch: Handle both [B, C, H, W] and [B, H, W, C] formats
+            is_channels_first = img.shape[1] == 3  # Check if channels are in dimension 1
+
+            if is_channels_first:
+                # Convert [B, C, H, W] to [B, H, W, C] for processing
+                img = img.permute(0, 2, 3, 1)
+
+            # from openpi preprocess_observation_pytorch: Resize with padding if needed
+            if img.shape[1:3] != self.config.image_resolution:
+                img = resize_with_pad_torch(img, *self.config.image_resolution)
+
+            # Normalize from [0,1] to [-1,1] as expected by siglip
+            img = img * 2.0 - 1.0
+
+            # from openpi preprocess_observation_pytorch: Convert back to [B, C, H, W] format if it was originally channels-first
+            if is_channels_first:
+                img = img.permute(0, 3, 1, 2)  # [B, H, W, C] -> [B, C, H, W]
+
+            images.append(img)
+            # Create mask (all ones for real images)
+            bsize = img.shape[0]
+            mask = torch.ones(bsize, dtype=torch.bool, device=device)
+            img_masks.append(mask)
+
+        # Create image features not present in the batch as fully 0 padded images
+        for _num_empty_cameras in range(len(missing_img_keys)):
+            img = torch.ones_like(img) * -1  # Padded with -1 for SigLIP
+            mask = torch.zeros_like(mask)  # Mask is zero for empty cameras
+            images.append(img)
+            img_masks.append(mask)
+
+        return images, img_masks
+
+    def prepare_action(self, batch):
+        """Pad action"""
+        actions = pad_vector(batch[ACTION], self.config.max_action_dim)
+        return actions
+
+    @torch.no_grad()
+    def select_action(self, batch: dict[str, Tensor]) -> Tensor:
+        """Select a single action given environment observations."""
+        assert not self._rtc_enabled(), (
+            "RTC is not supported for select_action, use it with predict_action_chunk"
+        )
+
+        self.eval()
+
+        # Action queue logic for n_action_steps > 1
+        if len(self._action_queue) == 0:
+            actions = self.predict_action_chunk(batch)[:, : self.config.n_action_steps]
+            # Transpose to get shape (n_action_steps, batch_size, action_dim)
+            self._action_queue.extend(actions.transpose(0, 1))
+
+        return self._action_queue.popleft()
+
+    @torch.no_grad()
+    def predict_action_chunk(self, batch: dict[str, Tensor], **kwargs: Unpack[ActionSelectKwargs]) -> Tensor:
+        """Predict a chunk of actions given environment observations."""
+        self.eval()
+
+        # Prepare inputs
+        images, img_masks = self._preprocess_images(batch)
+        tokens, masks = batch[f"{OBS_LANGUAGE_TOKENS}"], batch[f"{OBS_LANGUAGE_ATTENTION_MASK}"]
+
+        # Sample actions using the model (pass through RTC kwargs, no separate state needed for PI05)
+        actions = self.model.sample_actions(images, img_masks, tokens, masks, **kwargs)
+
+        # Unpad actions to actual action dimension
+        original_action_dim = self.config.output_features[ACTION].shape[0]
+        actions = actions[:, :, :original_action_dim]
+
+        return actions
+
+    def forward(self, batch: dict[str, Tensor], reduction: str = "mean") -> tuple[Tensor, dict]:
+        """Run the batch through the model and compute the loss for training.
+
+        Args:
+            batch: Training batch containing observations and actions.
+            reduction: How to reduce the loss. Options:
+                - "mean": Return scalar mean loss (default, backward compatible)
+                - "none": Return per-sample losses of shape (batch_size,) for RA-BC weighting
+        """
+        # Prepare inputs
+        images, img_masks = self._preprocess_images(batch)
+        tokens, masks = batch[f"{OBS_LANGUAGE_TOKENS}"], batch[f"{OBS_LANGUAGE_ATTENTION_MASK}"]
+
+        actions = self.prepare_action(batch)
+
+        # Compute loss (no separate state needed for PI05)
+        losses = self.model.forward(images, img_masks, tokens, masks, actions)
+
+        # Truncate losses to actual action dimensions
+        original_action_dim = self.config.output_features[ACTION].shape[0]
+        losses = losses[:, :, :original_action_dim]
+
+        loss_dict = {
+            "loss_per_dim": losses.mean(dim=[0, 1]).detach().cpu().numpy().tolist(),
+        }
+
+        if reduction == "none":
+            # Return per-sample losses (B,) by averaging over time and action dims
+            per_sample_loss = losses.mean(dim=(1, 2))
+            loss_dict["loss"] = per_sample_loss.mean().item()
+            return per_sample_loss, loss_dict
+        else:
+            # Default: return scalar mean loss
+            loss = losses.mean()
+            loss_dict["loss"] = loss.item()
+            return loss, loss_dict
+
+    def _get_default_peft_targets(self) -> dict[str, any]:
+        """Return default PEFT target modules for PI0.5 fine-tuning."""
+        common_projections = (
+            "state_proj|action_in_proj|action_out_proj|action_time_mlp_in|action_time_mlp_out"
+        )
+        target_modules = rf"(.*\.gemma_expert\..*\.self_attn\.(q|v)_proj|model\.({common_projections}))"
+        return {
+            "target_modules": target_modules,
+            "modules_to_save": [],
+        }
diff --git a/lerobot_patches/src/lerobot/robots/so_follower/so_follower.py b/lerobot_patches/src/lerobot/robots/so_follower/so_follower.py
new file mode 100644
index 0000000000000000000000000000000000000000..464e4bcf72201c8f7f6d5e56b2902be310e8fe4b
--- /dev/null
+++ b/lerobot_patches/src/lerobot/robots/so_follower/so_follower.py
@@ -0,0 +1,307 @@
+#!/usr/bin/env python
+
+# Copyright 2026 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+import logging
+import time
+from functools import cached_property
+from typing import TypeAlias
+
+from lerobot.cameras.utils import make_cameras_from_configs
+from lerobot.motors import Motor, MotorCalibration, MotorNormMode
+from lerobot.motors.feetech import (
+    FeetechMotorsBus,
+    OperatingMode,
+)
+from lerobot.processor import RobotAction, RobotObservation
+from lerobot.utils.decorators import check_if_already_connected, check_if_not_connected
+
+from ..robot import Robot
+from ..utils import ensure_safe_goal_position
+from .config_so_follower import SOFollowerRobotConfig
+
+logger = logging.getLogger(__name__)
+
+
+class SOFollower(Robot):
+    """
+    Generic SO follower base implementing common functionality for SO-100/101/10X.
+    Designed to be subclassed with a per-hardware-model `config_class` and `name`.
+    """
+
+    config_class = SOFollowerRobotConfig
+    name = "so_follower"
+
+    def __init__(self, config: SOFollowerRobotConfig):
+        super().__init__(config)
+        self.config = config
+        # choose normalization mode depending on config if available
+        norm_mode_body = MotorNormMode.DEGREES if config.use_degrees else MotorNormMode.RANGE_M100_100
+        self.bus = FeetechMotorsBus(
+            port=self.config.port,
+            motors={
+                "shoulder_pan": Motor(1, "sts3215", norm_mode_body),
+                "shoulder_lift": Motor(2, "sts3215", norm_mode_body),
+                "elbow_flex": Motor(3, "sts3215", norm_mode_body),
+                "wrist_flex": Motor(4, "sts3215", norm_mode_body),
+                "wrist_roll": Motor(5, "sts3215", norm_mode_body),
+                "gripper": Motor(6, "sts3215", MotorNormMode.RANGE_0_100),
+            },
+            calibration=self.calibration,
+        )
+        self.cameras = make_cameras_from_configs(config.cameras)
+
+    @property
+    def _motors_ft(self) -> dict[str, type]:
+        return {f"{motor}.pos": float for motor in self.bus.motors}
+
+    @property
+    def _cameras_ft(self) -> dict[str, tuple]:
+        return {
+            cam: (self.config.cameras[cam].height, self.config.cameras[cam].width, 3) for cam in self.cameras
+        }
+
+    @cached_property
+    def observation_features(self) -> dict[str, type | tuple]:
+        return {**self._motors_ft, **self._cameras_ft}
+
+    @cached_property
+    def action_features(self) -> dict[str, type]:
+        return self._motors_ft
+
+    @property
+    def is_connected(self) -> bool:
+        return self.bus.is_connected and all(cam.is_connected for cam in self.cameras.values())
+
+    @check_if_already_connected
+    def connect(self, calibrate: bool = True) -> None:
+        """
+        We assume that at connection time, arm is in a rest position,
+        and torque can be safely disabled to run calibration.
+        """
+
+        self.bus.connect()
+        if not self.is_calibrated and calibrate:
+            logger.info(
+                "Mismatch between calibration values in the motor and the calibration file or no calibration file found"
+            )
+            self.calibrate()
+
+        for cam in self.cameras.values():
+            cam.connect()
+
+        self.configure()
+        logger.info(f"{self} connected.")
+
+    @property
+    def is_calibrated(self) -> bool:
+        return self.bus.is_calibrated
+
+    def _record_ranges_wraparound_safe(self) -> tuple[dict, dict, dict]:
+        """Record range of motion for all joints, handling encoder wraparound.
+
+        Uses cumulative delta tracking between consecutive readings to correctly
+        measure displacement even when encoder values wrap past 0/4095. Then
+        computes corrected homing offsets so that each joint's range fits within
+        0-4095 with min < max.
+
+        Returns:
+            (homing_offsets, range_mins, range_maxes) with corrected values.
+            homing_offsets are signed (-2047..+2047) ready for bus.write().
+        """
+        from lerobot.motors.encoding_utils import decode_sign_magnitude
+        from lerobot.utils.utils import enter_pressed, move_cursor_up
+
+        ENCODER_STEPS = 4096
+        HOMING_OFFSET_MAX = 2047
+
+        motor_names = list(self.bus.motors.keys())
+        prev = self.bus.sync_read("Present_Position", motor_names, normalize=False)
+        start = dict(prev)
+        cumulative = {m: 0 for m in motor_names}
+        min_offsets = {m: 0 for m in motor_names}
+        max_offsets = {m: 0 for m in motor_names}
+
+        while True:
+            positions = self.bus.sync_read("Present_Position", motor_names, normalize=False)
+            for motor in motor_names:
+                delta = positions[motor] - prev[motor]
+                if delta > 2048:
+                    delta -= ENCODER_STEPS
+                elif delta < -2048:
+                    delta += ENCODER_STEPS
+                cumulative[motor] += delta
+                min_offsets[motor] = min(cumulative[motor], min_offsets[motor])
+                max_offsets[motor] = max(cumulative[motor], max_offsets[motor])
+            prev = dict(positions)
+
+            print("\n-------------------------------------------")
+            print(f"{'NAME':<15} | {'MIN':>6} | {'POS':>6} | {'MAX':>6}")
+            for motor in motor_names:
+                print(f"{motor:<15} | {min_offsets[motor]:>6} | {cumulative[motor]:>6} | {max_offsets[motor]:>6}")
+
+            if enter_pressed():
+                break
+            move_cursor_up(len(motor_names) + 3)
+
+        # Compute corrected homing offsets and ranges
+        # Read raw register values and decode sign-magnitude (bit 11)
+        original_offsets_raw = {m: self.bus.sync_read("Homing_Offset", [m], normalize=False)[m] for m in motor_names}
+        homing_offsets = {}
+        range_mins = {}
+        range_maxes = {}
+        for motor in motor_names:
+            orig_offset_signed = decode_sign_magnitude(original_offsets_raw[motor], 11)
+            actual_start = start[motor] + orig_offset_signed
+            actual_min = actual_start + min_offsets[motor]
+            actual_max = actual_start + max_offsets[motor]
+            range_size = actual_max - actual_min
+            desired_present_min = (ENCODER_STEPS - range_size) // 2
+            ideal_offset = actual_min - desired_present_min
+            new_offset = ((ideal_offset + 2048) % ENCODER_STEPS) - 2048
+            new_offset = max(-HOMING_OFFSET_MAX, min(HOMING_OFFSET_MAX, new_offset))
+            homing_offsets[motor] = new_offset
+            range_mins[motor] = (actual_min - new_offset) % ENCODER_STEPS
+            range_maxes[motor] = (actual_max - new_offset) % ENCODER_STEPS
+
+        return homing_offsets, range_mins, range_maxes
+
+    def calibrate(self) -> None:
+        if self.calibration:
+            # Calibration file exists, ask user whether to use it or run new calibration
+            user_input = input(
+                f"Press ENTER to use provided calibration file associated with the id {self.id}, or type 'c' and press ENTER to run calibration: "
+            )
+            if user_input.strip().lower() != "c":
+                logger.info(f"Writing calibration file associated with the id {self.id} to the motors")
+                self.bus.write_calibration(self.calibration)
+                return
+
+        logger.info(f"\nRunning calibration of {self}")
+        self.bus.disable_torque()
+        for motor in self.bus.motors:
+            self.bus.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+
+        input(f"Move {self} to the middle of its range of motion and press ENTER....")
+        self.bus.set_half_turn_homings()
+
+        print(
+            f"Move all joints sequentially through their "
+            "entire ranges of motion.\nRecording positions. Press ENTER to stop..."
+        )
+        homing_offsets, range_mins, range_maxes = self._record_ranges_wraparound_safe()
+
+        # Write corrected homing offsets to motors
+        for motor, offset in homing_offsets.items():
+            self.bus.write("Homing_Offset", motor, offset)
+
+        self.calibration = {}
+        for motor, m in self.bus.motors.items():
+            self.calibration[motor] = MotorCalibration(
+                id=m.id,
+                drive_mode=0,
+                homing_offset=homing_offsets[motor],
+                range_min=range_mins[motor],
+                range_max=range_maxes[motor],
+            )
+
+        self.bus.write_calibration(self.calibration)
+        self._save_calibration()
+        print("Calibration saved to", self.calibration_fpath)
+
+    def configure(self) -> None:
+        with self.bus.torque_disabled():
+            self.bus.configure_motors()
+            for motor in self.bus.motors:
+                self.bus.write("Operating_Mode", motor, OperatingMode.POSITION.value)
+                # Set P_Coefficient to lower value to avoid shakiness (Default is 32)
+                self.bus.write("P_Coefficient", motor, 16)
+                # Set I_Coefficient and D_Coefficient to default value 0 and 32
+                self.bus.write("I_Coefficient", motor, 0)
+                self.bus.write("D_Coefficient", motor, 32)
+                # Limit velocity to prevent slamming into mechanical stops.
+                # STS3215 units: ~0.0146 RPM/step, 600 ≈ 8.8 RPM ≈ 53°/s
+                self.bus.write("Goal_Velocity", motor, 600)
+                # Gentler acceleration (default 254 = instant, 50 = smooth ramp)
+                self.bus.write("Acceleration", motor, 50)
+
+                if motor == "gripper":
+                    self.bus.write("Max_Torque_Limit", motor, 500)  # 50% of max torque to avoid burnout
+                    self.bus.write("Protection_Current", motor, 250)  # 50% of max current to avoid burnout
+                    self.bus.write("Overload_Torque", motor, 25)  # 25% torque when overloaded
+
+    def setup_motors(self) -> None:
+        for motor in reversed(self.bus.motors):
+            input(f"Connect the controller board to the '{motor}' motor only and press enter.")
+            self.bus.setup_motor(motor)
+            print(f"'{motor}' motor id set to {self.bus.motors[motor].id}")
+
+    @check_if_not_connected
+    def get_observation(self) -> RobotObservation:
+        # Read arm position
+        start = time.perf_counter()
+        obs_dict = self.bus.sync_read("Present_Position", num_retry=10)
+        obs_dict = {f"{motor}.pos": val for motor, val in obs_dict.items()}
+        dt_ms = (time.perf_counter() - start) * 1e3
+        logger.debug(f"{self} read state: {dt_ms:.1f}ms")
+
+        # Capture images from cameras
+        for cam_key, cam in self.cameras.items():
+            start = time.perf_counter()
+            obs_dict[cam_key] = cam.async_read()
+            dt_ms = (time.perf_counter() - start) * 1e3
+            logger.debug(f"{self} read {cam_key}: {dt_ms:.1f}ms")
+
+        return obs_dict
+
+    @check_if_not_connected
+    def send_action(self, action: RobotAction) -> RobotAction:
+        """Command arm to move to a target joint configuration.
+
+        The relative action magnitude may be clipped depending on the configuration parameter
+        `max_relative_target`. In this case, the action sent differs from original action.
+        Thus, this function always returns the action actually sent.
+
+        Raises:
+            RobotDeviceNotConnectedError: if robot is not connected.
+
+        Returns:
+            RobotAction: the action sent to the motors, potentially clipped.
+        """
+
+        goal_pos = {key.removesuffix(".pos"): val for key, val in action.items() if key.endswith(".pos")}
+
+        # Cap goal position when too far away from present position.
+        # /!\ Slower fps expected due to reading from the follower.
+        if self.config.max_relative_target is not None:
+            present_pos = self.bus.sync_read("Present_Position")
+            goal_present_pos = {key: (g_pos, present_pos[key]) for key, g_pos in goal_pos.items()}
+            goal_pos = ensure_safe_goal_position(goal_present_pos, self.config.max_relative_target)
+
+        # Send goal position to the arm
+        self.bus.sync_write("Goal_Position", goal_pos)
+        return {f"{motor}.pos": val for motor, val in goal_pos.items()}
+
+    @check_if_not_connected
+    def disconnect(self):
+        self.bus.disconnect(self.config.disable_torque_on_disconnect)
+        for cam in self.cameras.values():
+            cam.disconnect()
+
+        logger.info(f"{self} disconnected.")
+
+
+SO100Follower: TypeAlias = SOFollower
+SO101Follower: TypeAlias = SOFollower
diff --git a/lerobot_patches/src/lerobot/scripts/lerobot_record.py b/lerobot_patches/src/lerobot/scripts/lerobot_record.py
new file mode 100644
index 0000000000000000000000000000000000000000..037884818f3e0bcff1ba12925cf851f771868f7f
--- /dev/null
+++ b/lerobot_patches/src/lerobot/scripts/lerobot_record.py
@@ -0,0 +1,642 @@
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""
+Records a dataset. Actions for the robot can be either generated by teleoperation or by a policy.
+
+Example:
+
+```shell
+lerobot-record \
+    --robot.type=so100_follower \
+    --robot.port=/dev/tty.usbmodem58760431541 \
+    --robot.cameras="{laptop: {type: opencv, index_or_path: 0, width: 640, height: 480, fps: 30}}" \
+    --robot.id=black \
+    --dataset.repo_id=<my_username>/<my_dataset_name> \
+    --dataset.num_episodes=2 \
+    --dataset.single_task="Grab the cube" \
+    --display_data=true
+    # <- Optional: specify video codec (h264, hevc, libsvtav1). Default is libsvtav1. \
+    # --dataset.vcodec=h264 \
+    # <- Teleop optional if you want to teleoperate to record or in between episodes with a policy \
+    # --teleop.type=so100_leader \
+    # --teleop.port=/dev/tty.usbmodem58760431551 \
+    # --teleop.id=blue \
+    # <- Policy optional if you want to record with a policy \
+    # --policy.path=${HF_USER}/my_policy \
+```
+
+Example recording with bimanual so100:
+```shell
+lerobot-record \
+  --robot.type=bi_so_follower \
+  --robot.left_arm_config.port=/dev/tty.usbmodem5A460822851 \
+  --robot.right_arm_config.port=/dev/tty.usbmodem5A460814411 \
+  --robot.id=bimanual_follower \
+  --robot.left_arm_config.cameras='{
+    wrist: {"type": "opencv", "index_or_path": 1, "width": 640, "height": 480, "fps": 30},
+    top: {"type": "opencv", "index_or_path": 3, "width": 640, "height": 480, "fps": 30},
+  }' --robot.right_arm_config.cameras='{
+    wrist: {"type": "opencv", "index_or_path": 2, "width": 640, "height": 480, "fps": 30},
+    front: {"type": "opencv", "index_or_path": 4, "width": 640, "height": 480, "fps": 30},
+  }' \
+  --teleop.type=bi_so_leader \
+  --teleop.left_arm_config.port=/dev/tty.usbmodem5A460852721 \
+  --teleop.right_arm_config.port=/dev/tty.usbmodem5A460819811 \
+  --teleop.id=bimanual_leader \
+  --display_data=true \
+  --dataset.repo_id=${HF_USER}/bimanual-so-handover-cube \
+  --dataset.num_episodes=25 \
+  --dataset.single_task="Grab and handover the red cube to the other arm"
+```
+"""
+
+import logging
+import time
+from dataclasses import asdict, dataclass, field
+from pathlib import Path
+from pprint import pformat
+from typing import Any
+
+from lerobot.cameras import (  # noqa: F401
+    CameraConfig,  # noqa: F401
+)
+from lerobot.cameras.opencv.configuration_opencv import OpenCVCameraConfig  # noqa: F401
+from lerobot.cameras.reachy2_camera.configuration_reachy2_camera import Reachy2CameraConfig  # noqa: F401
+from lerobot.cameras.realsense.configuration_realsense import RealSenseCameraConfig  # noqa: F401
+from lerobot.cameras.zmq.configuration_zmq import ZMQCameraConfig  # noqa: F401
+from lerobot.configs import parser
+from lerobot.configs.policies import PreTrainedConfig
+from lerobot.datasets.image_writer import safe_stop_image_writer
+from lerobot.datasets.lerobot_dataset import LeRobotDataset
+from lerobot.datasets.pipeline_features import aggregate_pipeline_dataset_features, create_initial_features
+from lerobot.datasets.utils import build_dataset_frame, combine_feature_dicts
+from lerobot.datasets.video_utils import VideoEncodingManager
+from lerobot.policies.factory import make_policy, make_pre_post_processors
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.policies.utils import make_robot_action
+from lerobot.processor import (
+    PolicyAction,
+    PolicyProcessorPipeline,
+    RobotAction,
+    RobotObservation,
+    RobotProcessorPipeline,
+    make_default_processors,
+)
+from lerobot.processor.rename_processor import rename_stats
+from lerobot.robots import (  # noqa: F401
+    Robot,
+    RobotConfig,
+    bi_openarm_follower,
+    bi_so_follower,
+    earthrover_mini_plus,
+    hope_jr,
+    koch_follower,
+    make_robot_from_config,
+    omx_follower,
+    openarm_follower,
+    reachy2,
+    so_follower,
+    unitree_g1 as unitree_g1_robot,
+)
+from lerobot.teleoperators import (  # noqa: F401
+    Teleoperator,
+    TeleoperatorConfig,
+    bi_openarm_leader,
+    bi_so_leader,
+    homunculus,
+    koch_leader,
+    make_teleoperator_from_config,
+    omx_leader,
+    openarm_leader,
+    reachy2_teleoperator,
+    so_leader,
+    unitree_g1,
+)
+from lerobot.teleoperators.keyboard.teleop_keyboard import KeyboardTeleop
+from lerobot.utils.constants import ACTION, HF_LEROBOT_HOME, OBS_STR
+from lerobot.utils.control_utils import (
+    init_keyboard_listener,
+    is_headless,
+    predict_action,
+    sanity_check_dataset_name,
+    sanity_check_dataset_robot_compatibility,
+)
+from lerobot.utils.import_utils import register_third_party_plugins
+from lerobot.utils.robot_utils import precise_sleep
+from lerobot.utils.utils import (
+    get_safe_torch_device,
+    init_logging,
+    log_say,
+)
+from lerobot.utils.visualization_utils import init_rerun, log_rerun_data
+
+
+@dataclass
+class DatasetRecordConfig:
+    # Dataset identifier. By convention it should match '{hf_username}/{dataset_name}' (e.g. `lerobot/test`).
+    repo_id: str
+    # A short but accurate description of the task performed during the recording (e.g. "Pick the Lego block and drop it in the box on the right.")
+    single_task: str
+    # Root directory where the dataset will be stored (e.g. 'dataset/path').
+    root: str | Path | None = None
+    # Limit the frames per second.
+    fps: int = 30
+    # Number of seconds for data recording for each episode.
+    episode_time_s: int | float = 60
+    # Number of seconds for resetting the environment after each episode.
+    reset_time_s: int | float = 60
+    # Number of episodes to record.
+    num_episodes: int = 50
+    # Encode frames in the dataset into video
+    video: bool = True
+    # Upload dataset to Hugging Face hub.
+    push_to_hub: bool = True
+    # Upload on private repository on the Hugging Face hub.
+    private: bool = False
+    # Add tags to your dataset on the hub.
+    tags: list[str] | None = None
+    # Number of subprocesses handling the saving of frames as PNG. Set to 0 to use threads only;
+    # set to ≥1 to use subprocesses, each using threads to write images. The best number of processes
+    # and threads depends on your system. We recommend 4 threads per camera with 0 processes.
+    # If fps is unstable, adjust the thread count. If still unstable, try using 1 or more subprocesses.
+    num_image_writer_processes: int = 0
+    # Number of threads writing the frames as png images on disk, per camera.
+    # Too many threads might cause unstable teleoperation fps due to main thread being blocked.
+    # Not enough threads might cause low camera fps.
+    num_image_writer_threads_per_camera: int = 4
+    # Number of episodes to record before batch encoding videos
+    # Set to 1 for immediate encoding (default behavior), or higher for batched encoding
+    video_encoding_batch_size: int = 1
+    # Video codec for encoding videos. Options: 'h264', 'hevc', 'libsvtav1'.
+    # Use 'h264' for faster encoding on systems where AV1 encoding is CPU-heavy.
+    vcodec: str = "libsvtav1"
+    # Rename map for the observation to override the image and state keys
+    rename_map: dict[str, str] = field(default_factory=dict)
+
+    def __post_init__(self):
+        if self.single_task is None:
+            raise ValueError("You need to provide a task as argument in `single_task`.")
+
+
+@dataclass
+class RecordConfig:
+    robot: RobotConfig
+    dataset: DatasetRecordConfig
+    # Whether to control the robot with a teleoperator
+    teleop: TeleoperatorConfig | None = None
+    # Whether to control the robot with a policy
+    policy: PreTrainedConfig | None = None
+    # Display all cameras on screen
+    display_data: bool = False
+    # Display data on a remote Rerun server
+    display_ip: str | None = None
+    # Port of the remote Rerun server
+    display_port: int | None = None
+    # Whether to  display compressed images in Rerun
+    display_compressed_images: bool = False
+    # Use vocal synthesis to read events.
+    play_sounds: bool = True
+    # Resume recording on an existing dataset.
+    resume: bool = False
+
+    def __post_init__(self):
+        # HACK: We parse again the cli args here to get the pretrained path if there was one.
+        policy_path = parser.get_path_arg("policy")
+
+        if policy_path:
+            cli_overrides = parser.get_cli_overrides("policy")
+
+            self.policy = PreTrainedConfig.from_pretrained(policy_path, cli_overrides=cli_overrides)
+            self.policy.pretrained_path = policy_path
+
+        if self.teleop is None and self.policy is None:
+            raise ValueError("Choose a policy, a teleoperator or both to control the robot")
+
+    @classmethod
+    def __get_path_fields__(cls) -> list[str]:
+        """This enables the parser to load config from the policy using `--policy.path=local/dir`"""
+        return ["policy"]
+
+
+""" --------------- record_loop() data flow --------------------------
+       [ Robot ]
+           V
+     [ robot.get_observation() ] ---> raw_obs
+           V
+     [ robot_observation_processor ] ---> processed_obs
+           V
+     .-----( ACTION LOGIC )------------------.
+     V                                       V
+     [ From Teleoperator ]                   [ From Policy ]
+     |                                       |
+     |  [teleop.get_action] -> raw_action    |   [predict_action]
+     |          |                            |          |
+     |          V                            |          V
+     | [teleop_action_processor]             |          |
+     |          |                            |          |
+     '---> processed_teleop_action           '---> processed_policy_action
+     |                                       |
+     '-------------------------.-------------'
+                               V
+                  [ robot_action_processor ] --> robot_action_to_send
+                               V
+                    [ robot.send_action() ] -- (Robot Executes)
+                               V
+                    ( Save to Dataset )
+                               V
+                  ( Rerun Log / Loop Wait )
+"""
+
+
+@safe_stop_image_writer
+def record_loop(
+    robot: Robot,
+    events: dict,
+    fps: int,
+    teleop_action_processor: RobotProcessorPipeline[
+        tuple[RobotAction, RobotObservation], RobotAction
+    ],  # runs after teleop
+    robot_action_processor: RobotProcessorPipeline[
+        tuple[RobotAction, RobotObservation], RobotAction
+    ],  # runs before robot
+    robot_observation_processor: RobotProcessorPipeline[
+        RobotObservation, RobotObservation
+    ],  # runs after robot
+    dataset: LeRobotDataset | None = None,
+    teleop: Teleoperator | list[Teleoperator] | None = None,
+    policy: PreTrainedPolicy | None = None,
+    preprocessor: PolicyProcessorPipeline[dict[str, Any], dict[str, Any]] | None = None,
+    postprocessor: PolicyProcessorPipeline[PolicyAction, PolicyAction] | None = None,
+    control_time_s: int | None = None,
+    single_task: str | None = None,
+    display_data: bool = False,
+    display_compressed_images: bool = False,
+):
+    if dataset is not None and dataset.fps != fps:
+        raise ValueError(f"The dataset fps should be equal to requested fps ({dataset.fps} != {fps}).")
+
+    teleop_arm = teleop_keyboard = None
+    if isinstance(teleop, list):
+        teleop_keyboard = next((t for t in teleop if isinstance(t, KeyboardTeleop)), None)
+        teleop_arm = next(
+            (
+                t
+                for t in teleop
+                if isinstance(
+                    t,
+                    (
+                        so_leader.SO100Leader
+                        | so_leader.SO101Leader
+                        | koch_leader.KochLeader
+                        | omx_leader.OmxLeader
+                    ),
+                )
+            ),
+            None,
+        )
+
+        if not (teleop_arm and teleop_keyboard and len(teleop) == 2 and robot.name == "lekiwi_client"):
+            raise ValueError(
+                "For multi-teleop, the list must contain exactly one KeyboardTeleop and one arm teleoperator. Currently only supported for LeKiwi robot."
+            )
+
+    # Reset policy and processor if they are provided
+    if policy is not None and preprocessor is not None and postprocessor is not None:
+        policy.reset()
+        preprocessor.reset()
+        postprocessor.reset()
+
+    # DEBUG: trace every step between here and the first observation
+    logging.warning(f"[DEBUG] record_loop entry. policy={policy is not None}, teleop={teleop is not None}, control_time_s={control_time_s}")
+
+    logging.warning(f"[DEBUG] Pre-loop bus test...")
+    try:
+        _test = robot.get_observation()
+        logging.warning(f"[DEBUG] Pre-loop bus test: OK")
+    except Exception as _e:
+        logging.warning(f"[DEBUG] Pre-loop bus test: FAIL - {_e}")
+
+    # Check serial port state
+    if hasattr(robot, 'bus') and hasattr(robot.bus, 'port_handler'):
+        _ser = robot.bus.port_handler.ser
+        logging.warning(f"[DEBUG] Serial state: is_open={_ser.is_open}, in_waiting={_ser.in_waiting}, timeout={_ser.timeout}")
+
+    timestamp = 0
+    start_episode_t = time.perf_counter()
+    logging.warning(f"[DEBUG] Entering while loop. timestamp={timestamp}, control_time_s={control_time_s}")
+
+    while timestamp < control_time_s:
+        start_loop_t = time.perf_counter()
+
+        logging.warning(f"[DEBUG] Loop iteration. events['exit_early']={events.get('exit_early')}, events['stop_recording']={events.get('stop_recording')}")
+
+        if events["exit_early"]:
+            logging.warning(f"[DEBUG] exit_early is True, breaking")
+            events["exit_early"] = False
+            break
+
+        # Check serial state right before the read
+        if hasattr(robot, 'bus') and hasattr(robot.bus, 'port_handler'):
+            _ser = robot.bus.port_handler.ser
+            _in_waiting = _ser.in_waiting
+            if _in_waiting > 0:
+                logging.warning(f"[DEBUG] {_in_waiting} stale bytes in serial buffer before get_observation!")
+                _stale = _ser.read(_in_waiting)
+                logging.warning(f"[DEBUG] Flushed: {_stale.hex()[:200]}")
+
+        logging.warning(f"[DEBUG] Calling robot.get_observation()...")
+        # Get robot observation
+        obs = robot.get_observation()
+        logging.warning(f"[DEBUG] get_observation() succeeded")
+
+        # Applies a pipeline to the raw robot observation, default is IdentityProcessor
+        obs_processed = robot_observation_processor(obs)
+
+        if policy is not None or dataset is not None:
+            observation_frame = build_dataset_frame(dataset.features, obs_processed, prefix=OBS_STR)
+
+        # Get action from either policy or teleop
+        if policy is not None and preprocessor is not None and postprocessor is not None:
+            action_values = predict_action(
+                observation=observation_frame,
+                policy=policy,
+                device=get_safe_torch_device(policy.config.device),
+                preprocessor=preprocessor,
+                postprocessor=postprocessor,
+                use_amp=policy.config.use_amp,
+                task=single_task,
+                robot_type=robot.robot_type,
+            )
+
+            act_processed_policy: RobotAction = make_robot_action(action_values, dataset.features)
+
+        elif policy is None and isinstance(teleop, Teleoperator):
+            act = teleop.get_action()
+
+            # Applies a pipeline to the raw teleop action, default is IdentityProcessor
+            act_processed_teleop = teleop_action_processor((act, obs))
+
+        elif policy is None and isinstance(teleop, list):
+            arm_action = teleop_arm.get_action()
+            arm_action = {f"arm_{k}": v for k, v in arm_action.items()}
+            keyboard_action = teleop_keyboard.get_action()
+            base_action = robot._from_keyboard_to_base_action(keyboard_action)
+            act = {**arm_action, **base_action} if len(base_action) > 0 else arm_action
+            act_processed_teleop = teleop_action_processor((act, obs))
+        else:
+            logging.info(
+                "No policy or teleoperator provided, skipping action generation."
+                "This is likely to happen when resetting the environment without a teleop device."
+                "The robot won't be at its rest position at the start of the next episode."
+            )
+            continue
+
+        # Applies a pipeline to the action, default is IdentityProcessor
+        if policy is not None and act_processed_policy is not None:
+            action_values = act_processed_policy
+            robot_action_to_send = robot_action_processor((act_processed_policy, obs))
+        else:
+            action_values = act_processed_teleop
+            robot_action_to_send = robot_action_processor((act_processed_teleop, obs))
+
+        # Send action to robot
+        # Action can eventually be clipped using `max_relative_target`,
+        # so action actually sent is saved in the dataset. action = postprocessor.process(action)
+        # TODO(steven, pepijn, adil): we should use a pipeline step to clip the action, so the sent action is the action that we input to the robot.
+        _sent_action = robot.send_action(robot_action_to_send)
+
+        # Write to dataset
+        if dataset is not None:
+            action_frame = build_dataset_frame(dataset.features, action_values, prefix=ACTION)
+            frame = {**observation_frame, **action_frame, "task": single_task}
+            dataset.add_frame(frame)
+
+        if display_data:
+            log_rerun_data(
+                observation=obs_processed, action=action_values, compress_images=display_compressed_images
+            )
+
+        dt_s = time.perf_counter() - start_loop_t
+        precise_sleep(max(1 / fps - dt_s, 0.0))
+
+        timestamp = time.perf_counter() - start_episode_t
+
+
+@parser.wrap()
+def record(cfg: RecordConfig) -> LeRobotDataset:
+    init_logging()
+    logging.info(pformat(asdict(cfg)))
+    if cfg.display_data:
+        init_rerun(session_name="recording", ip=cfg.display_ip, port=cfg.display_port)
+    display_compressed_images = (
+        True
+        if (cfg.display_data and cfg.display_ip is not None and cfg.display_port is not None)
+        else cfg.display_compressed_images
+    )
+
+    robot = make_robot_from_config(cfg.robot)
+    teleop = make_teleoperator_from_config(cfg.teleop) if cfg.teleop is not None else None
+
+    teleop_action_processor, robot_action_processor, robot_observation_processor = make_default_processors()
+
+    dataset_features = combine_feature_dicts(
+        aggregate_pipeline_dataset_features(
+            pipeline=teleop_action_processor,
+            initial_features=create_initial_features(
+                action=robot.action_features
+            ),  # TODO(steven, pepijn): in future this should be come from teleop or policy
+            use_videos=cfg.dataset.video,
+        ),
+        aggregate_pipeline_dataset_features(
+            pipeline=robot_observation_processor,
+            initial_features=create_initial_features(observation=robot.observation_features),
+            use_videos=cfg.dataset.video,
+        ),
+    )
+
+    dataset = None
+    listener = None
+
+    try:
+        if cfg.resume:
+            dataset = LeRobotDataset(
+                cfg.dataset.repo_id,
+                root=cfg.dataset.root,
+                batch_encoding_size=cfg.dataset.video_encoding_batch_size,
+                vcodec=cfg.dataset.vcodec,
+            )
+
+            if hasattr(robot, "cameras") and len(robot.cameras) > 0:
+                dataset.start_image_writer(
+                    num_processes=cfg.dataset.num_image_writer_processes,
+                    num_threads=cfg.dataset.num_image_writer_threads_per_camera * len(robot.cameras),
+                )
+            sanity_check_dataset_robot_compatibility(dataset, robot, cfg.dataset.fps, dataset_features)
+        else:
+            # Create empty dataset or load existing saved episodes
+            sanity_check_dataset_name(cfg.dataset.repo_id, cfg.policy)
+            # Auto-increment repo_id if directory already exists
+            base_repo_id = cfg.dataset.repo_id
+            root = Path(cfg.dataset.root) if cfg.dataset.root else HF_LEROBOT_HOME
+            counter = 0
+            repo_id = base_repo_id
+            while (root / repo_id).exists():
+                counter += 1
+                repo_id = f"{base_repo_id}_{counter}"
+            if repo_id != base_repo_id:
+                logging.info(f"Dataset '{base_repo_id}' already exists, using '{repo_id}'")
+                cfg.dataset.repo_id = repo_id
+            dataset = LeRobotDataset.create(
+                cfg.dataset.repo_id,
+                cfg.dataset.fps,
+                root=cfg.dataset.root,
+                robot_type=robot.name,
+                features=dataset_features,
+                use_videos=cfg.dataset.video,
+                image_writer_processes=cfg.dataset.num_image_writer_processes,
+                image_writer_threads=cfg.dataset.num_image_writer_threads_per_camera * len(robot.cameras),
+                batch_encoding_size=cfg.dataset.video_encoding_batch_size,
+                vcodec=cfg.dataset.vcodec,
+            )
+
+        # Load pretrained policy
+        policy = None if cfg.policy is None else make_policy(cfg.policy, ds_meta=dataset.meta, rename_map=cfg.dataset.rename_map)
+        preprocessor = None
+        postprocessor = None
+        if cfg.policy is not None:
+            preprocessor, postprocessor = make_pre_post_processors(
+                policy_cfg=cfg.policy,
+                pretrained_path=cfg.policy.pretrained_path,
+                dataset_stats=rename_stats(dataset.meta.stats, cfg.dataset.rename_map),
+                preprocessor_overrides={
+                    "device_processor": {"device": cfg.policy.device},
+                    "rename_observations_processor": {"rename_map": cfg.dataset.rename_map},
+                },
+            )
+
+        robot.connect()
+
+        # DEBUG: test motor bus immediately after connect
+        logging.warning("[DEBUG] Testing motor bus after robot.connect()...")
+        try:
+            _test_obs = robot.get_observation()
+            logging.warning(f"[DEBUG] Motor bus OK after connect. State keys: {[k for k in _test_obs if 'pos' in k or 'state' in k]}")
+        except Exception as _e:
+            logging.warning(f"[DEBUG] Motor bus FAILED after connect: {_e}")
+
+        if teleop is not None:
+            teleop.connect()
+
+        listener, events = init_keyboard_listener()
+
+        # DEBUG: test again after keyboard listener init
+        logging.warning("[DEBUG] Testing motor bus after keyboard listener init...")
+        try:
+            _test_obs = robot.get_observation()
+            logging.warning(f"[DEBUG] Motor bus OK after keyboard init.")
+        except Exception as _e:
+            logging.warning(f"[DEBUG] Motor bus FAILED after keyboard init: {_e}")
+
+        with VideoEncodingManager(dataset):
+            # DEBUG: test again inside VideoEncodingManager
+            logging.warning("[DEBUG] Testing motor bus inside VideoEncodingManager...")
+            try:
+                _test_obs = robot.get_observation()
+                logging.warning(f"[DEBUG] Motor bus OK inside VideoEncodingManager.")
+            except Exception as _e:
+                logging.warning(f"[DEBUG] Motor bus FAILED inside VideoEncodingManager: {_e}")
+
+            recorded_episodes = 0
+            while recorded_episodes < cfg.dataset.num_episodes and not events["stop_recording"]:
+                log_say(f"Recording episode {dataset.num_episodes}", cfg.play_sounds)
+                record_loop(
+                    robot=robot,
+                    events=events,
+                    fps=cfg.dataset.fps,
+                    teleop_action_processor=teleop_action_processor,
+                    robot_action_processor=robot_action_processor,
+                    robot_observation_processor=robot_observation_processor,
+                    teleop=teleop,
+                    policy=policy,
+                    preprocessor=preprocessor,
+                    postprocessor=postprocessor,
+                    dataset=dataset,
+                    control_time_s=cfg.dataset.episode_time_s,
+                    single_task=cfg.dataset.single_task,
+                    display_data=cfg.display_data,
+                    display_compressed_images=display_compressed_images,
+                )
+
+                # Execute a few seconds without recording to give time to manually reset the environment
+                # Skip reset for the last episode to be recorded
+                if not events["stop_recording"] and (
+                    (recorded_episodes < cfg.dataset.num_episodes - 1) or events["rerecord_episode"]
+                ):
+                    log_say("Reset the environment", cfg.play_sounds)
+
+                    # reset g1 robot
+                    if robot.name == "unitree_g1":
+                        robot.reset()
+
+                    record_loop(
+                        robot=robot,
+                        events=events,
+                        fps=cfg.dataset.fps,
+                        teleop_action_processor=teleop_action_processor,
+                        robot_action_processor=robot_action_processor,
+                        robot_observation_processor=robot_observation_processor,
+                        teleop=teleop,
+                        control_time_s=cfg.dataset.reset_time_s,
+                        single_task=cfg.dataset.single_task,
+                        display_data=cfg.display_data,
+                    )
+
+                if events["rerecord_episode"]:
+                    log_say("Re-record episode", cfg.play_sounds)
+                    events["rerecord_episode"] = False
+                    events["exit_early"] = False
+                    dataset.clear_episode_buffer()
+                    continue
+
+                dataset.save_episode()
+                recorded_episodes += 1
+    finally:
+        log_say("Stop recording", cfg.play_sounds, blocking=True)
+
+        if dataset:
+            dataset.finalize()
+
+        if robot.is_connected:
+            robot.disconnect()
+        if teleop and teleop.is_connected:
+            teleop.disconnect()
+
+        if not is_headless() and listener:
+            listener.stop()
+
+        if cfg.dataset.push_to_hub:
+            dataset.push_to_hub(tags=cfg.dataset.tags, private=cfg.dataset.private)
+
+        log_say("Exiting", cfg.play_sounds)
+    return dataset
+
+
+def main():
+    register_third_party_plugins()
+    record()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/lerobot_patches/src/lerobot/scripts/lerobot_train.py b/lerobot_patches/src/lerobot/scripts/lerobot_train.py
new file mode 100644
index 0000000000000000000000000000000000000000..a53a436cf43a9ab83f595b94638788bdbff9667e
--- /dev/null
+++ b/lerobot_patches/src/lerobot/scripts/lerobot_train.py
@@ -0,0 +1,564 @@
+#!/usr/bin/env python
+
+# Copyright 2024 The HuggingFace Inc. team. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+import dataclasses
+import logging
+import shutil
+import time
+from contextlib import nullcontext
+from pprint import pformat
+from typing import Any
+
+import torch
+from accelerate import Accelerator
+from termcolor import colored
+from torch.optim import Optimizer
+
+from lerobot.configs import parser
+from lerobot.configs.train import TrainPipelineConfig
+from lerobot.datasets.factory import make_dataset
+from lerobot.datasets.sampler import EpisodeAwareSampler
+from lerobot.datasets.utils import cycle
+from lerobot.envs.factory import make_env, make_env_pre_post_processors
+from lerobot.envs.utils import close_envs
+from lerobot.optim.factory import make_optimizer_and_scheduler
+from lerobot.policies.factory import make_policy, make_pre_post_processors
+from lerobot.policies.pretrained import PreTrainedPolicy
+from lerobot.rl.wandb_utils import WandBLogger
+from lerobot.scripts.lerobot_eval import eval_policy_all
+from lerobot.utils.import_utils import register_third_party_plugins
+from lerobot.utils.logging_utils import AverageMeter, MetricsTracker
+from lerobot.utils.random_utils import set_seed
+from lerobot.utils.train_utils import (
+    get_step_checkpoint_dir,
+    get_step_identifier,
+    load_training_state,
+    save_checkpoint,
+    update_last_checkpoint,
+)
+from lerobot.utils.utils import (
+    format_big_number,
+    has_method,
+    init_logging,
+)
+
+
+def update_policy(
+    train_metrics: MetricsTracker,
+    policy: PreTrainedPolicy,
+    batch: Any,
+    optimizer: Optimizer,
+    grad_clip_norm: float,
+    accelerator: Accelerator,
+    lr_scheduler=None,
+    lock=None,
+    rabc_weights_provider=None,
+) -> tuple[MetricsTracker, dict]:
+    """
+    Performs a single training step to update the policy's weights.
+
+    This function executes the forward and backward passes, clips gradients, and steps the optimizer and
+    learning rate scheduler. Accelerator handles mixed-precision training automatically.
+
+    Args:
+        train_metrics: A MetricsTracker instance to record training statistics.
+        policy: The policy model to be trained.
+        batch: A batch of training data.
+        optimizer: The optimizer used to update the policy's parameters.
+        grad_clip_norm: The maximum norm for gradient clipping.
+        accelerator: The Accelerator instance for distributed training and mixed precision.
+        lr_scheduler: An optional learning rate scheduler.
+        lock: An optional lock for thread-safe optimizer updates.
+        rabc_weights_provider: Optional RABCWeights instance for sample weighting.
+
+    Returns:
+        A tuple containing:
+        - The updated MetricsTracker with new statistics for this step.
+        - A dictionary of outputs from the policy's forward pass, for logging purposes.
+    """
+    start_time = time.perf_counter()
+    policy.train()
+
+    # Get RA-BC weights if enabled
+    rabc_batch_weights = None
+    rabc_batch_stats = None
+    if rabc_weights_provider is not None:
+        rabc_batch_weights, rabc_batch_stats = rabc_weights_provider.compute_batch_weights(batch)
+
+    # Let accelerator handle mixed precision
+    with accelerator.autocast():
+        # Use per-sample loss when RA-BC is enabled for proper weighting
+        if rabc_batch_weights is not None:
+            # Get per-sample losses
+            per_sample_loss, output_dict = policy.forward(batch, reduction="none")
+
+            # Apply RA-BC weights: L_RA-BC = Σ(w_i * l_i) / (Σw_i + ε)
+            # rabc_batch_weights is already normalized to sum to batch_size
+            epsilon = 1e-6
+            loss = (per_sample_loss * rabc_batch_weights).sum() / (rabc_batch_weights.sum() + epsilon)
+            # Log raw mean weight (before normalization) - this is the meaningful metric
+            output_dict["rabc_mean_weight"] = rabc_batch_stats["raw_mean_weight"]
+            output_dict["rabc_num_zero_weight"] = rabc_batch_stats["num_zero_weight"]
+            output_dict["rabc_num_full_weight"] = rabc_batch_stats["num_full_weight"]
+        else:
+            loss, output_dict = policy.forward(batch)
+
+        # TODO(rcadene): policy.unnormalize_outputs(out_dict)
+
+    # Use accelerator's backward method
+    accelerator.backward(loss)
+
+    # Clip gradients if specified
+    if grad_clip_norm > 0:
+        grad_norm = accelerator.clip_grad_norm_(policy.parameters(), grad_clip_norm)
+    else:
+        grad_norm = torch.nn.utils.clip_grad_norm_(
+            policy.parameters(), float("inf"), error_if_nonfinite=False
+        )
+
+    # Optimizer step
+    with lock if lock is not None else nullcontext():
+        optimizer.step()
+
+    optimizer.zero_grad()
+
+    # Step through pytorch scheduler at every batch instead of epoch
+    if lr_scheduler is not None:
+        lr_scheduler.step()
+
+    # Update internal buffers if policy has update method
+    if has_method(accelerator.unwrap_model(policy, keep_fp32_wrapper=True), "update"):
+        accelerator.unwrap_model(policy, keep_fp32_wrapper=True).update()
+
+    train_metrics.loss = loss.item()
+    train_metrics.grad_norm = grad_norm.item()
+    train_metrics.lr = optimizer.param_groups[0]["lr"]
+    train_metrics.update_s = time.perf_counter() - start_time
+    return train_metrics, output_dict
+
+
+@parser.wrap()
+def train(cfg: TrainPipelineConfig, accelerator: Accelerator | None = None):
+    """
+    Main function to train a policy.
+
+    This function orchestrates the entire training pipeline, including:
+    - Setting up logging, seeding, and device configuration.
+    - Creating the dataset, evaluation environment (if applicable), policy, and optimizer.
+    - Handling resumption from a checkpoint.
+    - Running the main training loop, which involves fetching data batches and calling `update_policy`.
+    - Periodically logging metrics, saving model checkpoints, and evaluating the policy.
+    - Pushing the final trained model to the Hugging Face Hub if configured.
+
+    Args:
+        cfg: A `TrainPipelineConfig` object containing all training configurations.
+        accelerator: Optional Accelerator instance. If None, one will be created automatically.
+    """
+    cfg.validate()
+
+    # Create Accelerator if not provided
+    # It will automatically detect if running in distributed mode or single-process mode
+    # We set step_scheduler_with_optimizer=False to prevent accelerate from adjusting the lr_scheduler steps based on the num_processes
+    # We set find_unused_parameters=True to handle models with conditional computation
+    if accelerator is None:
+        from accelerate.utils import DistributedDataParallelKwargs
+
+        ddp_kwargs = DistributedDataParallelKwargs(find_unused_parameters=True)
+        # Accelerate auto-detects the device based on the available hardware and ignores the policy.device setting.
+        # Force the device to be CPU when policy.device is set to CPU.
+        force_cpu = cfg.policy.device == "cpu"
+        accelerator = Accelerator(
+            step_scheduler_with_optimizer=False,
+            kwargs_handlers=[ddp_kwargs],
+            cpu=force_cpu,
+        )
+
+    init_logging(accelerator=accelerator)
+
+    # Determine if this is the main process (for logging and checkpointing)
+    # When using accelerate, only the main process should log to avoid duplicate outputs
+    is_main_process = accelerator.is_main_process
+
+    # Only log on main process
+    if is_main_process:
+        logging.info(pformat(cfg.to_dict()))
+
+    # Initialize wandb only on main process
+    if cfg.wandb.enable and cfg.wandb.project and is_main_process:
+        wandb_logger = WandBLogger(cfg)
+    else:
+        wandb_logger = None
+        if is_main_process:
+            logging.info(colored("Logs will be saved locally.", "yellow", attrs=["bold"]))
+
+    if cfg.seed is not None:
+        set_seed(cfg.seed, accelerator=accelerator)
+
+    # Use accelerator's device
+    device = accelerator.device
+    torch.backends.cudnn.benchmark = True
+    torch.backends.cuda.matmul.allow_tf32 = True
+
+    # Dataset loading synchronization: main process downloads first to avoid race conditions
+    if is_main_process:
+        logging.info("Creating dataset")
+        dataset = make_dataset(cfg)
+
+    accelerator.wait_for_everyone()
+
+    # Now all other processes can safely load the dataset
+    if not is_main_process:
+        dataset = make_dataset(cfg)
+
+    # Create environment used for evaluating checkpoints during training on simulation data.
+    # On real-world data, no need to create an environment as evaluations are done outside train.py,
+    # using the eval.py instead, with gym_dora environment and dora-rs.
+    eval_env = None
+    if cfg.eval_freq > 0 and cfg.env is not None and is_main_process:
+        logging.info("Creating env")
+        eval_env = make_env(cfg.env, n_envs=cfg.eval.batch_size, use_async_envs=cfg.eval.use_async_envs)
+
+    if is_main_process:
+        logging.info("Creating policy")
+    policy = make_policy(
+        cfg=cfg.policy,
+        ds_meta=dataset.meta,
+        rename_map=cfg.rename_map,
+    )
+
+    if cfg.peft is not None:
+        logging.info("Using PEFT! Wrapping model.")
+        # Convert CLI peft config to dict for overrides
+        peft_cli_overrides = dataclasses.asdict(cfg.peft)
+        policy = policy.wrap_with_peft(peft_cli_overrides=peft_cli_overrides)
+
+    # Wait for all processes to finish policy creation before continuing
+    accelerator.wait_for_everyone()
+
+    # Create processors - only provide dataset_stats if not resuming from saved processors
+    processor_kwargs = {}
+    postprocessor_kwargs = {}
+    if (cfg.policy.pretrained_path and not cfg.resume) or not cfg.policy.pretrained_path:
+        # Only provide dataset_stats when not resuming from saved processor state
+        processor_kwargs["dataset_stats"] = dataset.meta.stats
+
+    # For SARM, always provide dataset_meta for progress normalization
+    if cfg.policy.type == "sarm":
+        processor_kwargs["dataset_meta"] = dataset.meta
+
+    if cfg.policy.pretrained_path is not None:
+        processor_kwargs["preprocessor_overrides"] = {
+            "device_processor": {"device": device.type},
+            "normalizer_processor": {
+                "stats": dataset.meta.stats,
+                "features": {**policy.config.input_features, **policy.config.output_features},
+                "norm_map": policy.config.normalization_mapping,
+            },
+        }
+        processor_kwargs["preprocessor_overrides"]["rename_observations_processor"] = {
+            "rename_map": cfg.rename_map
+        }
+        postprocessor_kwargs["postprocessor_overrides"] = {
+            "unnormalizer_processor": {
+                "stats": dataset.meta.stats,
+                "features": policy.config.output_features,
+                "norm_map": policy.config.normalization_mapping,
+            },
+        }
+
+    preprocessor, postprocessor = make_pre_post_processors(
+        policy_cfg=cfg.policy,
+        pretrained_path=cfg.policy.pretrained_path,
+        **processor_kwargs,
+        **postprocessor_kwargs,
+    )
+
+    if is_main_process:
+        logging.info("Creating optimizer and scheduler")
+    optimizer, lr_scheduler = make_optimizer_and_scheduler(cfg, policy)
+
+    # Load precomputed SARM progress for RA-BC if enabled
+    # Generate progress using: src/lerobot/policies/sarm/compute_rabc_weights.py
+    rabc_weights = None
+    if cfg.use_rabc:
+        from lerobot.utils.rabc import RABCWeights
+
+        # Get chunk_size from policy config
+        chunk_size = getattr(policy.config, "chunk_size", None)
+        if chunk_size is None:
+            raise ValueError("Chunk size is not found in policy config")
+
+        head_mode = getattr(cfg, "rabc_head_mode", "sparse")
+        logging.info(f"Loading SARM progress for RA-BC from {cfg.rabc_progress_path}")
+        logging.info(f"Using chunk_size={chunk_size} from policy config, head_mode={head_mode}")
+        rabc_weights = RABCWeights(
+            progress_path=cfg.rabc_progress_path,
+            chunk_size=chunk_size,
+            head_mode=head_mode,
+            kappa=getattr(cfg, "rabc_kappa", 0.01),
+            epsilon=getattr(cfg, "rabc_epsilon", 1e-6),
+            device=device,
+        )
+
+    step = 0  # number of policy updates (forward + backward + optim)
+
+    if cfg.resume:
+        step, optimizer, lr_scheduler = load_training_state(cfg.checkpoint_path, optimizer, lr_scheduler)
+
+    num_learnable_params = sum(p.numel() for p in policy.parameters() if p.requires_grad)
+    num_total_params = sum(p.numel() for p in policy.parameters())
+
+    if is_main_process:
+        logging.info(colored("Output dir:", "yellow", attrs=["bold"]) + f" {cfg.output_dir}")
+        if cfg.env is not None:
+            logging.info(f"{cfg.env.task=}")
+            logging.info("Creating environment processors")
+            env_preprocessor, env_postprocessor = make_env_pre_post_processors(
+                env_cfg=cfg.env, policy_cfg=cfg.policy
+            )
+        logging.info(f"{cfg.steps=} ({format_big_number(cfg.steps)})")
+        logging.info(f"{dataset.num_frames=} ({format_big_number(dataset.num_frames)})")
+        logging.info(f"{dataset.num_episodes=}")
+        num_processes = accelerator.num_processes
+        effective_bs = cfg.batch_size * num_processes
+        logging.info(f"Effective batch size: {cfg.batch_size} x {num_processes} = {effective_bs}")
+        logging.info(f"{num_learnable_params=} ({format_big_number(num_learnable_params)})")
+        logging.info(f"{num_total_params=} ({format_big_number(num_total_params)})")
+
+    # create dataloader for offline training
+    if hasattr(cfg.policy, "drop_n_last_frames"):
+        shuffle = False
+        sampler = EpisodeAwareSampler(
+            dataset.meta.episodes["dataset_from_index"],
+            dataset.meta.episodes["dataset_to_index"],
+            episode_indices_to_use=dataset.episodes,
+            drop_n_last_frames=cfg.policy.drop_n_last_frames,
+            shuffle=True,
+        )
+    else:
+        shuffle = True
+        sampler = None
+
+    dataloader = torch.utils.data.DataLoader(
+        dataset,
+        num_workers=cfg.num_workers,
+        batch_size=cfg.batch_size,
+        shuffle=shuffle and not cfg.dataset.streaming,
+        sampler=sampler,
+        pin_memory=device.type == "cuda",
+        drop_last=False,
+        prefetch_factor=2 if cfg.num_workers > 0 else None,
+    )
+
+    # Prepare everything with accelerator
+    accelerator.wait_for_everyone()
+    policy, optimizer, dataloader, lr_scheduler = accelerator.prepare(
+        policy, optimizer, dataloader, lr_scheduler
+    )
+    dl_iter = cycle(dataloader)
+
+    policy.train()
+
+    train_metrics = {
+        "loss": AverageMeter("loss", ":.3f"),
+        "grad_norm": AverageMeter("grdn", ":.3f"),
+        "lr": AverageMeter("lr", ":0.1e"),
+        "update_s": AverageMeter("updt_s", ":.3f"),
+        "dataloading_s": AverageMeter("data_s", ":.3f"),
+    }
+
+    # Use effective batch size for proper epoch calculation in distributed training
+    effective_batch_size = cfg.batch_size * accelerator.num_processes
+    train_tracker = MetricsTracker(
+        effective_batch_size,
+        dataset.num_frames,
+        dataset.num_episodes,
+        train_metrics,
+        initial_step=step,
+        accelerator=accelerator,
+    )
+
+    if is_main_process:
+        logging.info(
+            f"Start offline training on a fixed dataset, with effective batch size: {effective_batch_size}"
+        )
+
+    prev_checkpoint_dir = None
+    for _ in range(step, cfg.steps):
+        start_time = time.perf_counter()
+        batch = next(dl_iter)
+        batch = preprocessor(batch)
+        train_tracker.dataloading_s = time.perf_counter() - start_time
+
+        train_tracker, output_dict = update_policy(
+            train_tracker,
+            policy,
+            batch,
+            optimizer,
+            cfg.optimizer.grad_clip_norm,
+            accelerator=accelerator,
+            lr_scheduler=lr_scheduler,
+            rabc_weights_provider=rabc_weights,
+        )
+
+        # Note: eval and checkpoint happens *after* the `step`th training update has completed, so we
+        # increment `step` here.
+        step += 1
+        train_tracker.step()
+
+        is_log_step = cfg.log_freq > 0 and step % cfg.log_freq == 0 and is_main_process
+        is_early_stop = cfg.early_stop_steps is not None and step >= cfg.early_stop_steps
+        is_saving_step = step % cfg.save_freq == 0 or step == cfg.steps or is_early_stop
+        is_eval_step = cfg.eval_freq > 0 and step % cfg.eval_freq == 0
+
+        if is_log_step:
+            logging.info(train_tracker)
+            if wandb_logger:
+                wandb_log_dict = train_tracker.to_dict()
+                if output_dict:
+                    wandb_log_dict.update(output_dict)
+                # Log RA-BC statistics if enabled
+                if rabc_weights is not None:
+                    rabc_stats = rabc_weights.get_stats()
+                    wandb_log_dict.update(
+                        {
+                            "rabc_delta_mean": rabc_stats["delta_mean"],
+                            "rabc_delta_std": rabc_stats["delta_std"],
+                            "rabc_num_frames": rabc_stats["num_frames"],
+                        }
+                    )
+                wandb_logger.log_dict(wandb_log_dict, step)
+            train_tracker.reset_averages()
+
+        if cfg.save_checkpoint and is_saving_step:
+            if is_main_process:
+                logging.info(f"Checkpoint policy after step {step}")
+                checkpoint_dir = get_step_checkpoint_dir(cfg.output_dir, cfg.steps, step)
+                save_checkpoint(
+                    checkpoint_dir=checkpoint_dir,
+                    step=step,
+                    cfg=cfg,
+                    policy=accelerator.unwrap_model(policy),
+                    optimizer=optimizer,
+                    scheduler=lr_scheduler,
+                    preprocessor=preprocessor,
+                    postprocessor=postprocessor,
+                )
+                update_last_checkpoint(checkpoint_dir)
+                # Upload checkpoint to HF then delete previous local copy
+                uploaded = False
+                if cfg.policy.push_to_hub:
+                    try:
+                        from huggingface_hub import HfApi
+                        api = HfApi()
+                        api.upload_folder(
+                            folder_path=checkpoint_dir,
+                            repo_id=cfg.policy.repo_id,
+                            path_in_repo=f"checkpoints/step_{step:06d}",
+                        )
+                        logging.info(f"Uploaded checkpoint step {step} to HF")
+                        uploaded = True
+                    except Exception as e:
+                        logging.warning(f"Failed to upload checkpoint step {step}: {e}")
+                # Only delete previous if current was uploaded successfully
+                if uploaded and prev_checkpoint_dir is not None and Path(prev_checkpoint_dir).exists():
+                    shutil.rmtree(prev_checkpoint_dir)
+                prev_checkpoint_dir = checkpoint_dir
+                if wandb_logger:
+                    wandb_logger.log_policy(checkpoint_dir)
+
+            accelerator.wait_for_everyone()
+
+        if cfg.env and is_eval_step:
+            if is_main_process:
+                step_id = get_step_identifier(step, cfg.steps)
+                logging.info(f"Eval policy at step {step}")
+                with torch.no_grad(), accelerator.autocast():
+                    eval_info = eval_policy_all(
+                        envs=eval_env,  # dict[suite][task_id] -> vec_env
+                        policy=accelerator.unwrap_model(policy),
+                        env_preprocessor=env_preprocessor,
+                        env_postprocessor=env_postprocessor,
+                        preprocessor=preprocessor,
+                        postprocessor=postprocessor,
+                        n_episodes=cfg.eval.n_episodes,
+                        videos_dir=cfg.output_dir / "eval" / f"videos_step_{step_id}",
+                        max_episodes_rendered=4,
+                        start_seed=cfg.seed,
+                        max_parallel_tasks=cfg.env.max_parallel_tasks,
+                    )
+                # overall metrics (suite-agnostic)
+                aggregated = eval_info["overall"]
+
+                # optional: per-suite logging
+                for suite, suite_info in eval_info.items():
+                    logging.info("Suite %s aggregated: %s", suite, suite_info)
+
+                # meters/tracker
+                eval_metrics = {
+                    "avg_sum_reward": AverageMeter("∑rwrd", ":.3f"),
+                    "pc_success": AverageMeter("success", ":.1f"),
+                    "eval_s": AverageMeter("eval_s", ":.3f"),
+                }
+                eval_tracker = MetricsTracker(
+                    cfg.batch_size,
+                    dataset.num_frames,
+                    dataset.num_episodes,
+                    eval_metrics,
+                    initial_step=step,
+                    accelerator=accelerator,
+                )
+                eval_tracker.eval_s = aggregated.pop("eval_s")
+                eval_tracker.avg_sum_reward = aggregated.pop("avg_sum_reward")
+                eval_tracker.pc_success = aggregated.pop("pc_success")
+                if wandb_logger:
+                    wandb_log_dict = {**eval_tracker.to_dict(), **eval_info}
+                    wandb_logger.log_dict(wandb_log_dict, step, mode="eval")
+                    wandb_logger.log_video(eval_info["overall"]["video_paths"][0], step, mode="eval")
+
+            accelerator.wait_for_everyone()
+
+        if is_early_stop:
+            if is_main_process:
+                logging.info(f"Early stop at step {step} (early_stop_steps={cfg.early_stop_steps})")
+            break
+
+    if eval_env:
+        close_envs(eval_env)
+
+    if is_main_process:
+        logging.info("End of training")
+
+        if cfg.policy.push_to_hub:
+            unwrapped_policy = accelerator.unwrap_model(policy)
+            if cfg.policy.use_peft:
+                unwrapped_policy.push_model_to_hub(cfg, peft_model=unwrapped_policy)
+            else:
+                unwrapped_policy.push_model_to_hub(cfg)
+            preprocessor.push_to_hub(cfg.policy.repo_id)
+            postprocessor.push_to_hub(cfg.policy.repo_id)
+
+    # Properly clean up the distributed process group
+    accelerator.wait_for_everyone()
+    accelerator.end_training()
+
+
+def main():
+    register_third_party_plugins()
+    train()
+
+
+if __name__ == "__main__":
+    main()
diff --git a/preflight.sh b/preflight.sh
new file mode 100755
index 0000000000000000000000000000000000000000..434479da3dc6ab99ee5e5ac150061e441724ceab
--- /dev/null
+++ b/preflight.sh
@@ -0,0 +1,108 @@
+#!/bin/bash
+# Preflight check: validates the training environment without GPUs or data
+# Run inside the Docker container:
+#   docker run --rm pi05-training ./preflight.sh
+
+PASS=0
+FAIL=0
+
+check() {
+    local name="$1"
+    shift
+    if "$@" > /dev/null 2>&1; then
+        echo "  PASS  $name"
+        PASS=$((PASS + 1))
+    else
+        echo "  FAIL  $name"
+        FAIL=$((FAIL + 1))
+    fi
+}
+
+echo "=== Preflight Checks ==="
+echo ""
+
+echo "-- Python & Core Packages --"
+check "python 3.10"          python -c "import sys; assert sys.version_info[:2] == (3,10)"
+check "torch imports"        python -c "import torch"
+check "transformers >= 4.45" python -c "import transformers; v=transformers.__version__; assert tuple(int(x) for x in v.split('.')[:2]) >= (4,45), v"
+check "accelerate imports"   python -c "import accelerate"
+check "lerobot imports"      python -c "import lerobot"
+check "wandb imports"        python -c "import wandb"
+check "huggingface_hub"      python -c "import huggingface_hub"
+
+echo ""
+echo "-- PaliGemma Config (the previous crash) --"
+check "PaliGemma registered" python -c "
+from transformers import AutoConfig
+# This is what crashed before - CONFIG_MAPPING['paligemma'] was None
+c = AutoConfig.for_model('paligemma')
+assert c is not None
+"
+
+echo ""
+echo "-- FFmpeg --"
+check "ffmpeg available"     ffmpeg -version
+check "ffmpeg version >= 6"  python -c "
+import subprocess, re
+out = subprocess.check_output(['ffmpeg', '-version']).decode()
+ver = int(re.search(r'ffmpeg version (\d+)', out).group(1))
+assert ver >= 6, f'ffmpeg {ver} < 6'
+"
+
+echo ""
+echo "-- Project Files --"
+check "filtered_index.json"  test -f /workspace/pi05-so100-diverse/filtered_index.json
+check "norm_stats.json"      test -f /workspace/pi05-so100-diverse/norm_stats.json
+check "train_cloud.sh"       test -f /workspace/pi05-so100-diverse/train_cloud.sh
+check "so100_dataset.py"     test -f /workspace/pi05-so100-diverse/so100_dataset.py
+
+echo ""
+echo "-- LeRobot Patches Applied --"
+check "patched train script" python -c "
+import lerobot.scripts.lerobot_train
+import inspect
+src = inspect.getsource(lerobot.scripts.lerobot_train)
+assert 'early_stop_steps' in src, 'train patch not applied'
+"
+check "patched factory"      python -c "
+import lerobot.datasets.factory
+import inspect
+src = inspect.getsource(lerobot.datasets.factory)
+assert 'so100:' in src, 'factory patch not applied'
+"
+
+echo ""
+echo "-- Accelerate Multi-GPU Config --"
+check "accelerate launch"    accelerate launch --help
+
+echo ""
+echo "-- HuggingFace Auth --"
+if [ -n "$HF_TOKEN" ]; then
+    check "HF_TOKEN valid" python -c "
+from huggingface_hub import HfApi
+api = HfApi(token='$HF_TOKEN')
+api.whoami()
+"
+else
+    echo "  SKIP  HF_TOKEN not set (set it to validate auth + Gemma license)"
+fi
+
+echo ""
+echo "-- Weights Download (dry check) --"
+if [ -n "$HF_TOKEN" ]; then
+    check "pi05_base accessible" python -c "
+from huggingface_hub import HfApi
+api = HfApi(token='$HF_TOKEN')
+info = api.model_info('lerobot/pi05_base')
+assert info is not None
+"
+else
+    echo "  SKIP  Need HF_TOKEN to check model access"
+fi
+
+echo ""
+echo "================================"
+echo "  Results: $PASS passed, $FAIL failed"
+echo "================================"
+
+[ "$FAIL" -eq 0 ] && exit 0 || exit 1
diff --git a/prep_dataset.sh b/prep_dataset.sh
new file mode 100644
index 0000000000000000000000000000000000000000..a75cac8181b6e4810bef33606c0cc1e24ab57dc4
--- /dev/null
+++ b/prep_dataset.sh
@@ -0,0 +1,58 @@
+#!/bin/bash
+# Run on a prep instance to: download the training subset, split-tar it, upload to HF.
+# Usage: HF_TOKEN=xxx bash prep_dataset.sh
+
+set -e
+
+if [ -z "$HF_TOKEN" ]; then echo "ERROR: export HF_TOKEN first"; exit 1; fi
+
+DATA_DIR="$HOME/data/community_dataset_v3"
+TAR_DIR="$HOME/data/tar_chunks"
+CHUNK_SIZE="2G"
+REPO_ID="StrongRoboticsLab/pi05-so100-diverse"
+
+echo "=== Step 1: Clone project repo ==="
+if [ ! -d $HOME/pi05-so100-diverse ]; then
+    git clone https://huggingface.co/$REPO_ID $HOME/pi05-so100-diverse
+fi
+cd $HOME/pi05-so100-diverse
+
+echo "=== Step 2: Install deps ==="
+pip install -q huggingface_hub pandas
+
+echo "=== Step 3: Download training subset ==="
+python download_subset.py \
+    --index filtered_index.json \
+    --output "$DATA_DIR" \
+    --token "$HF_TOKEN"
+
+echo "=== Step 4: Split-tar the dataset ==="
+mkdir -p "$TAR_DIR"
+echo "Tarring $(du -sh $DATA_DIR | cut -f1) into ${CHUNK_SIZE} chunks..."
+time tar cf - -C "$(dirname $DATA_DIR)" "$(basename $DATA_DIR)" \
+    | split -b "$CHUNK_SIZE" -d -a 3 - "$TAR_DIR/training_subset.tar."
+echo "Chunks created:"
+ls -lh "$TAR_DIR"/training_subset.tar.*
+
+echo "=== Step 5: Upload chunks to HF ==="
+python -c "
+from huggingface_hub import HfApi
+import glob
+
+api = HfApi(token='$HF_TOKEN')
+chunks = sorted(glob.glob('$TAR_DIR/training_subset.tar.*'))
+print(f'Uploading {len(chunks)} chunks...')
+for i, chunk in enumerate(chunks):
+    name = chunk.split('/')[-1]
+    print(f'  [{i+1}/{len(chunks)}] {name}')
+    api.upload_file(
+        path_or_fileobj=chunk,
+        path_in_repo=f'dataset/{name}',
+        repo_id='$REPO_ID',
+        repo_type='model',
+    )
+print('All chunks uploaded')
+"
+
+echo "=== Done! ==="
+echo "Chunks are at dataset/training_subset.tar.* in the HF repo."
diff --git a/train_cloud.sh b/train_cloud.sh
new file mode 100644
index 0000000000000000000000000000000000000000..48882762236406325d4f8a0be09eda8dad649b2f
--- /dev/null
+++ b/train_cloud.sh
@@ -0,0 +1,69 @@
+#!/bin/bash
+
+# Run training on cloud instance, then auto-stop
+
+# Prerequisites: set WANDB_API_KEY and HF_TOKEN
+
+set +e
+
+LOG_FILE="training_$(date +%Y%m%d_%H%M%S).log"
+NUM_GPUS="${NUM_GPUS:-1}"
+DATASET_DIR="${DATASET_DIR:-/ephemeral/community_dataset_v3}"
+
+if [ -z "$WANDB_API_KEY" ]; then echo "ERROR: WANDB_API_KEY not set"; exit 1; fi
+
+if [ -z "$HF_TOKEN" ]; then echo "ERROR: HF_TOKEN not set"; exit 1; fi
+
+if [ ! -d "$DATASET_DIR" ]; then echo "ERROR: Dataset not at $DATASET_DIR"; exit 1; fi
+
+echo "=== Starting Training ===" | tee "$LOG_FILE"
+
+# Activate conda env if available (bare metal), otherwise assume deps are global (Docker)
+if command -v conda &> /dev/null; then
+    eval "$(conda shell.bash hook)"
+    conda activate lerobot
+fi
+
+ACCEL_FLAGS=""
+if [ "$NUM_GPUS" -gt 1 ]; then
+    ACCEL_FLAGS="--multi_gpu --num_processes $NUM_GPUS"
+fi
+
+accelerate launch $ACCEL_FLAGS \
+-m lerobot.scripts.lerobot_train \
+--dataset.repo_id="so100:$DATASET_DIR:/workspace/pi05-so100-diverse/filtered_index.json:/workspace/pi05-so100-diverse/norm_stats.json" \
+--policy.path=lerobot/pi05_base \
+--policy.train_expert_only=true \
+--policy.dtype=bfloat16 \
+--policy.gradient_checkpointing=true \
+--policy.push_to_hub=true \
+--policy.repo_id=StrongRoboticsLab/pi05-so100-diverse \
+--policy.normalization_mapping='{"VISUAL": "IDENTITY", "STATE": "MEAN_STD", "ACTION": "MEAN_STD"}' \
+--policy.scheduler_warmup_steps=1000 \
+--policy.scheduler_decay_steps=85000 \
+--rename_map='{"observation.images.image": "observation.images.base_0_rgb", "observation.images.image2": "observation.images.left_wrist_0_rgb"}' \
+--batch_size=64 \
+--steps=85000 \
+--save_freq=500 \
+--log_freq=50 \
+--num_workers=4 \
+--wandb.enable=true \
+--wandb.project=pi05-so100-diverse \
+--output_dir=/ephemeral/production_run \
+2>&1 | tee -a "$LOG_FILE"
+
+TRAIN_EXIT=${PIPESTATUS[0]}
+
+echo "=== Training Complete (exit: $TRAIN_EXIT) ===" | tee -a "$LOG_FILE"
+
+python -c "
+from huggingface_hub import HfApi
+HfApi().upload_file(path_or_fileobj='$LOG_FILE', path_in_repo='logs/$LOG_FILE',
+repo_id='StrongRoboticsLab/pi05-so100-diverse', repo_type='model')
+print('Log uploaded')
+" 2>&1 | tee -a "$LOG_FILE"
+
+# Auto-shutdown if running on cloud (sudo available)
+if command -v sudo &> /dev/null; then
+    sudo shutdown -h now
+fi