menhguin
diff --git a/‎download_and_upload_dataset.py
-779 b/‎download_and_upload_dataset.py
-779
diff --git a/‎lerobot/__init__.py
+7 b/‎lerobot/__init__.py
+7
diff --git a/‎lerobot/common/datasets/_diffusion_policy_replay_buffer.py ‎lerobot/common/datasets/push_dataset_to_hub/_diffusion_policy_replay_buffer.py b/‎lerobot/common/datasets/_diffusion_policy_replay_buffer.py ‎lerobot/common/datasets/push_dataset_to_hub/_diffusion_policy_replay_buffer.py
diff --git a/‎lerobot/common/datasets/push_dataset_to_hub/_download_raw.py
+179 b/‎lerobot/common/datasets/push_dataset_to_hub/_download_raw.py
+179
diff --git a/‎lerobot/common/datasets/_umi_imagecodecs_numcodecs.py ‎lerobot/common/datasets/push_dataset_to_hub/_umi_imagecodecs_numcodecs.py b/‎lerobot/common/datasets/_umi_imagecodecs_numcodecs.py ‎lerobot/common/datasets/push_dataset_to_hub/_umi_imagecodecs_numcodecs.py
diff --git a/‎lerobot/common/datasets/push_dataset_to_hub/aloha_processor.py
+199 b/‎lerobot/common/datasets/push_dataset_to_hub/aloha_processor.py
+199
@@ -61,6 +61,13 @@
     itertools.chain(*available_datasets_per_env.values(), available_datasets_without_env)
 )
 
+# TODO(rcadene, aliberts, alexander-soare): Add real-world env with a gym API
+available_datasets_without_env = ["lerobot/umi_cup_in_the_wild"]
+
+available_datasets = list(
+    itertools.chain(*available_datasets_per_env.values(), available_datasets_without_env)
+)
+
 available_policies = [
     "act",
     "diffusion",
 
@@ -0,0 +1,179 @@
+"""
+This file contains all obsolete download scripts. They are centralized here to not have to load
+useless dependencies when using datasets.
+"""
+
+import io
+from pathlib import Path
+
+import tqdm
+
+
+def download_raw(root, dataset_id) -> Path:
+    if "pusht" in dataset_id:
+        return download_pusht(root=root, dataset_id=dataset_id)
+    elif "xarm" in dataset_id:
+        return download_xarm(root=root, dataset_id=dataset_id)
+    elif "aloha" in dataset_id:
+        return download_aloha(root=root, dataset_id=dataset_id)
+    elif "umi" in dataset_id:
+        return download_umi(root=root, dataset_id=dataset_id)
+    else:
+        raise ValueError(dataset_id)
+
+
+def download_and_extract_zip(url: str, destination_folder: Path) -> bool:
+    import zipfile
+
+    import requests
+
+    print(f"downloading from {url}")
+    response = requests.get(url, stream=True)
+    if response.status_code == 200:
+        total_size = int(response.headers.get("content-length", 0))
+        progress_bar = tqdm.tqdm(total=total_size, unit="B", unit_scale=True)
+
+        zip_file = io.BytesIO()
+        for chunk in response.iter_content(chunk_size=1024):
+            if chunk:
+                zip_file.write(chunk)
+                progress_bar.update(len(chunk))
+
+        progress_bar.close()
+
+        zip_file.seek(0)
+
+        with zipfile.ZipFile(zip_file, "r") as zip_ref:
+            zip_ref.extractall(destination_folder)
+        return True
+    else:
+        return False
+
+
+def download_pusht(root: str, dataset_id: str = "pusht", fps: int = 10) -> Path:
+    pusht_url = "https://diffusion-policy.cs.columbia.edu/data/training/pusht.zip"
+    pusht_zarr = Path("pusht/pusht_cchi_v7_replay.zarr")
+
+    root = Path(root)
+    raw_dir: Path = root / f"{dataset_id}_raw"
+    zarr_path: Path = (raw_dir / pusht_zarr).resolve()
+    if not zarr_path.is_dir():
+        raw_dir.mkdir(parents=True, exist_ok=True)
+        download_and_extract_zip(pusht_url, raw_dir)
+    return zarr_path
+
+
+def download_xarm(root: str, dataset_id: str, fps: int = 15) -> Path:
+    root = Path(root)
+    raw_dir: Path = root / "xarm_datasets_raw"
+    if not raw_dir.exists():
+        import zipfile
+
+        import gdown
+
+        raw_dir.mkdir(parents=True, exist_ok=True)
+        # from https://github.com/fyhMer/fowm/blob/main/scripts/download_datasets.py
+        url = "https://drive.google.com/uc?id=1nhxpykGtPDhmQKm-_B8zBSywVRdgeVya"
+        zip_path = raw_dir / "data.zip"
+        gdown.download(url, str(zip_path), quiet=False)
+        print("Extracting...")
+        with zipfile.ZipFile(str(zip_path), "r") as zip_f:
+            for member in zip_f.namelist():
+                if member.startswith("data/xarm") and member.endswith(".pkl"):
+                    print(member)
+                    zip_f.extract(member=member)
+        zip_path.unlink()
+
+    dataset_path: Path = root / f"{dataset_id}"
+    return dataset_path
+
+
+def download_aloha(root: str, dataset_id: str) -> Path:
+    folder_urls = {
+        "aloha_sim_insertion_human": "https://drive.google.com/drive/folders/1RgyD0JgTX30H4IM5XZn8I3zSV_mr8pyF",
+        "aloha_sim_insertion_scripted": "https://drive.google.com/drive/folders/1TsojQQSXtHEoGnqgJ3gmpPQR2DPLtS2N",
+        "aloha_sim_transfer_cube_human": "https://drive.google.com/drive/folders/1sc-E4QYW7A0o23m1u2VWNGVq5smAsfCo",
+        "aloha_sim_transfer_cube_scripted": "https://drive.google.com/drive/folders/1aRyoOhQwxhyt1J8XgEig4s6kzaw__LXj",
+    }
+
+    ep48_urls = {
+        "aloha_sim_insertion_human": "https://drive.google.com/file/d/18Cudl6nikDtgRolea7je8iF_gGKzynOP/view?usp=drive_link",
+        "aloha_sim_insertion_scripted": "https://drive.google.com/file/d/1wfMSZ24oOh5KR_0aaP3Cnu_c4ZCveduB/view?usp=drive_link",
+        "aloha_sim_transfer_cube_human": "https://drive.google.com/file/d/18smMymtr8tIxaNUQ61gW6dG50pt3MvGq/view?usp=drive_link",
+        "aloha_sim_transfer_cube_scripted": "https://drive.google.com/file/d/1pnGIOd-E4-rhz2P3VxpknMKRZCoKt6eI/view?usp=drive_link",
+    }
+
+    ep49_urls = {
+        "aloha_sim_insertion_human": "https://drive.google.com/file/d/1C1kZYyROzs-PrLc0SkDgUgMi4-L3lauE/view?usp=drive_link",
+        "aloha_sim_insertion_scripted": "https://drive.google.com/file/d/17EuCUWS6uCCr6yyNzpXdcdE-_TTNCKtf/view?usp=drive_link",
+        "aloha_sim_transfer_cube_human": "https://drive.google.com/file/d/1Nk7l53d9sJoGDBKAOnNrExX5nLacATc6/view?usp=drive_link",
+        "aloha_sim_transfer_cube_scripted": "https://drive.google.com/file/d/1GKReZHrXU73NMiC5zKCq_UtqPVtYq8eo/view?usp=drive_link",
+    }
+    num_episodes = {  # noqa: F841 # we keep this for reference
+        "aloha_sim_insertion_human": 50,
+        "aloha_sim_insertion_scripted": 50,
+        "aloha_sim_transfer_cube_human": 50,
+        "aloha_sim_transfer_cube_scripted": 50,
+    }
+
+    episode_len = {  # noqa: F841 # we keep this for reference
+        "aloha_sim_insertion_human": 500,
+        "aloha_sim_insertion_scripted": 400,
+        "aloha_sim_transfer_cube_human": 400,
+        "aloha_sim_transfer_cube_scripted": 400,
+    }
+
+    cameras = {  # noqa: F841 # we keep this for reference
+        "aloha_sim_insertion_human": ["top"],
+        "aloha_sim_insertion_scripted": ["top"],
+        "aloha_sim_transfer_cube_human": ["top"],
+        "aloha_sim_transfer_cube_scripted": ["top"],
+    }
+    root = Path(root)
+    raw_dir: Path = root / f"{dataset_id}_raw"
+    if not raw_dir.is_dir():
+        import gdown
+
+        assert dataset_id in folder_urls
+        assert dataset_id in ep48_urls
+        assert dataset_id in ep49_urls
+
+        raw_dir.mkdir(parents=True, exist_ok=True)
+
+        gdown.download_folder(folder_urls[dataset_id], output=str(raw_dir))
+
+        # because of the 50 files limit per directory, two files episode 48 and 49 were missing
+        gdown.download(ep48_urls[dataset_id], output=str(raw_dir / "episode_48.hdf5"), fuzzy=True)
+        gdown.download(ep49_urls[dataset_id], output=str(raw_dir / "episode_49.hdf5"), fuzzy=True)
+    return raw_dir
+
+
+def download_umi(root: str, dataset_id: str) -> Path:
+    url_cup_in_the_wild = "https://real.stanford.edu/umi/data/zarr_datasets/cup_in_the_wild.zarr.zip"
+    cup_in_the_wild_zarr = Path("umi/cup_in_the_wild/cup_in_the_wild.zarr")
+
+    root = Path(root)
+    raw_dir: Path = root / f"{dataset_id}_raw"
+    zarr_path: Path = (raw_dir / cup_in_the_wild_zarr).resolve()
+    if not zarr_path.is_dir():
+        raw_dir.mkdir(parents=True, exist_ok=True)
+        download_and_extract_zip(url_cup_in_the_wild, zarr_path)
+    return zarr_path
+
+
+if __name__ == "__main__":
+    root = "data"
+    dataset_ids = [
+        "pusht",
+        "xarm_lift_medium",
+        "xarm_lift_medium_replay",
+        "xarm_push_medium",
+        "xarm_push_medium_replay",
+        "aloha_sim_insertion_human",
+        "aloha_sim_insertion_scripted",
+        "aloha_sim_transfer_cube_human",
+        "aloha_sim_transfer_cube_scripted",
+        "umi_cup_in_the_wild",
+    ]
+    for dataset_id in dataset_ids:
+        download_raw(root=root, dataset_id=dataset_id)
@@ -0,0 +1,199 @@
+import re
+from pathlib import Path
+
+import h5py
+import torch
+import tqdm
+from datasets import Dataset, Features, Image, Sequence, Value
+from PIL import Image as PILImage
+
+from lerobot.common.datasets.push_dataset_to_hub.utils import concatenate_episodes
+from lerobot.common.datasets.utils import (
+    hf_transform_to_torch,
+)
+
+
+class AlohaProcessor:
+    """
+    Process HDF5 files formatted like in: https://github.com/tonyzhaozh/act
+
+    Attributes:
+        folder_path (Path): Path to the directory containing HDF5 files.
+        cameras (list[str]): List of camera identifiers to check in the files.
+        fps (int): Frames per second used in timestamp calculations.
+
+    Methods:
+        is_valid() -> bool:
+            Validates if each HDF5 file within the folder contains all required datasets.
+        preprocess() -> dict:
+            Processes the files and returns structured data suitable for further analysis.
+        to_hf_dataset(data_dict: dict) -> Dataset:
+            Converts processed data into a Hugging Face Dataset object.
+    """
+
+    def __init__(self, folder_path: Path, cameras: list[str] | None = None, fps: int | None = None):
+        """
+        Initializes the AlohaProcessor with a specified directory path containing HDF5 files,
+        an optional list of cameras, and a frame rate.
+
+        Args:
+            folder_path (Path): The directory path where HDF5 files are stored.
+            cameras (list[str] | None): Optional list of cameras to validate within the files. Defaults to ['top'] if None.
+            fps (int): Frame rate for the datasets, used in time calculations. Default is 50.
+
+        Examples:
+            >>> processor = AlohaProcessor(Path("path_to_hdf5_directory"), ["camera1", "camera2"])
+            >>> processor.is_valid()
+            True
+        """
+        self.folder_path = folder_path
+        if cameras is None:
+            cameras = ["top"]
+        self.cameras = cameras
+        if fps is None:
+            fps = 50
+        self._fps = fps
+
+    @property
+    def fps(self) -> int:
+        return self._fps
+
+    def is_valid(self) -> bool:
+        """
+        Validates the HDF5 files in the specified folder to ensure they contain the required datasets
+        for actions, positions, and images for each specified camera.
+
+        Returns:
+            bool: True if all files are valid HDF5 files with all required datasets, False otherwise.
+        """
+        hdf5_files: list[Path] = list(self.folder_path.glob("episode_*.hdf5"))
+        if len(hdf5_files) == 0:
+            return False
+        try:
+            hdf5_files = sorted(
+                hdf5_files, key=lambda x: int(re.search(r"episode_(\d+).hdf5", x.name).group(1))
+            )
+        except AttributeError:
+            # All file names must contain a numerical identifier matching 'episode_(\\d+).hdf5
+            return False
+
+        # Check if the sequence is consecutive eg episode_0, episode_1, episode_2, etc.
+        # If not, return False
+        previous_number = None
+        for file in hdf5_files:
+            current_number = int(re.search(r"episode_(\d+).hdf5", file.name).group(1))
+            if previous_number is not None and current_number - previous_number != 1:
+                return False
+            previous_number = current_number
+
+        for file in hdf5_files:
+            try:
+                with h5py.File(file, "r") as file:
+                    # Check for the expected datasets within the HDF5 file
+                    required_datasets = ["/action", "/observations/qpos"]
+                    # Add camera-specific image datasets to the required datasets
+                    camera_datasets = [f"/observations/images/{cam}" for cam in self.cameras]
+                    required_datasets.extend(camera_datasets)
+
+                    if not all(dataset in file for dataset in required_datasets):
+                        return False
+            except OSError:
+                return False
+        return True
+
+    def preprocess(self):
+        """
+        Collects episode data from the HDF5 file and returns it as an AlohaStep named tuple.
+
+        Returns:
+            AlohaStep: Named tuple containing episode data.
+
+        Raises:
+            ValueError: If the file is not valid.
+        """
+        if not self.is_valid():
+            raise ValueError("The HDF5 file is invalid or does not contain the required datasets.")
+
+        hdf5_files = list(self.folder_path.glob("*.hdf5"))
+        hdf5_files = sorted(hdf5_files, key=lambda x: int(re.search(r"episode_(\d+)", x.name).group(1)))
+        ep_dicts = []
+        episode_data_index = {"from": [], "to": []}
+
+        id_from = 0
+
+        for ep_path in tqdm.tqdm(hdf5_files):
+            with h5py.File(ep_path, "r") as ep:
+                ep_id = int(re.search(r"episode_(\d+)", ep_path.name).group(1))
+                num_frames = ep["/action"].shape[0]
+
+                # last step of demonstration is considered done
+                done = torch.zeros(num_frames, dtype=torch.bool)
+                done[-1] = True
+
+                state = torch.from_numpy(ep["/observations/qpos"][:])
+                action = torch.from_numpy(ep["/action"][:])
+
+                ep_dict = {}
+
+                for cam in self.cameras:
+                    image = torch.from_numpy(ep[f"/observations/images/{cam}"][:])  # b h w c
+                    ep_dict[f"observation.images.{cam}"] = [PILImage.fromarray(x.numpy()) for x in image]
+
+                ep_dict.update(
+                    {
+                        "observation.state": state,
+                        "action": action,
+                        "episode_index": torch.tensor([ep_id] * num_frames),
+                        "frame_index": torch.arange(0, num_frames, 1),
+                        "timestamp": torch.arange(0, num_frames, 1) / self.fps,
+                        # TODO(rcadene): compute reward and success
+                        # "next.reward": reward,
+                        "next.done": done,
+                        # "next.success": success,
+                    }
+                )
+
+                assert isinstance(ep_id, int)
+                ep_dicts.append(ep_dict)
+
+                episode_data_index["from"].append(id_from)
+                episode_data_index["to"].append(id_from + num_frames)
+
+            id_from += num_frames
+
+        data_dict = concatenate_episodes(ep_dicts)
+        return data_dict, episode_data_index
+
+    def to_hf_dataset(self, data_dict) -> Dataset:
+        """
+        Converts a dictionary of data into a Hugging Face Dataset object.
+
+        Args:
+            data_dict (dict): A dictionary containing the data to be converted.
+
+        Returns:
+            Dataset: The converted Hugging Face Dataset object.
+        """
+        image_features = {f"observation.images.{cam}": Image() for cam in self.cameras}
+        features = {
+            "observation.state": Sequence(
+                length=data_dict["observation.state"].shape[1], feature=Value(dtype="float32", id=None)
+            ),
+            "action": Sequence(length=data_dict["action"].shape[1], feature=Value(dtype="float32", id=None)),
+            "episode_index": Value(dtype="int64", id=None),
+            "frame_index": Value(dtype="int64", id=None),
+            "timestamp": Value(dtype="float32", id=None),
+            # "next.reward": Value(dtype="float32", id=None),
+            "next.done": Value(dtype="bool", id=None),
+            # "next.success": Value(dtype="bool", id=None),
+            "index": Value(dtype="int64", id=None),
+        }
+        update_features = {**image_features, **features}
+        features = Features(update_features)
+        hf_dataset = Dataset.from_dict(data_dict, features=features)
+        hf_dataset.set_transform(hf_transform_to_torch)
+
+        return hf_dataset
+
+    def cleanup(self):
+        pass