Add in-memory byte index and manifest-driven episode MP4 cache.

Build moov-derived byte ranges in RAM or from sidecar parquet, fetch tight mdat slices over the network, and decode via TorchCodec custom_frame_mappings to skip full-file metadata scans.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
pepijn
2026-06-16 15:03:17 +00:00
parent 4940281120
commit 7b6f4f2b11
10 changed files with 1682 additions and 5 deletions
+8 -3
View File
@@ -335,9 +335,14 @@ torch = [{ index = "pytorch-cu128", marker = "sys_platform == 'linux'" }]
torchvision = [{ index = "pytorch-cu128", marker = "sys_platform == 'linux'" }]
# Temporary: the native streaming pipeline needs batch(by_column=...) to survive shard/shuffle
# re-creation (datasets#8259), reshard() per row group (#8193), and shuffle(max_buffer_input_shards=...)
# (#8194) — all merged, not yet in a tagged 5.0 release. Pin to the merge commit until the next
# datasets release ships them, then drop this and rely on the `datasets>=5.0.0` floor in `dependencies`.
datasets = { git = "https://github.com/huggingface/datasets.git", rev = "2c45eab1bb975ac3d846f2aa6217b82adec8eba3" }
# (#8194) — all merged, not yet in a tagged 5.0 release. Track main until the next datasets release ships
# them, then drop this and rely on the `datasets>=5.0.0` floor in `dependencies`.
datasets = { git = "https://github.com/huggingface/datasets.git", branch = "main" }
# Temporary: huggingface_hub main carries the 408-retry fix (not yet released). NOTE: main still closes the
# shared httpx.Client on every ConnectError, which races with concurrent streaming requests
# ("Cannot send a request, as the client has been closed"); we patch that out locally in
# huggingface_hub/utils/_http.py. A fresh `uv sync` re-installs main *without* that local patch.
huggingface-hub = { git = "https://github.com/huggingface/huggingface_hub.git", branch = "main" }
[tool.setuptools.package-data]
lerobot = ["envs/*.json"]