davanstrien/prelinger-sample / scripts /chunk_prelinger.py
davanstrien's picture
download
raw
3.18 kB
# /// script
# requires-python = ">=3.11"
# dependencies = ["imageio-ffmpeg"]
# ///
"""Pre-chunk mirrored Prelinger films into ~60s segments for the caption lane.
Stream-copy via ffmpeg's segment muxer (no re-encode): boundaries snap to
keyframes near each 60s mark, and the muxer's segment_list records the ACTUAL
start/end of every chunk. Timestamps stay exact because we offset by recorded
actuals, not assumed multiples — this is why stream-copy is safe here even
though naive -ss stream-copy is not.
In: /films/films/{id}.mp4 (mirror bucket, ro)
Out: /out/chunks/{id}/c0000.mp4 ... (chunks bucket)
/out/chunkmeta/{id}.json (per-film chunk list w/ actual times;
written last = completion marker)
"""
import csv
import json
import shutil
import subprocess
import tempfile
import time
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
import imageio_ffmpeg
FFMPEG = imageio_ffmpeg.get_ffmpeg_exe() # default uv image ships no ffmpeg
IN = Path("/films/films")
OUT = Path("/out")
CHUNKS, CMETA = OUT / "chunks", OUT / "chunkmeta"
SEG = 60
WORKERS = 8
def chunk_one(src: Path) -> str:
ident = src.stem
marker = CMETA / f"{ident}.json"
if marker.exists():
return "skip"
with tempfile.TemporaryDirectory() as td:
tdp = Path(td)
listfile = tdp / "segs.csv"
try:
subprocess.run(
[FFMPEG, "-hide_banner", "-loglevel", "error", "-i", str(src),
"-c", "copy", "-an", "-f", "segment", "-segment_time", str(SEG),
"-reset_timestamps", "1", "-segment_list", str(listfile),
"-segment_list_type", "csv", str(tdp / "c%04d.mp4")],
check=True, stdin=subprocess.DEVNULL, timeout=600)
except (subprocess.SubprocessError, OSError) as e:
print(f"[err] {ident}: {type(e).__name__}", flush=True)
return "err"
rows = list(csv.reader(listfile.open()))
dest = CHUNKS / ident
dest.mkdir(parents=True, exist_ok=True)
chunks = []
for fname, start, end in rows:
shutil.copyfile(tdp / fname, dest / fname)
chunks.append({"chunk": fname, "start": round(float(start), 3),
"end": round(float(end), 3)})
marker.write_text(json.dumps({"identifier": ident, "chunks": chunks}))
return "ok"
def main():
CHUNKS.mkdir(parents=True, exist_ok=True)
CMETA.mkdir(parents=True, exist_ok=True)
films = sorted(IN.glob("*.mp4"))
print(f"[plan] {len(films)} films to chunk", flush=True)
counts = {"ok": 0, "skip": 0, "err": 0}
t0 = time.time()
with ThreadPoolExecutor(WORKERS) as ex:
for n, res in enumerate(ex.map(chunk_one, films), 1):
counts[res] += 1
if n % 50 == 0 or n == len(films):
print(f"[progress] {n}/{len(films)} {counts} {time.time()-t0:.0f}s",
flush=True)
total = sum(1 for _ in CHUNKS.rglob("*.mp4"))
print(f"[CHUNKING DONE] {counts} total_chunks={total}", flush=True)
if __name__ == "__main__":
main()

Xet Storage Details

Size:
3.18 kB
·
Xet hash:
e210e46f04fdd758cfae99e345985df099118cbf4b083ee5d51f8e98504b0665

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.