"""Register an extracted bundle as tier 3. python manage.py ingest_bundle # ./manifest.json python manage.py ingest_bundle scratch/my-take # that bundle WHAT THIS REPLACES. Until step 9 the page fetched `/manifest.json` and then built `frames/0001.png` itself, with shadow-cljs's `:dev-http` serving the repo root. So the frame layout was a shared secret between a shell script and a ClojureScript namespace, and "where the frames are" was answered by a directory listing. Now the server names every frame, and the client asks it. The frames go into the content-addressed blob store — by hard link, so 112MB of PNGs is not copied — and the manifest the client receives carries a URL per frame. That is the whole of what makes the frames the backend's to serve, and it is what the in-browser wasm-ffmpeg extraction docs/architecture.md describes will upload INTO, without the client learning anything new when it arrives: the same blobs, the same manifest, a different producer. `extract.sh` still does the decoding. It is out of step 9's scope, it works, and it is the only part of this that needs a terminal. """ import json from pathlib import Path from django.core.management.base import BaseCommand, CommandError from django.db import transaction from clips import blobs from clips.models import Blob, Footage, FootageFrame class Command(BaseCommand): help = "Register an extracted frames+audio+manifest bundle as footage." def add_arguments(self, parser): parser.add_argument( "bundle", nargs="?", default=".", help="a directory holding manifest.json, or the manifest itself", ) parser.add_argument("--label", default="", help="what to call it in the UI") def handle(self, *args, **options): manifest_path = Path(options["bundle"]) if manifest_path.is_dir(): manifest_path = manifest_path / "manifest.json" if not manifest_path.exists(): raise CommandError(f"{manifest_path} does not exist — run ./extract.sh first") manifest = json.loads(manifest_path.read_text()) root = manifest_path.parent frames_dir = root / manifest["dir"] audio_path = root / manifest["audio"] count = int(manifest["frames"]) pngs = sorted(frames_dir.glob("*.png")) if len(pngs) != count: raise CommandError( f"the manifest says {count} frames and {frames_dir} holds {len(pngs)}; " "refusing an inaccurate footage" ) if not audio_path.exists(): raise CommandError(f"{audio_path} does not exist") width, height = blobs.png_size(pngs[0]) self.stdout.write(f"hashing {len(pngs)} frames…") frame_blobs = [] for i, png in enumerate(pngs): digest, size = blobs.adopt(png) frame_blobs.append((i, digest, size)) if (i + 1) % 25 == 0 or i + 1 == len(pngs): self.stdout.write(f" {i + 1}/{len(pngs)}") audio_digest, audio_size = blobs.adopt(audio_path) # The footage's own identity: every frame in order, plus the audio and the # rate. Two extractions of one clip at one rate are one footage, so an # analysis over it is reusable across both. import hashlib h = hashlib.sha256() h.update(f"arthur-footage-1/{manifest['fps']}/{count}/{width}x{height}\n".encode()) for _, digest, _ in frame_blobs: h.update(digest.encode()) h.update(audio_digest.encode()) footage_digest = h.hexdigest() if existing := Footage.objects.filter(digest=footage_digest).first(): self.stdout.write(self.style.SUCCESS(f"already ingested: {existing.id}")) return with transaction.atomic(): audio_blob, _ = Blob.objects.get_or_create( digest=audio_digest, defaults={"size": audio_size, "media_type": "audio/wav"}, ) footage = Footage.objects.create( digest=footage_digest, label=options["label"] or manifest.get("source") or frames_dir.name, source=manifest.get("source") or "", fps=float(manifest["fps"]), frames=count, width=width, height=height, audio=audio_blob, feature_absence=manifest.get("feature-absence") or {}, ) rows = [] for index, digest, size in frame_blobs: blob, _ = Blob.objects.get_or_create( digest=digest, defaults={"size": size, "media_type": "image/png"} ) rows.append(FootageFrame(footage=footage, index=index, blob=blob)) FootageFrame.objects.bulk_create(rows) self.stdout.write( self.style.SUCCESS( f"{count} frames at {manifest['fps']}fps, {width}x{height} -> footage {footage.id}" ) )