121 lines
4.9 KiB
Python
121 lines
4.9 KiB
Python
|
|
"""Register an extracted bundle as tier 3.
|
||
|
|
|
||
|
|
python manage.py ingest_bundle # ./manifest.json
|
||
|
|
python manage.py ingest_bundle scratch/my-take # that bundle
|
||
|
|
|
||
|
|
WHAT THIS REPLACES. Until step 9 the page fetched `/manifest.json` and then built
|
||
|
|
`frames/0001.png` itself, with shadow-cljs's `:dev-http` serving the repo root. So
|
||
|
|
the frame layout was a shared secret between a shell script and a ClojureScript
|
||
|
|
namespace, and "where the frames are" was answered by a directory listing.
|
||
|
|
|
||
|
|
Now the server names every frame, and the client asks it. The frames go into the
|
||
|
|
content-addressed blob store — by hard link, so 112MB of PNGs is not copied — and
|
||
|
|
the manifest the client receives carries a URL per frame. That is the whole of what
|
||
|
|
makes the frames the backend's to serve, and it is what the in-browser wasm-ffmpeg
|
||
|
|
extraction docs/architecture.md describes will upload INTO, without the client
|
||
|
|
learning anything new when it arrives: the same blobs, the same manifest, a
|
||
|
|
different producer.
|
||
|
|
|
||
|
|
`extract.sh` still does the decoding. It is out of step 9's scope, it works, and it
|
||
|
|
is the only part of this that needs a terminal.
|
||
|
|
"""
|
||
|
|
import json
|
||
|
|
from pathlib import Path
|
||
|
|
|
||
|
|
from django.core.management.base import BaseCommand, CommandError
|
||
|
|
from django.db import transaction
|
||
|
|
|
||
|
|
from clips import blobs
|
||
|
|
from clips.models import Blob, Footage, FootageFrame
|
||
|
|
|
||
|
|
|
||
|
|
class Command(BaseCommand):
|
||
|
|
help = "Register an extracted frames+audio+manifest bundle as footage."
|
||
|
|
|
||
|
|
def add_arguments(self, parser):
|
||
|
|
parser.add_argument(
|
||
|
|
"bundle", nargs="?", default=".",
|
||
|
|
help="a directory holding manifest.json, or the manifest itself",
|
||
|
|
)
|
||
|
|
parser.add_argument("--label", default="", help="what to call it in the UI")
|
||
|
|
|
||
|
|
def handle(self, *args, **options):
|
||
|
|
manifest_path = Path(options["bundle"])
|
||
|
|
if manifest_path.is_dir():
|
||
|
|
manifest_path = manifest_path / "manifest.json"
|
||
|
|
if not manifest_path.exists():
|
||
|
|
raise CommandError(f"{manifest_path} does not exist — run ./extract.sh first")
|
||
|
|
|
||
|
|
manifest = json.loads(manifest_path.read_text())
|
||
|
|
root = manifest_path.parent
|
||
|
|
frames_dir = root / manifest["dir"]
|
||
|
|
audio_path = root / manifest["audio"]
|
||
|
|
count = int(manifest["frames"])
|
||
|
|
|
||
|
|
pngs = sorted(frames_dir.glob("*.png"))
|
||
|
|
if len(pngs) != count:
|
||
|
|
raise CommandError(
|
||
|
|
f"the manifest says {count} frames and {frames_dir} holds {len(pngs)}; "
|
||
|
|
"refusing an inaccurate footage"
|
||
|
|
)
|
||
|
|
if not audio_path.exists():
|
||
|
|
raise CommandError(f"{audio_path} does not exist")
|
||
|
|
|
||
|
|
width, height = blobs.png_size(pngs[0])
|
||
|
|
|
||
|
|
self.stdout.write(f"hashing {len(pngs)} frames…")
|
||
|
|
frame_blobs = []
|
||
|
|
for i, png in enumerate(pngs):
|
||
|
|
digest, size = blobs.adopt(png)
|
||
|
|
frame_blobs.append((i, digest, size))
|
||
|
|
if (i + 1) % 25 == 0 or i + 1 == len(pngs):
|
||
|
|
self.stdout.write(f" {i + 1}/{len(pngs)}")
|
||
|
|
|
||
|
|
audio_digest, audio_size = blobs.adopt(audio_path)
|
||
|
|
|
||
|
|
# The footage's own identity: every frame in order, plus the audio and the
|
||
|
|
# rate. Two extractions of one clip at one rate are one footage, so an
|
||
|
|
# analysis over it is reusable across both.
|
||
|
|
import hashlib
|
||
|
|
|
||
|
|
h = hashlib.sha256()
|
||
|
|
h.update(f"arthur-footage-1/{manifest['fps']}/{count}/{width}x{height}\n".encode())
|
||
|
|
for _, digest, _ in frame_blobs:
|
||
|
|
h.update(digest.encode())
|
||
|
|
h.update(audio_digest.encode())
|
||
|
|
footage_digest = h.hexdigest()
|
||
|
|
|
||
|
|
if existing := Footage.objects.filter(digest=footage_digest).first():
|
||
|
|
self.stdout.write(self.style.SUCCESS(f"already ingested: {existing.id}"))
|
||
|
|
return
|
||
|
|
|
||
|
|
with transaction.atomic():
|
||
|
|
audio_blob, _ = Blob.objects.get_or_create(
|
||
|
|
digest=audio_digest,
|
||
|
|
defaults={"size": audio_size, "media_type": "audio/wav"},
|
||
|
|
)
|
||
|
|
footage = Footage.objects.create(
|
||
|
|
digest=footage_digest,
|
||
|
|
label=options["label"] or manifest.get("source") or frames_dir.name,
|
||
|
|
source=manifest.get("source") or "",
|
||
|
|
fps=float(manifest["fps"]),
|
||
|
|
frames=count,
|
||
|
|
width=width,
|
||
|
|
height=height,
|
||
|
|
audio=audio_blob,
|
||
|
|
feature_absence=manifest.get("feature-absence") or {},
|
||
|
|
)
|
||
|
|
rows = []
|
||
|
|
for index, digest, size in frame_blobs:
|
||
|
|
blob, _ = Blob.objects.get_or_create(
|
||
|
|
digest=digest, defaults={"size": size, "media_type": "image/png"}
|
||
|
|
)
|
||
|
|
rows.append(FootageFrame(footage=footage, index=index, blob=blob))
|
||
|
|
FootageFrame.objects.bulk_create(rows)
|
||
|
|
|
||
|
|
self.stdout.write(
|
||
|
|
self.style.SUCCESS(
|
||
|
|
f"{count} frames at {manifest['fps']}fps, {width}x{height} -> footage {footage.id}"
|
||
|
|
)
|
||
|
|
)
|