From 65ad67c129daf52d25cde494cf1896e545c531f1 Mon Sep 17 00:00:00 2001 From: Olive Vaughn Date: Mon, 28 Sep 2026 14:18:09 -0400 Subject: [PATCH] Decode uploaded footage in order with WebCodecs --- README.md | 25 +- clips/extraction.py | 49 ++- ...0005_footage_stream_alter_footage_video.py | 24 ++ clips/models.py | 7 +- clips/views.py | 4 +- docs/architecture.md | 25 +- frontend/README.md | 15 +- frontend/src/arthur/events/footage.cljs | 128 ++++-- frontend/src/arthur/events/project.cljs | 51 +-- frontend/src/arthur/flow/ingest.cljs | 392 +++++++++--------- frontend/src/arthur/flow/source.cljs | 18 +- frontend/test/arthur/flow/ingest_test.cljs | 80 ++-- 12 files changed, 454 insertions(+), 364 deletions(-) create mode 100644 clips/migrations/0005_footage_stream_alter_footage_video.py diff --git a/README.md b/README.md index 3bb4523..6f178d2 100644 --- a/README.md +++ b/README.md @@ -17,7 +17,7 @@ modern conveniences belong in the workflow, not the output. See ## ClojureScript port The active port plays the synthetic take, accepts video uploads, transcodes them -to a browser-seekable proxy plus audio and tracing stills, analyzes real footage +to an H.264 proxy and decodable stream plus audio and tracing stills, analyzes real footage for mouth, eyes, brows and pixel-derived teeth, and saves the project with reusable analysis data. The step 8 data model represents persistent feature IDs, eye pairs and feature-level observation gaps; @@ -64,7 +64,7 @@ Three tiers, cut by mutability and size — the full argument is in | --- | --- | --- | | 1 **authored** | the scene: nodes, channels, features, time maps | the database, as independently addressed leaves. Kilobytes | | 2 **derived** | detected landmarks, raw mouth crops, and dense channel blocks | `var/blobs`, addressed by analysis and block inputs, including the detector version | -| 3 **source** | the uploaded video, the H.264 proxy measured from it, its tracing stills, and audio | the same blob store, by the hash of their bytes | +| 3 **source** | the uploaded video, H.264 proxy and elementary stream, tracing stills, and audio | the same blob store, by the hash of their bytes | Only tier 1 is the document. Tier 2 is a pure function of tiers 1 and 3, so a saved project names its blocks rather than carrying them, and a knob change gives @@ -90,28 +90,25 @@ wasm, which is fetched from a CDN on first use. For real footage: Upload it in the app. `./extract.sh` still writes the old PNG-sequence bundle and -`ingest_bundle` still registers it, but footage ingested that way has no proxy and -the loader will say so — the measured pixels come out of the video now. +`ingest_bundle` still registers it, but footage ingested that way has no decodable +stream and the loader will say so — the measured pixels come out of the video now. MediaPipe's wasm and `face_landmarker.task` are both local; nothing in detection touches the network. -Detection reads the VIDEO, not a frame per file. The page seeks the proxy to the -MIDDLE of each frame — `(i + 0.5) / fps` — and waits for -`requestVideoFrameCallback` to hand the frame over, then checks the `mediaTime` it -reports against the frame it asked for. Both halves are load-bearing and both were -measured against the same footage decoded to PNGs: aiming at `i / fps` sits on a -frame boundary and landed one frame early 31 times in 91, and aiming at the middle -was exact on all 91. A run that gets a frame it did not ask for stops and says so, -because a one-frame slip between the landmarks and the audio is not something -anyone finds by looking at the result. +Detection reads the H.264 elementary stream with WebCodecs, one coded frame at a +time. The proxy has no B-frames, so decode order is frame order. Each decoded +frame reaches MediaPipe in VIDEO running mode at its footage timestamp. The +decoder and detector advance together, with a pause between frames so progress +can paint. Saved analyses reuse their stored crop pixels and measure them with +the same pauses. This is what replaced the PNG sequence, which was 112MB for 7.6 seconds and would be 1.1GB at the 900-frame limit. The proxy is 6MB, and the landmarks barely notice: detected off decoded H.264 rather than off the PNGs, they moved at most 0.0033 of frame width. -`manifest.json` records the source rate. The extractor keeps every source frame; +The server's footage manifest records the proxy's frame rate and frame count; the page reads that rate because a guessed fps desynchronises audio from picture. Choosing a lower picture rate happens after analysis. diff --git a/clips/extraction.py b/clips/extraction.py index 2863aae..ff31bb6 100644 --- a/clips/extraction.py +++ b/clips/extraction.py @@ -106,7 +106,10 @@ def _encode_proxy(job, source_path, proxy_path, facts, root): # step that makes the thing the page measures not be. "-fps_mode", "cfr", "-r", facts.get("rate") or str(facts["fps"]), "-c:v", "libx264", "-preset", "veryfast", "-crf", PROXY_CRF, - # NO B-FRAMES, AND THIS IS THE LOAD-BEARING FLAG. With them x264 has a + # NO B-FRAMES, AND THIS IS THE LOAD-BEARING FLAG. It is what makes + # decode order presentation order, so the page can treat access unit k + # of the elementary stream as frame k without demuxing a container or + # consulting a timestamp. With them x264 has a # two-frame reordering delay, ffmpeg compensates by writing an edit list # (`elst` media_time 1024 at timebase 1/15360 — exactly two frames), and # the browser then lives on two timelines at once: `currentTime` obeys the @@ -125,6 +128,25 @@ def _encode_proxy(job, source_path, proxy_path, facts, root): root, "proxy", total, (0, 55)) +def _elementary_stream(proxy_path, out_path): + """The proxy's video, unwrapped into a raw Annex-B H.264 stream. + + A STREAM COPY, not a second encode: the same coded frames as the MP4, with + the container's length-prefixed NAL units rewritten as start-code-delimited + ones. It costs a file read and nothing else. + + This exists because the page decodes with WebCodecs, and `VideoDecoder` takes + demuxed chunks rather than a container. Handing it Annex-B means the client + needs no demuxer: NAL start codes are findable in a loop, and because the + proxy is encoded with no B-frames, decode order is presentation order — so + access unit k IS frame k, with no container timing to consult and no clock to + reconcile. That is the whole reason this file is worth the bytes it costs. + """ + _command(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-i", str(proxy_path), "-an", "-c:v", "copy", + "-bsf:v", "h264_mp4toannexb", "-f", "h264", str(out_path)]) + + def _extract_stills(job, proxy_path, frames_dir, frames, root): """The proxy -> one tracing JPEG per frame, long edge capped.""" _run_with_progress( @@ -234,13 +256,16 @@ def count_frames(path): def extraction_key(source, settings): - text = json.dumps({"scheme": 2, "source": source.blob_id, "settings": settings}, + # Scheme 3: the extraction now also produces the elementary stream the page + # decodes, so a job run under scheme 2 did not make everything this one does. + text = json.dumps({"scheme": 3, "source": source.blob_id, "settings": settings}, sort_keys=True, separators=(",", ":")) return "sha256:" + hashlib.sha256(text.encode()).hexdigest() -def _register(job, proxy_path, stills, audio_path, facts): +def _register(job, proxy_path, stream_path, stills, audio_path, facts): proxy_digest, proxy_size = blobs.adopt(proxy_path) + stream_digest, stream_size = blobs.adopt(stream_path) audio_digest, audio_size = blobs.adopt(audio_path) still_blobs = [(index, *blobs.adopt(path)) for index, path in enumerate(stills)] width, height, fps, frames = facts["width"], facts["height"], facts["fps"], facts["frames"] @@ -256,13 +281,24 @@ def _register(job, proxy_path, stills, audio_path, facts): with transaction.atomic(): proxy_blob, _ = Blob.objects.get_or_create( digest=proxy_digest, defaults={"size": proxy_size, "media_type": "video/mp4"}) + stream_blob, _ = Blob.objects.get_or_create( + digest=stream_digest, defaults={"size": stream_size, "media_type": "video/h264"}) audio_blob, _ = Blob.objects.get_or_create( digest=audio_digest, defaults={"size": audio_size, "media_type": "audio/wav"}) footage, created = Footage.objects.get_or_create( digest=h.hexdigest(), defaults={"label": job.source.filename[:200], "source": job.source.filename[:200], "fps": fps, "frames": frames, "width": width, "height": height, - "audio": audio_blob, "video": proxy_blob}) + "audio": audio_blob, "video": proxy_blob, "stream": stream_blob}) + if not created and not footage.stream_id: + # The same footage by identity, extracted before the elementary + # stream existed. Its digest is over the proxy and the audio, which + # have not changed — so this is the same footage gaining a file it + # was always entitled to, not a different one. + footage.stream = stream_blob + if not footage.video_id: + footage.video = proxy_blob + footage.save(update_fields=["stream", "video"]) if created: rows = [] for index, digest, size in still_blobs: @@ -306,6 +342,9 @@ def run(key): "audio would drift") proxy_facts["frames"] = frames + stream_path = root / "proxy.h264" + _elementary_stream(proxy_path, stream_path) + frames_dir = root / "stills" frames_dir.mkdir() _extract_stills(job, proxy_path, frames_dir, frames, root) @@ -325,7 +364,7 @@ def run(key): "-f", "lavfi", "-i", "anullsrc=r=44100:cl=mono", "-t", str(frames / proxy_facts["fps"]), "-c:a", "pcm_s16le", str(audio_path)]) - footage = _register(job, proxy_path, stills, audio_path, proxy_facts) + footage = _register(job, proxy_path, stream_path, stills, audio_path, proxy_facts) job.footage, job.state, job.progress = footage, "done", 100 job.save(update_fields=["footage", "state", "progress", "updated"]) except Exception as exc: diff --git a/clips/migrations/0005_footage_stream_alter_footage_video.py b/clips/migrations/0005_footage_stream_alter_footage_video.py new file mode 100644 index 0000000..00976af --- /dev/null +++ b/clips/migrations/0005_footage_stream_alter_footage_video.py @@ -0,0 +1,24 @@ +# Generated by Django 5.2.17 on 2026-09-28 17:11 + +import django.db.models.deletion +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0004_footage_video_alter_footageframe_index'), + ] + + operations = [ + migrations.AddField( + model_name='footage', + name='stream', + field=models.ForeignKey(blank=True, help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, one access unit per frame; null on footage extracted before it", null=True, on_delete=django.db.models.deletion.PROTECT, related_name='stream_for', to='clips.blob'), + ), + migrations.AlterField( + model_name='footage', + name='video', + field=models.ForeignKey(blank=True, help_text='the browser-safe proxy, playable and seekable', null=True, on_delete=django.db.models.deletion.PROTECT, related_name='video_for', to='clips.blob'), + ), + ] diff --git a/clips/models.py b/clips/models.py index 031d5ae..2f268df 100644 --- a/clips/models.py +++ b/clips/models.py @@ -102,7 +102,12 @@ class Footage(models.Model): audio = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="audio_for") video = models.ForeignKey( Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="video_for", - help_text="the browser-safe proxy the page detects from; null on pre-proxy footage", + help_text="the browser-safe proxy, playable and seekable", + ) + stream = models.ForeignKey( + Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="stream_for", + help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, " + "one access unit per frame; null on footage extracted before it", ) feature_absence = models.JSONField(default=dict, blank=True) created = models.DateTimeField(auto_now_add=True) diff --git a/clips/views.py b/clips/views.py index b58a0ba..b8a1c59 100644 --- a/clips/views.py +++ b/clips/views.py @@ -240,6 +240,8 @@ def _footage_json(footage: Footage, urls=True): # existed, which the loader reports as "re-extract this" rather than # failing somewhere inside MediaPipe. "video": f"/blob/{footage.video.digest}" if footage.video_id else None, + # What the page actually decodes: one access unit per frame, no container. + "stream": f"/blob/{footage.stream.digest}" if footage.stream_id else None, "feature-absence": footage.feature_absence or {}, } if urls: @@ -257,7 +259,7 @@ def footage_list(request): @require_http_methods(["GET"]) def footage_detail(request, footage_id): try: - footage = Footage.objects.select_related("audio", "video").get(id=footage_id) + footage = Footage.objects.select_related("audio", "video", "stream").get(id=footage_id) except Footage.DoesNotExist: return JsonResponse({"error": "no such footage"}, status=404) return JsonResponse(_footage_json(footage)) diff --git a/docs/architecture.md b/docs/architecture.md index e749caf..86f3f6f 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -145,8 +145,8 @@ boundary.** | # | Stage | In | Out | Cost | | --- | --- | --- | --- | --- | -| 1 | **ingest** | video | footage: a seekable H.264 proxy, tracing stills, audio, manifest | minutes, in-app | -| 2 | **detect** | the proxy, walked one frame at a time | raw landmarks per frame | minutes, **cached** | +| 1 | **ingest** | video | footage: an H.264 proxy and raw stream, tracing stills, audio, manifest | minutes, in-app | +| 2 | **detect** | the raw stream, decoded one frame at a time | raw landmarks per frame | minutes, **cached** | | 3 | **measure** | landmarks | anchor fit, residual, head-local rings, signals, interior pixels | seconds | | 4 | **condition** | measurements | smoothed transforms and contours | milliseconds | | 5 | **key** | conditioned signals + policy | channels: sparse keys, quantised holds, kept frames | milliseconds | @@ -575,8 +575,8 @@ PUT /api/analyses/ link dense landmarks, mask, crops POST /api/blocks/missing {keys} -> {missing} POST /api/blocks {key, descriptor, data, state} GET /api/blocks/ -GET /api/footage/ the manifest: the proxy to measure, audio, a URL per tracing still -GET /blob/ immutable bytes, and RANGE-capable so a