diff --git a/.gitignore b/.gitignore index ab2e70d..5f441bf 100644 --- a/.gitignore +++ b/.gitignore @@ -28,6 +28,8 @@ static/arthur/js/ # the Django half's own state: the document database, the content-addressed blob # store (tiers 2 and 3), and collectstatic's output db.sqlite3 +db.sqlite3-shm +db.sqlite3-wal /var/ # vim swap files diff --git a/README.md b/README.md index 3bb4523..6f178d2 100644 --- a/README.md +++ b/README.md @@ -17,7 +17,7 @@ modern conveniences belong in the workflow, not the output. See ## ClojureScript port The active port plays the synthetic take, accepts video uploads, transcodes them -to a browser-seekable proxy plus audio and tracing stills, analyzes real footage +to an H.264 proxy and decodable stream plus audio and tracing stills, analyzes real footage for mouth, eyes, brows and pixel-derived teeth, and saves the project with reusable analysis data. The step 8 data model represents persistent feature IDs, eye pairs and feature-level observation gaps; @@ -64,7 +64,7 @@ Three tiers, cut by mutability and size — the full argument is in | --- | --- | --- | | 1 **authored** | the scene: nodes, channels, features, time maps | the database, as independently addressed leaves. Kilobytes | | 2 **derived** | detected landmarks, raw mouth crops, and dense channel blocks | `var/blobs`, addressed by analysis and block inputs, including the detector version | -| 3 **source** | the uploaded video, the H.264 proxy measured from it, its tracing stills, and audio | the same blob store, by the hash of their bytes | +| 3 **source** | the uploaded video, H.264 proxy and elementary stream, tracing stills, and audio | the same blob store, by the hash of their bytes | Only tier 1 is the document. Tier 2 is a pure function of tiers 1 and 3, so a saved project names its blocks rather than carrying them, and a knob change gives @@ -90,28 +90,25 @@ wasm, which is fetched from a CDN on first use. For real footage: Upload it in the app. `./extract.sh` still writes the old PNG-sequence bundle and -`ingest_bundle` still registers it, but footage ingested that way has no proxy and -the loader will say so — the measured pixels come out of the video now. +`ingest_bundle` still registers it, but footage ingested that way has no decodable +stream and the loader will say so — the measured pixels come out of the video now. MediaPipe's wasm and `face_landmarker.task` are both local; nothing in detection touches the network. -Detection reads the VIDEO, not a frame per file. The page seeks the proxy to the -MIDDLE of each frame — `(i + 0.5) / fps` — and waits for -`requestVideoFrameCallback` to hand the frame over, then checks the `mediaTime` it -reports against the frame it asked for. Both halves are load-bearing and both were -measured against the same footage decoded to PNGs: aiming at `i / fps` sits on a -frame boundary and landed one frame early 31 times in 91, and aiming at the middle -was exact on all 91. A run that gets a frame it did not ask for stops and says so, -because a one-frame slip between the landmarks and the audio is not something -anyone finds by looking at the result. +Detection reads the H.264 elementary stream with WebCodecs, one coded frame at a +time. The proxy has no B-frames, so decode order is frame order. Each decoded +frame reaches MediaPipe in VIDEO running mode at its footage timestamp. The +decoder and detector advance together, with a pause between frames so progress +can paint. Saved analyses reuse their stored crop pixels and measure them with +the same pauses. This is what replaced the PNG sequence, which was 112MB for 7.6 seconds and would be 1.1GB at the 900-frame limit. The proxy is 6MB, and the landmarks barely notice: detected off decoded H.264 rather than off the PNGs, they moved at most 0.0033 of frame width. -`manifest.json` records the source rate. The extractor keeps every source frame; +The server's footage manifest records the proxy's frame rate and frame count; the page reads that rate because a guessed fps desynchronises audio from picture. Choosing a lower picture rate happens after analysis. diff --git a/clips/blobs.py b/clips/blobs.py index 7469e66..2216fcb 100644 --- a/clips/blobs.py +++ b/clips/blobs.py @@ -16,11 +16,13 @@ addressing that answers questions about work not yet done. import hashlib import os import tempfile +import zlib from pathlib import Path from django.conf import settings CHUNK = 1 << 20 +CROP_MEDIA_TYPE = "application/zlib" def digest_bytes(data: bytes) -> str: @@ -86,6 +88,20 @@ def write_stream(chunks) -> tuple[str, int]: return digest.hexdigest(), size +def write_compressed_stream(chunks) -> tuple[str, int]: + """Store a losslessly compressed stream; the digest names stored bytes.""" + compressor = zlib.compressobj() + + def compressed(): + for chunk in chunks: + if part := compressor.compress(chunk): + yield part + if part := compressor.flush(): + yield part + + return write_stream(compressed()) + + def adopt(source: Path) -> tuple[str, int]: """Store a file already on disk, by hard link where the filesystem allows it. diff --git a/clips/extraction.py b/clips/extraction.py index 2863aae..ff31bb6 100644 --- a/clips/extraction.py +++ b/clips/extraction.py @@ -106,7 +106,10 @@ def _encode_proxy(job, source_path, proxy_path, facts, root): # step that makes the thing the page measures not be. "-fps_mode", "cfr", "-r", facts.get("rate") or str(facts["fps"]), "-c:v", "libx264", "-preset", "veryfast", "-crf", PROXY_CRF, - # NO B-FRAMES, AND THIS IS THE LOAD-BEARING FLAG. With them x264 has a + # NO B-FRAMES, AND THIS IS THE LOAD-BEARING FLAG. It is what makes + # decode order presentation order, so the page can treat access unit k + # of the elementary stream as frame k without demuxing a container or + # consulting a timestamp. With them x264 has a # two-frame reordering delay, ffmpeg compensates by writing an edit list # (`elst` media_time 1024 at timebase 1/15360 — exactly two frames), and # the browser then lives on two timelines at once: `currentTime` obeys the @@ -125,6 +128,25 @@ def _encode_proxy(job, source_path, proxy_path, facts, root): root, "proxy", total, (0, 55)) +def _elementary_stream(proxy_path, out_path): + """The proxy's video, unwrapped into a raw Annex-B H.264 stream. + + A STREAM COPY, not a second encode: the same coded frames as the MP4, with + the container's length-prefixed NAL units rewritten as start-code-delimited + ones. It costs a file read and nothing else. + + This exists because the page decodes with WebCodecs, and `VideoDecoder` takes + demuxed chunks rather than a container. Handing it Annex-B means the client + needs no demuxer: NAL start codes are findable in a loop, and because the + proxy is encoded with no B-frames, decode order is presentation order — so + access unit k IS frame k, with no container timing to consult and no clock to + reconcile. That is the whole reason this file is worth the bytes it costs. + """ + _command(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-i", str(proxy_path), "-an", "-c:v", "copy", + "-bsf:v", "h264_mp4toannexb", "-f", "h264", str(out_path)]) + + def _extract_stills(job, proxy_path, frames_dir, frames, root): """The proxy -> one tracing JPEG per frame, long edge capped.""" _run_with_progress( @@ -234,13 +256,16 @@ def count_frames(path): def extraction_key(source, settings): - text = json.dumps({"scheme": 2, "source": source.blob_id, "settings": settings}, + # Scheme 3: the extraction now also produces the elementary stream the page + # decodes, so a job run under scheme 2 did not make everything this one does. + text = json.dumps({"scheme": 3, "source": source.blob_id, "settings": settings}, sort_keys=True, separators=(",", ":")) return "sha256:" + hashlib.sha256(text.encode()).hexdigest() -def _register(job, proxy_path, stills, audio_path, facts): +def _register(job, proxy_path, stream_path, stills, audio_path, facts): proxy_digest, proxy_size = blobs.adopt(proxy_path) + stream_digest, stream_size = blobs.adopt(stream_path) audio_digest, audio_size = blobs.adopt(audio_path) still_blobs = [(index, *blobs.adopt(path)) for index, path in enumerate(stills)] width, height, fps, frames = facts["width"], facts["height"], facts["fps"], facts["frames"] @@ -256,13 +281,24 @@ def _register(job, proxy_path, stills, audio_path, facts): with transaction.atomic(): proxy_blob, _ = Blob.objects.get_or_create( digest=proxy_digest, defaults={"size": proxy_size, "media_type": "video/mp4"}) + stream_blob, _ = Blob.objects.get_or_create( + digest=stream_digest, defaults={"size": stream_size, "media_type": "video/h264"}) audio_blob, _ = Blob.objects.get_or_create( digest=audio_digest, defaults={"size": audio_size, "media_type": "audio/wav"}) footage, created = Footage.objects.get_or_create( digest=h.hexdigest(), defaults={"label": job.source.filename[:200], "source": job.source.filename[:200], "fps": fps, "frames": frames, "width": width, "height": height, - "audio": audio_blob, "video": proxy_blob}) + "audio": audio_blob, "video": proxy_blob, "stream": stream_blob}) + if not created and not footage.stream_id: + # The same footage by identity, extracted before the elementary + # stream existed. Its digest is over the proxy and the audio, which + # have not changed — so this is the same footage gaining a file it + # was always entitled to, not a different one. + footage.stream = stream_blob + if not footage.video_id: + footage.video = proxy_blob + footage.save(update_fields=["stream", "video"]) if created: rows = [] for index, digest, size in still_blobs: @@ -306,6 +342,9 @@ def run(key): "audio would drift") proxy_facts["frames"] = frames + stream_path = root / "proxy.h264" + _elementary_stream(proxy_path, stream_path) + frames_dir = root / "stills" frames_dir.mkdir() _extract_stills(job, proxy_path, frames_dir, frames, root) @@ -325,7 +364,7 @@ def run(key): "-f", "lavfi", "-i", "anullsrc=r=44100:cl=mono", "-t", str(frames / proxy_facts["fps"]), "-c:a", "pcm_s16le", str(audio_path)]) - footage = _register(job, proxy_path, stills, audio_path, proxy_facts) + footage = _register(job, proxy_path, stream_path, stills, audio_path, proxy_facts) job.footage, job.state, job.progress = footage, "done", 100 job.save(update_fields=["footage", "state", "progress", "updated"]) except Exception as exc: diff --git a/clips/management/commands/compress_crop_blocks.py b/clips/management/commands/compress_crop_blocks.py new file mode 100644 index 0000000..6367b81 --- /dev/null +++ b/clips/management/commands/compress_crop_blocks.py @@ -0,0 +1,57 @@ +"""Compress existing raw mouth crop blocks without changing their public bytes.""" + +import hashlib +import zlib + +from django.core.management.base import BaseCommand, CommandError +from django.db import transaction +from django.db.models.deletion import ProtectedError + +from clips import blobs +from clips.models import Blob, Block + + +class Command(BaseCommand): + help = "Compress existing source/crops blobs and remove unreferenced raw copies" + + def handle(self, *args, **options): + converted = 0 + before = after = 0 + for block in Block.objects.filter(role="source/crops").select_related("data"): + old = block.data + if old.media_type == blobs.CROP_MEDIA_TYPE: + continue + old_digest = old.digest + with open(blobs.path_for(old_digest), "rb") as source: + digest, size = blobs.write_compressed_stream( + iter(lambda: source.read(blobs.CHUNK), b"") + ) + check = hashlib.sha256() + decompressor = zlib.decompressobj() + with open(blobs.path_for(digest), "rb") as compressed: + while chunk := compressed.read(blobs.CHUNK): + check.update(decompressor.decompress(chunk)) + check.update(decompressor.flush()) + if not decompressor.eof or check.hexdigest() != old_digest: + raise CommandError(f"crop compression failed verification: {block.key}") + with transaction.atomic(): + new, _ = Blob.objects.get_or_create( + digest=digest, + defaults={"size": size, "media_type": blobs.CROP_MEDIA_TYPE}, + ) + changed = Block.objects.filter(key=block.key, data=old).update(data=new) + if not changed: + continue + converted += 1 + before += old.size + after += size + if old_digest != new.digest: + try: + old.delete() + except ProtectedError: + pass + else: + blobs.path_for(old_digest).unlink(missing_ok=True) + self.stdout.write( + f"Compressed {converted} crop blocks: {before:,} -> {after:,} bytes" + ) diff --git a/clips/migrations/0005_footage_stream_alter_footage_video.py b/clips/migrations/0005_footage_stream_alter_footage_video.py new file mode 100644 index 0000000..00976af --- /dev/null +++ b/clips/migrations/0005_footage_stream_alter_footage_video.py @@ -0,0 +1,24 @@ +# Generated by Django 5.2.17 on 2026-09-28 17:11 + +import django.db.models.deletion +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0004_footage_video_alter_footageframe_index'), + ] + + operations = [ + migrations.AddField( + model_name='footage', + name='stream', + field=models.ForeignKey(blank=True, help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, one access unit per frame; null on footage extracted before it", null=True, on_delete=django.db.models.deletion.PROTECT, related_name='stream_for', to='clips.blob'), + ), + migrations.AlterField( + model_name='footage', + name='video', + field=models.ForeignKey(blank=True, help_text='the browser-safe proxy, playable and seekable', null=True, on_delete=django.db.models.deletion.PROTECT, related_name='video_for', to='clips.blob'), + ), + ] diff --git a/clips/migrations/0006_project_schema_version.py b/clips/migrations/0006_project_schema_version.py new file mode 100644 index 0000000..6697451 --- /dev/null +++ b/clips/migrations/0006_project_schema_version.py @@ -0,0 +1,15 @@ +from django.db import migrations, models + + +class Migration(migrations.Migration): + dependencies = [ + ("clips", "0005_footage_stream_alter_footage_video"), + ] + + operations = [ + migrations.AddField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=1), + ), + ] diff --git a/clips/models.py b/clips/models.py index 031d5ae..b120873 100644 --- a/clips/models.py +++ b/clips/models.py @@ -102,7 +102,12 @@ class Footage(models.Model): audio = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="audio_for") video = models.ForeignKey( Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="video_for", - help_text="the browser-safe proxy the page detects from; null on pre-proxy footage", + help_text="the browser-safe proxy, playable and seekable", + ) + stream = models.ForeignKey( + Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="stream_for", + help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, " + "one access unit per frame; null on footage extracted before it", ) feature_absence = models.JSONField(default=dict, blank=True) created = models.DateTimeField(auto_now_add=True) @@ -194,14 +199,14 @@ class Block(models.Model): class Project(models.Model): """Tier 1: the document's root. - `seq` is the monotonic project version docs/architecture.md asks for. Every - write bumps it, and a client that sees `seq > local + 1` refetches — which is - what makes staleness self-healing rather than permanent once there is a - broadcast to miss. + `schema_version` identifies the stored document format. `seq` counts writes + to this particular project; it is not a format version. Every write bumps + `seq`, and a client that sees `seq > local + 1` refetches once broadcasts exist. """ id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) name = models.CharField(max_length=200, default="untitled") + schema_version = models.PositiveIntegerField(default=1) seq = models.PositiveBigIntegerField(default=0) palette = models.CharField(max_length=64, default="arthur/default") created = models.DateTimeField(auto_now_add=True) diff --git a/clips/templates/clips/index.html b/clips/templates/clips/index.html index 2a75220..e08393d 100644 --- a/clips/templates/clips/index.html +++ b/clips/templates/clips/index.html @@ -27,6 +27,13 @@ and PUTs with ordinary CSRF protection — no endpoint in this app is exempt. canvas { image-rendering: pixelated; } h1 { font-size: 14px; font-weight: normal; opacity: .5; margin: 0 0 12px; } .stage { display: block; background: #12141c; } + .stage-wrap { position: relative; width: fit-content; } + .paint-overlay { position: absolute; inset: 0; touch-action: none; } + .paint-overlay circle { cursor: grab; } + .paint-tools { width: 640px; margin-top: 9px; font-size: 12px; } + .paint-tools .row { display: flex; flex-wrap: wrap; align-items: center; gap: 6px; margin: 4px 0; } + .paint-tools select { color: var(--fg); background: #1c1f2b; border: 1px solid #2b3040; font: inherit; } + .paint-tools .hint { color: #d0ba86; opacity: .8; } audio { display: none; } .transport { margin-top: 12px; width: 640px; } .transport .row { display: flex; flex-wrap: wrap; gap: 6px; align-items: center; } @@ -48,6 +55,27 @@ and PUTs with ordinary CSRF protection — no endpoint in this app is exempt. color: var(--fg); background: #1c1f2b; border: 1px solid #2b3040; font: inherit; max-width: 360px; } .load-status { margin-top: 6px; font-size: 12px; opacity: .75; } + .export { width: 640px; margin-top: 14px; padding-top: 12px; + border-top: 1px solid #2b3040; font-size: 12px; } + .export .row { display: flex; flex-wrap: wrap; gap: 6px; align-items: center; } + .export .gap { flex: 1; } + .export select { margin-left: 6px; padding: 3px 5px; color: var(--fg); + background: #1c1f2b; border: 1px solid #2b3040; font: inherit; } + .export .readout { margin-top: 7px; } + .export .note { margin: 7px 0 0; } + .controls { width: 640px; margin-top: 18px; padding-top: 12px; + border-top: 1px solid #2b3040; font-size: 12px; } + .controls select { margin-left: 8px; padding: 3px 5px; color: var(--fg); + background: #1c1f2b; border: 1px solid #2b3040; font: inherit; } + .shared-note { margin-top: 6px; color: #d0ba86; } + .control-list { display: grid; grid-template-columns: 1fr 1fr; gap: 6px 16px; + margin-top: 10px; } + .control-row { display: grid; grid-template-columns: 115px 1fr 42px; + align-items: center; gap: 6px; } + .control-row input { width: 100%; } + .control-row output { text-align: right; } + .regeneration-debug { padding: 8px; margin-top: 10px; background: #1c1f2b; + white-space: pre-wrap; color: #d0ba86; } .note { opacity: .35; font-size: 12px; max-width: 640px; } diff --git a/clips/tests/test_api.py b/clips/tests/test_api.py index 411c0d6..08554ad 100644 --- a/clips/tests/test_api.py +++ b/clips/tests/test_api.py @@ -16,6 +16,7 @@ of the system rather than a convention in ClojureScript: The rest is the load/save round trip, the conditional write, and the footage manifest that makes the frames the backend's to serve. """ +import base64 import hashlib import json import shutil @@ -23,11 +24,13 @@ import struct import subprocess import tempfile import zlib +from io import StringIO from pathlib import Path from unittest import skipUnless from unittest.mock import Mock, patch from django.core.files.uploadedfile import SimpleUploadedFile +from django.core.management import call_command from django.test import TestCase, override_settings from clips import blobs, extraction @@ -215,6 +218,47 @@ class Tier2Tests(TestCase): self.assertEqual("AAE=", fetched["state"]) self.assertEqual(descriptor, fetched["descriptor"]) + @override_settings(DATA_UPLOAD_MAX_MEMORY_SIZE=1024, FILE_UPLOAD_MAX_MEMORY_SIZE=1024) + def test_large_block_upload_streams_past_json_body_limit(self): + analysis = self.register_analysis() + descriptor = block_descriptor(analysis, role="source/crops") + key = key_for(descriptor) + payload = bytes(range(256)) * 16 + response = self.client.post("/api/blocks", { + "key": key, + "descriptor": descriptor, + "data": SimpleUploadedFile("block.bin", payload), + "state": SimpleUploadedFile("state.bin", b"\x00\x01"), + }) + self.assertEqual(201, response.status_code, response.content) + row = Block.objects.get(key=key) + self.assertEqual(blobs.CROP_MEDIA_TYPE, row.data.media_type) + self.assertLess(row.data.size, len(payload)) + self.assertEqual(payload, zlib.decompress(blobs.read(row.data_id))) + self.assertEqual(base64.b64encode(payload).decode(), + self.client.get(f"/api/blocks/{key}").json()["data"]) + self.assertEqual(b"\x00\x01", blobs.read(row.state_id)) + + def test_existing_raw_crop_block_is_compressed_without_changing_its_key_or_read(self): + analysis = self.register_analysis() + descriptor = block_descriptor(analysis, role="source/crops") + key = key_for(descriptor) + payload = b"raw crop pixels" * 100 + digest, size = blobs.write(payload) + old = Blob.objects.create(digest=digest, size=size) + Block.objects.create(key=key, descriptor=descriptor, role="source/crops", + analysis_id=analysis, data=old) + + call_command("compress_crop_blocks", stdout=StringIO()) + row = Block.objects.select_related("data").get(key=key) + self.assertEqual(blobs.CROP_MEDIA_TYPE, row.data.media_type) + self.assertEqual(base64.b64encode(payload).decode(), + self.client.get(f"/api/blocks/{key}").json()["data"]) + self.assertFalse(blobs.path_for(digest).exists()) + compressed_digest = row.data_id + call_command("compress_crop_blocks", stdout=StringIO()) + self.assertEqual(compressed_digest, Block.objects.get(key=key).data_id) + def test_a_block_whose_analysis_is_unknown_is_refused(self): descriptor = block_descriptor("sha256:" + "f" * 64) response = self.post("/api/blocks", { @@ -283,6 +327,34 @@ class Tier2Tests(TestCase): f"/api/analyses/{analysis}", json.dumps({"source_blocks": keys[:2]}), content_type="application/json").status_code) + def test_source_roles_are_complete_and_unique_per_subject(self): + analysis = self.register_analysis() + keys = [] + for subject in ("face-1", "face-2"): + for role in ("source/dense", "source/detected", "source/crops"): + desc = json.loads(block_descriptor(analysis, role=role)) + desc["features"] = [subject] + descriptor = json.dumps(desc, sort_keys=True, separators=(",", ":")) + key = key_for(descriptor) + self.assertEqual(201, self.post("/api/blocks", { + "key": key, "descriptor": descriptor, "data": "AA==", + }).status_code) + keys.append(key) + + def put(keys): + return self.client.put(f"/api/analyses/{analysis}", + json.dumps({"source_blocks": keys}), + content_type="application/json") + + self.assertEqual(400, put(keys[:-1]).status_code) + self.assertEqual(400, put(keys + keys[:1]).status_code) + self.assertEqual(200, put(keys).status_code) + self.assertEqual(200, put(list(reversed(keys))).status_code) + self.assertEqual(409, put(keys[:3]).status_code) + self.assertEqual(set(keys), set(self.client.get( + f"/api/analyses/{analysis}").json()["source_blocks"])) + + @override_settings(BLOB_ROOT=BLOB_DIR) class DocumentTests(TestCase): @@ -335,6 +407,7 @@ class DocumentTests(TestCase): self.assertEqual(5, len(response.json()["written"])) loaded = self.client.get(f"/api/projects/{self.project.id}").json() + self.assertEqual(1, loaded["schema_version"]) self.assertEqual(1, len(loaded["clips"])) clip = loaded["clips"][0] self.assertEqual("c1", clip["cid"]) diff --git a/clips/views.py b/clips/views.py index b58a0ba..ce8ce79 100644 --- a/clips/views.py +++ b/clips/views.py @@ -26,6 +26,7 @@ tool got worse", with no event to attach it to. import hashlib import json import re +import zlib from functools import lru_cache from pathlib import Path from uuid import UUID @@ -98,6 +99,22 @@ def _blob(b64, media_type="application/octet-stream"): return blob +def _uploaded_blob(upload, media_type="application/octet-stream"): + digest, size = blobs.write_stream(upload.chunks()) + blob, _ = Blob.objects.get_or_create( + digest=digest, defaults={"size": size, "media_type": media_type} + ) + return blob + + +def _crop_blob(chunks): + digest, size = blobs.write_compressed_stream(chunks) + blob, _ = Blob.objects.get_or_create( + digest=digest, defaults={"size": size, "media_type": blobs.CROP_MEDIA_TYPE} + ) + return blob + + # --------------------------------------------------------------------------- # the page @@ -240,6 +257,8 @@ def _footage_json(footage: Footage, urls=True): # existed, which the loader reports as "re-extract this" rather than # failing somewhere inside MediaPipe. "video": f"/blob/{footage.video.digest}" if footage.video_id else None, + # What the page actually decodes: one access unit per frame, no container. + "stream": f"/blob/{footage.stream.digest}" if footage.stream_id else None, "feature-absence": footage.feature_absence or {}, } if urls: @@ -257,7 +276,7 @@ def footage_list(request): @require_http_methods(["GET"]) def footage_detail(request, footage_id): try: - footage = Footage.objects.select_related("audio", "video").get(id=footage_id) + footage = Footage.objects.select_related("audio", "video", "stream").get(id=footage_id) except Footage.DoesNotExist: return JsonResponse({"error": "no such footage"}, status=404) return JsonResponse(_footage_json(footage)) @@ -414,12 +433,23 @@ def analysis_detail(request, key): try: keys = _body(request).get("source_blocks") roles = {"source/dense", "source/detected", "source/crops"} - if not isinstance(keys, list) or len(keys) != len(roles) or len(set(keys)) != len(roles): - raise Bad("an analysis needs one block for each source role") + if (not isinstance(keys, list) or not keys + or not all(isinstance(k, str) for k in keys) or len(set(keys)) != len(keys)): + raise Bad("an analysis needs distinct source block keys") blocks = list(Block.objects.filter(key__in=keys)) - if (len(blocks) != len(roles) or {b.role for b in blocks} != roles - or any(b.analysis_id != key for b in blocks)): - raise Bad("source blocks must have distinct source roles and name this analysis") + if len(blocks) != len(keys) or any(b.analysis_id != key for b in blocks): + raise Bad("source blocks must exist and name this analysis") + by_subject = {} + for block in blocks: + subjects = json.loads(block.descriptor).get("features", []) + if (not isinstance(subjects, list) or len(subjects) > 1 + or any(not isinstance(s, str) or not s for s in subjects)): + raise Bad("a source block must name one subject") + # Older single-face analyses used an empty feature list. + by_subject.setdefault(tuple(subjects), []).append(block.role) + if any(len(found) != len(roles) or set(found) != roles + for found in by_subject.values()): + raise Bad("each subject needs one block for each source role") with transaction.atomic(): row = Analysis.objects.select_for_update().get(key=key) current = set(row.source_blocks.values_list("key", flat=True)) @@ -454,7 +484,10 @@ def blocks(request): """Store one dense block: its bytes, its optional absence mask, and the descriptor its key is the hash of.""" try: - data = _body(request) + multipart = request.content_type == "multipart/form-data" + data = request.POST if multipart else _body(request) + upload = request.FILES.get("data") if multipart else None + state_upload = request.FILES.get("state") if multipart else None key = data.get("key") descriptor = data.get("descriptor") parsed = _check_key(key, descriptor) @@ -476,17 +509,27 @@ def blocks(request): "version that produced it", analysis=analysis_key, ) - if not data.get("data"): + if not (upload and upload.size) and not data.get("data"): raise Bad("a block with no bytes") with transaction.atomic(): + if role == "source/crops": + if upload: + data_blob = _crop_blob(upload.chunks()) + else: + import base64 + + data_blob = _crop_blob([base64.b64decode(data["data"])]) + else: + data_blob = _uploaded_blob(upload) if upload else _blob(data["data"]) row, created = Block.objects.get_or_create( key=key, defaults={ "descriptor": descriptor, "role": role, "analysis": analysis, - "data": _blob(data["data"]), - "state": _blob(data["state"]) if data.get("state") else None, + "data": data_blob, + "state": (_uploaded_blob(state_upload) if state_upload else + _blob(data["state"]) if data.get("state") else None), }, ) return JsonResponse({"key": row.key, "created": created}, status=201 if created else 200) @@ -502,10 +545,13 @@ def block_detail(request, key): row = Block.objects.select_related("data", "state").get(key=key) except Block.DoesNotExist: return JsonResponse({"error": "no such block"}, status=404) + data = blobs.read(row.data.digest) + if row.data.media_type == blobs.CROP_MEDIA_TYPE: + data = zlib.decompress(data) out = { "key": row.key, "descriptor": row.descriptor, - "data": base64.b64encode(blobs.read(row.data.digest)).decode("ascii"), + "data": base64.b64encode(data).decode("ascii"), } if row.state_id: out["state"] = base64.b64encode(blobs.read(row.state.digest)).decode("ascii") @@ -536,6 +582,7 @@ def _project_json(project: Project): return { "id": str(project.id), "name": project.name, + "schema_version": project.schema_version, "seq": project.seq, "palette": project.palette, "clips": clips, @@ -548,7 +595,8 @@ def projects(request): return JsonResponse( { "projects": [ - {"id": str(p.id), "name": p.name, "seq": p.seq, + {"id": str(p.id), "name": p.name, + "schema_version": p.schema_version, "seq": p.seq, "updated": p.updated.isoformat()} for p in Project.objects.all()[:100] ] @@ -655,6 +703,7 @@ def _save(project: Project, data): return JsonResponse( { "id": str(project.id), + "schema_version": project.schema_version, "seq": seq, "written": sorted(written), "removed": sorted(removed), diff --git a/docs/animation-model.md b/docs/animation-model.md index d7a407e..4c1a2bc 100644 --- a/docs/animation-model.md +++ b/docs/animation-model.md @@ -180,9 +180,10 @@ Three channel shapes, and the uniformity across them is the point: ``` `:interp` defaults to `:hold`, which `docs/design.md` requires of every cut part. -A key may carry its own `:interp` to override the channel's, which is how Lottie -and Blender both do per-key easing; nothing uses it yet and the door is cheap to -leave open. +An authored keyed channel may also carry `:segments {8 :linear}`: the key at 8 +tweens toward the next key, while other gaps use the channel default. The +transition belongs to the gap starting at a key, so a shape can cut into one +drawing and tween out of it. Per-key easing beyond hold and linear is deferred. ### Keys are a map by frame, not a list @@ -338,54 +339,47 @@ head is placed and scaled where it belongs and the rest of the frame is simply not on stage. The full frame stays *available* for tracing without being *visible*, and those are different requirements. -## The anchor: stabilisation is a channel, not a mode +## Head motion: free or anchored to measured frames -`stabilize` produces `{s, θ, tx, ty}` per frame, which is exactly -`[:xform :scale]`, `[:xform :rot]` and `[:xform :pos]`. So removing the head's -motion is not a pipeline setting — it is a question of **which node holds that -motion**, and the answer is one channel definition: +`stabilize` produces `{s, θ, tx, ty}` per source frame. Its inverse is stored +densely on `:head`'s position, rotation and scale channels. The same measured +track serves every placement choice: ```clojure -;; locked: the head sits still, for tracing and for judging articulation -[:xform :pos] {:animated? false :value [0.0 0.0]} - -;; as filmed: the head moves around the stage -[:xform :pos] {:animated? true :interp :hold - :dense {:store "sha256:…" :stride 2 :frames 600} - :generated {:by :anchor/similarity}} - -;; per plate: the head snaps at each selected frame and holds -[:xform :pos] {:animated? true :interp :hold :keys {0 […], 12 […], 23 […]}} +;; no :anchors — free: read the measured transform at the current frame +;; one key — lock to a chosen measured frame throughout +:anchors {0 12} +;; several keys — cut to another measured head transform at frame 40 +:anchors {0 12, 40 42} ``` -The three modes are the three channel shapes, on one channel, on one node. The -third is the one a plate strip wants — the head pose is stable for exactly as -long as a drawing is on screen — and it costs nothing because `:keys` already -exists. Its frame set is the kept-frame set, which is `suggestPlateFrames` in the -prototype and belongs to painting rather than to measurement. +The map is `local change frame -> measured source frame`. A single lock is a +one-key map. Position, rotation and scale read the same held source frame. The +frame set belongs to head placement, independently of plate drawings and stage +pose cuts. No measured block is copied into authored transform keys. -**Always measure, always store factored, toggle the parent.** The fit is computed -and the geometry is stored head-local in every mode, and only the parent's -channel changes. Two things downstream require it, and both would be lost by -making this an analysis-time switch: +**Always measure, always store factored.** The fit is computed and the geometry +is stored head-local in every mode. Only the frame address used to read the +head's measured transform changes. Two things downstream require that split: - *Smoothing.* "Smooth the transform, never the contour" only means anything while the two are separate. - *Key selection.* A velocity minimum is "articulation paused" in head-local space and "the head happened to be still" in image space. -It also makes the toggle an edit to the document rather than a reason to -re-analyse: tier 1, undoable, syncable, and instant. +This is a document edit, not a reason to re-analyse. A registered tracing photo +will use its own source frame's stabilising transform followed by the same +selected head placement, so it aligns with the vectors drawn over it. ### Two nodes, because two different things want that transform ``` :face group — AUTHORED. where the face sits on the stage, and how big. - :head group — MEASURED. the head's motion, or identity. + :head group — MEASURED. dense head motion read at the selected frame. :mouth :mouth-in :teeth :lid-r :lid-l :brow-r :brow-l … ``` -Switching modes rewrites `:head` and never touches `:face`, so it cannot move +Changing anchor keys edits `:head` and never touches `:face`, so it cannot move something that was placed by hand. A group node is free, and keeping the authored and the measured transform apart is the whole reason the transform is decomposed in the first place. @@ -495,6 +489,39 @@ reading head over the same channel. The resolver keys its caches by the instance path, not by node id — which is a detail of `Making it fast` below, and the one place symbol nesting is not free. +### Audio placements and controls + +Sound is placed on a timeline as a separate `:audio` node. It uses the same +`:span`, `:time`, and channel representation as a drawn node. A `:linked-to` id +records which picture instance it was placed with; it does not force the two +spans or source in-points to match. + +```clojure +{:id :voice-right :kind :audio :parent :root :z "a4" + :linked-to :right + :source {:footage "f8cace9e-..."} + :span [48 260] + :time {:mode :map :at 48 :in 0 :rate 1} + :channels {[:audio :gain] + {:animated? true :interp :linear + :keys {48 0.0, 60 1.0, 245 1.0, 259 0.0} :over []}}} +``` + +`[:audio :gain]`, `[:audio :pan]`, and `[:audio :rate]` are ordinary scalar +channels. They may be framed, keyed, or dense; numeric keyed channels can ramp +linearly. The time map sets the placement's base source rate, and +`[:audio :rate]` multiplies it. Audio is mixed from the referenced immutable +footage when the clip opens. The mix is derived output; the saved document holds +the nodes and channel keys, not another audio file. One audio element plays that +mix and remains the clock for both sound and picture. + +This is also the boundary for a future control surface. A control has a stable +target, such as a feature's `:verts` setting or an audio node's +`[:audio :gain]` channel. The UI and a MIDI binding can address both through the +same control interface. Their update costs differ: gain can be keyed over time; +changing the number of lip vertices changes topology and must regenerate its +dense geometry. A topology setting cannot be treated as a per-frame gain curve. + ## Evaluating a frame ```clojure @@ -593,12 +620,12 @@ Proof that it covers what exists, not just what is wanted: | brow ring + quantised raise | node `:brow-r`, `[:geom :pts]` dense (the traced ring with height removed), `[:xform :pos]` dense (the quantised raise). **The decomposition design.md insists on is two channels.** | | head plate, kept frames | node `:head`, `:symbol` per instance, keys on `[:symbol]` at kept frames | | `makeXform` face-oval crop | **gone.** Placement is `[:xform :*]` on `:face`; the stage clips | -| `stabilize` transforms | `[:xform :*]` on `:head` — framed identity, dense, or keyed at kept frames | +| `stabilize` transforms | dense `[:xform :*]` on `:head`, read through its optional `:anchors` map | | registered underlay | not data — a UI layer riding `(world-of resolver :head)` | | painted background cel | node per layer, `[:geom :pts]` **framed**, `[:style :color]` framed | | `mouth lead` | `:time {:offset k}` on performance nodes only | | `exposure` | `:time {:expose n}` on the clip root, inherited | -| picture fps | `:time {:source-fps s :sample-fps p}` on the clip root, applied after analysis | +| picture fps | resolver samples marked generated channels at the picture rate; authored keys keep their own time | | hand correction | an `:over` layer, `:offset` or `:replace` | The brow row is the one worth looking at twice. `docs/design.md` argues at length diff --git a/docs/architecture.md b/docs/architecture.md index e749caf..90d6536 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -145,8 +145,8 @@ boundary.** | # | Stage | In | Out | Cost | | --- | --- | --- | --- | --- | -| 1 | **ingest** | video | footage: a seekable H.264 proxy, tracing stills, audio, manifest | minutes, in-app | -| 2 | **detect** | the proxy, walked one frame at a time | raw landmarks per frame | minutes, **cached** | +| 1 | **ingest** | video | footage: an H.264 proxy and raw stream, tracing stills, audio, manifest | minutes, in-app | +| 2 | **detect** | the raw stream, decoded one frame at a time | raw landmarks per frame | minutes, **cached** | | 3 | **measure** | landmarks | anchor fit, residual, head-local rings, signals, interior pixels | seconds | | 4 | **condition** | measurements | smoothed transforms and contours | milliseconds | | 5 | **key** | conditioned signals + policy | channels: sparse keys, quantised holds, kept frames | milliseconds | @@ -444,9 +444,12 @@ must never run while the transport is moving. The mouth crops are the non-obvious entry, and they are what makes remote work possible at all. `extractTeeth` reads source pixels, so without them a -collaborator holding the analysis but not the 600 source PNGs cannot touch a -single teeth knob. A 40×30 crop is about 1.2KB; a 600-frame take is under a -megabyte against hundreds for the footage. +collaborator holding the analysis but not the source video cannot touch a +single teeth knob without decoding video again. These are RGBA crops: a 40×30 +crop is 4.8KB raw, and current 200-pixel-wide crops can total over 12MB for a +take. The blob store compresses `source/crops` losslessly with zlib; block reads +return the original pixels. Run `python manage.py compress_crop_blocks` once to +convert existing raw crop blobs and remove their unreferenced copies. ### Bake B — resolved geometry. For scale. @@ -573,10 +576,10 @@ POST /api/analyses {key, descriptor} idempotent GET /api/analyses/ metadata + source block keys PUT /api/analyses/ link dense landmarks, mask, crops POST /api/blocks/missing {keys} -> {missing} -POST /api/blocks {key, descriptor, data, state} +POST /api/blocks multipart: key, descriptor, data file, optional state file (JSON also accepted) GET /api/blocks/ -GET /api/footage/ the manifest: the proxy to measure, audio, a URL per tracing still -GET /blob/ immutable bytes, and RANGE-capable so a