diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..ef0e9d8 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,16 @@ +.git +.venv +**/node_modules +frontend/.shadow-cljs +frontend/out +frontend/.cpcache +frontend/test/browser/out +static/arthur/js +db.sqlite3 +var +frames +scratch +audio.wav +manifest.json +*.take +*.tflite diff --git a/.gitignore b/.gitignore index caf400b..5f441bf 100644 --- a/.gitignore +++ b/.gitignore @@ -1,5 +1,40 @@ +# extract.sh's output. It is TIER 3 — immutable, large, and the backend's to serve +# once `manage.py ingest_bundle` has hashed it into var/blobs — so none of it +# belongs in the repo. `audio.wav` was tracked before step 9 because the synthetic +# take borrowed it for a clock; that copy now lives at static/arthur/audio.wav, +# which is an asset the project owns rather than an extraction that churns. frames/ +/audio.wav +/manifest.json +# local extracted takes for comparing source cadences +/scratch/ *.task *.take *.tflite + +# CLJS build +frontend/node_modules/ +# screenshots from the browser suite; regenerated by `npm run browser` +frontend/test/browser/out/ +frontend/.shadow-cljs/ +frontend/out/ +frontend/.cpcache/ +static/arthur/js/ + +# mise-managed venv for the Django half +.venv/ + +# the Django half's own state: the document database, the content-addressed blob +# store (tiers 2 and 3), and collectstatic's output +db.sqlite3 +db.sqlite3-shm +db.sqlite3-wal +/var/ + +# vim swap files +*.swp + +# Python bytecode +__pycache__/ +*.py[cod] diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..805d9a6 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,43 @@ +FROM node:20-bookworm AS frontend + +RUN apt-get update \ + && apt-get install -y --no-install-recommends ca-certificates curl tar \ + && rm -rf /var/lib/apt/lists/* \ + && curl -fsSL 'https://api.adoptium.net/v3/binary/latest/21/ga/linux/x64/jdk/hotspot/normal/eclipse' -o /tmp/jdk.tar.gz \ + && mkdir -p /opt/java \ + && tar -xzf /tmp/jdk.tar.gz -C /opt/java --strip-components=1 \ + && rm /tmp/jdk.tar.gz + +ENV JAVA_HOME=/opt/java +ENV PATH="/opt/java/bin:${PATH}" +WORKDIR /app/frontend +COPY frontend/package.json frontend/package-lock.json ./ +RUN npm ci +COPY frontend/ ./ +RUN npx shadow-cljs release app + +FROM python:3.12-slim-bookworm + +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + DJANGO_DEBUG=0 \ + DJANGO_DB_PATH=/data/db.sqlite3 \ + DJANGO_BLOB_ROOT=/data/blobs + +RUN apt-get update \ + && apt-get install -y --no-install-recommends ffmpeg \ + && rm -rf /var/lib/apt/lists/* \ + && useradd --create-home --shell /usr/sbin/nologin app + +WORKDIR /app +COPY requirements.txt ./ +RUN pip install --no-cache-dir -r requirements.txt +COPY . ./ +COPY --from=frontend /app/static/arthur/js/ ./static/arthur/js/ +RUN python manage.py collectstatic --noinput \ + && mkdir -p /data \ + && chown -R app:app /app /data + +USER app +EXPOSE 8000 +CMD ["sh", "-c", "python manage.py migrate --noinput && exec gunicorn server.wsgi:application --bind 0.0.0.0:8000 --workers 1 --threads 4 --timeout 120"] diff --git a/README.md b/README.md index 8f13ce7..6f178d2 100644 --- a/README.md +++ b/README.md @@ -14,6 +14,62 @@ Animator Pro, where this started, but they are why the output looks right — modern conveniences belong in the workflow, not the output. See [docs/design.md](docs/design.md). +## ClojureScript port + +The active port plays the synthetic take, accepts video uploads, transcodes them +to an H.264 proxy and decodable stream plus audio and tracing stills, analyzes real footage +for mouth, eyes, brows and pixel-derived teeth, and saves the project with +reusable analysis data. The step 8 data model +represents persistent feature IDs, eye pairs and feature-level observation gaps; +its controls are still pending. See the [port plan](docs/port-plan.md). + +```sh +mise install # both halves +pip install -r requirements.txt +mise exec -- python manage.py migrate +./do start # Django + frontend watcher +``` + +In the app, upload a video, choose its footage, click **load frames**, then +**save**. Opening that project on another client reuses its saved landmarks and +mouth crops without detecting source frames again. The upload path derives its +footage response from database records; it does not create or consume a +`manifest.json` file. See [frontend/README.md](frontend/README.md) for details. +Anything under "## Run" and below describes the older JS prototype, which still +runs separately on port 8777. + +### Deploy to Fly.io + +The app uses a persistent Fly volume for its SQLite document database and +content-addressed footage blobs. Create the app once, set its Django secret, and +deploy from the repository root: + +```sh +fly apps create arthur --org personal +fly volumes create data --region iad --size 1 +fly secrets set DJANGO_SECRET_KEY="$(openssl rand -hex 32)" --app arthur +fly deploy --app arthur +``` + +The deployed app is at . The container builds the +ClojureScript frontend, collects static assets, and runs database migrations on +startup. + +### How it is stored + +Three tiers, cut by mutability and size — the full argument is in +[docs/architecture.md](docs/architecture.md): + +| Tier | What | Where | +| --- | --- | --- | +| 1 **authored** | the scene: nodes, channels, features, time maps | the database, as independently addressed leaves. Kilobytes | +| 2 **derived** | detected landmarks, raw mouth crops, and dense channel blocks | `var/blobs`, addressed by analysis and block inputs, including the detector version | +| 3 **source** | the uploaded video, H.264 proxy and elementary stream, tracing stills, and audio | the same blob store, by the hash of their bytes | + +Only tier 1 is the document. Tier 2 is a pure function of tiers 1 and 3, so a +saved project names its blocks rather than carrying them, and a knob change gives +a block a new name rather than overwriting an old one. + ## Run ```sh @@ -33,27 +89,35 @@ wasm, which is fetched from a CDN on first use. For real footage: -```sh -./extract.sh /path/to/clip.mov 12 # -> frames/*.png, audio.wav, manifest.json -``` +Upload it in the app. `./extract.sh` still writes the old PNG-sequence bundle and +`ingest_bundle` still registers it, but footage ingested that way has no decodable +stream and the loader will say so — the measured pixels come out of the video now. -then **Load frames**. MediaPipe's wasm is fetched from jsdelivr on first use; -`face_landmarker.task` is local. +MediaPipe's wasm and `face_landmarker.task` are both local; nothing in detection +touches the network. -Frames are pre-extracted rather than decoded in the page because browser video -seeking is approximate and `requestVideoFrameCallback` only delivers frames at -playback speed — neither gives a deterministic per-frame pass. +Detection reads the H.264 elementary stream with WebCodecs, one coded frame at a +time. The proxy has no B-frames, so decode order is frame order. Each decoded +frame reaches MediaPipe in VIDEO running mode at its footage timestamp. The +decoder and detector advance together, with a pause between frames so progress +can paint. Saved analyses reuse their stored crop pixels and measure them with +the same pauses. -`manifest.json` records the true extraction rate. The page reads it rather than -assuming, because a guessed fps desynchronises audio from picture — and sync is -the one thing this view exists to show. +This is what replaced the PNG sequence, which was 112MB for 7.6 seconds and would +be 1.1GB at the 900-frame limit. The proxy is 6MB, and the landmarks barely +notice: detected off decoded H.264 rather than off the PNGs, they moved at most +0.0033 of frame width. -**Exposure** decides how often the picture gets a new drawing: rip at 24 and -render `on 2s` for 12, `on 3s` for 8. The dense track and the audio are -untouched, so it is a dropdown rather than a re-rip, and the export emits keys -only on the grid instead of the same pose twice. Everything rides the same grid -— mouth, eyes, teeth, plate — because a head cutting on the odd frames while the -mouth cuts on the even ones reads as two performances laid over each other. +The server's footage manifest records the proxy's frame rate and frame count; +the page reads that rate because a guessed fps desynchronises audio from picture. +Choosing a lower picture rate happens after analysis. + +**Picture fps** decides how often the finished roto gets a new pose. Analyze all +source frames, then sample those frozen poses at 12, 24 or the source rate while +keeping the original duration and audio. **Exposure** can hold a drawing across +more than one picture slot. The tracing editor chooses source frames for cel +references separately. Shared timing is the useful default for mouth, eyes, +teeth and plate so their changes read as one performance. **Audio is the playback clock**: `frame = floor(audio.currentTime * fps)`. A slow render loop therefore drops frames instead of drifting, and ½x / ¼x work by @@ -314,13 +378,13 @@ the tool a person made by hand; everything else regenerates. They are not in the ## Two kinds of sparseness Sparseness has two unrelated causes, and conflating them was the original design -error here. **Aesthetic** sparseness is set by the extraction rate — pick 12fps and -you have already chosen your timing. **Labour** sparseness is a human drawing -each one, and it binds only on the plate. +error here. **Aesthetic** sparseness is chosen from the full analyzed source +track at rendering time. **Labour** sparseness is a human drawing each cel and +selecting which source frames to use as tracing references. -Aesthetic sparseness is the **exposure** control, not the extraction rate — -making it a render-time grid means auditioning 12 against 24 costs a dropdown -instead of a re-rip and a full re-detection. +Aesthetic sparseness is the **picture fps** control, with exposure available for +longer holds. Both happen after analysis, so auditioning 12 against 24 needs no +re-extraction or re-detection. So the mouth keeps **every** frame: it is traced, and therefore free. In limited animation lip sync is routinely the densest element, on 1s, while heads hold on @@ -373,3 +437,16 @@ performer→character calibration (currently identity, fitting the face oval to canvas); the override layer; anything on the Animator Pro side. The plate is a face-oval polygon per kept frame — it exists so the mouth has a face to read against, not to look good. + +In the port specifically: the parameter UI and scoped regeneration (the model is +built, the controls are not); automatic per-feature detection, so presence still +comes from the full-face mask plus a manifest annotation; multiplayer, for which +step 9 built the addressing and none of the socket; and in-browser extraction, so +`extract.sh` plus `manage.py ingest_bundle` is still how footage arrives. + +Two smaller things that are known and undecided. `measure/brows` takes no +`presence` where `measure/eyes` does, so an occluded brow affects the freeze mask +but not brow measurement, and occluded landmarks still enter contour smoothing — +asymmetric with the eyes, and it is not settled which way is right. And `open` +takes the most recently updated project and shows its first clip: there is no +project browser, and the runtime store holds one clip at a time. diff --git a/audio.wav b/audio.wav deleted file mode 100644 index f5793b0..0000000 Binary files a/audio.wav and /dev/null differ diff --git a/clips/__init__.py b/clips/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/admin.py b/clips/admin.py new file mode 100644 index 0000000..9e73270 --- /dev/null +++ b/clips/admin.py @@ -0,0 +1,63 @@ +"""The admin, which is here for one reason: tier 1 is readable. + +docs/architecture.md's argument against a CRDT is partly this — "the canonical +document moves into an opaque blob, and every server-side thing that reads the +document needs it materialised back out". A leaf is transit-as-JSON in a +JSONField, so it is legible here, and that is a property worth being able to see. +""" +from django.contrib import admin + +from .models import Analysis, Block, Blob, Clip, Footage, FootageFrame, Leaf, Project, Revision + + +@admin.register(Project) +class ProjectAdmin(admin.ModelAdmin): + list_display = ("name", "id", "seq", "updated") + search_fields = ("name", "id") + + +@admin.register(Clip) +class ClipAdmin(admin.ModelAdmin): + list_display = ("cid", "project", "name", "footage", "analysis") + list_filter = ("project",) + + +@admin.register(Leaf) +class LeafAdmin(admin.ModelAdmin): + list_display = ("path", "project", "version", "updated") + list_filter = ("project",) + search_fields = ("path",) + + +@admin.register(Revision) +class RevisionAdmin(admin.ModelAdmin): + list_display = ("project", "seq", "summary", "author", "created") + + +@admin.register(Footage) +class FootageAdmin(admin.ModelAdmin): + list_display = ("label", "source", "fps", "frames", "width", "height", "created") + + +@admin.register(FootageFrame) +class FootageFrameAdmin(admin.ModelAdmin): + list_display = ("footage", "index", "blob") + list_filter = ("footage",) + + +@admin.register(Analysis) +class AnalysisAdmin(admin.ModelAdmin): + list_display = ("key", "detector", "version", "footage", "created") + search_fields = ("key", "detector", "version") + + +@admin.register(Block) +class BlockAdmin(admin.ModelAdmin): + list_display = ("key", "role", "analysis", "data", "state", "created") + list_filter = ("role",) + search_fields = ("key",) + + +@admin.register(Blob) +class BlobAdmin(admin.ModelAdmin): + list_display = ("digest", "media_type", "size", "created") diff --git a/clips/apps.py b/clips/apps.py new file mode 100644 index 0000000..6b51579 --- /dev/null +++ b/clips/apps.py @@ -0,0 +1,14 @@ +from django.apps import AppConfig + + +class ClipsConfig(AppConfig): + """The one app. + + `clips` because the CLIP is the entity the whole tool is about and the one the + prototype had exactly one of and never named — `state` in `js/app.js` is a clip + with its analysis inlined and its palette global. Project, Footage, Analysis, + Block, Leaf and Revision all hang off it. + """ + + default_auto_field = "django.db.models.BigAutoField" + name = "clips" diff --git a/clips/blobs.py b/clips/blobs.py new file mode 100644 index 0000000..2216fcb --- /dev/null +++ b/clips/blobs.py @@ -0,0 +1,146 @@ +"""The content-addressed blob store: tiers 2 and 3 on disk. + +One store for both, and docs/architecture.md says why in a sentence: once tier 3 +is decoded by the app rather than by a shell script, frames and audio become "the +same kind of thing as tier 2 — a cache with a hash". So there is one place that +writes bytes, one that reads them, and one URL shape for both. + +TWO KINDS OF HASH, AND THEY ARE NOT THE SAME HASH. A blob is named by the sha256 +of its BYTES: that is what makes identical frames in two extractions one file. A +derived thing — an analysis artifact, a dense block — is named by a sha256 over +its INPUTS, which is what lets the client ask for the block the current settings +want before anything has computed it. So `Block.key` is an input hash and +`Block.data.digest` is a byte hash, and conflating them would break the half of +addressing that answers questions about work not yet done. +""" +import hashlib +import os +import tempfile +import zlib +from pathlib import Path + +from django.conf import settings + +CHUNK = 1 << 20 +CROP_MEDIA_TYPE = "application/zlib" + + +def digest_bytes(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +def digest_file(path: Path) -> str: + h = hashlib.sha256() + with open(path, "rb") as fh: + while chunk := fh.read(CHUNK): + h.update(chunk) + return h.hexdigest() + + +def path_for(digest: str) -> Path: + """Where a blob lives. + + Fanned out two levels, so that a take's worth of frames does not put a hundred + thousand entries in one directory — which is slow on every filesystem and + unusable on some. + """ + if len(digest) != 64 or any(c not in "0123456789abcdef" for c in digest): + raise ValueError(f"not a sha256: {digest!r}") + return Path(settings.BLOB_ROOT) / digest[:2] / digest[2:4] / digest + + +def write(data: bytes) -> tuple[str, int]: + """Store bytes, return (digest, size). Writing the same bytes twice is a + no-op, which is what content addressing is for.""" + digest = digest_bytes(data) + dest = path_for(digest) + if not dest.exists(): + dest.parent.mkdir(parents=True, exist_ok=True) + tmp = dest.with_suffix(".part") + with open(tmp, "wb") as fh: + fh.write(data) + os.replace(tmp, dest) + return digest, len(data) + + +def write_stream(chunks) -> tuple[str, int]: + """Store an uploaded file without reading the whole video into memory.""" + root = Path(settings.BLOB_ROOT) + root.mkdir(parents=True, exist_ok=True) + digest = hashlib.sha256() + size = 0 + with tempfile.NamedTemporaryFile(dir=root, prefix="upload-", delete=False) as out: + temporary = Path(out.name) + try: + for chunk in chunks: + digest.update(chunk) + size += len(chunk) + out.write(chunk) + except BaseException: + temporary.unlink(missing_ok=True) + raise + dest = path_for(digest.hexdigest()) + dest.parent.mkdir(parents=True, exist_ok=True) + if dest.exists(): + temporary.unlink() + else: + os.replace(temporary, dest) + return digest.hexdigest(), size + + +def write_compressed_stream(chunks) -> tuple[str, int]: + """Store a losslessly compressed stream; the digest names stored bytes.""" + compressor = zlib.compressobj() + + def compressed(): + for chunk in chunks: + if part := compressor.compress(chunk): + yield part + if part := compressor.flush(): + yield part + + return write_stream(compressed()) + + +def adopt(source: Path) -> tuple[str, int]: + """Store a file already on disk, by hard link where the filesystem allows it. + + 112MB of PNGs is a normal extraction and copying them into a second place in + the tree for no reason is not. A hard link is exact — the blob is immutable, so + two names for one inode is the whole of what is wanted — and a copy is the + fallback when `extract.sh` wrote to another volume. + """ + digest = digest_file(source) + dest = path_for(digest) + size = source.stat().st_size + if not dest.exists(): + dest.parent.mkdir(parents=True, exist_ok=True) + try: + os.link(source, dest) + except OSError: + tmp = dest.with_suffix(".part") + with open(source, "rb") as src, open(tmp, "wb") as out: + while chunk := src.read(CHUNK): + out.write(chunk) + os.replace(tmp, dest) + return digest, size + + +def read(digest: str) -> bytes: + with open(path_for(digest), "rb") as fh: + return fh.read() + + +def png_size(path: Path) -> tuple[int, int]: + """A PNG's dimensions, out of its IHDR. + + Twenty-four bytes rather than a dependency. The footage's width and height are + manifest data — docs/architecture.md's entity model puts them there — and + Pillow to read two integers out of a header that has held them in the same + place since 1996 is not a trade worth making. + """ + with open(path, "rb") as fh: + head = fh.read(24) + if head[:8] != b"\x89PNG\r\n\x1a\n" or head[12:16] != b"IHDR": + raise ValueError(f"{path} is not a PNG") + return int.from_bytes(head[16:20], "big"), int.from_bytes(head[20:24], "big") diff --git a/clips/extraction.py b/clips/extraction.py new file mode 100644 index 0000000..ff31bb6 --- /dev/null +++ b/clips/extraction.py @@ -0,0 +1,384 @@ +"""Upload a video once, then turn it into the two things the app actually reads. + +WHAT CHANGED AND WHY. This used to decode one PNG per source frame and store every +one of them. A 7.6-second 1440x1920 take is 112MB that way, and the 900-frame limit +is 1.1GB — for pixels whose only consumer was a canvas that MediaPipe then read +once. The page now detects from the video itself (see `frontend/src/arthur/flow/ +ingest.cljs`), so this produces: + + THE PROXY. One browser-safe H.264/yuv420p MP4, CFR, `+faststart`. The same take + is 6MB. This is the analysis source, and it is re-encoded RATHER THAN KEPT AS + UPLOADED even when the upload is already H.264, for two reasons that are both + about not guessing: an iPhone's HEVC is not decodable in every browser, and the + footage's identity is the digest of this file — one produced by one ffmpeg + invocation, not one that depends on which branch the source happened to take. + + THE TRACING STILLS. One JPEG per frame, long edge capped, for the tracing editor + to draw over. Reference images; nothing measures them. They are not in the + footage digest — see `models.Footage`. + +The proxy is probed after it is written rather than before. `width`, `height` and +`frames` are properties of the file the browser will decode, and taking them from +the source instead is how a scaler or a dropped frame becomes a silent one-frame +offset between the landmarks and the audio. +""" + +import hashlib +import json +import subprocess +import tempfile +import threading +import time +from fractions import Fraction +from pathlib import Path + +from django.db import close_old_connections, transaction + +from . import blobs +from .models import Blob, Extraction, Footage, FootageFrame + +_active = set() +_lock = threading.Lock() +TIMEOUT = 3600 + +# Visually lossless enough that landmarks do not move: measured against the same +# frames as PNGs, IMAGE-mode landmarks shifted at most 0.0033 of frame width. +PROXY_CRF = "18" +# The long edge of a tracing still. The proxy keeps full resolution because the +# detector reads it; a still only has to be good enough to draw a cel over. +TRACING_EDGE = 1280 +TRACING_QUALITY = "4" + + +def _command(args): + result = subprocess.run(args, capture_output=True, text=True, timeout=TIMEOUT) + if result.returncode: + raise ValueError((result.stderr or result.stdout or "media tool failed")[-1200:]) + return result.stdout + + +def _run_with_progress(job, args, root, name, total, span): + """Run one ffmpeg and publish its live frame count as `span` of the job. + + ffmpeg's `-progress` file is the only honest source for this: parsing its + stderr means parsing a format that is explicitly not an interface, and a + spinner that is not attached to frames is a spinner that lies on a long take. + """ + progress_path = root / f"{name}.progress" + log_path = root / f"{name}.log" + first, last = span + args = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-stats_period", "0.25", "-progress", str(progress_path)] + args + with open(log_path, "wb") as log: + proc = subprocess.Popen(args, stdout=log, stderr=subprocess.STDOUT) + deadline = time.monotonic() + TIMEOUT + try: + while proc.poll() is None: + if time.monotonic() >= deadline: + raise TimeoutError(f"{name} timed out") + if progress_path.exists(): + lines = progress_path.read_text(errors="replace").splitlines() + count = next((int(line[6:].strip()) for line in reversed(lines) + if line.startswith("frame=") and + line[6:].strip().isdigit()), 0) + if count and total: + reached = first + int((last - first) * min(1.0, count / total)) + if reached > job.progress: + job.progress = reached + job.save(update_fields=["progress", "updated"]) + time.sleep(0.2) + finally: + if proc.poll() is None: + proc.kill() + proc.wait() + if proc.returncode: + raise ValueError(log_path.read_text(errors="replace")[-1200:] or f"{name} failed") + + +def _encode_proxy(job, source_path, proxy_path, facts, root): + """The uploaded video -> one H.264 file every browser can decode and seek.""" + total = facts.get("reported_frames") or round(facts["duration"] * facts["fps"]) + _run_with_progress( + job, + ["-i", str(source_path), "-an", + # Constant frame rate at the rate `probe` chose. This RESAMPLES rather + # than asserts: the upload is allowed to be variable, and this is the + # step that makes the thing the page measures not be. + "-fps_mode", "cfr", "-r", facts.get("rate") or str(facts["fps"]), + "-c:v", "libx264", "-preset", "veryfast", "-crf", PROXY_CRF, + # NO B-FRAMES, AND THIS IS THE LOAD-BEARING FLAG. It is what makes + # decode order presentation order, so the page can treat access unit k + # of the elementary stream as frame k without demuxing a container or + # consulting a timestamp. With them x264 has a + # two-frame reordering delay, ffmpeg compensates by writing an edit list + # (`elst` media_time 1024 at timebase 1/15360 — exactly two frames), and + # the browser then lives on two timelines at once: `currentTime` obeys the + # edit list and the `mediaTime` reported by requestVideoFrameCallback does + # not. Seek to frame 0 and the browser correctly hands back a frame whose + # mediaTime says 2. Software decoding hides it; hardware decoding does + # not, which is the worst possible way for it to be wrong. Without + # B-frames DTS equals PTS, no edit list is written, and the two timelines + # are the same one. It also makes decode order presentation order, should + # this ever be fed to a WebCodecs VideoDecoder. + "-bf", "0", + # yuv420p and an even frame size are what makes this playable everywhere + # rather than only in the browser that happened to be tested. + "-pix_fmt", "yuv420p", "-vf", "scale=trunc(iw/2)*2:trunc(ih/2)*2", + "-movflags", "+faststart", str(proxy_path)], + root, "proxy", total, (0, 55)) + + +def _elementary_stream(proxy_path, out_path): + """The proxy's video, unwrapped into a raw Annex-B H.264 stream. + + A STREAM COPY, not a second encode: the same coded frames as the MP4, with + the container's length-prefixed NAL units rewritten as start-code-delimited + ones. It costs a file read and nothing else. + + This exists because the page decodes with WebCodecs, and `VideoDecoder` takes + demuxed chunks rather than a container. Handing it Annex-B means the client + needs no demuxer: NAL start codes are findable in a loop, and because the + proxy is encoded with no B-frames, decode order is presentation order — so + access unit k IS frame k, with no container timing to consult and no clock to + reconcile. That is the whole reason this file is worth the bytes it costs. + """ + _command(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-i", str(proxy_path), "-an", "-c:v", "copy", + "-bsf:v", "h264_mp4toannexb", "-f", "h264", str(out_path)]) + + +def _extract_stills(job, proxy_path, frames_dir, frames, root): + """The proxy -> one tracing JPEG per frame, long edge capped.""" + _run_with_progress( + job, + ["-i", str(proxy_path), "-fps_mode", "passthrough", + "-vf", f"scale='if(gt(iw,ih),min({TRACING_EDGE},iw),-2)':" + f"'if(gt(iw,ih),-2,min({TRACING_EDGE},ih))'", + "-q:v", TRACING_QUALITY, str(frames_dir / "%04d.jpg")], + root, "stills", frames, (55, 85)) + + +MAX_RATE = 120 # a capture rate; past this the container is describing something else + + +def probe(path): + """What the upload is, as far as choosing a proxy rate goes. + + IT NO LONGER REFUSES VARIABLE-FRAME-RATE INPUT, and the reason is the proxy. + That refusal was written when the page measured the source's own frames, where + a wandering frame duration really does break `frame = floor(t * fps)`. Nothing + measures the source now: ffmpeg resamples it onto a constant rate, and the + proxy — constant by construction, and re-probed after it is written — is the + only timeline anything downstream sees. + + Keeping the check would have been worse than useless, because the thing it + tested is not reliable. Ordinary iPhone footage, shot straight from the camera + app, reports `avg_frame_rate` 8670/299 and `nb_frames` 289 on a stream whose + decoded timestamps are 280 frames exactly 1/30s apart. The container's summary + of itself disagreed with the container's own contents, so the guard rejected + CFR video for being variable. + + THE RATE IS THE NOMINAL ONE. `r_frame_rate` is the rate every timestamp in the + stream can be expressed at, which is the rate that keeps every distinct source + frame; resampling to the average would drop some. Duration is preserved either + way — ffmpeg's CFR conversion is driven by timestamps, so the audio stays in + sync at any rate — so this trades a possible duplicated frame against a + certainly lost one. + """ + data = json.loads(_command(["ffprobe", "-v", "error", "-show_streams", + "-show_format", "-of", "json", str(path)])) + video = next((s for s in data.get("streams", []) if s.get("codec_type") == "video"), None) + if not video: + raise ValueError("the uploaded file has no video stream") + nominal = Fraction(video.get("r_frame_rate") or "0") + average = Fraction(video.get("avg_frame_rate") or "0") + if nominal <= 0 and average <= 0: + raise ValueError("the video's frame rate is unknown") + rate = nominal if 0 < nominal <= MAX_RATE else average + if not 0 < rate <= MAX_RATE: + raise ValueError(f"the video reports a frame rate of {float(rate):g}, which is " + "not a rate footage can be measured at") + duration = float(data.get("format", {}).get("duration") or 0) + if duration > 0 and duration * float(rate) > 901: + raise ValueError("video is longer than the 900-frame footage limit") + frames = video.get("nb_frames") + return {"fps": float(rate), + # The exact rate, for ffmpeg. 30000/1001 is not a float, and handing + # `-r` a rounded one is how a long take drifts out of sync. + "rate": f"{rate.numerator}/{rate.denominator}", + "nominal_fps": float(nominal), "average_fps": float(average), + "width": int(video["width"]), "height": int(video["height"]), + "duration": duration, + # KEPT, AND NO LONGER TRUSTED AS A COUNT. See the docstring: this is + # the container's claim about itself, it is wrong on ordinary phone + # footage, and `run` checks the proxy's DURATION instead. + "reported_frames": int(frames) if frames and frames.isdigit() else None, + "has_audio": any(s.get("codec_type") == "audio" for s in data.get("streams", [])), + "vfr": nominal != average} + + +def _refuse_a_shifted_timeline(path): + """The proxy must put frame `i` at `i / fps` on BOTH of the browser's clocks. + + Asserted rather than assumed, because the failure is silent and the symptom is + unrecognisable. An encoder delay makes ffmpeg write an edit list, `currentTime` + then obeys it while `requestVideoFrameCallback`'s `mediaTime` does not, and the + page's frame walk is uniformly off by the delay — on hardware decoding only. It + cost two wrong diagnoses to find, so it does not get to come back silently if + somebody changes an encoder flag. + """ + data = json.loads(_command(["ffprobe", "-v", "error", "-select_streams", "v:0", + "-show_streams", "-of", "json", str(path)])) + stream = data["streams"][0] + if int(stream.get("has_b_frames") or 0): + raise ValueError( + "the proxy was encoded with B-frames, whose reordering delay makes the " + "browser's seek clock and its frame-timestamp clock disagree") + if float(stream.get("start_time") or 0) != 0: + raise ValueError(f"the proxy starts at {stream['start_time']}s rather than 0") + + +def count_frames(path): + """How many frames a file really holds, counted rather than reported. + + `nb_frames` is a container's claim. This is the decoder's answer, and it is + what the page will get when it walks the proxy — so a disagreement between the + two has to be settled before the count reaches a manifest, not after it has + become a one-frame audio offset nobody can find. + """ + text = _command(["ffprobe", "-v", "error", "-select_streams", "v:0", + "-count_frames", "-show_entries", "stream=nb_read_frames", + "-of", "default=nokey=1:noprint_wrappers=1", str(path)]) + counted = text.strip() + if not counted.isdigit(): + raise ValueError("could not count the proxy's frames") + return int(counted) + + +def extraction_key(source, settings): + # Scheme 3: the extraction now also produces the elementary stream the page + # decodes, so a job run under scheme 2 did not make everything this one does. + text = json.dumps({"scheme": 3, "source": source.blob_id, "settings": settings}, + sort_keys=True, separators=(",", ":")) + return "sha256:" + hashlib.sha256(text.encode()).hexdigest() + + +def _register(job, proxy_path, stream_path, stills, audio_path, facts): + proxy_digest, proxy_size = blobs.adopt(proxy_path) + stream_digest, stream_size = blobs.adopt(stream_path) + audio_digest, audio_size = blobs.adopt(audio_path) + still_blobs = [(index, *blobs.adopt(path)) for index, path in enumerate(stills)] + width, height, fps, frames = facts["width"], facts["height"], facts["fps"], facts["frames"] + + # The footage's own identity: the bytes the page will measure, the audio it + # will clock against, and the rate that ties them together. Scheme 2 — scheme + # 1 hashed a PNG per frame, and those footages name pixels this no longer has. + h = hashlib.sha256() + h.update(f"arthur-footage-2/{fps}/{frames}/{width}x{height}\n".encode()) + h.update(proxy_digest.encode()) + h.update(audio_digest.encode()) + + with transaction.atomic(): + proxy_blob, _ = Blob.objects.get_or_create( + digest=proxy_digest, defaults={"size": proxy_size, "media_type": "video/mp4"}) + stream_blob, _ = Blob.objects.get_or_create( + digest=stream_digest, defaults={"size": stream_size, "media_type": "video/h264"}) + audio_blob, _ = Blob.objects.get_or_create( + digest=audio_digest, defaults={"size": audio_size, "media_type": "audio/wav"}) + footage, created = Footage.objects.get_or_create( + digest=h.hexdigest(), + defaults={"label": job.source.filename[:200], "source": job.source.filename[:200], + "fps": fps, "frames": frames, "width": width, "height": height, + "audio": audio_blob, "video": proxy_blob, "stream": stream_blob}) + if not created and not footage.stream_id: + # The same footage by identity, extracted before the elementary + # stream existed. Its digest is over the proxy and the audio, which + # have not changed — so this is the same footage gaining a file it + # was always entitled to, not a different one. + footage.stream = stream_blob + if not footage.video_id: + footage.video = proxy_blob + footage.save(update_fields=["stream", "video"]) + if created: + rows = [] + for index, digest, size in still_blobs: + blob, _ = Blob.objects.get_or_create( + digest=digest, defaults={"size": size, "media_type": "image/jpeg"}) + rows.append(FootageFrame(footage=footage, index=index, blob=blob)) + FootageFrame.objects.bulk_create(rows) + return footage + + +def run(key): + close_old_connections() + try: + job = Extraction.objects.select_related("source", "source__blob").get(key=key) + job.state, job.progress, job.error = "running", 0, "" + job.save(update_fields=["state", "progress", "error", "updated"]) + facts = job.source.probe + source_path = blobs.path_for(job.source.blob_id) + with tempfile.TemporaryDirectory(prefix="arthur-extract-") as directory: + root = Path(directory) + proxy_path = root / "proxy.mp4" + _encode_proxy(job, source_path, proxy_path, facts, root) + + # Everything downstream describes the PROXY, not the upload. + proxy_facts = probe(proxy_path) + _refuse_a_shifted_timeline(proxy_path) + frames = count_frames(proxy_path) + if not 1 <= frames <= 900: + raise ValueError(f"the proxy holds {frames} frames; the limit is 1–900") + # CHECKED AS A DURATION, not as a frame count. The page's clock is + # `frame = floor(audio.currentTime * fps)`, so what must not drift is + # how long the picture lasts against how long the audio lasts — and + # the source's own frame count is a number this has already caught + # lying. A resample to a constant rate legitimately changes the count + # and must not change the duration. + drift = abs(frames / proxy_facts["fps"] - facts["duration"]) + if facts["duration"] > 0 and drift > 0.5: + raise ValueError( + f"the proxy runs {frames / proxy_facts['fps']:.2f}s and the upload " + f"runs {facts['duration']:.2f}s; refusing footage whose picture and " + "audio would drift") + proxy_facts["frames"] = frames + + stream_path = root / "proxy.h264" + _elementary_stream(proxy_path, stream_path) + + frames_dir = root / "stills" + frames_dir.mkdir() + _extract_stills(job, proxy_path, frames_dir, frames, root) + stills = sorted(frames_dir.glob("*.jpg")) + if len(stills) != frames: + raise ValueError(f"wrote {len(stills)} tracing stills for {frames} frames") + + job.progress = 85 + job.save(update_fields=["progress", "updated"]) + audio_path = root / "audio.wav" + if facts["has_audio"]: + _command(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-i", str(source_path), "-vn", "-ac", "1", "-ar", "44100", + str(audio_path)]) + else: + _command(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-f", "lavfi", "-i", "anullsrc=r=44100:cl=mono", + "-t", str(frames / proxy_facts["fps"]), "-c:a", "pcm_s16le", + str(audio_path)]) + footage = _register(job, proxy_path, stream_path, stills, audio_path, proxy_facts) + job.footage, job.state, job.progress = footage, "done", 100 + job.save(update_fields=["footage", "state", "progress", "updated"]) + except Exception as exc: + Extraction.objects.filter(key=key).update(state="failed", error=str(exc)[:2000]) + finally: + with _lock: + _active.discard(key) + close_old_connections() + + +def enqueue(key): + with _lock: + if key in _active: + return + _active.add(key) + threading.Thread(target=run, args=(key,), daemon=True, + name=f"arthur-extract-{key[7:15]}").start() diff --git a/clips/management/__init__.py b/clips/management/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/management/commands/__init__.py b/clips/management/commands/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/management/commands/compress_crop_blocks.py b/clips/management/commands/compress_crop_blocks.py new file mode 100644 index 0000000..6367b81 --- /dev/null +++ b/clips/management/commands/compress_crop_blocks.py @@ -0,0 +1,57 @@ +"""Compress existing raw mouth crop blocks without changing their public bytes.""" + +import hashlib +import zlib + +from django.core.management.base import BaseCommand, CommandError +from django.db import transaction +from django.db.models.deletion import ProtectedError + +from clips import blobs +from clips.models import Blob, Block + + +class Command(BaseCommand): + help = "Compress existing source/crops blobs and remove unreferenced raw copies" + + def handle(self, *args, **options): + converted = 0 + before = after = 0 + for block in Block.objects.filter(role="source/crops").select_related("data"): + old = block.data + if old.media_type == blobs.CROP_MEDIA_TYPE: + continue + old_digest = old.digest + with open(blobs.path_for(old_digest), "rb") as source: + digest, size = blobs.write_compressed_stream( + iter(lambda: source.read(blobs.CHUNK), b"") + ) + check = hashlib.sha256() + decompressor = zlib.decompressobj() + with open(blobs.path_for(digest), "rb") as compressed: + while chunk := compressed.read(blobs.CHUNK): + check.update(decompressor.decompress(chunk)) + check.update(decompressor.flush()) + if not decompressor.eof or check.hexdigest() != old_digest: + raise CommandError(f"crop compression failed verification: {block.key}") + with transaction.atomic(): + new, _ = Blob.objects.get_or_create( + digest=digest, + defaults={"size": size, "media_type": blobs.CROP_MEDIA_TYPE}, + ) + changed = Block.objects.filter(key=block.key, data=old).update(data=new) + if not changed: + continue + converted += 1 + before += old.size + after += size + if old_digest != new.digest: + try: + old.delete() + except ProtectedError: + pass + else: + blobs.path_for(old_digest).unlink(missing_ok=True) + self.stdout.write( + f"Compressed {converted} crop blocks: {before:,} -> {after:,} bytes" + ) diff --git a/clips/management/commands/ingest_bundle.py b/clips/management/commands/ingest_bundle.py new file mode 100644 index 0000000..89ca5b5 --- /dev/null +++ b/clips/management/commands/ingest_bundle.py @@ -0,0 +1,120 @@ +"""Register an extracted bundle as tier 3. + + python manage.py ingest_bundle # ./manifest.json + python manage.py ingest_bundle scratch/my-take # that bundle + +WHAT THIS REPLACES. Until step 9 the page fetched `/manifest.json` and then built +`frames/0001.png` itself, with shadow-cljs's `:dev-http` serving the repo root. So +the frame layout was a shared secret between a shell script and a ClojureScript +namespace, and "where the frames are" was answered by a directory listing. + +Now the server names every frame, and the client asks it. The frames go into the +content-addressed blob store — by hard link, so 112MB of PNGs is not copied — and +the manifest the client receives carries a URL per frame. That is the whole of what +makes the frames the backend's to serve, and it is what the in-browser wasm-ffmpeg +extraction docs/architecture.md describes will upload INTO, without the client +learning anything new when it arrives: the same blobs, the same manifest, a +different producer. + +`extract.sh` still does the decoding. It is out of step 9's scope, it works, and it +is the only part of this that needs a terminal. +""" +import json +from pathlib import Path + +from django.core.management.base import BaseCommand, CommandError +from django.db import transaction + +from clips import blobs +from clips.models import Blob, Footage, FootageFrame + + +class Command(BaseCommand): + help = "Register an extracted frames+audio+manifest bundle as footage." + + def add_arguments(self, parser): + parser.add_argument( + "bundle", nargs="?", default=".", + help="a directory holding manifest.json, or the manifest itself", + ) + parser.add_argument("--label", default="", help="what to call it in the UI") + + def handle(self, *args, **options): + manifest_path = Path(options["bundle"]) + if manifest_path.is_dir(): + manifest_path = manifest_path / "manifest.json" + if not manifest_path.exists(): + raise CommandError(f"{manifest_path} does not exist — run ./extract.sh first") + + manifest = json.loads(manifest_path.read_text()) + root = manifest_path.parent + frames_dir = root / manifest["dir"] + audio_path = root / manifest["audio"] + count = int(manifest["frames"]) + + pngs = sorted(frames_dir.glob("*.png")) + if len(pngs) != count: + raise CommandError( + f"the manifest says {count} frames and {frames_dir} holds {len(pngs)}; " + "refusing an inaccurate footage" + ) + if not audio_path.exists(): + raise CommandError(f"{audio_path} does not exist") + + width, height = blobs.png_size(pngs[0]) + + self.stdout.write(f"hashing {len(pngs)} frames…") + frame_blobs = [] + for i, png in enumerate(pngs): + digest, size = blobs.adopt(png) + frame_blobs.append((i, digest, size)) + if (i + 1) % 25 == 0 or i + 1 == len(pngs): + self.stdout.write(f" {i + 1}/{len(pngs)}") + + audio_digest, audio_size = blobs.adopt(audio_path) + + # The footage's own identity: every frame in order, plus the audio and the + # rate. Two extractions of one clip at one rate are one footage, so an + # analysis over it is reusable across both. + import hashlib + + h = hashlib.sha256() + h.update(f"arthur-footage-1/{manifest['fps']}/{count}/{width}x{height}\n".encode()) + for _, digest, _ in frame_blobs: + h.update(digest.encode()) + h.update(audio_digest.encode()) + footage_digest = h.hexdigest() + + if existing := Footage.objects.filter(digest=footage_digest).first(): + self.stdout.write(self.style.SUCCESS(f"already ingested: {existing.id}")) + return + + with transaction.atomic(): + audio_blob, _ = Blob.objects.get_or_create( + digest=audio_digest, + defaults={"size": audio_size, "media_type": "audio/wav"}, + ) + footage = Footage.objects.create( + digest=footage_digest, + label=options["label"] or manifest.get("source") or frames_dir.name, + source=manifest.get("source") or "", + fps=float(manifest["fps"]), + frames=count, + width=width, + height=height, + audio=audio_blob, + feature_absence=manifest.get("feature-absence") or {}, + ) + rows = [] + for index, digest, size in frame_blobs: + blob, _ = Blob.objects.get_or_create( + digest=digest, defaults={"size": size, "media_type": "image/png"} + ) + rows.append(FootageFrame(footage=footage, index=index, blob=blob)) + FootageFrame.objects.bulk_create(rows) + + self.stdout.write( + self.style.SUCCESS( + f"{count} frames at {manifest['fps']}fps, {width}x{height} -> footage {footage.id}" + ) + ) diff --git a/clips/migrations/0001_initial.py b/clips/migrations/0001_initial.py new file mode 100644 index 0000000..15e07ba --- /dev/null +++ b/clips/migrations/0001_initial.py @@ -0,0 +1,149 @@ +# Generated by Django 5.2.17 on 2026-09-28 04:44 + +import django.db.models.deletion +import uuid +from django.db import migrations, models + + +class Migration(migrations.Migration): + + initial = True + + dependencies = [ + ] + + operations = [ + migrations.CreateModel( + name='Blob', + fields=[ + ('digest', models.CharField(max_length=64, primary_key=True, serialize=False)), + ('media_type', models.CharField(default='application/octet-stream', max_length=100)), + ('size', models.BigIntegerField()), + ('created', models.DateTimeField(auto_now_add=True)), + ], + ), + migrations.CreateModel( + name='Project', + fields=[ + ('id', models.UUIDField(default=uuid.uuid4, editable=False, primary_key=True, serialize=False)), + ('name', models.CharField(default='untitled', max_length=200)), + ('seq', models.PositiveBigIntegerField(default=0)), + ('palette', models.CharField(default='arthur/default', max_length=64)), + ('created', models.DateTimeField(auto_now_add=True)), + ('updated', models.DateTimeField(auto_now=True)), + ], + options={ + 'ordering': ['-updated'], + }, + ), + migrations.CreateModel( + name='Analysis', + fields=[ + ('key', models.CharField(max_length=71, primary_key=True, serialize=False)), + ('descriptor', models.TextField()), + ('detector', models.CharField(max_length=64)), + ('version', models.CharField(max_length=64)), + ('created', models.DateTimeField(auto_now_add=True)), + ('artifact', models.ForeignKey(blank=True, help_text='the dense landmark track, once bake A is uploaded', null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='analysis_for', to='clips.blob')), + ], + options={ + 'verbose_name_plural': 'analyses', + }, + ), + migrations.CreateModel( + name='Block', + fields=[ + ('key', models.CharField(max_length=71, primary_key=True, serialize=False)), + ('descriptor', models.TextField()), + ('role', models.CharField(max_length=32)), + ('created', models.DateTimeField(auto_now_add=True)), + ('analysis', models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='blocks', to='clips.analysis')), + ('data', models.ForeignKey(on_delete=django.db.models.deletion.PROTECT, related_name='block_data_for', to='clips.blob')), + ('state', models.ForeignKey(blank=True, help_text='the per-track absence mask, when the take has one', null=True, on_delete=django.db.models.deletion.PROTECT, related_name='block_state_for', to='clips.blob')), + ], + ), + migrations.CreateModel( + name='Footage', + fields=[ + ('id', models.UUIDField(default=uuid.uuid4, editable=False, primary_key=True, serialize=False)), + ('digest', models.CharField(max_length=64, unique=True)), + ('label', models.CharField(blank=True, max_length=200)), + ('source', models.CharField(blank=True, max_length=200)), + ('fps', models.FloatField()), + ('frames', models.PositiveIntegerField()), + ('width', models.PositiveIntegerField()), + ('height', models.PositiveIntegerField()), + ('feature_absence', models.JSONField(blank=True, default=dict)), + ('created', models.DateTimeField(auto_now_add=True)), + ('audio', models.ForeignKey(on_delete=django.db.models.deletion.PROTECT, related_name='audio_for', to='clips.blob')), + ], + options={ + 'ordering': ['-created'], + }, + ), + migrations.AddField( + model_name='analysis', + name='footage', + field=models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='analyses', to='clips.footage'), + ), + migrations.CreateModel( + name='Revision', + fields=[ + ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('seq', models.PositiveBigIntegerField()), + ('author', models.CharField(blank=True, max_length=200)), + ('summary', models.CharField(blank=True, max_length=500)), + ('document', models.JSONField(help_text='every leaf of the project, by path')), + ('created', models.DateTimeField(auto_now_add=True)), + ('project', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='revisions', to='clips.project')), + ], + options={ + 'ordering': ['-seq'], + }, + ), + migrations.CreateModel( + name='FootageFrame', + fields=[ + ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('index', models.PositiveIntegerField(help_text="0-based; source frame index + 1 is the PNG's name")), + ('blob', models.ForeignKey(on_delete=django.db.models.deletion.PROTECT, related_name='frame_for', to='clips.blob')), + ('footage', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='frame_set', to='clips.footage')), + ], + options={ + 'ordering': ['index'], + 'constraints': [models.UniqueConstraint(fields=('footage', 'index'), name='one_blob_per_frame')], + }, + ), + migrations.CreateModel( + name='Leaf', + fields=[ + ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('path', models.CharField(max_length=300)), + ('value', models.JSONField()), + ('version', models.PositiveBigIntegerField(default=1)), + ('updated', models.DateTimeField(auto_now=True)), + ('project', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='leaves', to='clips.project')), + ], + options={ + 'ordering': ['path'], + 'constraints': [models.UniqueConstraint(fields=('project', 'path'), name='one_leaf_per_path')], + }, + ), + migrations.CreateModel( + name='Clip', + fields=[ + ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('cid', models.SlugField(max_length=64)), + ('name', models.CharField(blank=True, max_length=200)), + ('order', models.IntegerField(default=0)), + ('analysis', models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='clips', to='clips.analysis')), + ('blocks', models.ManyToManyField(blank=True, help_text="the tier-2 blocks this clip's channels name", related_name='clips', to='clips.block')), + ('footage', models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='clips', to='clips.footage')), + ('project', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='clips', to='clips.project')), + ], + options={ + 'ordering': ['order', 'cid'], + 'constraints': [models.UniqueConstraint(fields=('project', 'cid'), name='one_cid_per_project')], + }, + ), + ] diff --git a/clips/migrations/0002_remove_analysis_artifact_analysis_source_blocks.py b/clips/migrations/0002_remove_analysis_artifact_analysis_source_blocks.py new file mode 100644 index 0000000..a4f3396 --- /dev/null +++ b/clips/migrations/0002_remove_analysis_artifact_analysis_source_blocks.py @@ -0,0 +1,22 @@ +# Generated by Django 5.2.17 on 2026-09-28 13:10 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0001_initial'), + ] + + operations = [ + migrations.RemoveField( + model_name='analysis', + name='artifact', + ), + migrations.AddField( + model_name='analysis', + name='source_blocks', + field=models.ManyToManyField(blank=True, help_text='pixel-dependent landmarks, detection mask and mouth crops', related_name='source_for', to='clips.block'), + ), + ] diff --git a/clips/migrations/0003_source_extraction.py b/clips/migrations/0003_source_extraction.py new file mode 100644 index 0000000..2bb2f51 --- /dev/null +++ b/clips/migrations/0003_source_extraction.py @@ -0,0 +1,39 @@ +# Generated by Django 5.2.17 on 2026-09-28 13:23 + +import django.db.models.deletion +import uuid +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0002_remove_analysis_artifact_analysis_source_blocks'), + ] + + operations = [ + migrations.CreateModel( + name='Source', + fields=[ + ('id', models.UUIDField(default=uuid.uuid4, editable=False, primary_key=True, serialize=False)), + ('filename', models.CharField(max_length=255)), + ('probe', models.JSONField(default=dict)), + ('created', models.DateTimeField(auto_now_add=True)), + ('blob', models.OneToOneField(on_delete=django.db.models.deletion.PROTECT, related_name='video_source', to='clips.blob')), + ], + ), + migrations.CreateModel( + name='Extraction', + fields=[ + ('key', models.CharField(max_length=71, primary_key=True, serialize=False)), + ('settings', models.JSONField(default=dict)), + ('state', models.CharField(default='queued', max_length=16)), + ('progress', models.PositiveIntegerField(default=0)), + ('error', models.TextField(blank=True)), + ('created', models.DateTimeField(auto_now_add=True)), + ('updated', models.DateTimeField(auto_now=True)), + ('footage', models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='extractions', to='clips.footage')), + ('source', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='extractions', to='clips.source')), + ], + ), + ] diff --git a/clips/migrations/0004_footage_video_alter_footageframe_index.py b/clips/migrations/0004_footage_video_alter_footageframe_index.py new file mode 100644 index 0000000..6e53732 --- /dev/null +++ b/clips/migrations/0004_footage_video_alter_footageframe_index.py @@ -0,0 +1,24 @@ +# Generated by Django 5.2.17 on 2026-09-28 15:10 + +import django.db.models.deletion +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0003_source_extraction'), + ] + + operations = [ + migrations.AddField( + model_name='footage', + name='video', + field=models.ForeignKey(blank=True, help_text='the browser-safe proxy the page detects from; null on pre-proxy footage', null=True, on_delete=django.db.models.deletion.PROTECT, related_name='video_for', to='clips.blob'), + ), + migrations.AlterField( + model_name='footageframe', + name='index', + field=models.PositiveIntegerField(help_text="0-based; source frame index + 1 is the JPEG's name"), + ), + ] diff --git a/clips/migrations/0005_footage_stream_alter_footage_video.py b/clips/migrations/0005_footage_stream_alter_footage_video.py new file mode 100644 index 0000000..00976af --- /dev/null +++ b/clips/migrations/0005_footage_stream_alter_footage_video.py @@ -0,0 +1,24 @@ +# Generated by Django 5.2.17 on 2026-09-28 17:11 + +import django.db.models.deletion +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0004_footage_video_alter_footageframe_index'), + ] + + operations = [ + migrations.AddField( + model_name='footage', + name='stream', + field=models.ForeignKey(blank=True, help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, one access unit per frame; null on footage extracted before it", null=True, on_delete=django.db.models.deletion.PROTECT, related_name='stream_for', to='clips.blob'), + ), + migrations.AlterField( + model_name='footage', + name='video', + field=models.ForeignKey(blank=True, help_text='the browser-safe proxy, playable and seekable', null=True, on_delete=django.db.models.deletion.PROTECT, related_name='video_for', to='clips.blob'), + ), + ] diff --git a/clips/migrations/0006_project_schema_version.py b/clips/migrations/0006_project_schema_version.py new file mode 100644 index 0000000..6697451 --- /dev/null +++ b/clips/migrations/0006_project_schema_version.py @@ -0,0 +1,15 @@ +from django.db import migrations, models + + +class Migration(migrations.Migration): + dependencies = [ + ("clips", "0005_footage_stream_alter_footage_video"), + ] + + operations = [ + migrations.AddField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=1), + ), + ] diff --git a/clips/migrations/__init__.py b/clips/migrations/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/models.py b/clips/models.py new file mode 100644 index 0000000..b120873 --- /dev/null +++ b/clips/models.py @@ -0,0 +1,311 @@ +"""The entity model, as tables. + +It follows docs/architecture.md's model exactly, and the one thing worth reading +it for is which tier each table is in, because that is what decides whether a row +is a document, a cache entry or a source. + + TIER 1, the document. Project, Clip, Leaf, Revision. Kilobytes, authored, + versioned, and the only tier anything will ever sync. + + TIER 2, derived. Analysis, Block. Content-addressed by a hash over every input + that produced them — including the detector version — so a stale bake is + unreachable rather than wrong, and a collaborator's bake is fetchable by the + same key. + + TIER 3, source. Footage, FootageFrame. Immutable, by hash. + + Blob is under all three of them: bytes, named by the sha256 of themselves. + +WHAT IS DELIBERATELY NOT HERE. `Clip` does not store fps, frames, width or height. +They are in the document — the `timing` and `stage` leaves — and a copy of them in +a column is a copy that comes to disagree with the scene it describes. The columns +`Clip` does have are the ones the SERVER needs to answer a question about a clip +without parsing its leaves: which footage, which analysis, which blocks. +""" +import uuid + +from django.db import models + + +class Blob(models.Model): + """Bytes, named by the sha256 of themselves. The file is on disk under + `BLOB_ROOT`; this row is the index and the size.""" + + digest = models.CharField(primary_key=True, max_length=64) + media_type = models.CharField(max_length=100, default="application/octet-stream") + size = models.BigIntegerField() + created = models.DateTimeField(auto_now_add=True) + + def __str__(self): + return f"{self.digest[:12]}… {self.size}B {self.media_type}" + + +class Source(models.Model): + """An uploaded video, identified by its byte digest.""" + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + blob = models.OneToOneField(Blob, on_delete=models.PROTECT, related_name="video_source") + filename = models.CharField(max_length=255) + probe = models.JSONField(default=dict) + created = models.DateTimeField(auto_now_add=True) + + +class Extraction(models.Model): + """One requested decode of a source into immutable footage.""" + + key = models.CharField(primary_key=True, max_length=71) + source = models.ForeignKey(Source, on_delete=models.CASCADE, related_name="extractions") + settings = models.JSONField(default=dict) + state = models.CharField(max_length=16, default="queued") + progress = models.PositiveIntegerField(default=0) + error = models.TextField(blank=True) + footage = models.ForeignKey( + "Footage", null=True, blank=True, on_delete=models.SET_NULL, + related_name="extractions", + ) + created = models.DateTimeField(auto_now_add=True) + updated = models.DateTimeField(auto_now=True) + + +class Footage(models.Model): + """Tier 3: the frames and audio of one extraction, immutable. + + `digest` is over the PROXY VIDEO's digest plus the audio's and the rate, so + two extractions of the same clip at the same settings are one footage and the + same analysis can be reused across both. + + THE PROXY IS THE ANALYSIS SOURCE AND THE FRAMES ARE NOT. `video` is one + browser-safe H.264 file, and it is what the page seeks through to detect + landmarks. `frame_set` is a JPEG per frame at tracing size: reference stills + for the tracing editor, never the thing measured. The two are not + interchangeable, and which one carries the pixels an analysis was computed + from is the difference between a 6MB take and a 1.1GB one. + + So the frame JPEGs are deliberately NOT in `digest`. They are a rendering of + this footage for a human to trace over; re-rendering them at another size does + not make it different footage, and putting them in the identity would throw + away every analysis when the tracing size changed. + + `feature_absence` is the manifest annotation step 8 introduced: known + occlusion intervals, one-based and inclusive, expanded into presence tracks by + the loader. An input format, not a control UI. + """ + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + digest = models.CharField(max_length=64, unique=True) + label = models.CharField(max_length=200, blank=True) + source = models.CharField(max_length=200, blank=True) + fps = models.FloatField() + frames = models.PositiveIntegerField() + width = models.PositiveIntegerField() + height = models.PositiveIntegerField() + audio = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="audio_for") + video = models.ForeignKey( + Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="video_for", + help_text="the browser-safe proxy, playable and seekable", + ) + stream = models.ForeignKey( + Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="stream_for", + help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, " + "one access unit per frame; null on footage extracted before it", + ) + feature_absence = models.JSONField(default=dict, blank=True) + created = models.DateTimeField(auto_now_add=True) + + class Meta: + ordering = ["-created"] + + def __str__(self): + return f"{self.label or self.source or self.id} ({self.frames}f @{self.fps})" + + +class FootageFrame(models.Model): + """One tracing still. A row rather than an entry in a JSON list, because a + frame is a thing the server serves, and because a blob's references have to be + countable before anything can be collected. + + A REFERENCE IMAGE, NOT A MEASUREMENT INPUT. See `Footage.video`.""" + + footage = models.ForeignKey(Footage, on_delete=models.CASCADE, related_name="frame_set") + index = models.PositiveIntegerField(help_text="0-based; source frame index + 1 is the JPEG's name") + blob = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="frame_for") + + class Meta: + ordering = ["index"] + constraints = [ + models.UniqueConstraint(fields=["footage", "index"], name="one_blob_per_frame"), + ] + + +class Analysis(models.Model): + """Tier 2: one detector, at one version, over one footage. + + `key` is a content address over every input, and `descriptor` is the exact + canonical text that key is the sha256 of — sent by the client and stored, not + recomputed here. `clips/views.py` says why that is the honest arrangement: JS + prints an integral double as `1` and Python as `1.0`, so a scheme where both + sides re-render the numbers breaks on the first one of them. + + `detector` and `version` are columns as well as descriptor fields so that the + question "which model produced this take" is answerable in the admin and in a + query, rather than only by parsing a hash's preimage. + """ + + key = models.CharField(primary_key=True, max_length=71) + descriptor = models.TextField() + detector = models.CharField(max_length=64) + version = models.CharField(max_length=64) + footage = models.ForeignKey( + Footage, null=True, blank=True, on_delete=models.SET_NULL, related_name="analyses" + ) + source_blocks = models.ManyToManyField( + "Block", blank=True, related_name="source_for", + help_text="pixel-dependent landmarks, detection mask and mouth crops", + ) + created = models.DateTimeField(auto_now_add=True) + + class Meta: + verbose_name_plural = "analyses" + + def __str__(self): + return f"{self.detector} {self.version} → {self.key[7:19]}…" + + +class Block(models.Model): + """Tier 2: one dense channel block. + + Two hashes, and they are not the same hash. `key` is over the block's INPUTS, + which is what lets a client ask for the block its current settings want before + anything has computed it. `data.digest` is over the bytes. See clips/blobs.py. + """ + + key = models.CharField(primary_key=True, max_length=71) + descriptor = models.TextField() + role = models.CharField(max_length=32) + analysis = models.ForeignKey( + Analysis, null=True, blank=True, on_delete=models.SET_NULL, related_name="blocks" + ) + data = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="block_data_for") + state = models.ForeignKey( + Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="block_state_for", + help_text="the per-track absence mask, when the take has one", + ) + created = models.DateTimeField(auto_now_add=True) + + def __str__(self): + return f"{self.role} {self.key[7:19]}…" + + +class Project(models.Model): + """Tier 1: the document's root. + + `schema_version` identifies the stored document format. `seq` counts writes + to this particular project; it is not a format version. Every write bumps + `seq`, and a client that sees `seq > local + 1` refetches once broadcasts exist. + """ + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + name = models.CharField(max_length=200, default="untitled") + schema_version = models.PositiveIntegerField(default=1) + seq = models.PositiveBigIntegerField(default=0) + palette = models.CharField(max_length=64, default="arthur/default") + created = models.DateTimeField(auto_now_add=True) + updated = models.DateTimeField(auto_now=True) + + class Meta: + ordering = ["-updated"] + + def __str__(self): + return f"{self.name} ({self.id})" + + def bump(self): + self.seq += 1 + self.save(update_fields=["seq", "updated"]) + return self.seq + + +class Clip(models.Model): + """Tier 1: the unit of work, and the thing leaf paths are scoped by. + + `cid` is what appears in `clip//...`, so it is the clip's identity as far + as addressing is concerned and it does not change. + """ + + project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="clips") + cid = models.SlugField(max_length=64) + name = models.CharField(max_length=200, blank=True) + order = models.IntegerField(default=0) + footage = models.ForeignKey( + Footage, null=True, blank=True, on_delete=models.SET_NULL, related_name="clips" + ) + analysis = models.ForeignKey( + Analysis, null=True, blank=True, on_delete=models.SET_NULL, related_name="clips" + ) + blocks = models.ManyToManyField( + Block, blank=True, related_name="clips", + help_text="the tier-2 blocks this clip's channels name", + ) + + class Meta: + ordering = ["order", "cid"] + constraints = [ + models.UniqueConstraint(fields=["project", "cid"], name="one_cid_per_project"), + ] + + def __str__(self): + return f"{self.cid} of {self.project.name}" + + +class Leaf(models.Model): + """Tier 1: one independently addressed, independently versioned piece of the + document. + + The value is transit-as-JSON in a JSONField, so the column holds JSON rather + than a string containing JSON: the admin can read a leaf, and the field-wise + merge of a channel leaf that docs/architecture.md describes as fifteen lines of + Python is possible over it. `version` is the entity tag a conditional write + compares — RFC 7232, not a bespoke invention. + """ + + project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="leaves") + path = models.CharField(max_length=300) + value = models.JSONField() + version = models.PositiveBigIntegerField(default=1) + updated = models.DateTimeField(auto_now=True) + + class Meta: + ordering = ["path"] + constraints = [ + models.UniqueConstraint(fields=["project", "path"], name="one_leaf_per_path"), + ] + + @property + def etag(self): + return f'"{self.version}"' + + def __str__(self): + return f"{self.path}@{self.version}" + + +class Revision(models.Model): + """Tier 1: a snapshot of the authored layer, with a user and a summary. + + ON AN EXPLICIT TRIGGER, not on every save. tl snapshots a small annotation + layer; arthur's tier 1 will contain cel polygons, so a snapshot per save bloats + the table — docs/architecture.md's "revisions need a coarser trigger". So this + is written by `POST /api/projects//revisions`, which is a "mark version" + button, and never by a save. + """ + + project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="revisions") + seq = models.PositiveBigIntegerField() + author = models.CharField(max_length=200, blank=True) + summary = models.CharField(max_length=500, blank=True) + document = models.JSONField(help_text="every leaf of the project, by path") + created = models.DateTimeField(auto_now_add=True) + + class Meta: + ordering = ["-seq"] + + def __str__(self): + return f"{self.project.name} r{self.seq}: {self.summary}" diff --git a/clips/templates/clips/index.html b/clips/templates/clips/index.html new file mode 100644 index 0000000..e08393d --- /dev/null +++ b/clips/templates/clips/index.html @@ -0,0 +1,88 @@ +{% load static %} +{% comment %} +The host page, served by Django since port-plan step 9. + +It was `frontend/public/index.html`, served by shadow-cljs's `:dev-http`, and that +key is gone. The bundle is unchanged: shadow-cljs writes it into +`static/arthur/js` and staticfiles serves it from there, so `manage.py runserver` +and `shadow-cljs watch app` are the whole dev loop with nothing copying files +between them. + +The CSRF token is rendered so that Django sets its cookie, which is what +`arthur.fx.http` reads to write the `X-CSRFToken` header. Saves are ordinary POSTs +and PUTs with ordinary CSRF protection — no endpoint in this app is exempt. +{% endcomment %} + + + + + arthur + + + + {% csrf_token %} +
+ + + + diff --git a/clips/tests/__init__.py b/clips/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/tests/test_api.py b/clips/tests/test_api.py new file mode 100644 index 0000000..08554ad --- /dev/null +++ b/clips/tests/test_api.py @@ -0,0 +1,816 @@ +"""What the server guarantees, as opposed to what the client intends. + +The two interesting groups here are the ones that make the tier split a property +of the system rather than a convention in ClojureScript: + + A KEY DESCRIBES ITS BYTES. The server recomputes every tier-2 key it is handed + and refuses a mismatch, so nothing can store a block under a name that is not + the hash of its own descriptor. + + A BLOCK CAN NAME ITS DETECTOR VERSION. Every block names an analysis and every + analysis declares a detector and a version, both enforced here. That chain is + what docs/architecture.md asks for: without it, a model upgrade that silently + reuses old landmarks presents as "the tool got worse" with no event to attach it + to. + +The rest is the load/save round trip, the conditional write, and the footage +manifest that makes the frames the backend's to serve. +""" +import base64 +import hashlib +import json +import shutil +import struct +import subprocess +import tempfile +import zlib +from io import StringIO +from pathlib import Path +from unittest import skipUnless +from unittest.mock import Mock, patch + +from django.core.files.uploadedfile import SimpleUploadedFile +from django.core.management import call_command +from django.test import TestCase, override_settings + +from clips import blobs, extraction +from clips.models import Analysis, Block, Blob, Clip, Footage, Leaf, Project, Revision, Source + +BLOB_DIR = tempfile.mkdtemp(prefix="arthur-test-blobs-") + + +def key_for(descriptor: str) -> str: + return "sha256:" + hashlib.sha256(descriptor.encode("utf-8")).hexdigest() + + +def analysis_descriptor(version="1.0.1"): + # Canonical JSON, written the way arthur.domain.canon writes it: sorted keys, + # no spaces, integral doubles with no point. + return ('{"aspect":1,"detector":"mediapipe","frames":48,"fps":30,"scheme":1,' + f'"version":"{version}"}}') + + +def block_descriptor(analysis_key, role="geom", anchor_avg=2): + return (f'{{"analysis":"{analysis_key}","features":["mouth"],' + f'"layout":{{"frames":48,"scale":16384,"stride":16,"tracks":1,"type":"int16"}},' + f'"observation":null,"params":{{"anchor-avg":{anchor_avg}}},' + f'"role":"{role}","scheme":1,"tracks":["outer"]}}') + + +def png(width=4, height=3): + """The smallest valid PNG of a given size, written by hand. + + So that `blobs.png_size` and the ingest path are exercised without Pillow. The + one thing the backend needs from a PNG is its IHDR, and this is a PNG with one. + """ + def chunk(kind, payload): + return (struct.pack(">I", len(payload)) + kind + payload + + struct.pack(">I", zlib.crc32(kind + payload) & 0xFFFFFFFF)) + + ihdr = struct.pack(">IIBBBBB", width, height, 8, 2, 0, 0, 0) + raw = b"".join(b"\x00" + b"\x40\x40\x40" * width for _ in range(height)) + return (b"\x89PNG\r\n\x1a\n" + chunk(b"IHDR", ihdr) + + chunk(b"IDAT", zlib.compress(raw)) + chunk(b"IEND", b"")) + + +@override_settings(BLOB_ROOT=BLOB_DIR) +class BlobStoreTests(TestCase): + def test_the_same_bytes_are_stored_once(self): + a, size = blobs.write(b"the same bytes") + b, _ = blobs.write(b"the same bytes") + self.assertEqual(a, b) + self.assertEqual(size, 14) + self.assertEqual(blobs.read(a), b"the same bytes") + + def test_a_path_that_is_not_a_hash_is_refused(self): + # The blob route takes its digest from the URL, so this is the check that + # stops `/blob/../../etc/passwd` being a path at all. + with self.assertRaises(ValueError): + blobs.path_for("../../etc/passwd") + with self.assertRaises(ValueError): + blobs.path_for("deadbeef") + + def test_a_png_reports_its_own_size(self): + with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as fh: + fh.write(png(17, 5)) + self.assertEqual((17, 5), blobs.png_size(Path(fh.name))) + + def test_a_blob_is_served_immutable(self): + digest, size = blobs.write(b"bytes on the wire") + Blob.objects.create(digest=digest, size=size, media_type="application/octet-stream") + response = self.client.get(f"/blob/{digest}") + self.assertEqual(200, response.status_code) + self.assertIn("immutable", response["Cache-Control"]) + self.assertEqual(f'"{digest}"', response["ETag"]) + self.assertEqual(b"bytes on the wire", b"".join(response.streaming_content)) + + def test_an_unknown_blob_is_a_404_and_not_a_traceback(self): + self.assertEqual(404, self.client.get("/blob/" + "0" * 64).status_code) + self.assertEqual(404, self.client.get("/blob/nonsense").status_code) + + def test_a_blob_serves_byte_ranges(self): + # NOT AN OPTIMISATION. A