diff --git a/.claude/worktrees/tracing-layers b/.claude/worktrees/tracing-layers new file mode 160000 index 0000000..f4dd047 --- /dev/null +++ b/.claude/worktrees/tracing-layers @@ -0,0 +1 @@ +Subproject commit f4dd04764204506fc275180364d0366d693036f4 diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..dca80db --- /dev/null +++ b/.dockerignore @@ -0,0 +1,18 @@ +.git +.venv +**/node_modules +frontend/.shadow-cljs +frontend/out +frontend/.cpcache +frontend/test/browser/out +static/arthur/js +db.sqlite3 +var +frames +scratch +audio.wav +manifest.json +*.take +*.tflite +.claude +.venv* diff --git a/.gitignore b/.gitignore index caf400b..5f441bf 100644 --- a/.gitignore +++ b/.gitignore @@ -1,5 +1,40 @@ +# extract.sh's output. It is TIER 3 — immutable, large, and the backend's to serve +# once `manage.py ingest_bundle` has hashed it into var/blobs — so none of it +# belongs in the repo. `audio.wav` was tracked before step 9 because the synthetic +# take borrowed it for a clock; that copy now lives at static/arthur/audio.wav, +# which is an asset the project owns rather than an extraction that churns. frames/ +/audio.wav +/manifest.json +# local extracted takes for comparing source cadences +/scratch/ *.task *.take *.tflite + +# CLJS build +frontend/node_modules/ +# screenshots from the browser suite; regenerated by `npm run browser` +frontend/test/browser/out/ +frontend/.shadow-cljs/ +frontend/out/ +frontend/.cpcache/ +static/arthur/js/ + +# mise-managed venv for the Django half +.venv/ + +# the Django half's own state: the document database, the content-addressed blob +# store (tiers 2 and 3), and collectstatic's output +db.sqlite3 +db.sqlite3-shm +db.sqlite3-wal +/var/ + +# vim swap files +*.swp + +# Python bytecode +__pycache__/ +*.py[cod] diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..6e6d717 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,43 @@ +FROM node:20-bookworm AS frontend + +RUN apt-get update \ + && apt-get install -y --no-install-recommends ca-certificates curl tar \ + && rm -rf /var/lib/apt/lists/* \ + && curl -fsSL 'https://api.adoptium.net/v3/binary/latest/21/ga/linux/x64/jdk/hotspot/normal/eclipse' -o /tmp/jdk.tar.gz \ + && mkdir -p /opt/java \ + && tar -xzf /tmp/jdk.tar.gz -C /opt/java --strip-components=1 \ + && rm /tmp/jdk.tar.gz + +ENV JAVA_HOME=/opt/java +ENV PATH="/opt/java/bin:${PATH}" +WORKDIR /app/frontend +COPY frontend/package.json frontend/package-lock.json ./ +RUN npm ci +COPY frontend/ ./ +RUN npx shadow-cljs release app + +FROM python:3.12-slim-bookworm + +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + DJANGO_DEBUG=0 \ + DJANGO_DB_PATH=/data/db.sqlite3 \ + DJANGO_BLOB_ROOT=/data/blobs + +RUN apt-get update \ + && apt-get install -y --no-install-recommends ffmpeg \ + && rm -rf /var/lib/apt/lists/* \ + && useradd --create-home --shell /usr/sbin/nologin app + +WORKDIR /app +COPY requirements.txt ./ +RUN pip install --no-cache-dir -r requirements.txt +COPY . ./ +COPY --from=frontend /app/static/arthur/js/ ./static/arthur/js/ +RUN python manage.py collectstatic --noinput \ + && mkdir -p /data \ + && chown -R app:app /app /data + +USER app +EXPOSE 8000 +CMD ["sh", "-c", "python manage.py migrate --noinput && exec daphne --bind 0.0.0.0 --port 8000 server.asgi:application"] diff --git a/README.md b/README.md index 8f13ce7..7b70ce8 100644 --- a/README.md +++ b/README.md @@ -14,6 +14,62 @@ Animator Pro, where this started, but they are why the output looks right — modern conveniences belong in the workflow, not the output. See [docs/design.md](docs/design.md). +## ClojureScript port + +The active port plays the synthetic take, accepts video uploads, transcodes them +to an H.264 proxy and decodable stream plus audio and tracing stills, analyzes real footage +for mouth, eyes, brows and pixel-derived teeth, and saves the project with +reusable analysis data. The step 8 data model +represents persistent feature IDs, eye pairs and feature-level observation gaps; +its controls are still pending. See the [port plan](docs/port-plan.md). + +```sh +mise install # both halves +pip install -r requirements.txt +mise exec -- python manage.py migrate +./do start # Django + frontend watcher +``` + +In the app, drop a video on the media pool — it uploads, extracts and runs +detection — then **save**. Opening that project on another client reuses its saved landmarks and +mouth crops without detecting source frames again. The upload path derives its +footage response from database records; it does not create or consume a +`manifest.json` file. See [frontend/README.md](frontend/README.md) for details. +Anything under "## Run" and below describes the older JS prototype, which still +runs separately on port 8777. + +### Deploy to Fly.io + +The app uses a persistent Fly volume for its SQLite document database and +content-addressed footage blobs. Create the app once, set its Django secret, and +deploy from the repository root: + +```sh +fly apps create arthur --org personal +fly volumes create data --region iad --size 1 +fly secrets set DJANGO_SECRET_KEY="$(openssl rand -hex 32)" --app arthur +fly deploy --app arthur +``` + +The deployed app is at . The container builds the +ClojureScript frontend, collects static assets, and runs database migrations on +startup. + +### How it is stored + +Three tiers, cut by mutability and size — the full argument is in +[docs/architecture.md](docs/architecture.md): + +| Tier | What | Where | +| --- | --- | --- | +| 1 **authored** | the scene: nodes, channels, features, time maps | the database, as independently addressed leaves. Kilobytes | +| 2 **derived** | detected landmarks, raw mouth crops, and dense channel blocks | `var/blobs`, addressed by analysis and block inputs, including the detector version | +| 3 **source** | the uploaded video, H.264 proxy and elementary stream, tracing stills, and audio | the same blob store, by the hash of their bytes | + +Only tier 1 is the document. Tier 2 is a pure function of tiers 1 and 3, so a +saved project names its blocks rather than carrying them, and a knob change gives +a block a new name rather than overwriting an old one. + ## Run ```sh @@ -33,27 +89,35 @@ wasm, which is fetched from a CDN on first use. For real footage: -```sh -./extract.sh /path/to/clip.mov 12 # -> frames/*.png, audio.wav, manifest.json -``` +Upload it in the app. `./extract.sh` still writes the old PNG-sequence bundle and +`ingest_bundle` still registers it, but footage ingested that way has no decodable +stream and the loader will say so — the measured pixels come out of the video now. -then **Load frames**. MediaPipe's wasm is fetched from jsdelivr on first use; -`face_landmarker.task` is local. +MediaPipe's wasm and `face_landmarker.task` are both local; nothing in detection +touches the network. -Frames are pre-extracted rather than decoded in the page because browser video -seeking is approximate and `requestVideoFrameCallback` only delivers frames at -playback speed — neither gives a deterministic per-frame pass. +Detection reads the H.264 elementary stream with WebCodecs, one coded frame at a +time. The proxy has no B-frames, so decode order is frame order. Each decoded +frame reaches MediaPipe in VIDEO running mode at its footage timestamp. The +decoder and detector advance together, with a pause between frames so progress +can paint. Saved analyses reuse their stored crop pixels and measure them with +the same pauses. -`manifest.json` records the true extraction rate. The page reads it rather than -assuming, because a guessed fps desynchronises audio from picture — and sync is -the one thing this view exists to show. +This is what replaced the PNG sequence, which was 112MB for 7.6 seconds and would +be 1.1GB at the 900-frame limit. The proxy is 6MB, and the landmarks barely +notice: detected off decoded H.264 rather than off the PNGs, they moved at most +0.0033 of frame width. -**Exposure** decides how often the picture gets a new drawing: rip at 24 and -render `on 2s` for 12, `on 3s` for 8. The dense track and the audio are -untouched, so it is a dropdown rather than a re-rip, and the export emits keys -only on the grid instead of the same pose twice. Everything rides the same grid -— mouth, eyes, teeth, plate — because a head cutting on the odd frames while the -mouth cuts on the even ones reads as two performances laid over each other. +The server's footage manifest records the proxy's frame rate and frame count; +the page reads that rate because a guessed fps desynchronises audio from picture. +Choosing a lower picture rate happens after analysis. + +**Picture fps** decides how often the finished roto gets a new pose. Analyze all +source frames, then sample those frozen poses at 12, 24 or the source rate while +keeping the original duration and audio. **Exposure** can hold a drawing across +more than one picture slot. The tracing editor chooses source frames for cel +references separately. Shared timing is the useful default for mouth, eyes, +teeth and plate so their changes read as one performance. **Audio is the playback clock**: `frame = floor(audio.currentTime * fps)`. A slow render loop therefore drops frames instead of drifting, and ½x / ¼x work by @@ -314,13 +378,13 @@ the tool a person made by hand; everything else regenerates. They are not in the ## Two kinds of sparseness Sparseness has two unrelated causes, and conflating them was the original design -error here. **Aesthetic** sparseness is set by the extraction rate — pick 12fps and -you have already chosen your timing. **Labour** sparseness is a human drawing -each one, and it binds only on the plate. +error here. **Aesthetic** sparseness is chosen from the full analyzed source +track at rendering time. **Labour** sparseness is a human drawing each cel and +selecting which source frames to use as tracing references. -Aesthetic sparseness is the **exposure** control, not the extraction rate — -making it a render-time grid means auditioning 12 against 24 costs a dropdown -instead of a re-rip and a full re-detection. +Aesthetic sparseness is the **picture fps** control, with exposure available for +longer holds. Both happen after analysis, so auditioning 12 against 24 needs no +re-extraction or re-detection. So the mouth keeps **every** frame: it is traced, and therefore free. In limited animation lip sync is routinely the densest element, on 1s, while heads hold on @@ -373,3 +437,16 @@ performer→character calibration (currently identity, fitting the face oval to canvas); the override layer; anything on the Animator Pro side. The plate is a face-oval polygon per kept frame — it exists so the mouth has a face to read against, not to look good. + +In the port specifically: the parameter UI and scoped regeneration (the model is +built, the controls are not); automatic per-feature detection, so presence still +comes from the full-face mask plus a manifest annotation; multiplayer, for which +step 9 built the addressing and none of the socket; and in-browser extraction, so +`extract.sh` plus `manage.py ingest_bundle` is still how footage arrives. + +Two smaller things that are known and undecided. `measure/brows` takes no +`presence` where `measure/eyes` does, so an occluded brow affects the freeze mask +but not brow measurement, and occluded landmarks still enter contour smoothing — +asymmetric with the eyes, and it is not settled which way is right. And `open` +takes the most recently updated project and shows its first clip: there is no +project browser, and the runtime store holds one clip at a time. diff --git a/audio.wav b/audio.wav deleted file mode 100644 index f5793b0..0000000 Binary files a/audio.wav and /dev/null differ diff --git a/clips/__init__.py b/clips/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/admin.py b/clips/admin.py new file mode 100644 index 0000000..8b7add7 --- /dev/null +++ b/clips/admin.py @@ -0,0 +1,63 @@ +"""The admin, which is here for one reason: tier 1 is readable. + +docs/architecture.md's argument against a CRDT is partly this — "the canonical +document moves into an opaque blob, and every server-side thing that reads the +document needs it materialised back out". A leaf is transit-as-JSON in a +JSONField, so it is legible here, and that is a property worth being able to see. +""" +from django.contrib import admin + +from .models import Analysis, Block, Blob, Clip, Footage, FootageFrame, Leaf, Project, Revision + + +@admin.register(Project) +class ProjectAdmin(admin.ModelAdmin): + list_display = ("name", "id", "seq", "updated") + search_fields = ("name", "id") + + +@admin.register(Clip) +class ClipAdmin(admin.ModelAdmin): + list_display = ("cid", "project", "name") + list_filter = ("project",) + + +@admin.register(Leaf) +class LeafAdmin(admin.ModelAdmin): + list_display = ("path", "project", "version", "updated") + list_filter = ("project",) + search_fields = ("path",) + + +@admin.register(Revision) +class RevisionAdmin(admin.ModelAdmin): + list_display = ("project", "seq", "summary", "author", "created") + + +@admin.register(Footage) +class FootageAdmin(admin.ModelAdmin): + list_display = ("label", "source", "fps", "frames", "width", "height", "created") + + +@admin.register(FootageFrame) +class FootageFrameAdmin(admin.ModelAdmin): + list_display = ("footage", "index", "blob") + list_filter = ("footage",) + + +@admin.register(Analysis) +class AnalysisAdmin(admin.ModelAdmin): + list_display = ("key", "detector", "version", "footage", "created") + search_fields = ("key", "detector", "version") + + +@admin.register(Block) +class BlockAdmin(admin.ModelAdmin): + list_display = ("key", "role", "analysis", "data", "state", "created") + list_filter = ("role",) + search_fields = ("key",) + + +@admin.register(Blob) +class BlobAdmin(admin.ModelAdmin): + list_display = ("digest", "media_type", "size", "created") diff --git a/clips/apps.py b/clips/apps.py new file mode 100644 index 0000000..6b51579 --- /dev/null +++ b/clips/apps.py @@ -0,0 +1,14 @@ +from django.apps import AppConfig + + +class ClipsConfig(AppConfig): + """The one app. + + `clips` because the CLIP is the entity the whole tool is about and the one the + prototype had exactly one of and never named — `state` in `js/app.js` is a clip + with its analysis inlined and its palette global. Project, Footage, Analysis, + Block, Leaf and Revision all hang off it. + """ + + default_auto_field = "django.db.models.BigAutoField" + name = "clips" diff --git a/clips/blobs.py b/clips/blobs.py new file mode 100644 index 0000000..2216fcb --- /dev/null +++ b/clips/blobs.py @@ -0,0 +1,146 @@ +"""The content-addressed blob store: tiers 2 and 3 on disk. + +One store for both, and docs/architecture.md says why in a sentence: once tier 3 +is decoded by the app rather than by a shell script, frames and audio become "the +same kind of thing as tier 2 — a cache with a hash". So there is one place that +writes bytes, one that reads them, and one URL shape for both. + +TWO KINDS OF HASH, AND THEY ARE NOT THE SAME HASH. A blob is named by the sha256 +of its BYTES: that is what makes identical frames in two extractions one file. A +derived thing — an analysis artifact, a dense block — is named by a sha256 over +its INPUTS, which is what lets the client ask for the block the current settings +want before anything has computed it. So `Block.key` is an input hash and +`Block.data.digest` is a byte hash, and conflating them would break the half of +addressing that answers questions about work not yet done. +""" +import hashlib +import os +import tempfile +import zlib +from pathlib import Path + +from django.conf import settings + +CHUNK = 1 << 20 +CROP_MEDIA_TYPE = "application/zlib" + + +def digest_bytes(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +def digest_file(path: Path) -> str: + h = hashlib.sha256() + with open(path, "rb") as fh: + while chunk := fh.read(CHUNK): + h.update(chunk) + return h.hexdigest() + + +def path_for(digest: str) -> Path: + """Where a blob lives. + + Fanned out two levels, so that a take's worth of frames does not put a hundred + thousand entries in one directory — which is slow on every filesystem and + unusable on some. + """ + if len(digest) != 64 or any(c not in "0123456789abcdef" for c in digest): + raise ValueError(f"not a sha256: {digest!r}") + return Path(settings.BLOB_ROOT) / digest[:2] / digest[2:4] / digest + + +def write(data: bytes) -> tuple[str, int]: + """Store bytes, return (digest, size). Writing the same bytes twice is a + no-op, which is what content addressing is for.""" + digest = digest_bytes(data) + dest = path_for(digest) + if not dest.exists(): + dest.parent.mkdir(parents=True, exist_ok=True) + tmp = dest.with_suffix(".part") + with open(tmp, "wb") as fh: + fh.write(data) + os.replace(tmp, dest) + return digest, len(data) + + +def write_stream(chunks) -> tuple[str, int]: + """Store an uploaded file without reading the whole video into memory.""" + root = Path(settings.BLOB_ROOT) + root.mkdir(parents=True, exist_ok=True) + digest = hashlib.sha256() + size = 0 + with tempfile.NamedTemporaryFile(dir=root, prefix="upload-", delete=False) as out: + temporary = Path(out.name) + try: + for chunk in chunks: + digest.update(chunk) + size += len(chunk) + out.write(chunk) + except BaseException: + temporary.unlink(missing_ok=True) + raise + dest = path_for(digest.hexdigest()) + dest.parent.mkdir(parents=True, exist_ok=True) + if dest.exists(): + temporary.unlink() + else: + os.replace(temporary, dest) + return digest.hexdigest(), size + + +def write_compressed_stream(chunks) -> tuple[str, int]: + """Store a losslessly compressed stream; the digest names stored bytes.""" + compressor = zlib.compressobj() + + def compressed(): + for chunk in chunks: + if part := compressor.compress(chunk): + yield part + if part := compressor.flush(): + yield part + + return write_stream(compressed()) + + +def adopt(source: Path) -> tuple[str, int]: + """Store a file already on disk, by hard link where the filesystem allows it. + + 112MB of PNGs is a normal extraction and copying them into a second place in + the tree for no reason is not. A hard link is exact — the blob is immutable, so + two names for one inode is the whole of what is wanted — and a copy is the + fallback when `extract.sh` wrote to another volume. + """ + digest = digest_file(source) + dest = path_for(digest) + size = source.stat().st_size + if not dest.exists(): + dest.parent.mkdir(parents=True, exist_ok=True) + try: + os.link(source, dest) + except OSError: + tmp = dest.with_suffix(".part") + with open(source, "rb") as src, open(tmp, "wb") as out: + while chunk := src.read(CHUNK): + out.write(chunk) + os.replace(tmp, dest) + return digest, size + + +def read(digest: str) -> bytes: + with open(path_for(digest), "rb") as fh: + return fh.read() + + +def png_size(path: Path) -> tuple[int, int]: + """A PNG's dimensions, out of its IHDR. + + Twenty-four bytes rather than a dependency. The footage's width and height are + manifest data — docs/architecture.md's entity model puts them there — and + Pillow to read two integers out of a header that has held them in the same + place since 1996 is not a trade worth making. + """ + with open(path, "rb") as fh: + head = fh.read(24) + if head[:8] != b"\x89PNG\r\n\x1a\n" or head[12:16] != b"IHDR": + raise ValueError(f"{path} is not a PNG") + return int.from_bytes(head[16:20], "big"), int.from_bytes(head[20:24], "big") diff --git a/clips/consumers.py b/clips/consumers.py new file mode 100644 index 0000000..98de12f --- /dev/null +++ b/clips/consumers.py @@ -0,0 +1,87 @@ +"""One socket per open project, and it is tl's, nearly line for line. + +Two things ride it. DELTAS, which the server sends after a write commits — the +socket is read-only for the document, and a dropped socket cannot lose a write. +PRESENCE, which peers gossip between themselves: all the server does is hand out +a connection id and stamp the sender's identity onto every message, so nobody can +post as somebody else. +""" +import json +import uuid + +from asgiref.sync import async_to_sync +from channels.generic.websocket import AsyncWebsocketConsumer +from channels.layers import get_channel_layer + +# Who is connected, per project: {group: {cid: presence}}. A cache of what has +# already been relayed, so a joiner gets the room in one message. Process-local, +# like the in-memory channel layer this runs on. +ROOMS = {} + + +def group(project_id): + return f"project_{project_id}" + + +def broadcast(project_id, delta, kind="delta"): + """Send a committed write to everyone in the project's room. `access` says + only that who may write has changed, and each client asks for itself.""" + async_to_sync(get_channel_layer().group_send)( + group(project_id), {"type": "project.delta", "delta": {"kind": kind, **delta}}, + ) + + +class ProjectConsumer(AsyncWebsocketConsumer): + RELAYED = ("state",) + + @property + def room(self): + return ROOMS.setdefault(self.group, {}) + + async def connect(self): + self.group = group(self.scope["url_route"]["kwargs"]["project_id"]) + self.cid = uuid.uuid4().hex[:12] + user = self.scope.get("user") + self.username = user.get_username() if user and user.is_authenticated else None + await self.channel_layer.group_add(self.group, self.channel_name) + await self.accept() + + me = {"cid": self.cid, "user": self.username} + others = list(self.room.values()) + self.room[self.cid] = me + await self.send(text_data=json.dumps({"kind": "welcome", **me})) + await self.send(text_data=json.dumps({"kind": "roster", "peers": others})) + await self._relay({"kind": "join"}) + + async def disconnect(self, code): + if hasattr(self, "cid"): + self.room.pop(self.cid, None) + if not self.room: + ROOMS.pop(self.group, None) + await self._relay({"kind": "leave"}) + await self.channel_layer.group_discard(self.group, self.channel_name) + + async def receive(self, text_data=None, bytes_data=None): + try: + msg = json.loads(text_data or "{}") + except ValueError: + return + if not isinstance(msg, dict) or msg.get("kind") not in self.RELAYED: + return + if self.cid in self.room: + self.room[self.cid].update( + {k: v for k, v in msg.items() if k not in ("kind", "cid", "user")} + ) + await self._relay(msg) + + async def _relay(self, msg): + await self.channel_layer.group_send( + self.group, + {"type": "peer.msg", "msg": {**msg, "cid": self.cid, "user": self.username}}, + ) + + async def peer_msg(self, event): + await self.send(text_data=json.dumps(event["msg"])) + + async def project_delta(self, event): + await self.send(text_data=json.dumps(event["delta"])) diff --git a/clips/extraction.py b/clips/extraction.py new file mode 100644 index 0000000..4c4c5d0 --- /dev/null +++ b/clips/extraction.py @@ -0,0 +1,501 @@ +"""Upload a video once, then turn it into the two things the app actually reads. + +WHAT CHANGED AND WHY. This used to decode one PNG per source frame and store every +one of them. A 7.6-second 1440x1920 take is 112MB that way, and the 900-frame limit +is 1.1GB — for pixels whose only consumer was a canvas that MediaPipe then read +once. The page now detects from the video itself (see `frontend/src/arthur/flow/ +ingest.cljs`), so this produces: + + THE PROXY. One browser-safe H.264/yuv420p MP4, CFR, `+faststart`. The same take + is 6MB. This is the analysis source, and it is re-encoded RATHER THAN KEPT AS + UPLOADED even when the upload is already H.264, for two reasons that are both + about not guessing: an iPhone's HEVC is not decodable in every browser, and the + footage's identity is the digest of this file — one produced by one ffmpeg + invocation, not one that depends on which branch the source happened to take. + + THE TRACING STILLS. One JPEG per frame, long edge capped, for the tracing editor + to draw over. Reference images; nothing measures them. They are not in the + footage digest — see `models.Footage`. + +The proxy is probed after it is written rather than before. `width`, `height` and +`frames` are properties of the file the browser will decode, and taking them from +the source instead is how a scaler or a dropped frame becomes a silent one-frame +offset between the landmarks and the audio. +""" + +import hashlib +import json +import subprocess +import tempfile +import threading +import time +from fractions import Fraction +from pathlib import Path + +from django.db import close_old_connections, transaction + +from . import blobs +from .models import Blob, Extraction, Footage, FootageFrame + +_active = set() +_lock = threading.Lock() +TIMEOUT = 3600 + +# Visually lossless enough that landmarks do not move: measured against the same +# frames as PNGs, IMAGE-mode landmarks shifted at most 0.0033 of frame width. +PROXY_CRF = "18" +# The long edge of a tracing still. The proxy keeps full resolution because the +# detector reads it; a still only has to be good enough to draw a cel over. +TRACING_EDGE = 1280 +TRACING_QUALITY = "4" + + +def _command(args): + result = subprocess.run(args, capture_output=True, text=True, timeout=TIMEOUT) + if result.returncode: + raise ValueError((result.stderr or result.stdout or "media tool failed")[-1200:]) + return result.stdout + + +def _run_with_progress(job, args, root, name, total, span): + """Run one ffmpeg and publish its live frame count as `span` of the job. + + ffmpeg's `-progress` file is the only honest source for this: parsing its + stderr means parsing a format that is explicitly not an interface, and a + spinner that is not attached to frames is a spinner that lies on a long take. + """ + progress_path = root / f"{name}.progress" + log_path = root / f"{name}.log" + first, last = span + args = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-stats_period", "0.25", "-progress", str(progress_path)] + args + with open(log_path, "wb") as log: + proc = subprocess.Popen(args, stdout=log, stderr=subprocess.STDOUT) + deadline = time.monotonic() + TIMEOUT + try: + while proc.poll() is None: + if time.monotonic() >= deadline: + raise TimeoutError(f"{name} timed out") + if progress_path.exists(): + lines = progress_path.read_text(errors="replace").splitlines() + count = next((int(line[6:].strip()) for line in reversed(lines) + if line.startswith("frame=") and + line[6:].strip().isdigit()), 0) + if count and total: + reached = first + int((last - first) * min(1.0, count / total)) + if reached > job.progress: + job.progress = reached + job.save(update_fields=["progress", "updated"]) + time.sleep(0.2) + finally: + if proc.poll() is None: + proc.kill() + proc.wait() + if proc.returncode: + raise ValueError(log_path.read_text(errors="replace")[-1200:] or f"{name} failed") + + +def _encode_proxy(job, source_path, proxy_path, facts, root): + """The uploaded video -> one H.264 file every browser can decode and seek.""" + total = facts.get("reported_frames") or round(facts["duration"] * facts["fps"]) + _run_with_progress( + job, + ["-i", str(source_path), "-an", + # Constant frame rate at the rate `probe` chose. This RESAMPLES rather + # than asserts: the upload is allowed to be variable, and this is the + # step that makes the thing the page measures not be. + "-fps_mode", "cfr", "-r", facts.get("rate") or str(facts["fps"]), + "-c:v", "libx264", "-preset", "veryfast", "-crf", PROXY_CRF, + # NO B-FRAMES, AND THIS IS THE LOAD-BEARING FLAG. It is what makes + # decode order presentation order, so the page can treat access unit k + # of the elementary stream as frame k without demuxing a container or + # consulting a timestamp. With them x264 has a + # two-frame reordering delay, ffmpeg compensates by writing an edit list + # (`elst` media_time 1024 at timebase 1/15360 — exactly two frames), and + # the browser then lives on two timelines at once: `currentTime` obeys the + # edit list and the `mediaTime` reported by requestVideoFrameCallback does + # not. Seek to frame 0 and the browser correctly hands back a frame whose + # mediaTime says 2. Software decoding hides it; hardware decoding does + # not, which is the worst possible way for it to be wrong. Without + # B-frames DTS equals PTS, no edit list is written, and the two timelines + # are the same one. It also makes decode order presentation order, should + # this ever be fed to a WebCodecs VideoDecoder. + "-bf", "0", + # yuv420p and an even frame size are what makes this playable everywhere + # rather than only in the browser that happened to be tested. + "-pix_fmt", "yuv420p", "-vf", "scale=trunc(iw/2)*2:trunc(ih/2)*2", + "-movflags", "+faststart", str(proxy_path)], + root, "proxy", total, (0, 55)) + + +def _elementary_stream(proxy_path, out_path): + """The proxy's video, unwrapped into a raw Annex-B H.264 stream. + + A STREAM COPY, not a second encode: the same coded frames as the MP4, with + the container's length-prefixed NAL units rewritten as start-code-delimited + ones. It costs a file read and nothing else. + + This exists because the page decodes with WebCodecs, and `VideoDecoder` takes + demuxed chunks rather than a container. Handing it Annex-B means the client + needs no demuxer: NAL start codes are findable in a loop, and because the + proxy is encoded with no B-frames, decode order is presentation order — so + access unit k IS frame k, with no container timing to consult and no clock to + reconcile. That is the whole reason this file is worth the bytes it costs. + """ + _command(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-i", str(proxy_path), "-an", "-c:v", "copy", + "-bsf:v", "h264_mp4toannexb", "-f", "h264", str(out_path)]) + + +def _extract_stills(job, proxy_path, frames_dir, frames, root): + """The proxy -> one tracing JPEG per frame, long edge capped.""" + _run_with_progress( + job, + ["-i", str(proxy_path), "-fps_mode", "passthrough", + "-vf", f"scale='if(gt(iw,ih),min({TRACING_EDGE},iw),-2)':" + f"'if(gt(iw,ih),-2,min({TRACING_EDGE},ih))'", + "-q:v", TRACING_QUALITY, str(frames_dir / "%04d.jpg")], + root, "stills", frames, (55, 85)) + + +MAX_RATE = 120 # a capture rate; past this the container is describing something else +# How many packet timestamps `_measured_rate` reads, and the fewest intervals it +# will draw a conclusion from. 300 is a flat cost on a long take and still a +# wide enough sample for a median; below 8 intervals there is not enough of a +# stream to outvote one odd timestamp, so the metadata is left to speak. +RATE_SAMPLE = 300 +RATE_MINIMUM = 8 +# How far a declared rate may sit from the measured one and still be taken as +# what the stream is: 2% covers 30 against 30000/1001 and nothing like 120 +# against 30. +RATE_TOLERANCE = 0.02 + + +def probe_image(path): + """An uploaded still's pixel size as (width, height), refusing anything that + is not one picture.""" + data = json.loads(_command(["ffprobe", "-v", "error", "-show_streams", + "-of", "json", str(path)])) + video = [s for s in data.get("streams", []) if s.get("codec_type") == "video"] + if len(video) != 1 or not (video[0].get("width") and video[0].get("height")): + raise ValueError("the uploaded file is not an image") + return int(video[0]["width"]), int(video[0]["height"]) + + +def probe_audio(path): + """The length of an uploaded sound in seconds, refusing a file with no audio.""" + data = json.loads(_command(["ffprobe", "-v", "error", "-show_streams", + "-show_format", "-of", "json", str(path)])) + if not any(s.get("codec_type") == "audio" for s in data.get("streams", [])): + raise ValueError("the uploaded file has no audio stream") + duration = float(data.get("format", {}).get("duration") or 0) + if duration <= 0: + raise ValueError("the sound's length is unknown") + return duration + + +def _measured_rate(path): + """The rate the stream's own packet timestamps imply, or None. + + THE CONTAINER'S SUMMARY OF ITSELF IS NOT EVIDENCE, and this is the function + that goes and looks. An iPhone's `r_frame_rate` is 120 on footage whose + timestamps are 1/30s apart, which is the difference between 323 frames and + 1293 — four times the encode, four times the tracing stills, four times the + blobs, for 970 frames that are copies of their neighbours. + + It reads TIMESTAMPS, not frames: `-show_entries packet=pts_time` demuxes + without decoding, so this costs a file read and no pixels. The times are + SORTED before differencing because a stream with B-frames arrives in decode + order — an HEVC clip's first packets come out 0, 0.133, 0.067, 0.033 — and + differencing that order measures the reordering rather than the rate. + + THE MEDIAN INTERVAL, which is what makes this safe on genuinely variable + input. It answers "how far apart are two frames normally", so a take held on + one frame for a second still reports the rate of the parts that move, and + choosing it keeps every distinct frame — the property `probe` used to reach + for by taking the nominal rate. Only the last few intervals of the sample are + unreliable (a frame whose turn comes after the window is missing from it), and + a median does not care. + + Returning None is the honest answer for a clip too short to sample, and this + also swallows a probe that fails outright: the rate the metadata declares is + the documented fallback, so an optimisation must not be able to refuse an + upload that would otherwise have been accepted. + """ + try: + text = _command(["ffprobe", "-v", "error", "-select_streams", "v:0", + "-show_entries", "packet=pts_time", "-of", "json", + "-read_intervals", f"%+#{RATE_SAMPLE}", str(path)]) + packets = json.loads(text).get("packets") or [] + times = sorted(float(packet["pts_time"]) for packet in packets + if (packet.get("pts_time") or "N/A") != "N/A") + except (ValueError, OSError): + return None + intervals = sorted(b - a for a, b in zip(times, times[1:]) if b > a) + if len(intervals) < RATE_MINIMUM: + return None + median = intervals[len(intervals) // 2] + return 1.0 / median if median > 0 else None + + +def _choose_rate(nominal, average, measured): + """The rate to resample onto, as an exact Fraction. + + A DECLARED RATE IS PREFERRED WHEN IT AGREES WITH THE TIMESTAMPS, because it is + the exact rational the stream was authored at — 30000/1001 is not a float, and + `limit_denominator` on a measured 29.97 is a guess at a number the container + already states. So the measured rate is used to CHOOSE between what the + container declares, and only stands in itself when neither declaration + describes the stream. + """ + candidates = [rate for rate in (nominal, average) if 0 < rate <= MAX_RATE] + if measured: + agreeing = [rate for rate in candidates + if abs(float(rate) - measured) <= RATE_TOLERANCE * measured] + if agreeing: + return min(agreeing, key=lambda rate: abs(float(rate) - measured)) + from_timestamps = Fraction(measured).limit_denominator(1001) + if 0 < from_timestamps <= MAX_RATE: + return from_timestamps + # Nothing to go on but the metadata, and nominal first keeps the rate that + # drops no distinct frame. An unusable pair falls through to the refusal + # below, which names the rate the file claimed rather than one of these. + return nominal if 0 < nominal <= MAX_RATE else average + + +def probe(path): + """What the upload is, as far as choosing a proxy rate goes. + + IT NO LONGER REFUSES VARIABLE-FRAME-RATE INPUT, and the reason is the proxy. + That refusal was written when the page measured the source's own frames, where + a wandering frame duration really does break `frame = floor(t * fps)`. Nothing + measures the source now: ffmpeg resamples it onto a constant rate, and the + proxy — constant by construction, and re-probed after it is written — is the + only timeline anything downstream sees. + + Keeping the check would have been worse than useless, because the thing it + tested is not reliable. Ordinary iPhone footage, shot straight from the camera + app, reports `avg_frame_rate` 8670/299 and `nb_frames` 289 on a stream whose + decoded timestamps are 280 frames exactly 1/30s apart. The container's summary + of itself disagreed with the container's own contents, so the guard rejected + CFR video for being variable. + + THE RATE IS MEASURED AND THE DECLARATIONS ARE VOTED ON, which is the same + distrust applied to the one number that still comes from here. This used to + take `r_frame_rate` outright — the rate every timestamp in the stream can be + expressed at, and so the rate that keeps every distinct source frame. The + trouble is that it is not a claim about frames at all: the file above declares + 120 and holds 30, and resampling it up cost four times the encode, four times + the tracing stills and four times the blobs for 970 duplicated frames. So + `_measured_rate` reads the timestamps, `_choose_rate` keeps whichever declared + rate they bear out, and the nominal rate is believed when it is true rather + than because it is nominal. + + Duration is preserved either way — ffmpeg's CFR conversion is driven by + timestamps, so the audio stays in sync at any rate — and the median interval + keeps the no-distinct-frame-dropped property that taking the nominal rate was + reaching for. See `_measured_rate`. + """ + data = json.loads(_command(["ffprobe", "-v", "error", "-show_streams", + "-show_format", "-of", "json", str(path)])) + video = next((s for s in data.get("streams", []) if s.get("codec_type") == "video"), None) + if not video: + raise ValueError("the uploaded file has no video stream") + nominal = Fraction(video.get("r_frame_rate") or "0") + average = Fraction(video.get("avg_frame_rate") or "0") + if nominal <= 0 and average <= 0: + raise ValueError("the video's frame rate is unknown") + measured = _measured_rate(path) + rate = _choose_rate(nominal, average, measured) + if not 0 < rate <= MAX_RATE: + raise ValueError(f"the video reports a frame rate of {float(rate):g}, which is " + "not a rate footage can be measured at") + duration = float(data.get("format", {}).get("duration") or 0) + # if duration > 0 and duration * float(rate) > 901: + # raise ValueError("video is longer than the 900-frame footage limit") + frames = video.get("nb_frames") + return {"fps": float(rate), + # The exact rate, for ffmpeg. 30000/1001 is not a float, and handing + # `-r` a rounded one is how a long take drifts out of sync. + "rate": f"{rate.numerator}/{rate.denominator}", + "nominal_fps": float(nominal), "average_fps": float(average), + # What the timestamps said, and null when there were too few to ask. + # Recorded because it is the input to a decision this file used not to + # make, and the one number that explains a chosen rate matching + # neither declaration. + "measured_fps": measured, + "width": int(video["width"]), "height": int(video["height"]), + "duration": duration, + # KEPT, AND NO LONGER TRUSTED AS A COUNT. See the docstring: this is + # the container's claim about itself, it is wrong on ordinary phone + # footage, and `run` checks the proxy's DURATION instead. + "reported_frames": int(frames) if frames and frames.isdigit() else None, + "has_audio": any(s.get("codec_type") == "audio" for s in data.get("streams", [])), + "vfr": nominal != average} + + +def _refuse_a_shifted_timeline(path): + """The proxy must put frame `i` at `i / fps` on BOTH of the browser's clocks. + + Asserted rather than assumed, because the failure is silent and the symptom is + unrecognisable. An encoder delay makes ffmpeg write an edit list, `currentTime` + then obeys it while `requestVideoFrameCallback`'s `mediaTime` does not, and the + page's frame walk is uniformly off by the delay — on hardware decoding only. It + cost two wrong diagnoses to find, so it does not get to come back silently if + somebody changes an encoder flag. + """ + data = json.loads(_command(["ffprobe", "-v", "error", "-select_streams", "v:0", + "-show_streams", "-of", "json", str(path)])) + stream = data["streams"][0] + if int(stream.get("has_b_frames") or 0): + raise ValueError( + "the proxy was encoded with B-frames, whose reordering delay makes the " + "browser's seek clock and its frame-timestamp clock disagree") + if float(stream.get("start_time") or 0) != 0: + raise ValueError(f"the proxy starts at {stream['start_time']}s rather than 0") + + +def count_frames(path): + """How many frames a file really holds, counted rather than reported. + + `nb_frames` is a container's claim. This is the decoder's answer, and it is + what the page will get when it walks the proxy — so a disagreement between the + two has to be settled before the count reaches a manifest, not after it has + become a one-frame audio offset nobody can find. + """ + text = _command(["ffprobe", "-v", "error", "-select_streams", "v:0", + "-count_frames", "-show_entries", "stream=nb_read_frames", + "-of", "default=nokey=1:noprint_wrappers=1", str(path)]) + counted = text.strip() + if not counted.isdigit(): + raise ValueError("could not count the proxy's frames") + return int(counted) + + +def extraction_key(source, settings): + # Scheme 3: the extraction now also produces the elementary stream the page + # decodes, so a job run under scheme 2 did not make everything this one does. + text = json.dumps({"scheme": 3, "source": source.blob_id, "settings": settings}, + sort_keys=True, separators=(",", ":")) + return "sha256:" + hashlib.sha256(text.encode()).hexdigest() + + +def _register(job, proxy_path, stream_path, stills, audio_path, facts): + proxy_digest, proxy_size = blobs.adopt(proxy_path) + stream_digest, stream_size = blobs.adopt(stream_path) + audio_digest, audio_size = blobs.adopt(audio_path) + still_blobs = [(index, *blobs.adopt(path)) for index, path in enumerate(stills)] + width, height, fps, frames = facts["width"], facts["height"], facts["fps"], facts["frames"] + + # The footage's own identity: the bytes the page will measure, the audio it + # will clock against, and the rate that ties them together. Scheme 2 — scheme + # 1 hashed a PNG per frame, and those footages name pixels this no longer has. + h = hashlib.sha256() + h.update(f"arthur-footage-2/{fps}/{frames}/{width}x{height}\n".encode()) + h.update(proxy_digest.encode()) + h.update(audio_digest.encode()) + + with transaction.atomic(): + proxy_blob, _ = Blob.objects.get_or_create( + digest=proxy_digest, defaults={"size": proxy_size, "media_type": "video/mp4"}) + stream_blob, _ = Blob.objects.get_or_create( + digest=stream_digest, defaults={"size": stream_size, "media_type": "video/h264"}) + audio_blob, _ = Blob.objects.get_or_create( + digest=audio_digest, defaults={"size": audio_size, "media_type": "audio/wav"}) + footage, created = Footage.objects.get_or_create( + digest=h.hexdigest(), + defaults={"label": job.source.filename[:200], "source": job.source.filename[:200], + "fps": fps, "frames": frames, "width": width, "height": height, + "audio": audio_blob, "video": proxy_blob, "stream": stream_blob}) + if not created and not footage.stream_id: + # The same footage by identity, extracted before the elementary + # stream existed. Its digest is over the proxy and the audio, which + # have not changed — so this is the same footage gaining a file it + # was always entitled to, not a different one. + footage.stream = stream_blob + if not footage.video_id: + footage.video = proxy_blob + footage.save(update_fields=["stream", "video"]) + if created: + rows = [] + for index, digest, size in still_blobs: + blob, _ = Blob.objects.get_or_create( + digest=digest, defaults={"size": size, "media_type": "image/jpeg"}) + rows.append(FootageFrame(footage=footage, index=index, blob=blob)) + FootageFrame.objects.bulk_create(rows) + return footage + + +def run(key): + close_old_connections() + try: + job = Extraction.objects.select_related("source", "source__blob").get(key=key) + job.state, job.progress, job.error = "running", 0, "" + job.save(update_fields=["state", "progress", "error", "updated"]) + facts = job.source.probe + source_path = blobs.path_for(job.source.blob_id) + with tempfile.TemporaryDirectory(prefix="arthur-extract-") as directory: + root = Path(directory) + proxy_path = root / "proxy.mp4" + _encode_proxy(job, source_path, proxy_path, facts, root) + + # Everything downstream describes the PROXY, not the upload. + proxy_facts = probe(proxy_path) + _refuse_a_shifted_timeline(proxy_path) + frames = count_frames(proxy_path) + #if not 1 <= frames <= 900: + # raise ValueError(f"the proxy holds {frames} frames; the limit is 1–900") + # CHECKED AS A DURATION, not as a frame count. The page's clock is + # `frame = floor(audio.currentTime * fps)`, so what must not drift is + # how long the picture lasts against how long the audio lasts — and + # the source's own frame count is a number this has already caught + # lying. A resample to a constant rate legitimately changes the count + # and must not change the duration. + drift = abs(frames / proxy_facts["fps"] - facts["duration"]) + if facts["duration"] > 0 and drift > 0.5: + raise ValueError( + f"the proxy runs {frames / proxy_facts['fps']:.2f}s and the upload " + f"runs {facts['duration']:.2f}s; refusing footage whose picture and " + "audio would drift") + proxy_facts["frames"] = frames + + stream_path = root / "proxy.h264" + _elementary_stream(proxy_path, stream_path) + + frames_dir = root / "stills" + frames_dir.mkdir() + _extract_stills(job, proxy_path, frames_dir, frames, root) + stills = sorted(frames_dir.glob("*.jpg")) + if len(stills) != frames: + raise ValueError(f"wrote {len(stills)} tracing stills for {frames} frames") + + job.progress = 85 + job.save(update_fields=["progress", "updated"]) + audio_path = root / "audio.wav" + if facts["has_audio"]: + _command(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-i", str(source_path), "-vn", "-ac", "1", "-ar", "44100", + str(audio_path)]) + else: + _command(["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", + "-f", "lavfi", "-i", "anullsrc=r=44100:cl=mono", + "-t", str(frames / proxy_facts["fps"]), "-c:a", "pcm_s16le", + str(audio_path)]) + footage = _register(job, proxy_path, stream_path, stills, audio_path, proxy_facts) + job.footage, job.state, job.progress = footage, "done", 100 + job.save(update_fields=["footage", "state", "progress", "updated"]) + except Exception as exc: + Extraction.objects.filter(key=key).update(state="failed", error=str(exc)[:2000]) + finally: + with _lock: + _active.discard(key) + close_old_connections() + + +def enqueue(key): + with _lock: + if key in _active: + return + _active.add(key) + threading.Thread(target=run, args=(key,), daemon=True, + name=f"arthur-extract-{key[7:15]}").start() diff --git a/clips/management/__init__.py b/clips/management/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/management/commands/__init__.py b/clips/management/commands/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/management/commands/compress_crop_blocks.py b/clips/management/commands/compress_crop_blocks.py new file mode 100644 index 0000000..6367b81 --- /dev/null +++ b/clips/management/commands/compress_crop_blocks.py @@ -0,0 +1,57 @@ +"""Compress existing raw mouth crop blocks without changing their public bytes.""" + +import hashlib +import zlib + +from django.core.management.base import BaseCommand, CommandError +from django.db import transaction +from django.db.models.deletion import ProtectedError + +from clips import blobs +from clips.models import Blob, Block + + +class Command(BaseCommand): + help = "Compress existing source/crops blobs and remove unreferenced raw copies" + + def handle(self, *args, **options): + converted = 0 + before = after = 0 + for block in Block.objects.filter(role="source/crops").select_related("data"): + old = block.data + if old.media_type == blobs.CROP_MEDIA_TYPE: + continue + old_digest = old.digest + with open(blobs.path_for(old_digest), "rb") as source: + digest, size = blobs.write_compressed_stream( + iter(lambda: source.read(blobs.CHUNK), b"") + ) + check = hashlib.sha256() + decompressor = zlib.decompressobj() + with open(blobs.path_for(digest), "rb") as compressed: + while chunk := compressed.read(blobs.CHUNK): + check.update(decompressor.decompress(chunk)) + check.update(decompressor.flush()) + if not decompressor.eof or check.hexdigest() != old_digest: + raise CommandError(f"crop compression failed verification: {block.key}") + with transaction.atomic(): + new, _ = Blob.objects.get_or_create( + digest=digest, + defaults={"size": size, "media_type": blobs.CROP_MEDIA_TYPE}, + ) + changed = Block.objects.filter(key=block.key, data=old).update(data=new) + if not changed: + continue + converted += 1 + before += old.size + after += size + if old_digest != new.digest: + try: + old.delete() + except ProtectedError: + pass + else: + blobs.path_for(old_digest).unlink(missing_ok=True) + self.stdout.write( + f"Compressed {converted} crop blocks: {before:,} -> {after:,} bytes" + ) diff --git a/clips/management/commands/ingest_bundle.py b/clips/management/commands/ingest_bundle.py new file mode 100644 index 0000000..89ca5b5 --- /dev/null +++ b/clips/management/commands/ingest_bundle.py @@ -0,0 +1,120 @@ +"""Register an extracted bundle as tier 3. + + python manage.py ingest_bundle # ./manifest.json + python manage.py ingest_bundle scratch/my-take # that bundle + +WHAT THIS REPLACES. Until step 9 the page fetched `/manifest.json` and then built +`frames/0001.png` itself, with shadow-cljs's `:dev-http` serving the repo root. So +the frame layout was a shared secret between a shell script and a ClojureScript +namespace, and "where the frames are" was answered by a directory listing. + +Now the server names every frame, and the client asks it. The frames go into the +content-addressed blob store — by hard link, so 112MB of PNGs is not copied — and +the manifest the client receives carries a URL per frame. That is the whole of what +makes the frames the backend's to serve, and it is what the in-browser wasm-ffmpeg +extraction docs/architecture.md describes will upload INTO, without the client +learning anything new when it arrives: the same blobs, the same manifest, a +different producer. + +`extract.sh` still does the decoding. It is out of step 9's scope, it works, and it +is the only part of this that needs a terminal. +""" +import json +from pathlib import Path + +from django.core.management.base import BaseCommand, CommandError +from django.db import transaction + +from clips import blobs +from clips.models import Blob, Footage, FootageFrame + + +class Command(BaseCommand): + help = "Register an extracted frames+audio+manifest bundle as footage." + + def add_arguments(self, parser): + parser.add_argument( + "bundle", nargs="?", default=".", + help="a directory holding manifest.json, or the manifest itself", + ) + parser.add_argument("--label", default="", help="what to call it in the UI") + + def handle(self, *args, **options): + manifest_path = Path(options["bundle"]) + if manifest_path.is_dir(): + manifest_path = manifest_path / "manifest.json" + if not manifest_path.exists(): + raise CommandError(f"{manifest_path} does not exist — run ./extract.sh first") + + manifest = json.loads(manifest_path.read_text()) + root = manifest_path.parent + frames_dir = root / manifest["dir"] + audio_path = root / manifest["audio"] + count = int(manifest["frames"]) + + pngs = sorted(frames_dir.glob("*.png")) + if len(pngs) != count: + raise CommandError( + f"the manifest says {count} frames and {frames_dir} holds {len(pngs)}; " + "refusing an inaccurate footage" + ) + if not audio_path.exists(): + raise CommandError(f"{audio_path} does not exist") + + width, height = blobs.png_size(pngs[0]) + + self.stdout.write(f"hashing {len(pngs)} frames…") + frame_blobs = [] + for i, png in enumerate(pngs): + digest, size = blobs.adopt(png) + frame_blobs.append((i, digest, size)) + if (i + 1) % 25 == 0 or i + 1 == len(pngs): + self.stdout.write(f" {i + 1}/{len(pngs)}") + + audio_digest, audio_size = blobs.adopt(audio_path) + + # The footage's own identity: every frame in order, plus the audio and the + # rate. Two extractions of one clip at one rate are one footage, so an + # analysis over it is reusable across both. + import hashlib + + h = hashlib.sha256() + h.update(f"arthur-footage-1/{manifest['fps']}/{count}/{width}x{height}\n".encode()) + for _, digest, _ in frame_blobs: + h.update(digest.encode()) + h.update(audio_digest.encode()) + footage_digest = h.hexdigest() + + if existing := Footage.objects.filter(digest=footage_digest).first(): + self.stdout.write(self.style.SUCCESS(f"already ingested: {existing.id}")) + return + + with transaction.atomic(): + audio_blob, _ = Blob.objects.get_or_create( + digest=audio_digest, + defaults={"size": audio_size, "media_type": "audio/wav"}, + ) + footage = Footage.objects.create( + digest=footage_digest, + label=options["label"] or manifest.get("source") or frames_dir.name, + source=manifest.get("source") or "", + fps=float(manifest["fps"]), + frames=count, + width=width, + height=height, + audio=audio_blob, + feature_absence=manifest.get("feature-absence") or {}, + ) + rows = [] + for index, digest, size in frame_blobs: + blob, _ = Blob.objects.get_or_create( + digest=digest, defaults={"size": size, "media_type": "image/png"} + ) + rows.append(FootageFrame(footage=footage, index=index, blob=blob)) + FootageFrame.objects.bulk_create(rows) + + self.stdout.write( + self.style.SUCCESS( + f"{count} frames at {manifest['fps']}fps, {width}x{height} -> footage {footage.id}" + ) + ) diff --git a/clips/migrations/0001_initial.py b/clips/migrations/0001_initial.py new file mode 100644 index 0000000..15e07ba --- /dev/null +++ b/clips/migrations/0001_initial.py @@ -0,0 +1,149 @@ +# Generated by Django 5.2.17 on 2026-09-28 04:44 + +import django.db.models.deletion +import uuid +from django.db import migrations, models + + +class Migration(migrations.Migration): + + initial = True + + dependencies = [ + ] + + operations = [ + migrations.CreateModel( + name='Blob', + fields=[ + ('digest', models.CharField(max_length=64, primary_key=True, serialize=False)), + ('media_type', models.CharField(default='application/octet-stream', max_length=100)), + ('size', models.BigIntegerField()), + ('created', models.DateTimeField(auto_now_add=True)), + ], + ), + migrations.CreateModel( + name='Project', + fields=[ + ('id', models.UUIDField(default=uuid.uuid4, editable=False, primary_key=True, serialize=False)), + ('name', models.CharField(default='untitled', max_length=200)), + ('seq', models.PositiveBigIntegerField(default=0)), + ('palette', models.CharField(default='arthur/default', max_length=64)), + ('created', models.DateTimeField(auto_now_add=True)), + ('updated', models.DateTimeField(auto_now=True)), + ], + options={ + 'ordering': ['-updated'], + }, + ), + migrations.CreateModel( + name='Analysis', + fields=[ + ('key', models.CharField(max_length=71, primary_key=True, serialize=False)), + ('descriptor', models.TextField()), + ('detector', models.CharField(max_length=64)), + ('version', models.CharField(max_length=64)), + ('created', models.DateTimeField(auto_now_add=True)), + ('artifact', models.ForeignKey(blank=True, help_text='the dense landmark track, once bake A is uploaded', null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='analysis_for', to='clips.blob')), + ], + options={ + 'verbose_name_plural': 'analyses', + }, + ), + migrations.CreateModel( + name='Block', + fields=[ + ('key', models.CharField(max_length=71, primary_key=True, serialize=False)), + ('descriptor', models.TextField()), + ('role', models.CharField(max_length=32)), + ('created', models.DateTimeField(auto_now_add=True)), + ('analysis', models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='blocks', to='clips.analysis')), + ('data', models.ForeignKey(on_delete=django.db.models.deletion.PROTECT, related_name='block_data_for', to='clips.blob')), + ('state', models.ForeignKey(blank=True, help_text='the per-track absence mask, when the take has one', null=True, on_delete=django.db.models.deletion.PROTECT, related_name='block_state_for', to='clips.blob')), + ], + ), + migrations.CreateModel( + name='Footage', + fields=[ + ('id', models.UUIDField(default=uuid.uuid4, editable=False, primary_key=True, serialize=False)), + ('digest', models.CharField(max_length=64, unique=True)), + ('label', models.CharField(blank=True, max_length=200)), + ('source', models.CharField(blank=True, max_length=200)), + ('fps', models.FloatField()), + ('frames', models.PositiveIntegerField()), + ('width', models.PositiveIntegerField()), + ('height', models.PositiveIntegerField()), + ('feature_absence', models.JSONField(blank=True, default=dict)), + ('created', models.DateTimeField(auto_now_add=True)), + ('audio', models.ForeignKey(on_delete=django.db.models.deletion.PROTECT, related_name='audio_for', to='clips.blob')), + ], + options={ + 'ordering': ['-created'], + }, + ), + migrations.AddField( + model_name='analysis', + name='footage', + field=models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='analyses', to='clips.footage'), + ), + migrations.CreateModel( + name='Revision', + fields=[ + ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('seq', models.PositiveBigIntegerField()), + ('author', models.CharField(blank=True, max_length=200)), + ('summary', models.CharField(blank=True, max_length=500)), + ('document', models.JSONField(help_text='every leaf of the project, by path')), + ('created', models.DateTimeField(auto_now_add=True)), + ('project', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='revisions', to='clips.project')), + ], + options={ + 'ordering': ['-seq'], + }, + ), + migrations.CreateModel( + name='FootageFrame', + fields=[ + ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('index', models.PositiveIntegerField(help_text="0-based; source frame index + 1 is the PNG's name")), + ('blob', models.ForeignKey(on_delete=django.db.models.deletion.PROTECT, related_name='frame_for', to='clips.blob')), + ('footage', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='frame_set', to='clips.footage')), + ], + options={ + 'ordering': ['index'], + 'constraints': [models.UniqueConstraint(fields=('footage', 'index'), name='one_blob_per_frame')], + }, + ), + migrations.CreateModel( + name='Leaf', + fields=[ + ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('path', models.CharField(max_length=300)), + ('value', models.JSONField()), + ('version', models.PositiveBigIntegerField(default=1)), + ('updated', models.DateTimeField(auto_now=True)), + ('project', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='leaves', to='clips.project')), + ], + options={ + 'ordering': ['path'], + 'constraints': [models.UniqueConstraint(fields=('project', 'path'), name='one_leaf_per_path')], + }, + ), + migrations.CreateModel( + name='Clip', + fields=[ + ('id', models.BigAutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('cid', models.SlugField(max_length=64)), + ('name', models.CharField(blank=True, max_length=200)), + ('order', models.IntegerField(default=0)), + ('analysis', models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='clips', to='clips.analysis')), + ('blocks', models.ManyToManyField(blank=True, help_text="the tier-2 blocks this clip's channels name", related_name='clips', to='clips.block')), + ('footage', models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='clips', to='clips.footage')), + ('project', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='clips', to='clips.project')), + ], + options={ + 'ordering': ['order', 'cid'], + 'constraints': [models.UniqueConstraint(fields=('project', 'cid'), name='one_cid_per_project')], + }, + ), + ] diff --git a/clips/migrations/0002_remove_analysis_artifact_analysis_source_blocks.py b/clips/migrations/0002_remove_analysis_artifact_analysis_source_blocks.py new file mode 100644 index 0000000..a4f3396 --- /dev/null +++ b/clips/migrations/0002_remove_analysis_artifact_analysis_source_blocks.py @@ -0,0 +1,22 @@ +# Generated by Django 5.2.17 on 2026-09-28 13:10 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0001_initial'), + ] + + operations = [ + migrations.RemoveField( + model_name='analysis', + name='artifact', + ), + migrations.AddField( + model_name='analysis', + name='source_blocks', + field=models.ManyToManyField(blank=True, help_text='pixel-dependent landmarks, detection mask and mouth crops', related_name='source_for', to='clips.block'), + ), + ] diff --git a/clips/migrations/0003_source_extraction.py b/clips/migrations/0003_source_extraction.py new file mode 100644 index 0000000..2bb2f51 --- /dev/null +++ b/clips/migrations/0003_source_extraction.py @@ -0,0 +1,39 @@ +# Generated by Django 5.2.17 on 2026-09-28 13:23 + +import django.db.models.deletion +import uuid +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0002_remove_analysis_artifact_analysis_source_blocks'), + ] + + operations = [ + migrations.CreateModel( + name='Source', + fields=[ + ('id', models.UUIDField(default=uuid.uuid4, editable=False, primary_key=True, serialize=False)), + ('filename', models.CharField(max_length=255)), + ('probe', models.JSONField(default=dict)), + ('created', models.DateTimeField(auto_now_add=True)), + ('blob', models.OneToOneField(on_delete=django.db.models.deletion.PROTECT, related_name='video_source', to='clips.blob')), + ], + ), + migrations.CreateModel( + name='Extraction', + fields=[ + ('key', models.CharField(max_length=71, primary_key=True, serialize=False)), + ('settings', models.JSONField(default=dict)), + ('state', models.CharField(default='queued', max_length=16)), + ('progress', models.PositiveIntegerField(default=0)), + ('error', models.TextField(blank=True)), + ('created', models.DateTimeField(auto_now_add=True)), + ('updated', models.DateTimeField(auto_now=True)), + ('footage', models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.SET_NULL, related_name='extractions', to='clips.footage')), + ('source', models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, related_name='extractions', to='clips.source')), + ], + ), + ] diff --git a/clips/migrations/0004_footage_video_alter_footageframe_index.py b/clips/migrations/0004_footage_video_alter_footageframe_index.py new file mode 100644 index 0000000..6e53732 --- /dev/null +++ b/clips/migrations/0004_footage_video_alter_footageframe_index.py @@ -0,0 +1,24 @@ +# Generated by Django 5.2.17 on 2026-09-28 15:10 + +import django.db.models.deletion +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0003_source_extraction'), + ] + + operations = [ + migrations.AddField( + model_name='footage', + name='video', + field=models.ForeignKey(blank=True, help_text='the browser-safe proxy the page detects from; null on pre-proxy footage', null=True, on_delete=django.db.models.deletion.PROTECT, related_name='video_for', to='clips.blob'), + ), + migrations.AlterField( + model_name='footageframe', + name='index', + field=models.PositiveIntegerField(help_text="0-based; source frame index + 1 is the JPEG's name"), + ), + ] diff --git a/clips/migrations/0005_footage_stream_alter_footage_video.py b/clips/migrations/0005_footage_stream_alter_footage_video.py new file mode 100644 index 0000000..00976af --- /dev/null +++ b/clips/migrations/0005_footage_stream_alter_footage_video.py @@ -0,0 +1,24 @@ +# Generated by Django 5.2.17 on 2026-09-28 17:11 + +import django.db.models.deletion +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0004_footage_video_alter_footageframe_index'), + ] + + operations = [ + migrations.AddField( + model_name='footage', + name='stream', + field=models.ForeignKey(blank=True, help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, one access unit per frame; null on footage extracted before it", null=True, on_delete=django.db.models.deletion.PROTECT, related_name='stream_for', to='clips.blob'), + ), + migrations.AlterField( + model_name='footage', + name='video', + field=models.ForeignKey(blank=True, help_text='the browser-safe proxy, playable and seekable', null=True, on_delete=django.db.models.deletion.PROTECT, related_name='video_for', to='clips.blob'), + ), + ] diff --git a/clips/migrations/0006_project_schema_version.py b/clips/migrations/0006_project_schema_version.py new file mode 100644 index 0000000..6697451 --- /dev/null +++ b/clips/migrations/0006_project_schema_version.py @@ -0,0 +1,15 @@ +from django.db import migrations, models + + +class Migration(migrations.Migration): + dependencies = [ + ("clips", "0005_footage_stream_alter_footage_video"), + ] + + operations = [ + migrations.AddField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=1), + ), + ] diff --git a/clips/migrations/0007_symbols_not_timelines.py b/clips/migrations/0007_symbols_not_timelines.py new file mode 100644 index 0000000..1f726c8 --- /dev/null +++ b/clips/migrations/0007_symbols_not_timelines.py @@ -0,0 +1,75 @@ +"""Schema 2: a document holds symbols, not timelines, and no symbol is reserved. + +Three renames, each in the stored transit and nowhere else: + + clip//timeline/... -> clip//symbol/... + a node leaf's :kind :symbol -> :kind :instance + a feature leaf's :timeline key -> :symbol + +A leaf value is transit's map form, ["^ ", k1, v1, k2, v2, ...]. Only TOP-LEVEL +pairs are rewritten, and only literal ones: transit caches a repeated keyword as +"^N", and a rename that met a cache reference where it expected the keyword would +be guessing. Every saved leaf at the time of writing had these as literals; if one +does not, the migration stops rather than writing a document that decodes to +something else. + +Renaming a cached keyword in place is safe because the cache is positional: the +literal keeps its slot, so any later "^N" that referred to it now refers to the +new name, which is what it meant. +""" + +import re + +from django.db import migrations, models + +PATH = re.compile(r"^(clip/[^/]+/)timeline(/|$)") + + +def _rename_pair(value, key, old, new, path): + if not (isinstance(value, list) and value[:1] == ["^ "]): + return value + out = list(value) + for i in range(1, len(out) - 1, 2): + if out[i] != key: + continue + if old is None: + out[i] = new + elif out[i + 1] == old: + out[i + 1] = new + elif isinstance(out[i + 1], str) and out[i + 1].startswith("^") and out[i + 1] != "^ ": + raise RuntimeError(f"leaf {path!r} has a cached {key} value; migrate it by hand") + return out + + +def forwards(apps, schema_editor): + Leaf = apps.get_model("clips", "Leaf") + Project = apps.get_model("clips", "Project") + for leaf in Leaf.objects.all(): + path = PATH.sub(r"\1symbol\2", leaf.path) + value = leaf.value + parts = path.split("/") + if len(parts) == 6 and parts[2] == "symbol" and parts[4] == "node": + value = _rename_pair(value, "~:kind", "~:symbol", "~:instance", leaf.path) + if len(parts) == 4 and parts[2] == "feature": + value = _rename_pair(value, "~:timeline", None, "~:symbol", leaf.path) + if path != leaf.path or value != leaf.value: + leaf.path = path + leaf.value = value + leaf.version += 1 + leaf.save(update_fields=["path", "value", "version"]) + Project.objects.update(schema_version=2) + + +class Migration(migrations.Migration): + dependencies = [ + ("clips", "0006_project_schema_version"), + ] + + operations = [ + migrations.AlterField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=2), + ), + migrations.RunPython(forwards, migrations.RunPython.noop), + ] diff --git a/clips/migrations/0008_owners_editors_leaf_seq.py b/clips/migrations/0008_owners_editors_leaf_seq.py new file mode 100644 index 0000000..66aaa89 --- /dev/null +++ b/clips/migrations/0008_owners_editors_leaf_seq.py @@ -0,0 +1,39 @@ +from django.conf import settings +from django.db import migrations, models +import django.db.models.deletion + + +def orphans(apps, schema_editor): + # Every project has an owner, and none of the ones saved before owners did. + apps.get_model("clips", "Project").objects.all().delete() + + +class Migration(migrations.Migration): + + dependencies = [ + ("clips", "0007_symbols_not_timelines"), + migrations.swappable_dependency(settings.AUTH_USER_MODEL), + ] + + operations = [ + migrations.RunPython(orphans, migrations.RunPython.noop), + migrations.AddField( + model_name="leaf", + name="seq", + field=models.PositiveBigIntegerField( + default=0, help_text="the project seq of the write that last changed it"), + ), + migrations.AddField( + model_name="project", + name="editors", + field=models.ManyToManyField(blank=True, related_name="shared_projects", + to=settings.AUTH_USER_MODEL), + ), + migrations.AddField( + model_name="project", + name="owner", + field=models.ForeignKey(on_delete=django.db.models.deletion.CASCADE, + related_name="projects", to=settings.AUTH_USER_MODEL), + preserve_default=False, + ), + ] diff --git a/clips/migrations/0009_revision_blocks.py b/clips/migrations/0009_revision_blocks.py new file mode 100644 index 0000000..3bd1303 --- /dev/null +++ b/clips/migrations/0009_revision_blocks.py @@ -0,0 +1,18 @@ +# Generated by Django 5.2.17 on 2026-09-30 01:54 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0008_owners_editors_leaf_seq'), + ] + + operations = [ + migrations.AddField( + model_name='revision', + name='blocks', + field=models.JSONField(default=dict, help_text="each clip's tier-2 block keys, by cid, so a restore can name them"), + ), + ] diff --git a/clips/migrations/0010_sounds.py b/clips/migrations/0010_sounds.py new file mode 100644 index 0000000..1f26395 --- /dev/null +++ b/clips/migrations/0010_sounds.py @@ -0,0 +1,25 @@ +# Generated by Django 5.2.17 on 2026-09-30 07:14 + +import django.db.models.deletion +import uuid +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0009_revision_blocks'), + ] + + operations = [ + migrations.CreateModel( + name='Sound', + fields=[ + ('id', models.UUIDField(default=uuid.uuid4, editable=False, primary_key=True, serialize=False)), + ('filename', models.CharField(max_length=255)), + ('duration', models.FloatField(help_text='seconds, as ffprobe reports it')), + ('created', models.DateTimeField(auto_now_add=True)), + ('blob', models.ForeignKey(on_delete=django.db.models.deletion.PROTECT, related_name='sound_for', to='clips.blob')), + ], + ), + ] diff --git a/clips/migrations/0011_occurrence_schema.py b/clips/migrations/0011_occurrence_schema.py new file mode 100644 index 0000000..cf9ad2e --- /dev/null +++ b/clips/migrations/0011_occurrence_schema.py @@ -0,0 +1,13 @@ +from django.db import migrations, models + + +class Migration(migrations.Migration): + dependencies = [("clips", "0010_sounds")] + + operations = [ + migrations.AlterField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=3), + ), + ] diff --git a/clips/migrations/0012_sound_label.py b/clips/migrations/0012_sound_label.py new file mode 100644 index 0000000..7baee07 --- /dev/null +++ b/clips/migrations/0012_sound_label.py @@ -0,0 +1,18 @@ +# Generated by Django 5.2.17 on 2026-10-01 04:38 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + + dependencies = [ + ('clips', '0011_occurrence_schema'), + ] + + operations = [ + migrations.AddField( + model_name='sound', + name='label', + field=models.CharField(blank=True, help_text='what a person called it; the filename when empty. Separate from `filename` because the name on disk is a fact about the upload and renaming must not rewrite it', max_length=200), + ), + ] diff --git a/clips/migrations/0013_palette_track.py b/clips/migrations/0013_palette_track.py new file mode 100644 index 0000000..ab00fc7 --- /dev/null +++ b/clips/migrations/0013_palette_track.py @@ -0,0 +1,45 @@ +"""Schema 4: split animated symbol palettes from the authoring palette. + +Before schema 4 a symbol's ``:palette`` leaf value was always a channel. It now +names the static palette used when that symbol is the viewed root, while the +old channel is retained as ``:palette-channel`` compatibility data. New edits +use a real lane symbol referenced by ``:palette-track``. +""" + +from django.db import migrations, models + + +def forwards(apps, schema_editor): + Leaf = apps.get_model("clips", "Leaf") + Project = apps.get_model("clips", "Project") + for leaf in Leaf.objects.filter(path__contains="/symbol/"): + parts = leaf.path.split("/") + if len(parts) != 4 or parts[2] != "symbol": + continue + value = leaf.value + if not (isinstance(value, list) and value[:1] == ["^ "]): + continue + out = list(value) + changed = False + for i in range(1, len(out) - 1, 2): + if out[i] == "~:palette" and isinstance(out[i + 1], list): + out[i] = "~:palette-channel" + changed = True + if changed: + leaf.value = out + leaf.version += 1 + leaf.save(update_fields=["value", "version"]) + Project.objects.update(schema_version=4) + + +class Migration(migrations.Migration): + dependencies = [("clips", "0012_sound_label")] + + operations = [ + migrations.AlterField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=4), + ), + migrations.RunPython(forwards, migrations.RunPython.noop), + ] diff --git a/clips/migrations/0014_multiple_analyses.py b/clips/migrations/0014_multiple_analyses.py new file mode 100644 index 0000000..97fe72b --- /dev/null +++ b/clips/migrations/0014_multiple_analyses.py @@ -0,0 +1,15 @@ +from django.db import migrations, models + + +class Migration(migrations.Migration): + dependencies = [("clips", "0013_palette_track")] + + operations = [ + migrations.RemoveField(model_name="clip", name="analysis"), + migrations.RemoveField(model_name="clip", name="footage"), + migrations.AlterField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=5), + ), + ] diff --git a/clips/migrations/0015_tracing_images.py b/clips/migrations/0015_tracing_images.py new file mode 100644 index 0000000..ab3aa3c --- /dev/null +++ b/clips/migrations/0015_tracing_images.py @@ -0,0 +1,48 @@ +"""Schema 6: tracing is a symbol. + +A face's `:head :trace` is gone: its footage is a placement of a `:type :trace` +symbol under the head, the trace keys are that placement's `:time :holds`, and the +head follows them with `:reads`. Nothing is converted. Every project is marked 6, +and one that still carries a `:trace` is refused when it is opened, by name and +with what to do about it; the rest open as they did. + +Images are stills to trace over, stored like sounds. +""" + +import uuid + +from django.db import migrations, models +import django.db.models.deletion + + +def forwards(apps, schema_editor): + apps.get_model("clips", "Project").objects.update(schema_version=6) + + +class Migration(migrations.Migration): + dependencies = [("clips", "0014_multiple_analyses")] + + operations = [ + migrations.CreateModel( + name="Image", + fields=[ + ("id", models.UUIDField(default=uuid.uuid4, editable=False, + primary_key=True, serialize=False)), + ("filename", models.CharField(max_length=255)), + ("label", models.CharField( + blank=True, max_length=200, + help_text="what a person called it; the filename when empty")), + ("width", models.PositiveIntegerField(help_text="pixels, as ffprobe reports them")), + ("height", models.PositiveIntegerField()), + ("created", models.DateTimeField(auto_now_add=True)), + ("blob", models.ForeignKey(on_delete=django.db.models.deletion.PROTECT, + related_name="image_for", to="clips.blob")), + ], + ), + migrations.AlterField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=6), + ), + migrations.RunPython(forwards, migrations.RunPython.noop), + ] diff --git a/clips/migrations/0016_an_anchor_is_a_peg.py b/clips/migrations/0016_an_anchor_is_a_peg.py new file mode 100644 index 0000000..208d603 --- /dev/null +++ b/clips/migrations/0016_an_anchor_is_a_peg.py @@ -0,0 +1,42 @@ +"""Schema 7: an anchor is a peg. + +`[:xform :anchor]` is gone from the transform. `T(a)·M·T(-a)` is a transform +conjugated by a translation — "do M in a frame shifted by a" — and a parent +already is a shifted frame, so an anchor was a peg written inline: one that could +not be selected, keyed, shared between nodes, or placed above a measured channel. +A pivot nobody chose is now derived from what the node draws, per drag, and stored +nowhere; a pivot to keep is a peg, an ordinary `:group` parent. + +Nothing is converted, as in schema 6. Every project is marked 7, and one that +still carries an anchor is refused when it is opened, by name and with what to do +about it — `node/problems` in the frontend. + +Not converted rather than not worth converting. Dropping an anchor is in fact +pixel-exact wherever rotation and scale are the identity, since the anchor +cancels out of the composition there — and that is everywhere a freeze, a drop or +a new drawing wrote one. It is NOT exact on anything a hand has since turned or +scaled, where the composed translation is `a + p - M·a`, and it cannot be made +exact at all where `pos` is dense, because tier 2 is content-addressed and not +rewritable here. A conversion would therefore be silent and right for most nodes +and silent and wrong for exactly the ones somebody had hand-placed, which is the +worse failure: a refusal names the document and says what to do. +""" + +from django.db import migrations, models + + +def forwards(apps, schema_editor): + apps.get_model("clips", "Project").objects.update(schema_version=7) + + +class Migration(migrations.Migration): + dependencies = [("clips", "0015_tracing_images")] + + operations = [ + migrations.AlterField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=7), + ), + migrations.RunPython(forwards, migrations.RunPython.noop), + ] diff --git a/clips/migrations/0017_a_node_has_a_pivot.py b/clips/migrations/0017_a_node_has_a_pivot.py new file mode 100644 index 0000000..2e71d6e --- /dev/null +++ b/clips/migrations/0017_a_node_has_a_pivot.py @@ -0,0 +1,45 @@ +"""Schema 8: a node has a pivot. + +`[:xform :pivot]` is back in the transform, as the point rotation and scale are +composed about: `local = T(pos)·T(piv)·R·K·S·T(-piv)`. Schema 7 deleted it, on +the argument that an anchor is a peg — true as algebra, and not true as a feature. +A peg is a node, and a turn about a point that is not the turning node's own +origin still has to solve for a position to hold that point still; that solution +is an arc in the angle while a position channel tweens along the chord, so it is +right on the frame it is written and wrong on every frame between two keys. A +drawing escaped it, since its origin is the middle of what it draws. A symbol +instance could not: its origin is its symbol's, which is the top-left corner of +the stage, so one keyed turn of an instance swung its drawing round that corner +on an orbit the size of the stage. + +CONVERTED, unlike 6 and 7, because adding this one is exact. A schema-7 node has +no pivot; an absent pivot reads as [0 0]; and T(pos)·T(0)·M·T(-0) is T(pos)·M to +the last bit of the mantissa. Every stored document therefore composes to exactly +the matrices it composed to before, dense tier-2 transforms included, so there is +nothing to guess at and no node a conversion could silently move. The version is +restamped and nothing else is touched. + +What a converted document does NOT get is a pivot somebody chose: nodes placed +before this carry none, so they still turn about their own origin until the first +turn or scale writes one — `gesture/with-pivot`, from the middle of what the node +draws at that moment — or until the cross is dragged (ctrl/cmd-drag on the stage). +""" + +from django.db import migrations, models + + +def forwards(apps, schema_editor): + apps.get_model("clips", "Project").objects.update(schema_version=8) + + +class Migration(migrations.Migration): + dependencies = [("clips", "0016_an_anchor_is_a_peg")] + + operations = [ + migrations.AlterField( + model_name="project", + name="schema_version", + field=models.PositiveIntegerField(default=8), + ), + migrations.RunPython(forwards, migrations.RunPython.noop), + ] diff --git a/clips/migrations/__init__.py b/clips/migrations/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/models.py b/clips/models.py new file mode 100644 index 0000000..c8cc44c --- /dev/null +++ b/clips/models.py @@ -0,0 +1,366 @@ +"""The entity model, as tables. + +It follows docs/architecture.md's model exactly, and the one thing worth reading +it for is which tier each table is in, because that is what decides whether a row +is a document, a cache entry or a source. + + TIER 1, the document. Project, Clip, Leaf, Revision. Kilobytes, authored, + versioned, and the only tier anything will ever sync. + + TIER 2, derived. Analysis, Block. Content-addressed by a hash over every input + that produced them — including the detector version — so a stale bake is + unreachable rather than wrong, and a collaborator's bake is fetchable by the + same key. + + TIER 3, source. Footage, FootageFrame. Immutable, by hash. + + Blob is under all three of them: bytes, named by the sha256 of themselves. + +WHAT IS DELIBERATELY NOT HERE. `Clip` does not store fps, frames, width or height. +They are in the document — the `timing` and `stage` leaves — and a copy of them in +a column is a copy that comes to disagree with the scene it describes. The columns +`Clip` does have are the ones the SERVER needs to answer a question about a clip +without parsing its leaves: which footage, which analysis, which blocks. +""" +import uuid + +from django.conf import settings +from django.db import models +from django.utils import timezone + + +class Blob(models.Model): + """Bytes, named by the sha256 of themselves. The file is on disk under + `BLOB_ROOT`; this row is the index and the size.""" + + digest = models.CharField(primary_key=True, max_length=64) + media_type = models.CharField(max_length=100, default="application/octet-stream") + size = models.BigIntegerField() + created = models.DateTimeField(auto_now_add=True) + + def __str__(self): + return f"{self.digest[:12]}… {self.size}B {self.media_type}" + + +class Source(models.Model): + """An uploaded video, identified by its byte digest.""" + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + blob = models.OneToOneField(Blob, on_delete=models.PROTECT, related_name="video_source") + filename = models.CharField(max_length=255) + probe = models.JSONField(default=dict) + created = models.DateTimeField(auto_now_add=True) + + +class Sound(models.Model): + """An uploaded sound file — mp3, wav, whatever the browser can decode — kept + as uploaded. Not footage: it has no frames and nothing measures it, so it + skips extraction and an audio node plays its bytes directly.""" + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + blob = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="sound_for") + filename = models.CharField(max_length=255) + label = models.CharField( + max_length=200, blank=True, + help_text="what a person called it; the filename when empty. Separate " + "from `filename` because the name on disk is a fact about the " + "upload and renaming must not rewrite it", + ) + duration = models.FloatField(help_text="seconds, as ffprobe reports it") + created = models.DateTimeField(auto_now_add=True) + + +class Image(models.Model): + """An uploaded still — a drawing, a photo, a model sheet — kept as uploaded, to + be traced over. Never part of the picture: a document names its blob as a + tracing symbol's `:media`, and the page draws it over the stage.""" + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + blob = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="image_for") + filename = models.CharField(max_length=255) + label = models.CharField(max_length=200, blank=True, + help_text="what a person called it; the filename when empty") + width = models.PositiveIntegerField(help_text="pixels, as ffprobe reports them") + height = models.PositiveIntegerField() + created = models.DateTimeField(auto_now_add=True) + + +class Extraction(models.Model): + """One requested decode of a source into immutable footage.""" + + key = models.CharField(primary_key=True, max_length=71) + source = models.ForeignKey(Source, on_delete=models.CASCADE, related_name="extractions") + settings = models.JSONField(default=dict) + state = models.CharField(max_length=16, default="queued") + progress = models.PositiveIntegerField(default=0) + error = models.TextField(blank=True) + footage = models.ForeignKey( + "Footage", null=True, blank=True, on_delete=models.SET_NULL, + related_name="extractions", + ) + created = models.DateTimeField(auto_now_add=True) + updated = models.DateTimeField(auto_now=True) + + +class Footage(models.Model): + """Tier 3: the frames and audio of one extraction, immutable. + + `digest` is over the PROXY VIDEO's digest plus the audio's and the rate, so + two extractions of the same clip at the same settings are one footage and the + same analysis can be reused across both. + + THE PROXY IS THE ANALYSIS SOURCE AND THE FRAMES ARE NOT. `video` is one + browser-safe H.264 file, and it is what the page seeks through to detect + landmarks. `frame_set` is a JPEG per frame at tracing size: reference stills + for the tracing editor, never the thing measured. The two are not + interchangeable, and which one carries the pixels an analysis was computed + from is the difference between a 6MB take and a 1.1GB one. + + So the frame JPEGs are deliberately NOT in `digest`. They are a rendering of + this footage for a human to trace over; re-rendering them at another size does + not make it different footage, and putting them in the identity would throw + away every analysis when the tracing size changed. + + `feature_absence` is the manifest annotation step 8 introduced: known + occlusion intervals, one-based and inclusive, expanded into presence tracks by + the loader. An input format, not a control UI. + """ + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + digest = models.CharField(max_length=64, unique=True) + label = models.CharField(max_length=200, blank=True) + source = models.CharField(max_length=200, blank=True) + fps = models.FloatField() + frames = models.PositiveIntegerField() + width = models.PositiveIntegerField() + height = models.PositiveIntegerField() + audio = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="audio_for") + video = models.ForeignKey( + Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="video_for", + help_text="the browser-safe proxy, playable and seekable", + ) + stream = models.ForeignKey( + Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="stream_for", + help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, " + "one access unit per frame; null on footage extracted before it", + ) + feature_absence = models.JSONField(default=dict, blank=True) + created = models.DateTimeField(auto_now_add=True) + + class Meta: + ordering = ["-created"] + + def __str__(self): + return f"{self.label or self.source or self.id} ({self.frames}f @{self.fps})" + + +class FootageFrame(models.Model): + """One tracing still. A row rather than an entry in a JSON list, because a + frame is a thing the server serves, and because a blob's references have to be + countable before anything can be collected. + + A REFERENCE IMAGE, NOT A MEASUREMENT INPUT. See `Footage.video`.""" + + footage = models.ForeignKey(Footage, on_delete=models.CASCADE, related_name="frame_set") + index = models.PositiveIntegerField(help_text="0-based; source frame index + 1 is the JPEG's name") + blob = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="frame_for") + + class Meta: + ordering = ["index"] + constraints = [ + models.UniqueConstraint(fields=["footage", "index"], name="one_blob_per_frame"), + ] + + +class Analysis(models.Model): + """Tier 2: one detector, at one version, over one footage. + + `key` is a content address over every input, and `descriptor` is the exact + canonical text that key is the sha256 of — sent by the client and stored, not + recomputed here. `clips/views.py` says why that is the honest arrangement: JS + prints an integral double as `1` and Python as `1.0`, so a scheme where both + sides re-render the numbers breaks on the first one of them. + + `detector` and `version` are columns as well as descriptor fields so that the + question "which model produced this take" is answerable in the admin and in a + query, rather than only by parsing a hash's preimage. + """ + + key = models.CharField(primary_key=True, max_length=71) + descriptor = models.TextField() + detector = models.CharField(max_length=64) + version = models.CharField(max_length=64) + footage = models.ForeignKey( + Footage, null=True, blank=True, on_delete=models.SET_NULL, related_name="analyses" + ) + source_blocks = models.ManyToManyField( + "Block", blank=True, related_name="source_for", + help_text="pixel-dependent landmarks, detection mask and mouth crops", + ) + created = models.DateTimeField(auto_now_add=True) + + class Meta: + verbose_name_plural = "analyses" + + def __str__(self): + return f"{self.detector} {self.version} → {self.key[7:19]}…" + + +class Block(models.Model): + """Tier 2: one dense channel block. + + Two hashes, and they are not the same hash. `key` is over the block's INPUTS, + which is what lets a client ask for the block its current settings want before + anything has computed it. `data.digest` is over the bytes. See clips/blobs.py. + """ + + key = models.CharField(primary_key=True, max_length=71) + descriptor = models.TextField() + role = models.CharField(max_length=32) + analysis = models.ForeignKey( + Analysis, null=True, blank=True, on_delete=models.SET_NULL, related_name="blocks" + ) + data = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="block_data_for") + state = models.ForeignKey( + Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="block_state_for", + help_text="the per-track absence mask, when the take has one", + ) + created = models.DateTimeField(auto_now_add=True) + + def __str__(self): + return f"{self.role} {self.key[7:19]}…" + + +class Project(models.Model): + """Tier 1: the document's root. + + `schema_version` identifies the stored document format. `seq` counts writes + to this particular project; it is not a format version. Every write bumps + `seq`, and a client that sees `seq > local + 1` refetches. + + ANYONE WITH THE LINK CAN VIEW; the owner and the editors can write. Every + project has an owner. + """ + + id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) + owner = models.ForeignKey( + settings.AUTH_USER_MODEL, on_delete=models.CASCADE, related_name="projects", + ) + editors = models.ManyToManyField( + settings.AUTH_USER_MODEL, blank=True, related_name="shared_projects", + ) + name = models.CharField(max_length=200, default="untitled") + schema_version = models.PositiveIntegerField(default=8) + seq = models.PositiveBigIntegerField(default=0) + palette = models.CharField(max_length=64, default="arthur/default") + created = models.DateTimeField(auto_now_add=True) + updated = models.DateTimeField(auto_now=True) + + class Meta: + ordering = ["-updated"] + + def __str__(self): + return f"{self.name} ({self.id})" + + def bump(self): + """The next seq, taken with an UPDATE so that inside a transaction it is + also the write lock: two concurrent saves cannot both get the same one.""" + Project.objects.filter(id=self.id).update( + seq=models.F("seq") + 1, updated=timezone.now() + ) + self.refresh_from_db(fields=["seq", "updated"]) + return self.seq + + def can_edit(self, user): + return user.is_authenticated and ( + user.id == self.owner_id or self.editors.filter(id=user.id).exists() + ) + + +class Clip(models.Model): + """Tier 1: the unit of work, and the thing leaf paths are scoped by. + + `cid` is what appears in `clip//...`, so it is the clip's identity as far + as addressing is concerned and it does not change. + """ + + project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="clips") + cid = models.SlugField(max_length=64) + name = models.CharField(max_length=200, blank=True) + order = models.IntegerField(default=0) + blocks = models.ManyToManyField( + Block, blank=True, related_name="clips", + help_text="the tier-2 blocks this clip's channels name", + ) + + class Meta: + ordering = ["order", "cid"] + constraints = [ + models.UniqueConstraint(fields=["project", "cid"], name="one_cid_per_project"), + ] + + def __str__(self): + return f"{self.cid} of {self.project.name}" + + +class Leaf(models.Model): + """Tier 1: one independently addressed, independently versioned piece of the + document. + + The value is transit-as-JSON in a JSONField, so the column holds JSON rather + than a string containing JSON: the admin can read a leaf, and the field-wise + merge of a channel leaf that docs/architecture.md describes as fifteen lines of + Python is possible over it. `version` is the entity tag a conditional write + compares — RFC 7232, not a bespoke invention. + """ + + project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="leaves") + path = models.CharField(max_length=300) + value = models.JSONField() + version = models.PositiveBigIntegerField(default=1) + seq = models.PositiveBigIntegerField( + default=0, help_text="the project seq of the write that last changed it", + ) + updated = models.DateTimeField(auto_now=True) + + class Meta: + ordering = ["path"] + constraints = [ + models.UniqueConstraint(fields=["project", "path"], name="one_leaf_per_path"), + ] + + @property + def etag(self): + return f'"{self.version}"' + + def __str__(self): + return f"{self.path}@{self.version}" + + +class Revision(models.Model): + """Tier 1: a snapshot of the authored layer, with a user and a summary — a + named snapshot, which is how a person marks a version now that every edit + saves itself. + + ON AN EXPLICIT TRIGGER, not on every save. tl snapshots a small annotation + layer; arthur's tier 1 will contain cel polygons, so a snapshot per save bloats + the table — docs/architecture.md's "revisions need a coarser trigger". So this + is written by `POST /api/projects//revisions`, which is a "mark version" + button, and never by a save. + """ + + project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="revisions") + seq = models.PositiveBigIntegerField() + author = models.CharField(max_length=200, blank=True) + summary = models.CharField(max_length=500, blank=True) + document = models.JSONField(help_text="every leaf of the project, by path") + blocks = models.JSONField( + default=dict, help_text="each clip's tier-2 block keys, by cid, so a restore can name them", + ) + created = models.DateTimeField(auto_now_add=True) + + class Meta: + ordering = ["-seq"] + + def __str__(self): + return f"{self.project.name} r{self.seq}: {self.summary}" diff --git a/clips/routing.py b/clips/routing.py new file mode 100644 index 0000000..b94cd1b --- /dev/null +++ b/clips/routing.py @@ -0,0 +1,7 @@ +from django.urls import path + +from .consumers import ProjectConsumer + +websocket_urlpatterns = [ + path("ws/projects/", ProjectConsumer.as_asgi()), +] diff --git a/clips/templates/clips/index.html b/clips/templates/clips/index.html new file mode 100644 index 0000000..d73c62e --- /dev/null +++ b/clips/templates/clips/index.html @@ -0,0 +1,31 @@ +{% load static %} +{% comment %} +The host page, served by Django since port-plan step 9. + +It carries no styles of its own any more. They are `static/arthur/app.css`, which +staticfiles serves from the same tree as the bundle — the page grew a five-pane +application chrome and "the styles" stopped being a thing you read in passing on +the way to the markup. + +The bundle is unchanged: shadow-cljs writes it into `static/arthur/js` and +staticfiles serves it from there, so `manage.py runserver` and `shadow-cljs watch +app` are the whole dev loop with nothing copying files between them. + +The CSRF token is rendered so that Django sets its cookie, which is what +`arthur.fx.http` reads to write the `X-CSRFToken` header. Saves are ordinary POSTs +and PUTs with ordinary CSRF protection — no endpoint in this app is exempt. +{% endcomment %} + + + + + arthur + + + + {% csrf_token %} +
+ + + + diff --git a/clips/tests/__init__.py b/clips/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clips/tests/test_api.py b/clips/tests/test_api.py new file mode 100644 index 0000000..706c89a --- /dev/null +++ b/clips/tests/test_api.py @@ -0,0 +1,1218 @@ +"""What the server guarantees, as opposed to what the client intends. + +The two interesting groups here are the ones that make the tier split a property +of the system rather than a convention in ClojureScript: + + A KEY DESCRIBES ITS BYTES. The server recomputes every tier-2 key it is handed + and refuses a mismatch, so nothing can store a block under a name that is not + the hash of its own descriptor. + + A BLOCK CAN NAME ITS DETECTOR VERSION. Every block names an analysis and every + analysis declares a detector and a version, both enforced here. That chain is + what docs/architecture.md asks for: without it, a model upgrade that silently + reuses old landmarks presents as "the tool got worse" with no event to attach it + to. + +The rest is the load/save round trip, the conditional write, and the footage +manifest that makes the frames the backend's to serve. +""" +import base64 +import hashlib +import json +import shutil +import struct +import subprocess +import tempfile +import zlib +from io import StringIO +from pathlib import Path +from unittest import skipUnless +from unittest.mock import Mock, patch + +from django.core.files.uploadedfile import SimpleUploadedFile +from django.core.management import call_command +from django.test import TestCase, override_settings + +from clips import blobs, extraction +from clips.models import Analysis, Block, Blob, Clip, Footage, Image, Leaf, Project, Revision, Sound, Source + +BLOB_DIR = tempfile.mkdtemp(prefix="arthur-test-blobs-") + + +def key_for(descriptor: str) -> str: + return "sha256:" + hashlib.sha256(descriptor.encode("utf-8")).hexdigest() + + +def analysis_descriptor(version="1.0.1"): + # Canonical JSON, written the way arthur.domain.canon writes it: sorted keys, + # no spaces, integral doubles with no point. + return ('{"aspect":1,"detector":"mediapipe","frames":48,"fps":30,"scheme":1,' + f'"version":"{version}"}}') + + +def block_descriptor(analysis_key, role="geom", anchor_avg=2): + return (f'{{"analysis":"{analysis_key}","features":["mouth"],' + f'"layout":{{"frames":48,"scale":16384,"stride":16,"tracks":1,"type":"int16"}},' + f'"observation":null,"params":{{"anchor-avg":{anchor_avg}}},' + f'"role":"{role}","scheme":1,"tracks":["outer"]}}') + + +def png(width=4, height=3): + """The smallest valid PNG of a given size, written by hand. + + So that `blobs.png_size` and the ingest path are exercised without Pillow. The + one thing the backend needs from a PNG is its IHDR, and this is a PNG with one. + """ + def chunk(kind, payload): + return (struct.pack(">I", len(payload)) + kind + payload + + struct.pack(">I", zlib.crc32(kind + payload) & 0xFFFFFFFF)) + + ihdr = struct.pack(">IIBBBBB", width, height, 8, 2, 0, 0, 0) + raw = b"".join(b"\x00" + b"\x40\x40\x40" * width for _ in range(height)) + return (b"\x89PNG\r\n\x1a\n" + chunk(b"IHDR", ihdr) + + chunk(b"IDAT", zlib.compress(raw)) + chunk(b"IEND", b"")) + + +@override_settings(BLOB_ROOT=BLOB_DIR) +class BlobStoreTests(TestCase): + def test_the_same_bytes_are_stored_once(self): + a, size = blobs.write(b"the same bytes") + b, _ = blobs.write(b"the same bytes") + self.assertEqual(a, b) + self.assertEqual(size, 14) + self.assertEqual(blobs.read(a), b"the same bytes") + + def test_a_path_that_is_not_a_hash_is_refused(self): + # The blob route takes its digest from the URL, so this is the check that + # stops `/blob/../../etc/passwd` being a path at all. + with self.assertRaises(ValueError): + blobs.path_for("../../etc/passwd") + with self.assertRaises(ValueError): + blobs.path_for("deadbeef") + + def test_a_png_reports_its_own_size(self): + with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as fh: + fh.write(png(17, 5)) + self.assertEqual((17, 5), blobs.png_size(Path(fh.name))) + + def test_a_blob_is_served_immutable(self): + digest, size = blobs.write(b"bytes on the wire") + Blob.objects.create(digest=digest, size=size, media_type="application/octet-stream") + response = self.client.get(f"/blob/{digest}") + self.assertEqual(200, response.status_code) + self.assertIn("immutable", response["Cache-Control"]) + self.assertEqual(f'"{digest}"', response["ETag"]) + self.assertEqual(b"bytes on the wire", b"".join(response.streaming_content)) + + def test_an_unknown_blob_is_a_404_and_not_a_traceback(self): + self.assertEqual(404, self.client.get("/blob/" + "0" * 64).status_code) + self.assertEqual(404, self.client.get("/blob/nonsense").status_code) + + def test_a_blob_serves_byte_ranges(self): + # NOT AN OPTIMISATION. A