"""The entity model, as tables. It follows docs/architecture.md's model exactly, and the one thing worth reading it for is which tier each table is in, because that is what decides whether a row is a document, a cache entry or a source. TIER 1, the document. Project, Clip, Leaf, Revision. Kilobytes, authored, versioned, and the only tier anything will ever sync. TIER 2, derived. Analysis, Block. Content-addressed by a hash over every input that produced them — including the detector version — so a stale bake is unreachable rather than wrong, and a collaborator's bake is fetchable by the same key. TIER 3, source. Footage, FootageFrame. Immutable, by hash. Blob is under all three of them: bytes, named by the sha256 of themselves. WHAT IS DELIBERATELY NOT HERE. `Clip` does not store fps, frames, width or height. They are in the document — the `timing` and `stage` leaves — and a copy of them in a column is a copy that comes to disagree with the scene it describes. The columns `Clip` does have are the ones the SERVER needs to answer a question about a clip without parsing its leaves: which footage, which analysis, which blocks. """ import uuid from django.conf import settings from django.db import models from django.utils import timezone class Blob(models.Model): """Bytes, named by the sha256 of themselves. The file is on disk under `BLOB_ROOT`; this row is the index and the size.""" digest = models.CharField(primary_key=True, max_length=64) media_type = models.CharField(max_length=100, default="application/octet-stream") size = models.BigIntegerField() created = models.DateTimeField(auto_now_add=True) def __str__(self): return f"{self.digest[:12]}… {self.size}B {self.media_type}" class Source(models.Model): """An uploaded video, identified by its byte digest.""" id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) blob = models.OneToOneField(Blob, on_delete=models.PROTECT, related_name="video_source") filename = models.CharField(max_length=255) probe = models.JSONField(default=dict) created = models.DateTimeField(auto_now_add=True) class Sound(models.Model): """An uploaded sound file — mp3, wav, whatever the browser can decode — kept as uploaded. Not footage: it has no frames and nothing measures it, so it skips extraction and an audio node plays its bytes directly.""" id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) blob = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="sound_for") filename = models.CharField(max_length=255) duration = models.FloatField(help_text="seconds, as ffprobe reports it") created = models.DateTimeField(auto_now_add=True) class Extraction(models.Model): """One requested decode of a source into immutable footage.""" key = models.CharField(primary_key=True, max_length=71) source = models.ForeignKey(Source, on_delete=models.CASCADE, related_name="extractions") settings = models.JSONField(default=dict) state = models.CharField(max_length=16, default="queued") progress = models.PositiveIntegerField(default=0) error = models.TextField(blank=True) footage = models.ForeignKey( "Footage", null=True, blank=True, on_delete=models.SET_NULL, related_name="extractions", ) created = models.DateTimeField(auto_now_add=True) updated = models.DateTimeField(auto_now=True) class Footage(models.Model): """Tier 3: the frames and audio of one extraction, immutable. `digest` is over the PROXY VIDEO's digest plus the audio's and the rate, so two extractions of the same clip at the same settings are one footage and the same analysis can be reused across both. THE PROXY IS THE ANALYSIS SOURCE AND THE FRAMES ARE NOT. `video` is one browser-safe H.264 file, and it is what the page seeks through to detect landmarks. `frame_set` is a JPEG per frame at tracing size: reference stills for the tracing editor, never the thing measured. The two are not interchangeable, and which one carries the pixels an analysis was computed from is the difference between a 6MB take and a 1.1GB one. So the frame JPEGs are deliberately NOT in `digest`. They are a rendering of this footage for a human to trace over; re-rendering them at another size does not make it different footage, and putting them in the identity would throw away every analysis when the tracing size changed. `feature_absence` is the manifest annotation step 8 introduced: known occlusion intervals, one-based and inclusive, expanded into presence tracks by the loader. An input format, not a control UI. """ id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) digest = models.CharField(max_length=64, unique=True) label = models.CharField(max_length=200, blank=True) source = models.CharField(max_length=200, blank=True) fps = models.FloatField() frames = models.PositiveIntegerField() width = models.PositiveIntegerField() height = models.PositiveIntegerField() audio = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="audio_for") video = models.ForeignKey( Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="video_for", help_text="the browser-safe proxy, playable and seekable", ) stream = models.ForeignKey( Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="stream_for", help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, " "one access unit per frame; null on footage extracted before it", ) feature_absence = models.JSONField(default=dict, blank=True) created = models.DateTimeField(auto_now_add=True) class Meta: ordering = ["-created"] def __str__(self): return f"{self.label or self.source or self.id} ({self.frames}f @{self.fps})" class FootageFrame(models.Model): """One tracing still. A row rather than an entry in a JSON list, because a frame is a thing the server serves, and because a blob's references have to be countable before anything can be collected. A REFERENCE IMAGE, NOT A MEASUREMENT INPUT. See `Footage.video`.""" footage = models.ForeignKey(Footage, on_delete=models.CASCADE, related_name="frame_set") index = models.PositiveIntegerField(help_text="0-based; source frame index + 1 is the JPEG's name") blob = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="frame_for") class Meta: ordering = ["index"] constraints = [ models.UniqueConstraint(fields=["footage", "index"], name="one_blob_per_frame"), ] class Analysis(models.Model): """Tier 2: one detector, at one version, over one footage. `key` is a content address over every input, and `descriptor` is the exact canonical text that key is the sha256 of — sent by the client and stored, not recomputed here. `clips/views.py` says why that is the honest arrangement: JS prints an integral double as `1` and Python as `1.0`, so a scheme where both sides re-render the numbers breaks on the first one of them. `detector` and `version` are columns as well as descriptor fields so that the question "which model produced this take" is answerable in the admin and in a query, rather than only by parsing a hash's preimage. """ key = models.CharField(primary_key=True, max_length=71) descriptor = models.TextField() detector = models.CharField(max_length=64) version = models.CharField(max_length=64) footage = models.ForeignKey( Footage, null=True, blank=True, on_delete=models.SET_NULL, related_name="analyses" ) source_blocks = models.ManyToManyField( "Block", blank=True, related_name="source_for", help_text="pixel-dependent landmarks, detection mask and mouth crops", ) created = models.DateTimeField(auto_now_add=True) class Meta: verbose_name_plural = "analyses" def __str__(self): return f"{self.detector} {self.version} → {self.key[7:19]}…" class Block(models.Model): """Tier 2: one dense channel block. Two hashes, and they are not the same hash. `key` is over the block's INPUTS, which is what lets a client ask for the block its current settings want before anything has computed it. `data.digest` is over the bytes. See clips/blobs.py. """ key = models.CharField(primary_key=True, max_length=71) descriptor = models.TextField() role = models.CharField(max_length=32) analysis = models.ForeignKey( Analysis, null=True, blank=True, on_delete=models.SET_NULL, related_name="blocks" ) data = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="block_data_for") state = models.ForeignKey( Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="block_state_for", help_text="the per-track absence mask, when the take has one", ) created = models.DateTimeField(auto_now_add=True) def __str__(self): return f"{self.role} {self.key[7:19]}…" class Project(models.Model): """Tier 1: the document's root. `schema_version` identifies the stored document format. `seq` counts writes to this particular project; it is not a format version. Every write bumps `seq`, and a client that sees `seq > local + 1` refetches. ANYONE WITH THE LINK CAN VIEW; the owner and the editors can write. Every project has an owner. """ id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False) owner = models.ForeignKey( settings.AUTH_USER_MODEL, on_delete=models.CASCADE, related_name="projects", ) editors = models.ManyToManyField( settings.AUTH_USER_MODEL, blank=True, related_name="shared_projects", ) name = models.CharField(max_length=200, default="untitled") schema_version = models.PositiveIntegerField(default=3) seq = models.PositiveBigIntegerField(default=0) palette = models.CharField(max_length=64, default="arthur/default") created = models.DateTimeField(auto_now_add=True) updated = models.DateTimeField(auto_now=True) class Meta: ordering = ["-updated"] def __str__(self): return f"{self.name} ({self.id})" def bump(self): """The next seq, taken with an UPDATE so that inside a transaction it is also the write lock: two concurrent saves cannot both get the same one.""" Project.objects.filter(id=self.id).update( seq=models.F("seq") + 1, updated=timezone.now() ) self.refresh_from_db(fields=["seq", "updated"]) return self.seq def can_edit(self, user): return user.is_authenticated and ( user.id == self.owner_id or self.editors.filter(id=user.id).exists() ) class Clip(models.Model): """Tier 1: the unit of work, and the thing leaf paths are scoped by. `cid` is what appears in `clip//...`, so it is the clip's identity as far as addressing is concerned and it does not change. """ project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="clips") cid = models.SlugField(max_length=64) name = models.CharField(max_length=200, blank=True) order = models.IntegerField(default=0) footage = models.ForeignKey( Footage, null=True, blank=True, on_delete=models.SET_NULL, related_name="clips" ) analysis = models.ForeignKey( Analysis, null=True, blank=True, on_delete=models.SET_NULL, related_name="clips" ) blocks = models.ManyToManyField( Block, blank=True, related_name="clips", help_text="the tier-2 blocks this clip's channels name", ) class Meta: ordering = ["order", "cid"] constraints = [ models.UniqueConstraint(fields=["project", "cid"], name="one_cid_per_project"), ] def __str__(self): return f"{self.cid} of {self.project.name}" class Leaf(models.Model): """Tier 1: one independently addressed, independently versioned piece of the document. The value is transit-as-JSON in a JSONField, so the column holds JSON rather than a string containing JSON: the admin can read a leaf, and the field-wise merge of a channel leaf that docs/architecture.md describes as fifteen lines of Python is possible over it. `version` is the entity tag a conditional write compares — RFC 7232, not a bespoke invention. """ project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="leaves") path = models.CharField(max_length=300) value = models.JSONField() version = models.PositiveBigIntegerField(default=1) seq = models.PositiveBigIntegerField( default=0, help_text="the project seq of the write that last changed it", ) updated = models.DateTimeField(auto_now=True) class Meta: ordering = ["path"] constraints = [ models.UniqueConstraint(fields=["project", "path"], name="one_leaf_per_path"), ] @property def etag(self): return f'"{self.version}"' def __str__(self): return f"{self.path}@{self.version}" class Revision(models.Model): """Tier 1: a snapshot of the authored layer, with a user and a summary — a named snapshot, which is how a person marks a version now that every edit saves itself. ON AN EXPLICIT TRIGGER, not on every save. tl snapshots a small annotation layer; arthur's tier 1 will contain cel polygons, so a snapshot per save bloats the table — docs/architecture.md's "revisions need a coarser trigger". So this is written by `POST /api/projects//revisions`, which is a "mark version" button, and never by a save. """ project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="revisions") seq = models.PositiveBigIntegerField() author = models.CharField(max_length=200, blank=True) summary = models.CharField(max_length=500, blank=True) document = models.JSONField(help_text="every leaf of the project, by path") blocks = models.JSONField( default=dict, help_text="each clip's tier-2 block keys, by cid, so a restore can name them", ) created = models.DateTimeField(auto_now_add=True) class Meta: ordering = ["-seq"] def __str__(self): return f"{self.project.name} r{self.seq}: {self.summary}"