351 lines
14 KiB
Python
351 lines
14 KiB
Python
"""The entity model, as tables.
|
|
|
|
It follows docs/architecture.md's model exactly, and the one thing worth reading
|
|
it for is which tier each table is in, because that is what decides whether a row
|
|
is a document, a cache entry or a source.
|
|
|
|
TIER 1, the document. Project, Clip, Leaf, Revision. Kilobytes, authored,
|
|
versioned, and the only tier anything will ever sync.
|
|
|
|
TIER 2, derived. Analysis, Block. Content-addressed by a hash over every input
|
|
that produced them — including the detector version — so a stale bake is
|
|
unreachable rather than wrong, and a collaborator's bake is fetchable by the
|
|
same key.
|
|
|
|
TIER 3, source. Footage, FootageFrame. Immutable, by hash.
|
|
|
|
Blob is under all three of them: bytes, named by the sha256 of themselves.
|
|
|
|
WHAT IS DELIBERATELY NOT HERE. `Clip` does not store fps, frames, width or height.
|
|
They are in the document — the `timing` and `stage` leaves — and a copy of them in
|
|
a column is a copy that comes to disagree with the scene it describes. The columns
|
|
`Clip` does have are the ones the SERVER needs to answer a question about a clip
|
|
without parsing its leaves: which footage, which analysis, which blocks.
|
|
"""
|
|
import uuid
|
|
|
|
from django.conf import settings
|
|
from django.db import models
|
|
from django.utils import timezone
|
|
|
|
|
|
class Blob(models.Model):
|
|
"""Bytes, named by the sha256 of themselves. The file is on disk under
|
|
`BLOB_ROOT`; this row is the index and the size."""
|
|
|
|
digest = models.CharField(primary_key=True, max_length=64)
|
|
media_type = models.CharField(max_length=100, default="application/octet-stream")
|
|
size = models.BigIntegerField()
|
|
created = models.DateTimeField(auto_now_add=True)
|
|
|
|
def __str__(self):
|
|
return f"{self.digest[:12]}… {self.size}B {self.media_type}"
|
|
|
|
|
|
class Source(models.Model):
|
|
"""An uploaded video, identified by its byte digest."""
|
|
|
|
id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False)
|
|
blob = models.OneToOneField(Blob, on_delete=models.PROTECT, related_name="video_source")
|
|
filename = models.CharField(max_length=255)
|
|
probe = models.JSONField(default=dict)
|
|
created = models.DateTimeField(auto_now_add=True)
|
|
|
|
|
|
class Sound(models.Model):
|
|
"""An uploaded sound file — mp3, wav, whatever the browser can decode — kept
|
|
as uploaded. Not footage: it has no frames and nothing measures it, so it
|
|
skips extraction and an audio node plays its bytes directly."""
|
|
|
|
id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False)
|
|
blob = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="sound_for")
|
|
filename = models.CharField(max_length=255)
|
|
duration = models.FloatField(help_text="seconds, as ffprobe reports it")
|
|
created = models.DateTimeField(auto_now_add=True)
|
|
|
|
|
|
class Extraction(models.Model):
|
|
"""One requested decode of a source into immutable footage."""
|
|
|
|
key = models.CharField(primary_key=True, max_length=71)
|
|
source = models.ForeignKey(Source, on_delete=models.CASCADE, related_name="extractions")
|
|
settings = models.JSONField(default=dict)
|
|
state = models.CharField(max_length=16, default="queued")
|
|
progress = models.PositiveIntegerField(default=0)
|
|
error = models.TextField(blank=True)
|
|
footage = models.ForeignKey(
|
|
"Footage", null=True, blank=True, on_delete=models.SET_NULL,
|
|
related_name="extractions",
|
|
)
|
|
created = models.DateTimeField(auto_now_add=True)
|
|
updated = models.DateTimeField(auto_now=True)
|
|
|
|
|
|
class Footage(models.Model):
|
|
"""Tier 3: the frames and audio of one extraction, immutable.
|
|
|
|
`digest` is over the PROXY VIDEO's digest plus the audio's and the rate, so
|
|
two extractions of the same clip at the same settings are one footage and the
|
|
same analysis can be reused across both.
|
|
|
|
THE PROXY IS THE ANALYSIS SOURCE AND THE FRAMES ARE NOT. `video` is one
|
|
browser-safe H.264 file, and it is what the page seeks through to detect
|
|
landmarks. `frame_set` is a JPEG per frame at tracing size: reference stills
|
|
for the tracing editor, never the thing measured. The two are not
|
|
interchangeable, and which one carries the pixels an analysis was computed
|
|
from is the difference between a 6MB take and a 1.1GB one.
|
|
|
|
So the frame JPEGs are deliberately NOT in `digest`. They are a rendering of
|
|
this footage for a human to trace over; re-rendering them at another size does
|
|
not make it different footage, and putting them in the identity would throw
|
|
away every analysis when the tracing size changed.
|
|
|
|
`feature_absence` is the manifest annotation step 8 introduced: known
|
|
occlusion intervals, one-based and inclusive, expanded into presence tracks by
|
|
the loader. An input format, not a control UI.
|
|
"""
|
|
|
|
id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False)
|
|
digest = models.CharField(max_length=64, unique=True)
|
|
label = models.CharField(max_length=200, blank=True)
|
|
source = models.CharField(max_length=200, blank=True)
|
|
fps = models.FloatField()
|
|
frames = models.PositiveIntegerField()
|
|
width = models.PositiveIntegerField()
|
|
height = models.PositiveIntegerField()
|
|
audio = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="audio_for")
|
|
video = models.ForeignKey(
|
|
Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="video_for",
|
|
help_text="the browser-safe proxy, playable and seekable",
|
|
)
|
|
stream = models.ForeignKey(
|
|
Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="stream_for",
|
|
help_text="the proxy's video as raw Annex-B H.264: what the page DECODES, "
|
|
"one access unit per frame; null on footage extracted before it",
|
|
)
|
|
feature_absence = models.JSONField(default=dict, blank=True)
|
|
created = models.DateTimeField(auto_now_add=True)
|
|
|
|
class Meta:
|
|
ordering = ["-created"]
|
|
|
|
def __str__(self):
|
|
return f"{self.label or self.source or self.id} ({self.frames}f @{self.fps})"
|
|
|
|
|
|
class FootageFrame(models.Model):
|
|
"""One tracing still. A row rather than an entry in a JSON list, because a
|
|
frame is a thing the server serves, and because a blob's references have to be
|
|
countable before anything can be collected.
|
|
|
|
A REFERENCE IMAGE, NOT A MEASUREMENT INPUT. See `Footage.video`."""
|
|
|
|
footage = models.ForeignKey(Footage, on_delete=models.CASCADE, related_name="frame_set")
|
|
index = models.PositiveIntegerField(help_text="0-based; source frame index + 1 is the JPEG's name")
|
|
blob = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="frame_for")
|
|
|
|
class Meta:
|
|
ordering = ["index"]
|
|
constraints = [
|
|
models.UniqueConstraint(fields=["footage", "index"], name="one_blob_per_frame"),
|
|
]
|
|
|
|
|
|
class Analysis(models.Model):
|
|
"""Tier 2: one detector, at one version, over one footage.
|
|
|
|
`key` is a content address over every input, and `descriptor` is the exact
|
|
canonical text that key is the sha256 of — sent by the client and stored, not
|
|
recomputed here. `clips/views.py` says why that is the honest arrangement: JS
|
|
prints an integral double as `1` and Python as `1.0`, so a scheme where both
|
|
sides re-render the numbers breaks on the first one of them.
|
|
|
|
`detector` and `version` are columns as well as descriptor fields so that the
|
|
question "which model produced this take" is answerable in the admin and in a
|
|
query, rather than only by parsing a hash's preimage.
|
|
"""
|
|
|
|
key = models.CharField(primary_key=True, max_length=71)
|
|
descriptor = models.TextField()
|
|
detector = models.CharField(max_length=64)
|
|
version = models.CharField(max_length=64)
|
|
footage = models.ForeignKey(
|
|
Footage, null=True, blank=True, on_delete=models.SET_NULL, related_name="analyses"
|
|
)
|
|
source_blocks = models.ManyToManyField(
|
|
"Block", blank=True, related_name="source_for",
|
|
help_text="pixel-dependent landmarks, detection mask and mouth crops",
|
|
)
|
|
created = models.DateTimeField(auto_now_add=True)
|
|
|
|
class Meta:
|
|
verbose_name_plural = "analyses"
|
|
|
|
def __str__(self):
|
|
return f"{self.detector} {self.version} → {self.key[7:19]}…"
|
|
|
|
|
|
class Block(models.Model):
|
|
"""Tier 2: one dense channel block.
|
|
|
|
Two hashes, and they are not the same hash. `key` is over the block's INPUTS,
|
|
which is what lets a client ask for the block its current settings want before
|
|
anything has computed it. `data.digest` is over the bytes. See clips/blobs.py.
|
|
"""
|
|
|
|
key = models.CharField(primary_key=True, max_length=71)
|
|
descriptor = models.TextField()
|
|
role = models.CharField(max_length=32)
|
|
analysis = models.ForeignKey(
|
|
Analysis, null=True, blank=True, on_delete=models.SET_NULL, related_name="blocks"
|
|
)
|
|
data = models.ForeignKey(Blob, on_delete=models.PROTECT, related_name="block_data_for")
|
|
state = models.ForeignKey(
|
|
Blob, null=True, blank=True, on_delete=models.PROTECT, related_name="block_state_for",
|
|
help_text="the per-track absence mask, when the take has one",
|
|
)
|
|
created = models.DateTimeField(auto_now_add=True)
|
|
|
|
def __str__(self):
|
|
return f"{self.role} {self.key[7:19]}…"
|
|
|
|
|
|
class Project(models.Model):
|
|
"""Tier 1: the document's root.
|
|
|
|
`schema_version` identifies the stored document format. `seq` counts writes
|
|
to this particular project; it is not a format version. Every write bumps
|
|
`seq`, and a client that sees `seq > local + 1` refetches.
|
|
|
|
ANYONE WITH THE LINK CAN VIEW; the owner and the editors can write. Every
|
|
project has an owner.
|
|
"""
|
|
|
|
id = models.UUIDField(primary_key=True, default=uuid.uuid4, editable=False)
|
|
owner = models.ForeignKey(
|
|
settings.AUTH_USER_MODEL, on_delete=models.CASCADE, related_name="projects",
|
|
)
|
|
editors = models.ManyToManyField(
|
|
settings.AUTH_USER_MODEL, blank=True, related_name="shared_projects",
|
|
)
|
|
name = models.CharField(max_length=200, default="untitled")
|
|
schema_version = models.PositiveIntegerField(default=2)
|
|
seq = models.PositiveBigIntegerField(default=0)
|
|
palette = models.CharField(max_length=64, default="arthur/default")
|
|
created = models.DateTimeField(auto_now_add=True)
|
|
updated = models.DateTimeField(auto_now=True)
|
|
|
|
class Meta:
|
|
ordering = ["-updated"]
|
|
|
|
def __str__(self):
|
|
return f"{self.name} ({self.id})"
|
|
|
|
def bump(self):
|
|
"""The next seq, taken with an UPDATE so that inside a transaction it is
|
|
also the write lock: two concurrent saves cannot both get the same one."""
|
|
Project.objects.filter(id=self.id).update(
|
|
seq=models.F("seq") + 1, updated=timezone.now()
|
|
)
|
|
self.refresh_from_db(fields=["seq", "updated"])
|
|
return self.seq
|
|
|
|
def can_edit(self, user):
|
|
return user.is_authenticated and (
|
|
user.id == self.owner_id or self.editors.filter(id=user.id).exists()
|
|
)
|
|
|
|
|
|
class Clip(models.Model):
|
|
"""Tier 1: the unit of work, and the thing leaf paths are scoped by.
|
|
|
|
`cid` is what appears in `clip/<cid>/...`, so it is the clip's identity as far
|
|
as addressing is concerned and it does not change.
|
|
"""
|
|
|
|
project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="clips")
|
|
cid = models.SlugField(max_length=64)
|
|
name = models.CharField(max_length=200, blank=True)
|
|
order = models.IntegerField(default=0)
|
|
footage = models.ForeignKey(
|
|
Footage, null=True, blank=True, on_delete=models.SET_NULL, related_name="clips"
|
|
)
|
|
analysis = models.ForeignKey(
|
|
Analysis, null=True, blank=True, on_delete=models.SET_NULL, related_name="clips"
|
|
)
|
|
blocks = models.ManyToManyField(
|
|
Block, blank=True, related_name="clips",
|
|
help_text="the tier-2 blocks this clip's channels name",
|
|
)
|
|
|
|
class Meta:
|
|
ordering = ["order", "cid"]
|
|
constraints = [
|
|
models.UniqueConstraint(fields=["project", "cid"], name="one_cid_per_project"),
|
|
]
|
|
|
|
def __str__(self):
|
|
return f"{self.cid} of {self.project.name}"
|
|
|
|
|
|
class Leaf(models.Model):
|
|
"""Tier 1: one independently addressed, independently versioned piece of the
|
|
document.
|
|
|
|
The value is transit-as-JSON in a JSONField, so the column holds JSON rather
|
|
than a string containing JSON: the admin can read a leaf, and the field-wise
|
|
merge of a channel leaf that docs/architecture.md describes as fifteen lines of
|
|
Python is possible over it. `version` is the entity tag a conditional write
|
|
compares — RFC 7232, not a bespoke invention.
|
|
"""
|
|
|
|
project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="leaves")
|
|
path = models.CharField(max_length=300)
|
|
value = models.JSONField()
|
|
version = models.PositiveBigIntegerField(default=1)
|
|
seq = models.PositiveBigIntegerField(
|
|
default=0, help_text="the project seq of the write that last changed it",
|
|
)
|
|
updated = models.DateTimeField(auto_now=True)
|
|
|
|
class Meta:
|
|
ordering = ["path"]
|
|
constraints = [
|
|
models.UniqueConstraint(fields=["project", "path"], name="one_leaf_per_path"),
|
|
]
|
|
|
|
@property
|
|
def etag(self):
|
|
return f'"{self.version}"'
|
|
|
|
def __str__(self):
|
|
return f"{self.path}@{self.version}"
|
|
|
|
|
|
class Revision(models.Model):
|
|
"""Tier 1: a snapshot of the authored layer, with a user and a summary — a
|
|
named snapshot, which is how a person marks a version now that every edit
|
|
saves itself.
|
|
|
|
ON AN EXPLICIT TRIGGER, not on every save. tl snapshots a small annotation
|
|
layer; arthur's tier 1 will contain cel polygons, so a snapshot per save bloats
|
|
the table — docs/architecture.md's "revisions need a coarser trigger". So this
|
|
is written by `POST /api/projects/<id>/revisions`, which is a "mark version"
|
|
button, and never by a save.
|
|
"""
|
|
|
|
project = models.ForeignKey(Project, on_delete=models.CASCADE, related_name="revisions")
|
|
seq = models.PositiveBigIntegerField()
|
|
author = models.CharField(max_length=200, blank=True)
|
|
summary = models.CharField(max_length=500, blank=True)
|
|
document = models.JSONField(help_text="every leaf of the project, by path")
|
|
blocks = models.JSONField(
|
|
default=dict, help_text="each clip's tier-2 block keys, by cid, so a restore can name them",
|
|
)
|
|
created = models.DateTimeField(auto_now_add=True)
|
|
|
|
class Meta:
|
|
ordering = ["-seq"]
|
|
|
|
def __str__(self):
|
|
return f"{self.project.name} r{self.seq}: {self.summary}"
|