arthur/clips/blobs.py

130 lines
4.7 KiB
Python

"""The content-addressed blob store: tiers 2 and 3 on disk.
One store for both, and docs/architecture.md says why in a sentence: once tier 3
is decoded by the app rather than by a shell script, frames and audio become "the
same kind of thing as tier 2 — a cache with a hash". So there is one place that
writes bytes, one that reads them, and one URL shape for both.
TWO KINDS OF HASH, AND THEY ARE NOT THE SAME HASH. A blob is named by the sha256
of its BYTES: that is what makes identical frames in two extractions one file. A
derived thing — an analysis artifact, a dense block — is named by a sha256 over
its INPUTS, which is what lets the client ask for the block the current settings
want before anything has computed it. So `Block.key` is an input hash and
`Block.data.digest` is a byte hash, and conflating them would break the half of
addressing that answers questions about work not yet done.
"""
import hashlib
import os
import tempfile
from pathlib import Path
from django.conf import settings
CHUNK = 1 << 20
def digest_bytes(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()
def digest_file(path: Path) -> str:
h = hashlib.sha256()
with open(path, "rb") as fh:
while chunk := fh.read(CHUNK):
h.update(chunk)
return h.hexdigest()
def path_for(digest: str) -> Path:
"""Where a blob lives.
Fanned out two levels, so that a take's worth of frames does not put a hundred
thousand entries in one directory — which is slow on every filesystem and
unusable on some.
"""
if len(digest) != 64 or any(c not in "0123456789abcdef" for c in digest):
raise ValueError(f"not a sha256: {digest!r}")
return Path(settings.BLOB_ROOT) / digest[:2] / digest[2:4] / digest
def write(data: bytes) -> tuple[str, int]:
"""Store bytes, return (digest, size). Writing the same bytes twice is a
no-op, which is what content addressing is for."""
digest = digest_bytes(data)
dest = path_for(digest)
if not dest.exists():
dest.parent.mkdir(parents=True, exist_ok=True)
tmp = dest.with_suffix(".part")
with open(tmp, "wb") as fh:
fh.write(data)
os.replace(tmp, dest)
return digest, len(data)
def write_stream(chunks) -> tuple[str, int]:
"""Store an uploaded file without reading the whole video into memory."""
root = Path(settings.BLOB_ROOT)
root.mkdir(parents=True, exist_ok=True)
digest = hashlib.sha256()
size = 0
with tempfile.NamedTemporaryFile(dir=root, prefix="upload-", delete=False) as out:
temporary = Path(out.name)
try:
for chunk in chunks:
digest.update(chunk)
size += len(chunk)
out.write(chunk)
except BaseException:
temporary.unlink(missing_ok=True)
raise
dest = path_for(digest.hexdigest())
dest.parent.mkdir(parents=True, exist_ok=True)
if dest.exists():
temporary.unlink()
else:
os.replace(temporary, dest)
return digest.hexdigest(), size
def adopt(source: Path) -> tuple[str, int]:
"""Store a file already on disk, by hard link where the filesystem allows it.
112MB of PNGs is a normal extraction and copying them into a second place in
the tree for no reason is not. A hard link is exact — the blob is immutable, so
two names for one inode is the whole of what is wanted — and a copy is the
fallback when `extract.sh` wrote to another volume.
"""
digest = digest_file(source)
dest = path_for(digest)
size = source.stat().st_size
if not dest.exists():
dest.parent.mkdir(parents=True, exist_ok=True)
try:
os.link(source, dest)
except OSError:
tmp = dest.with_suffix(".part")
with open(source, "rb") as src, open(tmp, "wb") as out:
while chunk := src.read(CHUNK):
out.write(chunk)
os.replace(tmp, dest)
return digest, size
def read(digest: str) -> bytes:
with open(path_for(digest), "rb") as fh:
return fh.read()
def png_size(path: Path) -> tuple[int, int]:
"""A PNG's dimensions, out of its IHDR.
Twenty-four bytes rather than a dependency. The footage's width and height are
manifest data — docs/architecture.md's entity model puts them there — and
Pillow to read two integers out of a header that has held them in the same
place since 1996 is not a trade worth making.
"""
with open(path, "rb") as fh:
head = fh.read(24)
if head[:8] != b"\x89PNG\r\n\x1a\n" or head[12:16] != b"IHDR":
raise ValueError(f"{path} is not a PNG")
return int.from_bytes(head[16:20], "big"), int.from_bytes(head[20:24], "big")