146 lines
5.1 KiB
Python
146 lines
5.1 KiB
Python
"""The content-addressed blob store: tiers 2 and 3 on disk.
|
|
|
|
One store for both, and docs/architecture.md says why in a sentence: once tier 3
|
|
is decoded by the app rather than by a shell script, frames and audio become "the
|
|
same kind of thing as tier 2 — a cache with a hash". So there is one place that
|
|
writes bytes, one that reads them, and one URL shape for both.
|
|
|
|
TWO KINDS OF HASH, AND THEY ARE NOT THE SAME HASH. A blob is named by the sha256
|
|
of its BYTES: that is what makes identical frames in two extractions one file. A
|
|
derived thing — an analysis artifact, a dense block — is named by a sha256 over
|
|
its INPUTS, which is what lets the client ask for the block the current settings
|
|
want before anything has computed it. So `Block.key` is an input hash and
|
|
`Block.data.digest` is a byte hash, and conflating them would break the half of
|
|
addressing that answers questions about work not yet done.
|
|
"""
|
|
import hashlib
|
|
import os
|
|
import tempfile
|
|
import zlib
|
|
from pathlib import Path
|
|
|
|
from django.conf import settings
|
|
|
|
CHUNK = 1 << 20
|
|
CROP_MEDIA_TYPE = "application/zlib"
|
|
|
|
|
|
def digest_bytes(data: bytes) -> str:
|
|
return hashlib.sha256(data).hexdigest()
|
|
|
|
|
|
def digest_file(path: Path) -> str:
|
|
h = hashlib.sha256()
|
|
with open(path, "rb") as fh:
|
|
while chunk := fh.read(CHUNK):
|
|
h.update(chunk)
|
|
return h.hexdigest()
|
|
|
|
|
|
def path_for(digest: str) -> Path:
|
|
"""Where a blob lives.
|
|
|
|
Fanned out two levels, so that a take's worth of frames does not put a hundred
|
|
thousand entries in one directory — which is slow on every filesystem and
|
|
unusable on some.
|
|
"""
|
|
if len(digest) != 64 or any(c not in "0123456789abcdef" for c in digest):
|
|
raise ValueError(f"not a sha256: {digest!r}")
|
|
return Path(settings.BLOB_ROOT) / digest[:2] / digest[2:4] / digest
|
|
|
|
|
|
def write(data: bytes) -> tuple[str, int]:
|
|
"""Store bytes, return (digest, size). Writing the same bytes twice is a
|
|
no-op, which is what content addressing is for."""
|
|
digest = digest_bytes(data)
|
|
dest = path_for(digest)
|
|
if not dest.exists():
|
|
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
tmp = dest.with_suffix(".part")
|
|
with open(tmp, "wb") as fh:
|
|
fh.write(data)
|
|
os.replace(tmp, dest)
|
|
return digest, len(data)
|
|
|
|
|
|
def write_stream(chunks) -> tuple[str, int]:
|
|
"""Store an uploaded file without reading the whole video into memory."""
|
|
root = Path(settings.BLOB_ROOT)
|
|
root.mkdir(parents=True, exist_ok=True)
|
|
digest = hashlib.sha256()
|
|
size = 0
|
|
with tempfile.NamedTemporaryFile(dir=root, prefix="upload-", delete=False) as out:
|
|
temporary = Path(out.name)
|
|
try:
|
|
for chunk in chunks:
|
|
digest.update(chunk)
|
|
size += len(chunk)
|
|
out.write(chunk)
|
|
except BaseException:
|
|
temporary.unlink(missing_ok=True)
|
|
raise
|
|
dest = path_for(digest.hexdigest())
|
|
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
if dest.exists():
|
|
temporary.unlink()
|
|
else:
|
|
os.replace(temporary, dest)
|
|
return digest.hexdigest(), size
|
|
|
|
|
|
def write_compressed_stream(chunks) -> tuple[str, int]:
|
|
"""Store a losslessly compressed stream; the digest names stored bytes."""
|
|
compressor = zlib.compressobj()
|
|
|
|
def compressed():
|
|
for chunk in chunks:
|
|
if part := compressor.compress(chunk):
|
|
yield part
|
|
if part := compressor.flush():
|
|
yield part
|
|
|
|
return write_stream(compressed())
|
|
|
|
|
|
def adopt(source: Path) -> tuple[str, int]:
|
|
"""Store a file already on disk, by hard link where the filesystem allows it.
|
|
|
|
112MB of PNGs is a normal extraction and copying them into a second place in
|
|
the tree for no reason is not. A hard link is exact — the blob is immutable, so
|
|
two names for one inode is the whole of what is wanted — and a copy is the
|
|
fallback when `extract.sh` wrote to another volume.
|
|
"""
|
|
digest = digest_file(source)
|
|
dest = path_for(digest)
|
|
size = source.stat().st_size
|
|
if not dest.exists():
|
|
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
try:
|
|
os.link(source, dest)
|
|
except OSError:
|
|
tmp = dest.with_suffix(".part")
|
|
with open(source, "rb") as src, open(tmp, "wb") as out:
|
|
while chunk := src.read(CHUNK):
|
|
out.write(chunk)
|
|
os.replace(tmp, dest)
|
|
return digest, size
|
|
|
|
|
|
def read(digest: str) -> bytes:
|
|
with open(path_for(digest), "rb") as fh:
|
|
return fh.read()
|
|
|
|
|
|
def png_size(path: Path) -> tuple[int, int]:
|
|
"""A PNG's dimensions, out of its IHDR.
|
|
|
|
Twenty-four bytes rather than a dependency. The footage's width and height are
|
|
manifest data — docs/architecture.md's entity model puts them there — and
|
|
Pillow to read two integers out of a header that has held them in the same
|
|
place since 1996 is not a trade worth making.
|
|
"""
|
|
with open(path, "rb") as fh:
|
|
head = fh.read(24)
|
|
if head[:8] != b"\x89PNG\r\n\x1a\n" or head[12:16] != b"IHDR":
|
|
raise ValueError(f"{path} is not a PNG")
|
|
return int.from_bytes(head[16:20], "big"), int.from_bytes(head[20:24], "big")
|