arthur/js/app.js
Olive Vaughn 35ef150b48 Give features stable identity, eye pairs and per-feature presence
Step 8's data model, ahead of its controls. Nothing here is a UI.

domain/params holds every knob's definition once — default, applicable area,
value constraints and the areas a change would force to regenerate. flow/take's
literal knob map becomes a view of it, so the take's defaults and the future
parameter panel cannot drift apart.

domain/feature adds subjects, features and groups as document data the renderer
never reads. A feature ID is stable for the whole clip, across occlusion: a run
of visible frames is not a new identity. An eye pair is an explicit group of one
or two eyes of the same subject, so a profile view with one identified eye needs
no invented partner. Settings resolve area -> subject -> group -> feature, and
dropping an eye from a pair materialises its effective values first so playback
does not jump. scene/problems now validates all of it.

Presence becomes per-feature rather than per-subject. freeze's :absent predicate
takes a track as well as a frame, so one occluded eye can be absent while its
partner still has a value; a full-face miss still marks everything absent. A
manifest may annotate known gaps as one-based inclusive intervals, which ingest
expands into observation tracks before measurement. An unobserved eye then gets
no vote in the iris pairing and cannot steer the shared gaze — gaze falls back to
whichever eye is visible. Temporal filters still see a sample on every frame,
held from the last observed one, because the numbers are a rectangular buffer;
the state mask, not the buffer, is what says the frame has no value.

js/app.js gets the same occlusion lesson: leading nulls from a face that starts
occluded used to throw away the whole take, and the neutral frame could be chosen
from a held duplicate pose.

Parameter editing, scoped regeneration and a feature-level detector remain. Until
one exists, footage without annotations falls back to the full-face mask rather
than claiming occlusions it cannot see.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01B87NVmiU36qQmN9gmFYnJ9
2026-09-27 22:44:36 -04:00

1459 lines
62 KiB
JavaScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import { FaceLandmarker, FilesetResolver } from 'https://cdn.jsdelivr.net/npm/@mediapipe/tasks-vision@1.0.1/vision_bundle.mjs';
import { LIPS_OUTER, LIPS_INNER, FACE_OVAL,
EYE_R_RING, EYE_L_RING, IRIS_A, IRIS_B,
BROW_A_RING, BROW_B_RING } from './landmarks.js';
import { stabilize, toRasterRing, smoothContours, suggestPlateFrames, heldFrame, shiftIndex,
exposeIndex, eyeSignals, gazeOrigin, quantizeSnap, resolveBlink,
browSignals } from './pipeline.js';
import { IndexedRaster } from './raster.js';
import { drawRegistered, posterizeInto } from './underlay.js';
import { extractTeeth } from './interior.js';
import { applySim, offsetRing } from './mathutil.js';
import { writeTake } from './take.js';
import { synthDense } from './synth.js';
import { PaintUI, drawCel, cloneCel, newCel } from './paint.js';
const RW = 320, RH = 200, ZOOM = 2, THUMB = 92;
const PALETTE = [
{ name: 'bg', hex: '#12141c' },
{ name: 'skin_base', hex: '#b07a5a' },
{ name: 'skin_dark', hex: '#7a4f3a' },
{ name: 'mouth_dark', hex: '#24161a' },
{ name: 'teeth', hex: '#d9cfc2' },
// Sclera is not white, and that is authored, not measured. A true white at
// 320x200 next to a warm skin ramp reads as a hole punched in the face; the
// eye sits in a socket, in shadow, so it is a dimmer and cooler tone than the
// teeth, which catch the light. The iris is one dark tone: at this size an
// iris is about five pixels across and a pupil inside it would be one, so the
// iris IS the pupil. Resolving it further would be drawing detail the format
// cannot hold.
{ name: 'eye_white', hex: '#c9c3b4' },
// Three tones for the eye - sclera, iris, pupil - which is the "two or three
// tones per part" budget, spent where it buys the most: an eye with no tonal
// step inside it reads as a hole.
{ name: 'iris', hex: '#4a5468' },
{ name: 'pupil', hex: '#171a22' },
// Brows get their own entry rather than sharing skin_dark with the lash line.
// They are hair, not shadow: when hair plates exist they want to match those,
// and tying them to the lash means you cannot change one without the other.
{ name: 'brow', hex: '#3a2a22' },
];
const IDX = { bg: 0, base: 1, dark: 2, mouth: 3, teeth: 4, white: 5, iris: 6, pupil: 7, brow: 8 };
const state = {
dense: null, images: [], stab: null, xform: null,
outer: null, inner: null, plates: null, hidden: null,
keep: new Set(), // frames that get their own plate drawing
frame: 0, playing: false, faceBox: null,
fps: 12, audio: null, // fps comes from manifest.json, never guessed
aspect: 1, // imgW/imgH; converts MediaPipe's anisotropic space
lead: 0, // performance-track offset in frames
exposure: 1, // 1 = on 1s, 2 = on 2s. Picture holds; audio does not.
interior: null, // per-frame teeth measurement from image content
detected: null, // per-frame: did MediaPipe really see a face here?
teeth: null, // resolved per-frame {show, t} after knobs
eyes: null, // resolved per-frame lid rings, shut flags, iris discs
brows: null, // resolved per-frame brow rings after quantised raise
cels: new Map(), // kept frame -> hand-painted background layers
eyeSig: null, // raw eye measurement, kept for the gaze readout
};
const el = (id) => {
const n = document.getElementById(id);
// A knob present in the code but missing from the markup used to throw during
// wiring and leave a blank page with nothing in the console worth reading.
if (!n) throw new Error(`missing element #${id} — knob wired in app.js but not in index.html`);
return n;
};
const opts = () => ({
verts: +el('verts').value,
lead: +el('lead').value,
teethOn: +el('teethOn').value / 100, // minimum Otsu class separation
teethDwell: +el('teethDwell').value,
teethSmooth: +el('teethSmooth').value,
cavityErode: +el('teethErode').value / 100,
tongueReject: +el('tongueReject').value / 100,
blobGrow: +el('blobGrow').value,
topBias: +el('topBias').value / 100,
teethVerts: +el('teethVerts').value,
smoothWin: +el('smoothWin').value,
contourSmooth: +el('contourSmooth').value,
apertureThresh: +el('apertureThresh').value / 1000,
tol: +el('tol').value / 1000,
exposure: +el('exposure').value,
browVerts: +el('browVerts').value,
browWeight: +el('browWeight').value,
browGain: +el('browGain').value / 100,
browStep: +el('browStep').value,
browDwell: +el('browDwell').value,
irisAnchor: el('irisAnchor').value,
gazeOrigin: el('gazeOrigin').value,
eyeVerts: +el('eyeVerts').value,
lashPx: +el('lashPx').value,
irisSize: +el('irisSize').value / 100,
gazeGain: +el('gazeGain').value / 100,
gazeStep: +el('gazeStep').value, // whole raster pixels
pupilPx: +el('pupilPx').value,
gazeDwell: +el('gazeDwell').value,
blinkCut: +el('blinkCut').value / 1000,
blinkHold: +el('blinkHold').value,
blinkDwell: +el('blinkDwell').value,
});
function status(msg, kind = '') {
el('status').textContent = msg;
el('status').className = kind;
}
/* ---------- loading ---------- */
const loadImage = (src) => new Promise((r) => {
const im = new Image();
im.onload = () => r(im); im.onerror = () => r(null); im.src = src;
});
// The extraction rate is read, not assumed. Guessing it would desynchronise
// audio from picture, which is the one thing this view exists to show.
async function loadManifest() {
try {
const r = await fetch('./manifest.json', { cache: 'no-store' });
if (!r.ok) return null;
return await r.json();
} catch { return null; }
}
function attachAudio(name) {
const a = el('audio');
if (!name) { a.removeAttribute('src'); a.hidden = true; state.audio = null; return; }
a.src = './' + name;
a.hidden = false;
state.audio = a;
}
async function loadFrameSequence() {
const dir = el('framedir').value.replace(/\/$/, '');
const imgs = [];
for (let i = 1; i <= 900; i++) {
const im = await loadImage(`${dir}/${String(i).padStart(4, '0')}.png`);
if (!im) break;
imgs.push(im);
if (i % 10 === 0) status(`loading frames… ${i}`);
}
return imgs;
}
let landmarker = null;
async function initLandmarker() {
if (landmarker) return landmarker;
status('loading MediaPipe wasm…');
const fileset = await FilesetResolver.forVisionTasks(
'https://cdn.jsdelivr.net/npm/@mediapipe/tasks-vision@1.0.1/wasm');
// GPU is faster but unavailable in some contexts; fall back rather than fail.
for (const delegate of ['GPU', 'CPU']) {
try {
landmarker = await FaceLandmarker.createFromOptions(fileset, {
baseOptions: { modelAssetPath: './face_landmarker.task', delegate },
runningMode: 'IMAGE', numFaces: 1,
});
status(`landmarker ready (${delegate})`);
return landmarker;
} catch (e) {
if (delegate === 'CPU') throw e;
console.warn('GPU delegate failed, falling back to CPU:', e.message);
}
}
}
async function detectAll(images) {
const lm = await initLandmarker();
const cv = document.createElement('canvas');
const dense = [], missing = [], detected = [];
for (let i = 0; i < images.length; i++) {
const im = images[i];
cv.width = im.naturalWidth; cv.height = im.naturalHeight;
cv.getContext('2d').drawImage(im, 0, 0);
const out = lm.detect(cv);
if (out.faceLandmarks && out.faceLandmarks.length) {
dense.push(out.faceLandmarks[0]); detected.push(true);
} else {
missing.push(i); detected.push(false);
dense.push(dense.length ? dense[dense.length - 1] : null);
}
if (i % 4 === 0) status(`detecting… ${i + 1}/${images.length}`);
}
// A gap mid-take holds the previous frame, but a gap at the TOP has no
// previous to hold - a face that starts occluded or walks in late left
// leading nulls, and this used to throw and discard the whole take. Back-fill
// from the first real detection: the mirror of hold-previous, and the only
// fill that is a real pose from this take rather than an invention.
const firstReal = dense.findIndex((d) => d !== null);
if (firstReal < 0) throw new Error('no face found in any frame — check framing and light');
for (let i = 0; i < firstReal; i++) dense[i] = dense[firstReal];
return { dense, missing, detected, firstReal };
}
// Interior measurement is a function of pixels alone, so it runs once with
// detection and the knobs re-resolve it instantly afterwards.
function measureAll(images, dense, o) {
const ctx = document.createElement('canvas').getContext('2d', { willReadFrequently: true });
return dense.map((lm, i) =>
extractTeeth(images[i], LIPS_INNER.map((k) => lm[k]), ctx, o));
}
// Extraction keys on every knob that changes the pixels examined, so the cache
// is keyed on exactly those and a change to anything else stays instant.
const extractKey = (o) =>
[o.cavityErode, o.tongueReject, o.blobGrow, o.topBias, o.teethVerts].join('|');
/* ---------- build ---------- */
function makeXform(stab, neutral) {
const oval = stab.oval[neutral];
let x0 = Infinity, y0 = Infinity, x1 = -Infinity, y1 = -Infinity;
for (const p of oval) {
x0 = Math.min(x0, p.x); y0 = Math.min(y0, p.y);
x1 = Math.max(x1, p.x); y1 = Math.max(y1, p.y);
}
const s = (RH * 0.80) / (y1 - y0);
const cx = (x0 + x1) / 2, cy = (y0 + y1) / 2;
return (p) => ({ x: (p.x - cx) * s + RW / 2, y: (p.y - cy) * s + RH / 2 });
}
function rebuild(resetKeep) {
if (!state.dense) return;
const o = opts();
state.lead = o.lead;
state.exposure = o.exposure;
const N = state.dense.length;
state.stab = stabilize(state.dense, o.smoothWin, state.aspect);
// The neutral drives calibration and the placeholder plate, so it has to be a
// frame the camera actually saw: a back-filled or held frame is a duplicate
// pose, and letting one win this contest would calibrate the whole take
// against a landmark set that belongs to some other moment.
const ap = state.stab.aperture;
const real = state.detected;
const shut = (a, b) => (b < 0 || ap[a] < ap[b] ? a : b);
const head = Math.max(1, Math.floor(N / 4));
let neutral = -1;
for (let i = 0; i < head; i++) if (!real || real[i]) neutral = shut(i, neutral);
// Whole head of the take occluded: widen to any real frame rather than give up.
if (neutral < 0) for (let i = 0; i < N; i++) if (!real || real[i]) neutral = shut(i, neutral);
if (neutral < 0) neutral = 0;
state.neutral = neutral;
state.xform = makeXform(state.stab, neutral);
// Mouth is traced, so it costs nothing: a key on EVERY frame. Only the plate,
// which a human draws, gets decimated.
state.outer = smoothContours(
state.stab.outer.map((r) => toRasterRing(r, LIPS_OUTER, o.verts, state.xform)), o.contourSmooth);
state.inner = smoothContours(
state.stab.inner.map((r) => toRasterRing(r, LIPS_INNER, o.verts, state.xform)), o.contourSmooth);
const apMax = Math.max(...ap);
state.hidden = ap.map((v) => v / apMax < o.apertureThresh);
if (state.images.length && state.extractKey !== extractKey(o)) {
state.interior = measureAll(state.images, state.dense, o);
state.extractKey = extractKey(o);
}
state.teeth = resolveTeeth(o);
state.eyes = buildEyes(o);
state.brows = buildBrows(o);
// Plate outline per frame, so a kept frame shows its own head shape.
state.plates = state.stab.oval.map((r) => r.map(state.xform));
if (resetKeep || !state.keep.size) {
state.keep = new Set(Array.from({ length: N }, (_, i) => i));
}
// Frame 0 must always be kept: something has to be on screen at the start.
state.keep.add(0);
state.faceBox = faceBoxes();
drawAll();
}
// Face bounding box per frame in image space, for legible strip thumbnails.
function faceBoxes() {
return state.dense.map((lm) => {
let x0 = 1, y0 = 1, x1 = 0, y1 = 0;
for (const i of FACE_OVAL) {
x0 = Math.min(x0, lm[i].x); y0 = Math.min(y0, lm[i].y);
x1 = Math.max(x1, lm[i].x); y1 = Math.max(y1, lm[i].y);
}
const mx = (x1 - x0) * 0.18, my = (y1 - y0) * 0.14;
return { x0: x0 - mx, y0: y0 - my, x1: x1 + mx, y1: y1 + my };
});
}
// Eyes: lid rings traced per frame, blinks resolved per eye, one gaze shared.
//
// Lids are a FEATURE in the part table - rotoscoped, open vocabulary, a key on
// every frame - so they get exactly the mouth's treatment, including the same
// bounded contour average. The iris is a PRIMITIVE: a disc whose position is
// quantised, which is where the stylisation lives.
function buildEyes(o) {
const st = state.stab, N = state.dense.length;
const sig = eyeSignals(st);
state.eyeSig = sig;
const blink = { cut: o.blinkCut, dwell: o.blinkDwell, hold: o.blinkHold };
const shutR = resolveBlink(sig.openR, blink);
const shutL = resolveBlink(sig.openL, blink);
// Head-local, subsampled, contour-averaged - the identical chain the mouth
// takes, with the identical knob. The eye tracks the face, because the face
// is what it is attached to; what gets removed is per-frame detector jitter,
// not the motion.
const lidR = smoothContours(
st.lidR.map((r) => toRasterRing(r, EYE_R_RING, o.eyeVerts, state.xform)), o.contourSmooth);
const lidL = smoothContours(
st.lidL.map((r) => toRasterRing(r, EYE_L_RING, o.eyeVerts, state.xform)), o.contourSmooth);
// Where the iris hangs. Three behaviours, because this turns out to be an
// aesthetic choice and not only a correctness one.
//
// STEADY (default) reads the socket back off the DRAWN ring. Slots 0 and 8 of
// a 16-slot lid ring are the two corners, and subsampling to any even budget n
// keeps them at output indices 0 and n/2 - so the ring that gets rendered
// carries its own corners with it. The iris is then placed in the frame of the
// exact polygon it sits inside, after smoothing, after subsampling: it cannot
// drift relative to its own eye, and it inherits the contour average for free.
//
// FREE reads the raw per-frame corners instead, jitter and all. It is what the
// eyes did before any of this, and it is not simply worse - the detector noise
// reads as liveliness, the eye never sits perfectly still, and against flat
// hand-drawn plates that restlessness can be the thing that sells it. It is
// also the honest baseline to compare the other two against.
//
// LOCKED pins the socket to the take's mean, so the eye never moves in the
// head at all. Watch it against a photo underlay and the drawn eyes hang still
// over a face whose eyes are moving - that is the registration cost, and it is
// real - but once the plate is a drawing rather than a photograph, nothing is
// being registered against and it reads as a deliberately locked-off stare.
const ringSocket = (ring) => {
const a = ring[0], b = ring[ring.length / 2];
return { cx: (a.x + b.x) / 2, cy: (a.y + b.y) / 2, w: Math.hypot(a.x - b.x, a.y - b.y) };
};
const rawSocket = (corners, f) => {
const a = state.xform(corners[f][0]), b = state.xform(corners[f][1]);
return { cx: (a.x + b.x) / 2, cy: (a.y + b.y) / 2, w: Math.hypot(a.x - b.x, a.y - b.y) };
};
const meanSocket = (rings) => {
const acc = rings.reduce((a, r) => {
const k = ringSocket(r);
return { cx: a.cx + k.cx, cy: a.cy + k.cy, w: a.w + k.w };
}, { cx: 0, cy: 0, w: 0 });
const n = rings.length;
return { cx: acc.cx / n, cy: acc.cy / n, w: acc.w / n };
};
const socketFor = (rings, corners) => {
if (o.irisAnchor === 'locked') { const k = meanSocket(rings); return () => k; }
if (o.irisAnchor === 'free') return (f) => rawSocket(corners, f);
return (f) => ringSocket(rings[f]);
};
const skR = socketFor(lidR, st.cornersR), skL = socketFor(lidL, st.cornersL);
const socket = (ring) => ringSocket(ring);
// Iris radius comes from the take's MEAN eye width, not the current frame's.
// Size is authored; only position is tracked. A radius recomputed per frame
// would breathe by a fraction of a pixel as the fit's depth-scale wanders,
// and at this resolution a fraction of a pixel is a pixel flicking on and off
// around the whole silhouette.
const meanW = (rings) => rings.reduce((a, r) => a + ringSocket(r).w, 0) / rings.length;
const wR = meanW(lidR), wL = meanW(lidL), w = (wR + wL) / 2;
// Calibrate against the neutral, apply the artist's gain, and only then
// quantise - the grid should be a grid of DRAWN positions, because that is
// what a viewer reads. Gain is an authored parameter: measured gaze excursion
// is small and a character's eye usually wants more throw than a performer's,
// which is a decision for a person and not for the detector.
const origin = gazeOrigin(sig.gazeRaw, o.gazeOrigin, state.neutral);
state.gazeOriginValue = origin;
const px = sig.gazeRaw.map((g) => ({
x: (g.x - origin.x) * o.gazeGain * w,
y: (g.y - origin.y) * o.gazeGain * w,
}));
const gaze = quantizeSnap(px, o.gazeStep, o.gazeDwell);
const eye = (sk, lids, shut, rad, f) => {
const e = sk(f);
return {
// The lash line is the lid ring pushed outward by a fixed number of
// pixels, exactly as the mouth's outer ring sits outside its inner one.
// When the eye shuts, the traced ring goes near-degenerate and this
// collapses to a lens - which is a closed eye, drawn correctly, for free.
lash: offsetRing(lids[f], o.lashPx),
lid: lids[f],
shut: shut[f],
// Rounded to whole pixels. The rasteriser quantises everything anyway, so
// this costs nothing - but it means the iris and the square pupil share
// one integer centre, so the pupil is exactly its nominal size on every
// frame instead of spilling to the next pixel on some and not others.
iris: { x: Math.round(e.cx + gaze[f].x), y: Math.round(e.cy + gaze[f].y), r: rad },
pupil: o.pupilPx,
};
};
return {
gazePx: px, gaze, shutR, shutL, hasIris: sig.hasIris,
frames: Array.from({ length: N }, (_, f) => ({
r: eye(skR, lidR, shutR, (wR * o.irisSize) / 2, f),
l: eye(skL, lidL, shutL, (wL * o.irisSize) / 2, f),
})),
};
}
// Brows: ring traced every frame, HEIGHT quantised.
//
// The decomposition is the point. The traced ring already contains the brow's
// real height, so adding a quantised raise on top would move it twice. Instead
// the height is measured out of the ring, quantised, and put back - the shape
// that renders is his, at a height that snaps between a few authored levels and
// holds. That is the same split the eyes got: lid traced as a feature, iris
// position quantised as a primitive.
//
// Two ends, not one height, warped linearly between them. Raise and tilt are
// different expressions out of one mechanism: both ends up is surprise, inner
// up alone is worry, inner down is anger.
function buildBrows(o) {
const st = state.stab, N = state.dense.length;
const sig = browSignals(st);
state.browSig = sig;
const ringOf = (side) => (side === 'R' ? sig.pairing.right : sig.pairing.left);
const table = (side) => (ringOf(side) === 'browA' ? BROW_A_RING : BROW_B_RING);
const build = (side, corners) => {
const rings = smoothContours(
st[ringOf(side)].map((r) => toRasterRing(r, table(side), o.browVerts, state.xform)),
o.contourSmooth);
// Eye width in raster pixels, so the raise converts from eye widths into the
// units the grid is expressed in and the knob means the same on any framing.
const wpx = (f) => {
const a = state.xform(corners[f][0]), b = state.xform(corners[f][1]);
return Math.hypot(a.x - b.x, a.y - b.y);
};
const meanW = st.cornersR.reduce((a, _, f) => a + wpx(f), 0) / N;
// Rest pose from the take MEDIAN, never from the neutral frame. That frame
// is chosen by minimum mouth aperture and says nothing about the brows, and
// the same mistake on the gaze origin re-pointed an entire performance.
const rest = gazeOrigin(sig[side], 'median');
const px = sig[side].map((g) => ({
x: (g.x - rest.x) * o.browGain * meanW,
y: (g.y - rest.y) * o.browGain * meanW,
}));
const q = quantizeSnap(px, o.browStep, o.browDwell);
const frames = rings.map((ring, f) => {
// Raise is measured upward but y grows downward, so a positive raise is a
// negative y offset.
const dOuter = -(q[f].x - px[f].x), dInner = -(q[f].y - px[f].y);
const a = state.xform(corners[f][0]), b = state.xform(corners[f][1]);
const span = b.x - a.x;
const warped = ring.map((p) => {
// Position along the brow's own axis, outer end to inner end. Taken from
// x against the eye corners rather than from ring slots, because
// subsampling does not keep the end slots at any given budget.
const t = span === 0 ? 0 : Math.min(1, Math.max(0, (p.x - a.x) / span));
return { x: p.x, y: p.y + dOuter + (dInner - dOuter) * t };
});
return offsetRing(warped, o.browWeight);
});
return { frames, px, q };
};
return { R: build('R', st.cornersR), L: build('L', st.cornersL), pairing: sig.pairing };
}
// Presence gets hysteresis and a minimum dwell, the same treatment plate
// selection gets: a teeth block that blinks on and off for single frames is
// worse than one that is simply absent. Appearing needs a clear signal, staying
// needs only a weak one.
function resolveTeeth(o) {
const N = state.dense.length;
if (!state.interior) return new Array(N).fill({ show: false, pts: null });
const raw = state.interior.map((m, f) =>
(state.hidden[f] || !m.contour ? 0 : m.contrast));
const on = o.teethOn, off = o.teethOn * 0.7;
const shown = new Array(N).fill(false);
let live = false, since = 0;
for (let f = 0; f < N; f++) {
const want = live ? raw[f] > off : raw[f] > on;
if (want !== live && since >= o.teethDwell) { live = want; since = 0; }
else since++;
shown[f] = live && !state.hidden[f] && !!state.interior[f].contour;
}
// Into raster space through the same chain the lips take, including the
// isotropic aspect conversion - a contour in MediaPipe's normalised space is
// in the same stretched coordinates the landmarks are.
const toRaster = (pts, f) => {
const tf = state.stab.transforms[f];
return pts.map((p) => state.xform(applySim(tf, { x: p.x * state.aspect, y: p.y })));
};
const rast = state.interior.map((m, f) => (m.contour ? toRaster(m.contour, f) : null));
// Radial sampling makes vertex k mean the same direction on every frame, so
// smoothing across time is well defined and cannot reorder anything.
const sm = rast.map((pts, f) => {
if (!pts || !shown[f]) return pts;
const acc = pts.map(() => ({ x: 0, y: 0 }));
let c = 0;
for (let j = f - o.teethSmooth; j <= f + o.teethSmooth; j++) {
const k = Math.min(N - 1, Math.max(0, j));
if (!shown[k] || !rast[k] || rast[k].length !== pts.length) continue;
for (let v = 0; v < pts.length; v++) { acc[v].x += rast[k][v].x; acc[v].y += rast[k][v].y; }
c++;
}
return c ? acc.map((p) => ({ x: p.x / c, y: p.y / c })) : pts;
});
return shown.map((show, f) => ({ show, pts: sm[f] }));
}
/* ---------- render ---------- */
// The plate layer has several representations because its job changes: a flat
// shape to judge the mouth against, or a registered photograph to draw over.
// Only the latter is any use as reference art, and the generated oval is only a
// stand-in until a drawing exists.
function renderFrame(f, mode = plateMode()) {
const r = new IndexedRaster(RW, RH);
const pf = plateIndex(f); // the plate frame on screen
if (mode === 'posterize' && state.images[pf]) {
posterizeInto(r, state.images[pf], state.stab.transforms[pf], state.xform,
PALETTE.map((p) => p.hex));
} else {
r.clear(IDX.bg);
}
// Painted cels sit BEHIND the face and hold on the same frames the plate
// does - pf is already "the most recent kept frame at or before f", which is
// exactly the rule the user draws against: a cel holds until the next frame
// that has its own drawing.
drawCel(r, state.cels.get(pf), (i) => i);
if (mode !== 'posterize' && (mode === 'oval' || mode === 'oval+photo')) {
r.fillPoly(state.plates[pf], IDX.base);
}
// Eyes run on the CLOCK, not on the mouth lead. The lead is a lip-sync
// device: it exists because a mouth shape anticipates the sound it makes.
// Nothing about a blink or a glance is tied to the audio, so shifting the
// eyes would only slide them off the head that carries them.
// Eyes and brows ride the exposure grid but NOT the mouth lead: the lead is a
// lip-sync device and nothing about a blink or a brow is tied to the audio.
const ef = perfIndex(f);
if (state.eyes) drawEyes(r, state.eyes.frames[ef]);
if (state.brows) {
r.fillPoly(state.brows.R.frames[ef], IDX.brow);
r.fillPoly(state.brows.L.frames[ef], IDX.brow);
}
const mf = leadIndex(f); // performance frame, possibly ahead
r.fillPoly(state.outer[mf], IDX.dark); // mouth keeps every frame
if (!state.hidden[mf]) {
r.fillPoly(state.inner[mf], IDX.mouth);
const te = state.teeth[mf];
if (te.show && te.pts && te.pts.length >= 3) r.fillPoly(te.pts, IDX.teeth);
}
return r;
}
// Lash ring, then sclera, then iris - the same three-layer structure the mouth
// has, for the same reason: the dark ring outside the pale interior is what
// makes a flat shape read as an opening rather than a blob.
//
// The iris is stencilled to the sclera it was just drawn over, so the lid crops
// it automatically. Nothing needs to clamp the gaze to keep the iris inside the
// eye, which matters because a clamp would flatten the performance at exactly
// the extremes that carry it.
function drawEyes(r, e) {
for (const s of [e.r, e.l]) {
r.fillPoly(s.lash, IDX.dark);
if (s.shut) continue; // a shut eye IS the lash line, alone
r.fillPoly(s.lid, IDX.white);
r.fillDisc(s.iris.x, s.iris.y, s.iris.r, IDX.iris, IDX.white);
// Stencilled to the iris, which is itself stencilled to the sclera - so the
// pupil is cropped by the lid transitively, and a blink or an extreme gaze
// takes the right bite out of it without anything having to compute where.
if (s.pupil) r.fillRect(s.iris.x, s.iris.y, s.pupil, IDX.pupil, IDX.iris);
}
}
const plateMode = () => el('plateMode').value;
// Photo modes composite under the indexed layer, so the flat shapes stay exactly
// as they render while the reference sits behind them.
function compositeRender(canvas, f, zoom) {
const mode = plateMode();
const pf = plateIndex(f);
const img = state.images[pf];
const showPhoto = img && (mode === 'photo' || mode === 'photo-dim' || mode === 'oval+photo');
canvas.width = RW * zoom; canvas.height = RH * zoom;
const g = canvas.getContext('2d');
g.fillStyle = PALETTE[IDX.bg].hex;
g.fillRect(0, 0, canvas.width, canvas.height);
if (showPhoto) {
drawRegistered(g, img, state.stab.transforms[pf], state.xform, zoom,
mode === 'photo-dim' ? 0.34 : 1);
}
blitIndexed(g, renderFrame(f, mode === 'oval+photo' ? 'oval' : (showPhoto ? 'off' : mode)),
zoom, showPhoto);
}
// Put an indexed raster onto a 2D context. With `keyBg`, background pixels go
// transparent instead of opaque, so whatever was painted underneath - a
// registered photograph, usually - stays visible through them.
//
// Split out of compositeRender so the paint canvas can draw the same pixels at
// a reduced globalAlpha. Nothing else is different about that path, which is
// what keeps the editing view honest: you are dimming the render, not looking
// at a second renderer that might disagree with it.
function blitIndexed(g, raster, zoom, keyBg) {
const img = raster.toImageData(PALETTE.map((p) => p.hex), zoom);
if (!keyBg) { g.putImageData(img, 0, 0); return; }
const bg = PALETTE[IDX.bg].hex.replace('#', '');
const br = parseInt(bg.slice(0, 2), 16), bgn = parseInt(bg.slice(2, 4), 16), bb = parseInt(bg.slice(4, 6), 16);
const d = img.data;
for (let i = 0; i < d.length; i += 4) {
if (d[i] === br && d[i + 1] === bgn && d[i + 2] === bb) d[i + 3] = 0;
}
const tmp = document.createElement('canvas');
tmp.width = img.width; tmp.height = img.height;
tmp.getContext('2d').putImageData(img, 0, 0);
g.drawImage(tmp, 0, 0);
}
const keptSorted = () => [...state.keep].sort((a, b) => a - b);
// Performance tracks can lead the clock.
//
// A centred moving average has no phase lag, so smoothing does not literally
// delay anything - but it blurs onsets, and the visually salient moment of a
// mouth opening moves later even though the mean does not. Animators also draw
// mouth shapes one or two frames ahead of the sound as a matter of course, so
// this is the normal control rather than a correction.
//
// Positive lead = the mouth arrives earlier. Only performance parts shift; the
// head stays with the audio, because it is the mouth that should anticipate.
function leadIndex(f) {
// Reads cached scalars, not opts(): this runs once per strip thumbnail, and
// calling opts() here meant ~14 DOM reads x 74 frames on every redraw.
//
// Exposure first, then lead. The grid decides WHICH frames get a new drawing;
// the lead then shifts which pose that drawing carries, by whole frames of the
// original track. Applying them the other way round would put the changes on
// the wrong beats - the picture would update on the odd frames instead of
// holding on the twos.
return shiftIndex(exposeIndex(f, state.exposure), state.lead, state.dense.length);
}
// The plate rides the same grid, so the whole picture updates together. On 2s
// means on 2s - a head that cut on the odd frames while the mouth cut on the
// even ones would read as two performances laid over each other.
const plateIndex = (f) => heldFrame(keptSorted(), exposeIndex(f, state.exposure));
// Performance tracks that do not take the mouth lead still ride the grid. This
// existing as a named thing is what stopped the eyes holding on 1s in the
// preview while the export held them on 2s - a preview that disagrees with the
// export is the one bug this tool cannot afford.
const perfIndex = (f) => exposeIndex(f, state.exposure);
function blit(canvas, raster, zoom) {
canvas.width = RW * zoom; canvas.height = RH * zoom;
canvas.getContext('2d').putImageData(raster.toImageData(PALETTE.map((p) => p.hex), zoom), 0, 0);
}
function drawAll() {
drawPanes();
drawStrip();
drawWorksheet();
drawReadout();
drawPaint();
}
function drawReadout() {
const kept = keptSorted();
const runs = kept.map((k, i) => (i + 1 < kept.length ? kept[i + 1] : state.dense.length) - k);
const lead = state.lead;
const teethFrames = state.teeth ? state.teeth.filter((t) => t.show).length : 0;
el('readout').textContent =
`${state.dense.length} frames → ${kept.length} drawings · ` +
`teeth on ${teethFrames}f · ` +
`${blinkRuns(state.eyes.shutR).length}/${blinkRuns(state.eyes.shutL).length} blinks R/L · ` +
`${gazeCells(state.eyes.gaze)} gaze cells · ` +
`${gazeCells(state.brows.R.q)} brow poses · ` +
(state.exposure > 1
? `on ${state.exposure}s = ${(state.fps / state.exposure).toFixed(4).replace(/\.?0+$/, '')}fps · `
: '') +
(lead ? `mouth leads ${lead}f (${(lead / state.fps * 1000).toFixed(0)}ms) · ` : '') +
`holds ${Math.min(...runs)}–${Math.max(...runs)} frames · ` +
`neutral f${state.neutral} · residual ` +
`${(state.stab.residual.reduce((a, b) => a + b, 0) / state.dense.length).toFixed(4)}`;
}
// Blinks as RUNS, not as shut frames: a three-frame blink is one blink, and the
// count is only useful as "did the performer blink six times or sixty".
function blinkRuns(shut) {
const runs = [];
for (let f = 0; f < shut.length; f++) {
if (shut[f] && !shut[f - 1]) runs.push(f);
}
return runs;
}
// How many distinct positions the iris ever occupies. This is the number the
// gaze knobs exist to control: two or three is a character who looks at things,
// forty is an unquantised iris sliding around, which is what the grid is for.
const gazeCells = (gaze) => new Set(gaze.map((g) => `${g.x},${g.y}`)).size;
function drawPanes() {
const f = state.frame, kept = keptSorted();
const pf = plateIndex(f);
// The mouth frame is always shown, not only when shifted, so the number can be
// watched diverging from f rather than taken on trust.
const lead = state.lead;
el('framelabel').textContent =
`f ${f} / ${state.dense.length - 1} · ${(f / state.fps).toFixed(2)}s · ` +
`plate f${pf} · mouth f${leadIndex(f)}` +
(state.exposure > 1 && f % state.exposure ? ' (held)' : '') +
(lead ? ` (${lead > 0 ? '+' : ''}${lead} = ${(lead / state.fps * 1000).toFixed(0)}ms)` : '') +
(state.keep.has(f) ? ' · KEPT' : ' · held');
const c1 = el('cv-source'), g1 = c1.getContext('2d');
c1.width = RW * ZOOM; c1.height = RH * ZOOM;
g1.fillStyle = '#000'; g1.fillRect(0, 0, c1.width, c1.height);
const im = state.images[f];
if (im) {
const b = state.faceBox[f];
const sx = b.x0 * im.naturalWidth, sy = b.y0 * im.naturalHeight;
const sw = (b.x1 - b.x0) * im.naturalWidth, sh = (b.y1 - b.y0) * im.naturalHeight;
const s = Math.min(c1.width / sw, c1.height / sh);
const dw = sw * s, dh = sh * s, dx = (c1.width - dw) / 2, dy = (c1.height - dh) / 2;
g1.drawImage(im, sx, sy, sw, sh, dx, dy, dw, dh);
const map = (p) => ({ x: dx + (p.x * im.naturalWidth - sx) * s, y: dy + (p.y * im.naturalHeight - sy) * s });
strokePts(g1, LIPS_OUTER.map((i) => map(state.dense[f][i])), '#4ade80');
strokePts(g1, LIPS_INNER.map((i) => map(state.dense[f][i])), '#f87171');
drawEyeOverlay(g1, map, f);
} else {
g1.fillStyle = '#555'; g1.font = '13px system-ui';
g1.fillText('synthetic — no source frames', 14, 24);
const sc = (p) => ({ x: p.x * c1.width, y: p.y * c1.height });
strokePts(g1, LIPS_OUTER.map((i) => sc(state.dense[f][i])), '#4ade80');
strokePts(g1, LIPS_INNER.map((i) => sc(state.dense[f][i])), '#f87171');
drawEyeOverlay(g1, sc, f);
}
const c2 = el('cv-stab'), g2 = c2.getContext('2d');
c2.width = RW * ZOOM; c2.height = RH * ZOOM;
g2.fillStyle = '#0d0f16'; g2.fillRect(0, 0, c2.width, c2.height);
g2.strokeStyle = '#2a2f3e'; g2.lineWidth = 1; g2.beginPath();
g2.moveTo(c2.width / 2, 0); g2.lineTo(c2.width / 2, c2.height);
g2.moveTo(0, c2.height / 2); g2.lineTo(c2.width, c2.height / 2); g2.stroke();
const z = (pts) => pts.map((p) => ({ x: p.x * ZOOM, y: p.y * ZOOM }));
const mf = leadIndex(f);
strokePts(g2, z(state.plates[pf]), '#3b4a63');
// With a lead set, the unshifted contour is drawn as a ghost so the offset is
// something you can see rather than something you have to trust.
if (mf !== f) strokePts(g2, z(state.outer[f]), '#2f6b46');
strokePts(g2, z(state.outer[mf]), '#4ade80');
if (!state.hidden[mf]) strokePts(g2, z(state.inner[mf]), '#f87171');
for (const e of [state.eyes.frames[f].r, state.eyes.frames[f].l]) {
strokePts(g2, z(e.lid), e.shut ? '#f87171' : '#60a5fa');
if (e.shut) continue;
g2.strokeStyle = '#fbbf24';
g2.beginPath();
g2.arc(e.iris.x * ZOOM, e.iris.y * ZOOM, e.iris.r * ZOOM, 0, Math.PI * 2);
g2.stroke();
}
compositeRender(el('cv-render'), f, ZOOM);
drawInteriorDebug(f);
drawGazeDebug(f);
}
// What the teeth measurement actually saw: sampled region, pixels above
// threshold in green, the resolved line in amber. Recomputed for the current
// frame only, so it costs nothing to keep on screen.
function drawInteriorDebug(fRaw) {
const f = leadIndex(fRaw);
const host = el('cv-teeth');
const img = state.images[f];
if (!img || state.hidden[f]) {
host.innerHTML = '';
el('teethinfo').textContent = state.images.length ? 'mouth closed' : 'no source frames';
return;
}
const o = opts();
const ctx = document.createElement('canvas').getContext('2d', { willReadFrequently: true });
const m = extractTeeth(img, LIPS_INNER.map((k) => state.dense[f][k]), ctx, o, true);
host.innerHTML = '';
if (m.debug) {
m.debug.style.width = '170px';
m.debug.style.imageRendering = 'pixelated';
host.append(m.debug);
}
const te = state.teeth[f];
el('teethinfo').textContent =
`contrast ${m.contrast.toFixed(3)} / gate ${o.teethOn.toFixed(2)} · ` +
`area ${m.area}px · ${te.show ? 'SHOWN' : 'hidden'}`;
}
// Lid rings and the iris, on the raw frame. Landmark overlays are how you tell
// a tracking failure from a knob set wrong, and the eyes need it more than the
// mouth does: an iris that has latched onto an eyebrow looks, in the flat
// render alone, exactly like a gaze gain that is too high.
function drawEyeOverlay(g, map, f) {
const lm = state.dense[f];
for (const ring of [EYE_R_RING, EYE_L_RING]) {
strokePts(g, ring.map((i) => map(lm[i])), '#60a5fa');
}
for (const ring of [BROW_A_RING, BROW_B_RING]) {
strokePts(g, ring.map((i) => map(lm[i])), '#c084fc');
}
if (!state.eyes.hasIris) return;
for (const iris of [IRIS_A, IRIS_B]) {
strokePts(g, iris.slice(1).map((i) => map(lm[i])), '#fbbf24');
}
}
// The gaze field: every position the iris takes over the whole take, plus where
// it is now. Tune against this, not against the numbers - "4 cells" tells you
// the quantisation is working, but only the picture tells you whether the four
// are the four looks the performance actually has.
function drawGazeDebug(f) {
const cv = el('cv-gaze'), S = 150;
cv.width = S; cv.height = S;
const g = cv.getContext('2d');
g.fillStyle = '#0d0f16'; g.fillRect(0, 0, S, S);
const ex = state.eyes;
// Scale so the widest excursion in the take fills the box, with a floor so a
// nearly-still gaze does not get magnified into a light show.
let m = 2;
for (const p of ex.gazePx) m = Math.max(m, Math.abs(p.x), Math.abs(p.y));
const k = (S / 2 - 8) / m;
const X = (v) => S / 2 + v * k, Y = (v) => S / 2 + v * k;
const o = opts();
if (o.gazeStep > 0) {
g.strokeStyle = '#1b2030'; g.lineWidth = 1;
for (let i = -20; i <= 20; i++) {
const v = i * o.gazeStep;
if (Math.abs(v) > m) continue;
g.beginPath(); g.moveTo(X(v), 0); g.lineTo(X(v), S); g.stroke();
g.beginPath(); g.moveTo(0, Y(v)); g.lineTo(S, Y(v)); g.stroke();
}
}
g.strokeStyle = '#2a2f3e';
g.beginPath(); g.moveTo(S / 2, 0); g.lineTo(S / 2, S);
g.moveTo(0, S / 2); g.lineTo(S, S / 2); g.stroke();
g.fillStyle = '#2f6b46';
for (const p of ex.gaze) g.fillRect(X(p.x) - 1.5, Y(p.y) - 1.5, 3, 3);
const raw = ex.gazePx[f], q = ex.gaze[f];
g.fillStyle = '#8891a5';
g.fillRect(X(raw.x) - 1, Y(raw.y) - 1, 2, 2);
g.fillStyle = '#fbbf24';
g.beginPath(); g.arc(X(q.x), Y(q.y), 4, 0, Math.PI * 2); g.fill();
const sig = state.eyeSig, fr = state.eyes.frames[f];
const og = state.gazeOriginValue;
// Per-eye raw gaze is the diagnostic for a wrong-looking eyeline. If the two
// agree and both point the wrong way, the ORIGIN is wrong. If they disagree in
// a sustained way, it is out-of-plane head rotation biasing the projection,
// which no 2D measurement can undo.
const sgn = (v) => `${v >= 0 ? '+' : ''}${v.toFixed(3)}`;
el('eyeinfo').textContent =
`open R ${sig.openR[f].toFixed(3)} L ${sig.openL[f].toFixed(3)} / cut ${o.blinkCut.toFixed(3)}\n` +
`${fr.r.shut ? 'R SHUT ' : ''}${fr.l.shut ? 'L SHUT' : ''}${!fr.r.shut && !fr.l.shut ? 'both open' : ''}\n` +
`gaze ${q.x >= 0 ? '+' : ''}${q.x.toFixed(1)}, ${q.y >= 0 ? '+' : ''}${q.y.toFixed(1)} px\n` +
`raw R ${sgn(sig.gazeR[f].x)} L ${sgn(sig.gazeL[f].x)} (x, eye widths)\n` +
`origin ${o.gazeOrigin} ${sgn(og.x)}, ${sgn(og.y)}` +
(ex.hasIris ? '' : ' — no iris landmarks');
}
function strokePts(g, pts, color, lw = 1) {
g.strokeStyle = color; g.lineWidth = lw;
g.beginPath();
pts.forEach((p, i) => (i ? g.lineTo(p.x, p.y) : g.moveTo(p.x, p.y)));
g.closePath(); g.stroke();
}
/* ---------- the frame strip: this is the editing surface ---------- */
function drawStrip() {
const host = el('strip');
host.innerHTML = '';
const N = state.dense.length;
for (let f = 0; f < N; f++) {
const cell = document.createElement('div');
cell.className = 'fr' + (state.keep.has(f) ? ' keep' : ' drop') + (f === state.frame ? ' cur' : '');
cell.dataset.f = f;
const cv = document.createElement('canvas');
const im = state.images[f];
cv.width = THUMB; cv.height = THUMB;
const g = cv.getContext('2d');
g.fillStyle = '#0d0f16'; g.fillRect(0, 0, THUMB, THUMB);
if (im) {
const b = state.faceBox[f];
const sx = b.x0 * im.naturalWidth, sy = b.y0 * im.naturalHeight;
const sw = (b.x1 - b.x0) * im.naturalWidth, sh = (b.y1 - b.y0) * im.naturalHeight;
const s = Math.min(THUMB / sw, THUMB / sh);
g.drawImage(im, sx, sy, sw, sh, (THUMB - sw * s) / 2, (THUMB - sh * s) / 2, sw * s, sh * s);
} else {
const r = new IndexedRaster(RW, RH);
r.clear(IDX.bg); r.fillPoly(state.plates[f], IDX.base);
const tmp = document.createElement('canvas');
blit(tmp, r, 1);
g.drawImage(tmp, 0, 0, RW, RH, 0, 0, THUMB, THUMB * (RH / RW));
}
const tag = document.createElement('span');
tag.textContent = f;
// Mark the frames that actually carry a drawing. Without it the only way to
// know where your cels are is to scrub and look, and "copy previous" then
// reaches back to somewhere you cannot see.
if ((state.cels.get(f) || []).length) cell.classList.add('cel');
cell.append(cv, tag);
cell.draggable = true;
cell.ondragstart = (ev) => ev.dataTransfer.setData('text/plain', String(f));
cell.onclick = (ev) => {
seekTo(f);
if (ev.shiftKey) toggle(f);
drawAll();
};
cell.ondblclick = () => { toggle(f); drawAll(); };
host.append(cell);
}
}
function toggle(f) {
if (f === 0) return; // frame 0 always has a drawing
if (state.keep.has(f)) state.keep.delete(f); else state.keep.add(f);
}
/* ---------- worksheet: the frames a human must draw ---------- */
function drawWorksheet() {
const host = el('sheet');
host.innerHTML = '';
const kept = keptSorted();
kept.forEach((f, i) => {
const until = (i + 1 < kept.length ? kept[i + 1] : state.dense.length) - 1;
const cell = document.createElement('div');
cell.className = 'cell';
// Registered, not raw-cropped: the worksheet frame is in raster space, so a
// drawing traced from it is already aligned to the mouth.
const cv = document.createElement('canvas');
compositeRender(cv, f, 1);
cv.style.width = '176px';
const cap = document.createElement('span');
cap.textContent = `f${f}` + (until > f ? ` → ${until}` : '') + ` (${until - f + 1}f)`;
cell.append(cv, cap);
cell.onclick = () => { seekTo(f); drawAll(); };
host.append(cell);
});
}
/* ---------- export ---------- */
// Six parts, three per eye, mirroring the mouth's lash/interior/content stack.
// `clip` is what tells the renderer the iris is stencilled by the sclera rather
// than merely drawn after it - without it an extreme gaze would put the iris on
// the cheek.
function eyeParts(grid) {
const out = [];
// Eyes ride the exposure grid but NOT the mouth lead: the lead is a lip-sync
// device and nothing about a blink is tied to the audio.
const src = perfIndex;
[['r', 20], ['l', 23]].forEach(([side, z]) => {
const at = (f) => state.eyes.frames[src(f)][side];
out.push(
{ name: `eye_${side}`, kind: 'poly', z, color: 'skin_dark', interp: 'hold',
keys: grid.map((f) => ({ f, src: src(f), pts: at(f).lash })) },
{ name: `eye_${side}_in`, kind: 'poly', z: z + 1, color: 'eye_white', interp: 'hold',
parent: `eye_${side}`,
keys: grid.map((f) =>
(at(f).shut ? { f, hidden: true } : { f, src: src(f), pts: at(f).lid })) },
{ name: `iris_${side}`, kind: 'disc', z: z + 2, color: 'iris', interp: 'hold',
parent: `eye_${side}_in`, clip: `eye_${side}_in`,
keys: grid.map((f) => {
const e = at(f);
return e.shut ? { f, hidden: true }
: { f, src: src(f), c: { x: e.iris.x, y: e.iris.y }, r: e.iris.r };
}) },
);
if (!state.eyes.frames[0].r.pupil) return;
out.push(
{ name: `pupil_${side}`, kind: 'rect', z: z + 3, color: 'pupil', interp: 'hold',
parent: `iris_${side}`, clip: `iris_${side}`,
keys: grid.map((f) => {
const e = at(f);
return e.shut ? { f, hidden: true }
: { f, src: src(f), c: { x: e.iris.x, y: e.iris.y }, size: e.pupil };
}) },
);
});
return out;
}
// Brows are a traced ring like the lids, so they keep every frame on the grid.
// The quantised raise is already baked into the points - the renderer is handed
// a polygon, not a shape plus an offset it would have to recombine.
function browParts(grid) {
return [['r', 26, 'R'], ['l', 27, 'L']].map(([name, z, side]) => ({
name: `brow_${name}`, kind: 'poly', z, color: 'brow', interp: 'hold',
keys: grid.map((f) => ({ f, src: perfIndex(f), pts: state.brows[side].frames[perfIndex(f)] })),
}));
}
function exportTake() {
const kept = keptSorted();
const N = state.dense.length;
// Output frames that actually carry a key. Everything between them is a hold,
// which the take format already expresses, so on 2s emits half the keys rather
// than emitting each pose twice.
const grid = [];
for (let f = 0; f < N; f += state.exposure) grid.push(f);
const take = {
name: el('takename').value || 'line_01',
frames: N, width: RW, height: RH, exposure: state.exposure, fps: state.fps,
palette: PALETTE,
slot: { x: RW / 2, y: RH / 2 },
parts: [
// Sparse: one plate key per frame a human draws.
{ name: 'head', kind: 'plate', z: 0, interp: 'hold',
keys: kept.map((f, i) => ({ f: i, plate: i, src: f })) },
// Dense: the traced mouth keeps every frame. The lead is baked in here -
// key f carries the pose from source frame f+lead - so the renderer never
// needs to know about it.
{ name: 'mouth', kind: 'poly', z: 30, color: 'skin_dark', interp: 'hold',
keys: grid.map((f) => ({ f, src: leadIndex(f), pts: state.outer[leadIndex(f)] })) },
{ name: 'mouth_in', kind: 'poly', z: 31, color: 'mouth_dark', interp: 'hold', parent: 'mouth',
keys: grid.map((f) => {
const m = leadIndex(f);
return state.hidden[m] ? { f, hidden: true } : { f, src: m, pts: state.inner[m] };
}) },
// Eyes. The lids are traced, so like the mouth they cost nothing and keep
// every frame. The iris is a primitive: its quantised position means the
// key stream is dense but the VALUES change only on saccades, so a
// hold-interpolating renderer cuts between fixations by itself.
...eyeParts(grid),
...browParts(grid),
{ name: 'teeth', kind: 'poly', z: 32, color: 'teeth', interp: 'hold', parent: 'mouth_in',
keys: grid.map((f) => {
const m = leadIndex(f), te = state.teeth[m];
return te.show && te.pts ? { f, src: m, pts: te.pts } : { f, hidden: true };
}) },
],
};
const text = writeTake(take)
+ `\n# plate drawings needed: ${kept.length} of ${N} frames\n`
+ kept.map((f, i) => {
const until = (i + 1 < kept.length ? kept[i + 1] : N) - 1;
return `# plate ${i} = source frame ${f}, holds f${f}..${until}`;
}).join('\n') + '\n';
const a = document.createElement('a');
a.href = URL.createObjectURL(new Blob([text], { type: 'text/plain' }));
a.download = `${take.name}.take`;
a.click();
status(`exported — ${kept.length} plate drawings, ${grid.length} mouth keys` +
(state.exposure > 1 ? ` on ${state.exposure}s` : '') + ', ' +
`${blinkRuns(state.eyes.shutR).length + blinkRuns(state.eyes.shutL).length} blinks`, 'ok');
}
/* ---------- wiring ---------- */
async function runFrames() {
try {
const man = await loadManifest();
if (man) {
state.fps = man.fps;
el('framedir').value = man.dir || 'frames';
attachAudio(man.audio);
} else {
attachAudio(null);
status('no manifest.json — assuming 12fps, no audio. Re-run extract.sh', 'warn');
}
const images = await loadFrameSequence();
if (!images.length) {
status(`no frames in ${el('framedir').value}/ — run extract.sh first`, 'err');
return;
}
const { dense, missing, detected, firstReal } = await detectAll(images);
state.images = images; state.dense = dense; state.detected = detected;
state.aspect = images[0].naturalWidth / images[0].naturalHeight;
status('measuring mouth interiors…');
state.interior = measureAll(images, dense, opts());
el('scrub').max = dense.length - 1;
state.frame = 0;
labelExposure();
loadCels();
rebuild(true);
const dur = (dense.length / state.fps).toFixed(2);
status(`${images.length} frames · ${images[0].naturalWidth}x${images[0].naturalHeight} · ` +
`${state.fps}fps · ${dur}s` +
(state.audio ? ' · audio loaded' : ' · no audio') +
(missing.length ? ` · no face on ${missing.length}` +
(firstReal ? ` (${firstReal} at the top back-filled from ${firstReal + 1}, rest held)`
: ' (held previous)') : ''),
missing.length ? 'warn' : 'ok');
} catch (e) { status(e.message, 'err'); console.error(e); }
}
function runSynthetic() {
state.images = [];
attachAudio(null);
state.fps = 12;
state.aspect = 1; // synthetic landmarks are generated square
state.detected = null; // no detection ran, so every frame counts as real
state.lead = 0;
state.interior = null; // no pixels, so no teeth
state.dense = synthDense(72);
el('scrub').max = 71;
state.frame = 0;
labelExposure();
loadCels();
rebuild(true);
status('synthetic — exercises everything below detection', 'ok');
}
// How each slider's raw value reads out. A table rather than the conditional
// chain this used to be: that chain grew a branch per knob and was one ternary
// away from being unreadable.
const FMT = {
apertureThresh: (v) => (v / 1000).toFixed(3),
tol: (v) => (v / 1000).toFixed(3),
blinkCut: (v) => (v / 1000).toFixed(3),
teethOn: (v) => (v / 100).toFixed(2),
teethErode: (v) => (v / 100).toFixed(2),
tongueReject: (v) => (v / 100).toFixed(2),
topBias: (v) => (v / 100).toFixed(2),
irisSize: (v) => `${v}%`,
gazeGain: (v) => (v / 100).toFixed(2),
gazeStep: (v) => (v ? `${v}px` : 'off'),
browStep: (v) => (v ? `${v}px` : 'off'),
browWeight: (v) => `${v}px`,
browGain: (v) => (v / 100).toFixed(2),
pupilPx: (v) => (v ? `${v}px` : 'off'),
lashPx: (v) => `${v}px`,
lead: (v) => (v > 0 ? `+${v}` : String(v)),
};
for (const id of ['verts', 'smoothWin', 'contourSmooth', 'apertureThresh', 'tol',
'teethOn', 'teethDwell', 'teethErode', 'tongueReject', 'blobGrow',
'topBias', 'teethVerts', 'teethSmooth', 'lead',
'eyeVerts', 'lashPx', 'irisSize', 'pupilPx', 'gazeGain',
'gazeStep', 'gazeDwell', 'blinkCut', 'blinkHold', 'blinkDwell',
'browVerts', 'browWeight', 'browGain', 'browStep', 'browDwell']) {
const show = () => {
el(id + 'v').textContent = FMT[id] ? FMT[id](+el(id).value) : el(id).value;
};
el(id).addEventListener('input', () => {
show();
if (id === 'tol') return; // tol only matters when you ask for a suggestion
rebuild(false);
});
show();
}
function seekTo(f) {
state.frame = f;
if (state.audio) state.audio.currentTime = f / state.fps;
el('scrub').value = f;
}
el('scrub').addEventListener('input', (e) => { seekTo(+e.target.value); drawAll(); });
el('btn-frames').onclick = runFrames;
el('btn-synth').onclick = runSynthetic;
el('btn-export').onclick = exportTake;
el('plateMode').addEventListener('change', () => { if (state.dense) drawAll(); });
el('exposure').addEventListener('change', () => { if (state.dense) rebuild(false); });
for (const id of ['irisAnchor', 'gazeOrigin']) {
el(id).addEventListener('change', () => { if (state.dense) rebuild(false); });
}
// Label the exposure options in the only units that mean anything here: the
// rate the picture actually changes at, which depends on the clip's own rate.
// "on 2s" is the animator's name for it and the number is what you hear against
// the audio, so the menu says both.
function labelExposure() {
for (const opt of el('exposure').options) {
const n = +opt.value;
const rate = (state.fps / n).toFixed(4).replace(/\.?0+$/, '');
opt.textContent = `${rate} fps · on ${n}s`;
}
}
el('btn-saveframe').onclick = () => {
if (!state.dense) return;
const cv = document.createElement('canvas');
compositeRender(cv, state.frame, 4); // 1280x800: enough to draw on
const a = document.createElement('a');
a.href = cv.toDataURL('image/png');
a.download = `${el('takename').value || 'take'}_f${String(state.frame).padStart(4, '0')}.png`;
a.click();
status(`saved registered frame f${state.frame} at 4x`, 'ok');
};
el('btn-keepall').onclick = () => { if (state.dense) { rebuild(true); } };
el('btn-suggest').onclick = () => {
if (!state.dense) return;
const kept = suggestPlateFrames(state.stab.rigid, opts().tol);
state.keep = new Set(kept);
drawAll();
status(`suggested ${kept.length} drawings at tolerance ${opts().tol.toFixed(3)} — now hand-correct`, 'ok');
};
el('btn-play').onclick = () => {
if (!state.dense) return;
state.playing = !state.playing;
el('btn-play').textContent = state.playing ? 'Stop' : 'Play';
if (state.playing) {
if (state.audio) {
state.audio.playbackRate = +el('speed').value;
// Restart from the top if we are sitting at the end.
if (state.frame >= state.dense.length - 1) state.frame = 0;
state.audio.currentTime = state.frame / state.fps;
state.audio.play().catch((e) => status('audio blocked: ' + e.message, 'warn'));
}
last = 0;
tick();
} else if (state.audio) {
state.audio.pause();
}
};
el('speed').addEventListener('change', () => {
// playbackRate retimes the clock, and the picture follows it for free.
if (state.audio) state.audio.playbackRate = +el('speed').value;
});
// Keyboard is the point: stepping and deleting 37 frames by mouse is miserable.
window.addEventListener('keydown', (e) => {
if (!state.dense || e.target.tagName === 'INPUT') return;
const N = state.dense.length;
if (e.key === 'ArrowRight') { seekTo(Math.min(N - 1, state.frame + 1)); drawAll(); }
else if (e.key === 'ArrowLeft') { seekTo(Math.max(0, state.frame - 1)); drawAll(); }
else if (e.key === 'Backspace' || e.key === 'Delete' || e.key === 'x') {
state.keep.delete(state.frame === 0 ? -1 : state.frame); drawAll();
} else if (e.key === '[' || e.key === ']') {
const n = el('lead');
n.value = Math.max(+n.min, Math.min(+n.max, +n.value + (e.key === ']' ? 1 : -1)));
state.lead = +n.value;
el('leadv').textContent = state.lead > 0 ? `+${state.lead}` : String(state.lead);
drawAll();
} else if (e.key === 'b') {
const sel = el('plateMode');
sel.selectedIndex = (sel.selectedIndex + 1) % sel.options.length;
drawAll();
} else if (e.key === 'k' || e.key === ' ') {
if (state.frame !== 0) state.keep.add(state.frame);
drawAll();
} else return;
e.preventDefault();
});
let last = 0;
function tick(ts = 0) {
if (!state.playing) return;
const N = state.dense.length;
let f;
if (state.audio) {
// Audio is the clock. Deriving the frame from currentTime rather than
// counting means a slow render loop drops frames instead of drifting out
// of sync, which is the behaviour you want when judging lip sync.
f = Math.floor(state.audio.currentTime * state.fps);
if (f >= N || state.audio.ended) {
state.audio.currentTime = 0;
state.audio.play().catch(() => {});
f = 0;
}
} else {
const rate = state.fps * +el('speed').value;
if (ts - last < 1000 / rate) { requestAnimationFrame(tick); return; }
last = ts;
f = (state.frame + 1) % N;
}
if (f !== state.frame) {
state.frame = f;
el('scrub').value = f;
drawPanes(); drawStrip();
}
requestAnimationFrame(tick);
}
PALETTE.forEach((p) => {
const sw = document.createElement('label');
sw.className = 'sw';
const inp = document.createElement('input');
inp.type = 'color'; inp.value = p.hex;
inp.oninput = () => { p.hex = inp.value; if (state.dense) drawAll(); };
sw.append(inp, document.createTextNode(p.name));
el('palette').append(sw);
});
/* ---------- paint ---------- */
// The cel being edited is the one ON SCREEN, which is the most recent kept
// frame at or before the playhead. You can scrub anywhere and keep drawing on
// the cel you can see, rather than having to land exactly on a kept frame.
const celFrame = () => (state.dense ? plateIndex(state.frame) : 0);
const paint = new PaintUI({
canvas: el('cv-paint'), list: el('paintlayers'), info: el('paintinfo'),
zoom: 3, RW, RH, palette: PALETTE,
getCel: () => (state.dense ? (state.cels.get(celFrame()) || []) : []),
setCel: (cel) => { if (state.dense) state.cels.set(celFrame(), cel); },
celAt: (f) => (state.dense ? state.cels.get(heldFrame(keptSorted(), f)) : null),
backing: (cv, z) => {
if (!state.dense) {
cv.width = RW * z; cv.height = RH * z;
const g = cv.getContext('2d');
g.fillStyle = '#0d0f16'; g.fillRect(0, 0, cv.width, cv.height);
return;
}
const op = +el('paintOpacity').value / 100;
// At 100% this is exactly the normal composite, so the slider changes
// nothing at all until you reach for it.
if (op >= 1) { compositeRender(cv, state.frame, z); return; }
paintGhost(cv, z, op);
},
// paintLabels, not drawPaint: drawPaint re-renders the canvas, and this runs
// from inside commit(), which renders immediately afterwards anyway.
onChange: () => { saveCels(); paintLabels(); drawPanes(); drawStrip(); drawWorksheet(); },
});
el('paintTool').addEventListener('change', () => {
paint.tool = el('paintTool').value;
paint.draft = null;
paint.render();
});
el('paintColor').addEventListener('change', () => { paint.color = +el('paintColor').value; });
el('paintOpacity').addEventListener('input', () => {
el('paintOpacityv').textContent = `${el('paintOpacity').value}%`;
paint.render();
});
el('paintOpacityv').textContent = `${el('paintOpacity').value}%`;
el('btn-celclear').onclick = () => {
if (!state.dense) return;
state.cels.set(celFrame(), newCel());
paint.sel = -1;
paint.commit();
};
el('btn-celprev').onclick = () => {
// Copy the last finished drawing onto this one - the case you reach for
// constantly, stepping forward and carrying the previous cel with you.
//
// "The last drawing" means the nearest earlier enabled frame that ACTUALLY
// HAS one, not simply the nearest earlier enabled frame. Every frame is
// enabled until you curate the strip, so the naive rule resolved to f-1,
// which is empty, and the button looked like it only ever copied the frame
// immediately to the left. Skipping the empties makes it behave the same
// before and after you thin the strip out.
if (!state.dense) return;
const here = celFrame();
const prev = keptSorted()
.filter((k) => k < here && (state.cels.get(k) || []).length)
.pop();
if (prev === undefined) { paint.say('no earlier drawing to copy'); return; }
state.cels.set(here, cloneCel(state.cels.get(prev)));
paint.sel = -1;
paint.commit();
paint.say(`copied f${prev} → f${here} — ${state.cels.get(here).length} layers`);
};
PALETTE.forEach((p, i) => {
const o = document.createElement('option');
o.value = i; o.textContent = p.name;
el('paintColor').append(o);
});
el('paintColor').value = 1;
paint.color = 1;
// Drawings are the only thing here a person made by hand, so losing them to a
// reload would be the worst failure in the tool. Everything else regenerates.
const celKey = () => `arthur.cels.${el('takename').value || 'line_01'}`;
function saveCels() {
try {
localStorage.setItem(celKey(), JSON.stringify([...state.cels]));
} catch { /* private window, quota - not worth failing a brush stroke over */ }
}
function loadCels() {
state.cels = new Map();
try {
const raw = localStorage.getItem(celKey());
if (raw) state.cels = new Map(JSON.parse(raw).map(([k, v]) => [+k, v]));
} catch { /* corrupt or absent: start empty */ }
}
function drawPaint() {
paintLabels();
paint.render();
}
// The drawing dimmed over the source frame, for tracing.
//
// An EDITING AID ONLY - opacity is never written to the raster, never exported,
// and the flat render is untouched. Nothing here may reach the output: a
// translucent fill is the one thing this format cannot express, so if it ever
// leaked into the rasteriser it would have to be flattened against a background
// and would silently become a colour that is not in the ramp.
//
// The plate oval is deliberately not drawn. It is a stand-in for art that does
// not exist yet, and it covers the performer's face in flat skin - which is
// exactly the part of the frame you turned the opacity down to look at.
function paintGhost(cv, z, op) {
cv.width = RW * z; cv.height = RH * z;
const g = cv.getContext('2d');
g.fillStyle = PALETTE[IDX.bg].hex;
g.fillRect(0, 0, cv.width, cv.height);
const f = state.frame, pf = plateIndex(f);
const img = state.images[pf];
if (img) drawRegistered(g, img, state.stab.transforms[pf], state.xform, z, 1);
g.globalAlpha = op;
blitIndexed(g, renderFrame(f, 'off'), z, !!img);
g.globalAlpha = 1;
}
function paintLabels() {
if (!state.dense) return;
const kept = keptSorted(), cf = celFrame();
const until = (kept[kept.indexOf(cf) + 1] ?? state.dense.length) - 1;
const drawn = kept.filter((k) => (state.cels.get(k) || []).length);
el('paintframe').textContent =
`drawing cel f${cf}` + (until > cf ? ` — holds to f${until}` : '') +
(state.frame !== cf ? ` · playhead f${state.frame}` : '');
el('paintcels').textContent = drawn.length
? `${drawn.length} drawn: ${drawn.join(' ')}`
: 'nothing drawn yet';
}
// #synth / #frames autorun, so the tool can be driven headlessly for smoke tests
// and deep-linked. Detection needs WebGL; the synthetic path does not.
window.addEventListener('error', (e) => {
const s = document.getElementById('status');
if (s) { s.textContent = e.message; s.className = 'err'; }
});
if (location.hash === '#synth') runSynthetic();
else if (location.hash === '#frames') runFrames();
else status('ready — Load frames, then step with \u2190 \u2192 and delete with X');
window.__roto = state; // headless smoke test reads this
window.__render = compositeRender; // ...and renders arbitrary frames off-screen
window.__lead = leadIndex; // ...and resolves the performance frame
window.__drawAll = drawAll; // ...and forces a full redraw