The test renderer turned out to be the product. Everything that decides how the work looks - stabilisation, reduction, timing, frame removal, palette - already happens here, and the flat indexed output already reads the way it should. The reason to leave is in the original design's own rule: never make a timing decision that requires a full render to evaluate. Honouring that moved every judgement out of Animator Pro, which left the host doing nothing but writing a file, in exchange for modal UI, minutes-long renders, one-level undo, FLX delta invariants, a single tween state and a cel singleton. What does NOT change is the constraint. 320x200, indexed palette, flat fills, no antialiasing - inherited, but load-bearing rather than accidental. The rasteriser writes palette indices and expands to RGBA only at the end precisely so nothing can soften an edge. Modern conveniences belong in the workflow. Adds docs/design.md: the principles, carried over without the Poco/FLX/cel machinery, plus architecture and an honest list of what is missing - the largest gap being that plates still have nowhere to be drawn. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
188 lines
7.8 KiB
JavaScript
188 lines
7.8 KiB
JavaScript
// Analysis: dense track -> stabilised head-local contours -> selected keys.
|
|
// All policy lives here, never in the renderer. See docs/design.md,
|
|
// "The take is the contract".
|
|
|
|
import { RIGID, LIPS_OUTER, LIPS_INNER, APERTURE, FACE_OVAL, EYE_INNER, subsampleSlots } from './landmarks.js';
|
|
import { fitSimilarity, applySimAll, applySim, fitResidual, procrustesMean, smoothTransforms, movingAverage } from './mathutil.js';
|
|
|
|
// MediaPipe normalises x by image WIDTH and y by image HEIGHT, so its normalised
|
|
// space is anisotropic: for a 1080x1920 frame, one unit of x is 1080px and one
|
|
// unit of y is 1920px. Treating those as comparable stretches everything
|
|
// horizontally by H/W, and worse, makes fitSimilarity fit a "rotation" in a
|
|
// sheared space, so head roll comes out subtly wrong as well.
|
|
//
|
|
// Multiplying x by aspect = W/H converts to an ISOTROPIC space whose unit is one
|
|
// image height, so equal numbers mean equal pixels. Everything downstream -
|
|
// Procrustes, the similarity fit, the raster transform - depends on that.
|
|
const pick = (lm, idx, aspect) => idx.map((i) => ({ x: lm[i].x * aspect, y: lm[i].y }));
|
|
|
|
// Stage 1-3: fit the rigid transform per frame, smooth its parameters, then map
|
|
// every contour through it into the reference frame. The result is head-local:
|
|
// translation, roll and depth-scale of the head are gone.
|
|
export function stabilize(dense, smoothRadius, aspect = 1) {
|
|
const rigid = dense.map((f) => pick(f, RIGID, aspect));
|
|
const ref = procrustesMean(rigid);
|
|
const raw = rigid.map((r) => fitSimilarity(r, ref));
|
|
const tfs = smoothTransforms(raw, smoothRadius);
|
|
|
|
return {
|
|
ref,
|
|
transforms: tfs,
|
|
// Rigid landmarks in IMAGE space: the head-pose signal. Frame removal is
|
|
// decided from head motion, not from the mouth, so this has to survive the
|
|
// fit rather than being consumed by it.
|
|
rigid,
|
|
// Residual rises with out-of-plane rotation, which no 2D similarity can
|
|
// remove. High values mean this section wants a different head plate.
|
|
residual: tfs.map((tf, i) => fitResidual(tf, rigid[i], ref)),
|
|
outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER, aspect))),
|
|
inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER, aspect))),
|
|
oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL, aspect))),
|
|
eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER, aspect))),
|
|
aperture: dense.map((f, i) => {
|
|
const a = applySimAll(tfs[i], pick(f, APERTURE, aspect));
|
|
return Math.hypot(a[0].x - a[1].x, a[0].y - a[1].y);
|
|
}),
|
|
};
|
|
}
|
|
|
|
// Stage 4: fixed-index subsample of a stabilised ring, then map from normalised
|
|
// face space into character raster space.
|
|
export function toRasterRing(stabRing, ringTable, n, xform) {
|
|
return subsampleSlots(ringTable.length, n).map((s) => xform(stabRing[s]));
|
|
}
|
|
|
|
// Stage 6: key selection.
|
|
//
|
|
// Keys go on velocity MINIMA, not on distance thresholds. A threshold fires at
|
|
// the frame it was crossed - partway through a transition - so every pose lands
|
|
// mushy and late. A minimum is where the shape is momentarily parked, which is
|
|
// the pose a viewer actually reads.
|
|
//
|
|
// Minima alone are not enough: during a long hold the velocity wobbles near zero
|
|
// and produces a key per wobble. So a candidate minimum is only accepted if the
|
|
// shape has actually moved since the last accepted key (distThresh) and the
|
|
// minimum hold has elapsed (minHold).
|
|
export function selectKeys(shapes, opts) {
|
|
const { minHold, distThresh, velSmooth, exposure } = opts;
|
|
const N = shapes.length;
|
|
if (N === 0) return { keys: [], velocity: [], candidates: [] };
|
|
|
|
const vel = new Array(N).fill(0);
|
|
for (let t = 1; t < N; t++) {
|
|
let acc = 0;
|
|
for (let i = 0; i < shapes[t].length; i++) {
|
|
acc += Math.hypot(shapes[t][i].x - shapes[t - 1][i].x, shapes[t][i].y - shapes[t - 1][i].y);
|
|
}
|
|
vel[t] = acc / shapes[t].length;
|
|
}
|
|
const sv = movingAverage(vel, velSmooth);
|
|
|
|
const candidates = [];
|
|
for (let t = 1; t < N - 1; t++) {
|
|
if (sv[t] <= sv[t - 1] && sv[t] <= sv[t + 1]) candidates.push(t);
|
|
}
|
|
|
|
const shapeDist = (a, b) => {
|
|
let acc = 0;
|
|
for (let i = 0; i < a.length; i++) acc += Math.hypot(a[i].x - b[i].x, a[i].y - b[i].y);
|
|
return acc / a.length;
|
|
};
|
|
|
|
const accepted = [0];
|
|
for (const t of candidates) {
|
|
const last = accepted[accepted.length - 1];
|
|
if (t - last < minHold) continue;
|
|
if (shapeDist(shapes[t], shapes[last]) < distThresh) continue;
|
|
accepted.push(t);
|
|
}
|
|
|
|
// Snap onto the exposure grid. f is what renders; src is provenance.
|
|
const keys = [];
|
|
for (const src of accepted) {
|
|
const f = Math.round(src / exposure) * exposure;
|
|
const prev = keys[keys.length - 1];
|
|
if (prev && prev.f === f) {
|
|
// Two extremes collapsed onto one grid slot: keep the stronger one.
|
|
if (sv[src] < sv[prev.src]) { prev.src = src; prev.frame = src; }
|
|
continue;
|
|
}
|
|
keys.push({ f, src, frame: src });
|
|
}
|
|
return { keys, velocity: sv, candidates };
|
|
}
|
|
|
|
// Resolve which key is live on a given output frame under interp=hold.
|
|
// "Most recent key at or before f" - lookup, not policy.
|
|
export function activeKey(keys, f) {
|
|
let hit = keys[0];
|
|
for (const k of keys) { if (k.f <= f) hit = k; else break; }
|
|
return hit;
|
|
}
|
|
|
|
// Temporal smoothing of a contour, per vertex, across time.
|
|
//
|
|
// docs/design.md says to smooth the transform and never the contour. That
|
|
// was correct while keys were sparse: sampling at velocity minima rejected
|
|
// per-frame detector noise for free. With a key on every frame the noise is
|
|
// visible as a shimmer along the lip edge, so a bounded exception applies -
|
|
// the window must stay SHORTER than the shortest articulation worth keeping.
|
|
// At 12fps, mouth movement spans 3-6 frames and detector noise is per-frame, so
|
|
// a radius of 1 separates them and a radius of 3 would start eating speech.
|
|
//
|
|
// `radius` in frames either side: 0 off, 1 = 3-frame average, 2 = 5-frame.
|
|
export function smoothContours(rings, radius) {
|
|
if (radius <= 0) return rings;
|
|
const half = Math.floor(radius), N = rings.length, V = rings[0].length;
|
|
const out = [];
|
|
for (let t = 0; t < N; t++) {
|
|
const frame = [];
|
|
for (let v = 0; v < V; v++) {
|
|
let sx = 0, sy = 0, c = 0;
|
|
for (let j = t - half; j <= t + half; j++) {
|
|
const k = Math.min(N - 1, Math.max(0, j));
|
|
sx += rings[k][v].x; sy += rings[k][v].y; c++;
|
|
}
|
|
frame.push({ x: sx / c, y: sy / c });
|
|
}
|
|
out.push(frame);
|
|
}
|
|
return out;
|
|
}
|
|
|
|
// Which frames need their own PLATE drawing.
|
|
//
|
|
// This is frame removal, not keyframe extraction: every frame is a candidate and
|
|
// the question is which can be dropped. Walk forward holding the current drawing
|
|
// until the head has moved further than `tol` from it, then a new drawing is
|
|
// required. The cost being managed is an artist drawing a head, which is why the
|
|
// signal is head pose and not the mouth - the mouth is traced and free.
|
|
export function suggestPlateFrames(rigid, tol) {
|
|
const dist = (a, b) => {
|
|
let m = 0;
|
|
for (let i = 0; i < a.length; i++) m = Math.max(m, Math.hypot(a[i].x - b[i].x, a[i].y - b[i].y));
|
|
return m;
|
|
};
|
|
const keep = [0];
|
|
let anchor = 0;
|
|
for (let f = 1; f < rigid.length; f++) {
|
|
if (dist(rigid[f], rigid[anchor]) > tol) { keep.push(f); anchor = f; }
|
|
}
|
|
return keep;
|
|
}
|
|
|
|
// Nearest kept frame at or before f - the plate that is on screen.
|
|
export function heldFrame(kept, f) {
|
|
let hit = kept[0];
|
|
for (const k of kept) { if (k <= f) hit = k; else break; }
|
|
return hit;
|
|
}
|
|
|
|
// Shift a performance track against the clock, clamped at the ends.
|
|
//
|
|
// Pure and exported so the shift can actually be asserted: "the slider feels
|
|
// like it does nothing" is otherwise indistinguishable from "the slider does
|
|
// nothing", and at 24fps a lead of 1 is 42ms, which is small enough to doubt.
|
|
export function shiftIndex(f, lead, n) {
|
|
return Math.min(n - 1, Math.max(0, f + lead));
|
|
}
|