2026-09-24 14:38:07 -04:00
|
|
|
// Analysis: dense track -> stabilised head-local contours -> selected keys.
|
|
|
|
|
// All policy lives here, never in the renderer. See docs/roto-puppet.md,
|
|
|
|
|
// "The take is the contract".
|
|
|
|
|
|
|
|
|
|
import { RIGID, LIPS_OUTER, LIPS_INNER, APERTURE, FACE_OVAL, EYE_INNER, subsampleSlots } from './landmarks.js';
|
|
|
|
|
import { fitSimilarity, applySimAll, applySim, fitResidual, procrustesMean, smoothTransforms, movingAverage } from './mathutil.js';
|
|
|
|
|
|
2026-09-24 15:03:12 -04:00
|
|
|
// MediaPipe normalises x by image WIDTH and y by image HEIGHT, so its normalised
|
|
|
|
|
// space is anisotropic: for a 1080x1920 frame, one unit of x is 1080px and one
|
|
|
|
|
// unit of y is 1920px. Treating those as comparable stretches everything
|
|
|
|
|
// horizontally by H/W, and worse, makes fitSimilarity fit a "rotation" in a
|
|
|
|
|
// sheared space, so head roll comes out subtly wrong as well.
|
|
|
|
|
//
|
|
|
|
|
// Multiplying x by aspect = W/H converts to an ISOTROPIC space whose unit is one
|
|
|
|
|
// image height, so equal numbers mean equal pixels. Everything downstream -
|
|
|
|
|
// Procrustes, the similarity fit, the raster transform - depends on that.
|
|
|
|
|
const pick = (lm, idx, aspect) => idx.map((i) => ({ x: lm[i].x * aspect, y: lm[i].y }));
|
2026-09-24 14:38:07 -04:00
|
|
|
|
|
|
|
|
// Stage 1-3: fit the rigid transform per frame, smooth its parameters, then map
|
|
|
|
|
// every contour through it into the reference frame. The result is head-local:
|
|
|
|
|
// translation, roll and depth-scale of the head are gone.
|
2026-09-24 15:03:12 -04:00
|
|
|
export function stabilize(dense, smoothRadius, aspect = 1) {
|
|
|
|
|
const rigid = dense.map((f) => pick(f, RIGID, aspect));
|
2026-09-24 14:38:07 -04:00
|
|
|
const ref = procrustesMean(rigid);
|
|
|
|
|
const raw = rigid.map((r) => fitSimilarity(r, ref));
|
2026-09-24 14:51:15 -04:00
|
|
|
const tfs = smoothTransforms(raw, smoothRadius);
|
2026-09-24 14:38:07 -04:00
|
|
|
|
|
|
|
|
return {
|
|
|
|
|
ref,
|
|
|
|
|
transforms: tfs,
|
2026-09-24 14:51:15 -04:00
|
|
|
// Rigid landmarks in IMAGE space: the head-pose signal. Frame removal is
|
|
|
|
|
// decided from head motion, not from the mouth, so this has to survive the
|
|
|
|
|
// fit rather than being consumed by it.
|
|
|
|
|
rigid,
|
2026-09-24 14:38:07 -04:00
|
|
|
// Residual rises with out-of-plane rotation, which no 2D similarity can
|
|
|
|
|
// remove. High values mean this section wants a different head plate.
|
|
|
|
|
residual: tfs.map((tf, i) => fitResidual(tf, rigid[i], ref)),
|
2026-09-24 15:03:12 -04:00
|
|
|
outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER, aspect))),
|
|
|
|
|
inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER, aspect))),
|
|
|
|
|
oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL, aspect))),
|
|
|
|
|
eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER, aspect))),
|
2026-09-24 14:38:07 -04:00
|
|
|
aperture: dense.map((f, i) => {
|
2026-09-24 15:03:12 -04:00
|
|
|
const a = applySimAll(tfs[i], pick(f, APERTURE, aspect));
|
2026-09-24 14:38:07 -04:00
|
|
|
return Math.hypot(a[0].x - a[1].x, a[0].y - a[1].y);
|
|
|
|
|
}),
|
|
|
|
|
};
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Stage 4: fixed-index subsample of a stabilised ring, then map from normalised
|
|
|
|
|
// face space into character raster space.
|
|
|
|
|
export function toRasterRing(stabRing, ringTable, n, xform) {
|
|
|
|
|
return subsampleSlots(ringTable.length, n).map((s) => xform(stabRing[s]));
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Stage 6: key selection.
|
|
|
|
|
//
|
|
|
|
|
// Keys go on velocity MINIMA, not on distance thresholds. A threshold fires at
|
|
|
|
|
// the frame it was crossed - partway through a transition - so every pose lands
|
|
|
|
|
// mushy and late. A minimum is where the shape is momentarily parked, which is
|
|
|
|
|
// the pose a viewer actually reads.
|
|
|
|
|
//
|
|
|
|
|
// Minima alone are not enough: during a long hold the velocity wobbles near zero
|
|
|
|
|
// and produces a key per wobble. So a candidate minimum is only accepted if the
|
|
|
|
|
// shape has actually moved since the last accepted key (distThresh) and the
|
|
|
|
|
// minimum hold has elapsed (minHold).
|
|
|
|
|
export function selectKeys(shapes, opts) {
|
|
|
|
|
const { minHold, distThresh, velSmooth, exposure } = opts;
|
|
|
|
|
const N = shapes.length;
|
|
|
|
|
if (N === 0) return { keys: [], velocity: [], candidates: [] };
|
|
|
|
|
|
|
|
|
|
const vel = new Array(N).fill(0);
|
|
|
|
|
for (let t = 1; t < N; t++) {
|
|
|
|
|
let acc = 0;
|
|
|
|
|
for (let i = 0; i < shapes[t].length; i++) {
|
|
|
|
|
acc += Math.hypot(shapes[t][i].x - shapes[t - 1][i].x, shapes[t][i].y - shapes[t - 1][i].y);
|
|
|
|
|
}
|
|
|
|
|
vel[t] = acc / shapes[t].length;
|
|
|
|
|
}
|
|
|
|
|
const sv = movingAverage(vel, velSmooth);
|
|
|
|
|
|
|
|
|
|
const candidates = [];
|
|
|
|
|
for (let t = 1; t < N - 1; t++) {
|
|
|
|
|
if (sv[t] <= sv[t - 1] && sv[t] <= sv[t + 1]) candidates.push(t);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
const shapeDist = (a, b) => {
|
|
|
|
|
let acc = 0;
|
|
|
|
|
for (let i = 0; i < a.length; i++) acc += Math.hypot(a[i].x - b[i].x, a[i].y - b[i].y);
|
|
|
|
|
return acc / a.length;
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
const accepted = [0];
|
|
|
|
|
for (const t of candidates) {
|
|
|
|
|
const last = accepted[accepted.length - 1];
|
|
|
|
|
if (t - last < minHold) continue;
|
|
|
|
|
if (shapeDist(shapes[t], shapes[last]) < distThresh) continue;
|
|
|
|
|
accepted.push(t);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Snap onto the exposure grid. f is what renders; src is provenance.
|
|
|
|
|
const keys = [];
|
|
|
|
|
for (const src of accepted) {
|
|
|
|
|
const f = Math.round(src / exposure) * exposure;
|
|
|
|
|
const prev = keys[keys.length - 1];
|
|
|
|
|
if (prev && prev.f === f) {
|
|
|
|
|
// Two extremes collapsed onto one grid slot: keep the stronger one.
|
|
|
|
|
if (sv[src] < sv[prev.src]) { prev.src = src; prev.frame = src; }
|
|
|
|
|
continue;
|
|
|
|
|
}
|
|
|
|
|
keys.push({ f, src, frame: src });
|
|
|
|
|
}
|
|
|
|
|
return { keys, velocity: sv, candidates };
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Resolve which key is live on a given output frame under interp=hold.
|
|
|
|
|
// "Most recent key at or before f" - lookup, not policy.
|
|
|
|
|
export function activeKey(keys, f) {
|
|
|
|
|
let hit = keys[0];
|
|
|
|
|
for (const k of keys) { if (k.f <= f) hit = k; else break; }
|
|
|
|
|
return hit;
|
|
|
|
|
}
|
2026-09-24 14:51:15 -04:00
|
|
|
|
|
|
|
|
// Temporal smoothing of a contour, per vertex, across time.
|
|
|
|
|
//
|
|
|
|
|
// docs/roto-puppet.md says to smooth the transform and never the contour. That
|
|
|
|
|
// was correct while keys were sparse: sampling at velocity minima rejected
|
|
|
|
|
// per-frame detector noise for free. With a key on every frame the noise is
|
|
|
|
|
// visible as a shimmer along the lip edge, so a bounded exception applies -
|
|
|
|
|
// the window must stay SHORTER than the shortest articulation worth keeping.
|
|
|
|
|
// At 12fps, mouth movement spans 3-6 frames and detector noise is per-frame, so
|
|
|
|
|
// a radius of 1 separates them and a radius of 3 would start eating speech.
|
|
|
|
|
//
|
|
|
|
|
// `radius` in frames either side: 0 off, 1 = 3-frame average, 2 = 5-frame.
|
|
|
|
|
export function smoothContours(rings, radius) {
|
|
|
|
|
if (radius <= 0) return rings;
|
|
|
|
|
const half = Math.floor(radius), N = rings.length, V = rings[0].length;
|
|
|
|
|
const out = [];
|
|
|
|
|
for (let t = 0; t < N; t++) {
|
|
|
|
|
const frame = [];
|
|
|
|
|
for (let v = 0; v < V; v++) {
|
|
|
|
|
let sx = 0, sy = 0, c = 0;
|
|
|
|
|
for (let j = t - half; j <= t + half; j++) {
|
|
|
|
|
const k = Math.min(N - 1, Math.max(0, j));
|
|
|
|
|
sx += rings[k][v].x; sy += rings[k][v].y; c++;
|
|
|
|
|
}
|
|
|
|
|
frame.push({ x: sx / c, y: sy / c });
|
|
|
|
|
}
|
|
|
|
|
out.push(frame);
|
|
|
|
|
}
|
|
|
|
|
return out;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Which frames need their own PLATE drawing.
|
|
|
|
|
//
|
|
|
|
|
// This is frame removal, not keyframe extraction: every frame is a candidate and
|
|
|
|
|
// the question is which can be dropped. Walk forward holding the current drawing
|
|
|
|
|
// until the head has moved further than `tol` from it, then a new drawing is
|
|
|
|
|
// required. The cost being managed is an artist drawing a head, which is why the
|
|
|
|
|
// signal is head pose and not the mouth - the mouth is traced and free.
|
|
|
|
|
export function suggestPlateFrames(rigid, tol) {
|
|
|
|
|
const dist = (a, b) => {
|
|
|
|
|
let m = 0;
|
|
|
|
|
for (let i = 0; i < a.length; i++) m = Math.max(m, Math.hypot(a[i].x - b[i].x, a[i].y - b[i].y));
|
|
|
|
|
return m;
|
|
|
|
|
};
|
|
|
|
|
const keep = [0];
|
|
|
|
|
let anchor = 0;
|
|
|
|
|
for (let f = 1; f < rigid.length; f++) {
|
|
|
|
|
if (dist(rigid[f], rigid[anchor]) > tol) { keep.push(f); anchor = f; }
|
|
|
|
|
}
|
|
|
|
|
return keep;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Nearest kept frame at or before f - the plate that is on screen.
|
|
|
|
|
export function heldFrame(kept, f) {
|
|
|
|
|
let hit = kept[0];
|
|
|
|
|
for (const k of kept) { if (k <= f) hit = k; else break; }
|
|
|
|
|
return hit;
|
|
|
|
|
}
|