arthur/js/pipeline.js

180 lines
7.4 KiB
JavaScript
Raw Normal View History

// Analysis: dense track -> stabilised head-local contours -> selected keys.
// All policy lives here, never in the renderer. See docs/roto-puppet.md,
// "The take is the contract".
import { RIGID, LIPS_OUTER, LIPS_INNER, APERTURE, FACE_OVAL, EYE_INNER, subsampleSlots } from './landmarks.js';
import { fitSimilarity, applySimAll, applySim, fitResidual, procrustesMean, smoothTransforms, movingAverage } from './mathutil.js';
// MediaPipe normalises x by image WIDTH and y by image HEIGHT, so its normalised
// space is anisotropic: for a 1080x1920 frame, one unit of x is 1080px and one
// unit of y is 1920px. Treating those as comparable stretches everything
// horizontally by H/W, and worse, makes fitSimilarity fit a "rotation" in a
// sheared space, so head roll comes out subtly wrong as well.
//
// Multiplying x by aspect = W/H converts to an ISOTROPIC space whose unit is one
// image height, so equal numbers mean equal pixels. Everything downstream -
// Procrustes, the similarity fit, the raster transform - depends on that.
const pick = (lm, idx, aspect) => idx.map((i) => ({ x: lm[i].x * aspect, y: lm[i].y }));
// Stage 1-3: fit the rigid transform per frame, smooth its parameters, then map
// every contour through it into the reference frame. The result is head-local:
// translation, roll and depth-scale of the head are gone.
export function stabilize(dense, smoothRadius, aspect = 1) {
const rigid = dense.map((f) => pick(f, RIGID, aspect));
const ref = procrustesMean(rigid);
const raw = rigid.map((r) => fitSimilarity(r, ref));
const tfs = smoothTransforms(raw, smoothRadius);
return {
ref,
transforms: tfs,
// Rigid landmarks in IMAGE space: the head-pose signal. Frame removal is
// decided from head motion, not from the mouth, so this has to survive the
// fit rather than being consumed by it.
rigid,
// Residual rises with out-of-plane rotation, which no 2D similarity can
// remove. High values mean this section wants a different head plate.
residual: tfs.map((tf, i) => fitResidual(tf, rigid[i], ref)),
outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER, aspect))),
inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER, aspect))),
oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL, aspect))),
eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER, aspect))),
aperture: dense.map((f, i) => {
const a = applySimAll(tfs[i], pick(f, APERTURE, aspect));
return Math.hypot(a[0].x - a[1].x, a[0].y - a[1].y);
}),
};
}
// Stage 4: fixed-index subsample of a stabilised ring, then map from normalised
// face space into character raster space.
export function toRasterRing(stabRing, ringTable, n, xform) {
return subsampleSlots(ringTable.length, n).map((s) => xform(stabRing[s]));
}
// Stage 6: key selection.
//
// Keys go on velocity MINIMA, not on distance thresholds. A threshold fires at
// the frame it was crossed - partway through a transition - so every pose lands
// mushy and late. A minimum is where the shape is momentarily parked, which is
// the pose a viewer actually reads.
//
// Minima alone are not enough: during a long hold the velocity wobbles near zero
// and produces a key per wobble. So a candidate minimum is only accepted if the
// shape has actually moved since the last accepted key (distThresh) and the
// minimum hold has elapsed (minHold).
export function selectKeys(shapes, opts) {
const { minHold, distThresh, velSmooth, exposure } = opts;
const N = shapes.length;
if (N === 0) return { keys: [], velocity: [], candidates: [] };
const vel = new Array(N).fill(0);
for (let t = 1; t < N; t++) {
let acc = 0;
for (let i = 0; i < shapes[t].length; i++) {
acc += Math.hypot(shapes[t][i].x - shapes[t - 1][i].x, shapes[t][i].y - shapes[t - 1][i].y);
}
vel[t] = acc / shapes[t].length;
}
const sv = movingAverage(vel, velSmooth);
const candidates = [];
for (let t = 1; t < N - 1; t++) {
if (sv[t] <= sv[t - 1] && sv[t] <= sv[t + 1]) candidates.push(t);
}
const shapeDist = (a, b) => {
let acc = 0;
for (let i = 0; i < a.length; i++) acc += Math.hypot(a[i].x - b[i].x, a[i].y - b[i].y);
return acc / a.length;
};
const accepted = [0];
for (const t of candidates) {
const last = accepted[accepted.length - 1];
if (t - last < minHold) continue;
if (shapeDist(shapes[t], shapes[last]) < distThresh) continue;
accepted.push(t);
}
// Snap onto the exposure grid. f is what renders; src is provenance.
const keys = [];
for (const src of accepted) {
const f = Math.round(src / exposure) * exposure;
const prev = keys[keys.length - 1];
if (prev && prev.f === f) {
// Two extremes collapsed onto one grid slot: keep the stronger one.
if (sv[src] < sv[prev.src]) { prev.src = src; prev.frame = src; }
continue;
}
keys.push({ f, src, frame: src });
}
return { keys, velocity: sv, candidates };
}
// Resolve which key is live on a given output frame under interp=hold.
// "Most recent key at or before f" - lookup, not policy.
export function activeKey(keys, f) {
let hit = keys[0];
for (const k of keys) { if (k.f <= f) hit = k; else break; }
return hit;
}
// Temporal smoothing of a contour, per vertex, across time.
//
// docs/roto-puppet.md says to smooth the transform and never the contour. That
// was correct while keys were sparse: sampling at velocity minima rejected
// per-frame detector noise for free. With a key on every frame the noise is
// visible as a shimmer along the lip edge, so a bounded exception applies -
// the window must stay SHORTER than the shortest articulation worth keeping.
// At 12fps, mouth movement spans 3-6 frames and detector noise is per-frame, so
// a radius of 1 separates them and a radius of 3 would start eating speech.
//
// `radius` in frames either side: 0 off, 1 = 3-frame average, 2 = 5-frame.
export function smoothContours(rings, radius) {
if (radius <= 0) return rings;
const half = Math.floor(radius), N = rings.length, V = rings[0].length;
const out = [];
for (let t = 0; t < N; t++) {
const frame = [];
for (let v = 0; v < V; v++) {
let sx = 0, sy = 0, c = 0;
for (let j = t - half; j <= t + half; j++) {
const k = Math.min(N - 1, Math.max(0, j));
sx += rings[k][v].x; sy += rings[k][v].y; c++;
}
frame.push({ x: sx / c, y: sy / c });
}
out.push(frame);
}
return out;
}
// Which frames need their own PLATE drawing.
//
// This is frame removal, not keyframe extraction: every frame is a candidate and
// the question is which can be dropped. Walk forward holding the current drawing
// until the head has moved further than `tol` from it, then a new drawing is
// required. The cost being managed is an artist drawing a head, which is why the
// signal is head pose and not the mouth - the mouth is traced and free.
export function suggestPlateFrames(rigid, tol) {
const dist = (a, b) => {
let m = 0;
for (let i = 0; i < a.length; i++) m = Math.max(m, Math.hypot(a[i].x - b[i].x, a[i].y - b[i].y));
return m;
};
const keep = [0];
let anchor = 0;
for (let f = 1; f < rigid.length; f++) {
if (dist(rigid[f], rigid[anchor]) > tol) { keep.push(f); anchor = f; }
}
return keep;
}
// Nearest kept frame at or before f - the plate that is on screen.
export function heldFrame(kept, f) {
let hit = kept[0];
for (const k of kept) { if (k <= f) hit = k; else break; }
return hit;
}