// Analysis: dense track -> stabilised head-local contours -> selected keys. // All policy lives here, never in the renderer. See docs/design.md, // "The take is the contract". import { RIGID, LIPS_OUTER, LIPS_INNER, APERTURE, FACE_OVAL, EYE_INNER, EYE_R_RING, EYE_L_RING, EYE_R_CORNERS, EYE_L_CORNERS, EYE_R_LIDS, EYE_L_LIDS, IRIS_A, IRIS_B, BROW_A_RING, BROW_B_RING, BROW_END_0, BROW_END_1, subsampleSlots } from './landmarks.js'; import { fitSimilarity, applySimAll, applySim, fitResidual, procrustesMean, smoothTransforms, movingAverage } from './mathutil.js'; // MediaPipe normalises x by image WIDTH and y by image HEIGHT, so its normalised // space is anisotropic: for a 1080x1920 frame, one unit of x is 1080px and one // unit of y is 1920px. Treating those as comparable stretches everything // horizontally by H/W, and worse, makes fitSimilarity fit a "rotation" in a // sheared space, so head roll comes out subtly wrong as well. // // Multiplying x by aspect = W/H converts to an ISOTROPIC space whose unit is one // image height, so equal numbers mean equal pixels. Everything downstream - // Procrustes, the similarity fit, the raster transform - depends on that. const pick = (lm, idx, aspect) => idx.map((i) => ({ x: lm[i].x * aspect, y: lm[i].y })); // Stage 1-3: fit the rigid transform per frame, smooth its parameters, then map // every contour through it into the reference frame. The result is head-local: // translation, roll and depth-scale of the head are gone. export function stabilize(dense, smoothRadius, aspect = 1) { const rigid = dense.map((f) => pick(f, RIGID, aspect)); const ref = procrustesMean(rigid); const raw = rigid.map((r) => fitSimilarity(r, ref)); const tfs = smoothTransforms(raw, smoothRadius); // The refined mesh appends ten iris points to the 468 face points, but a // plain mesh does not, and synthetic or hand-fed tracks need not. Checked // rather than assumed: reading past the end would surface as NaN gaze deep // downstream instead of as "this track carries no iris". const hasIris = dense.every((f) => f && f.length > IRIS_B[IRIS_B.length - 1]); const map = (table) => dense.map((f, i) => applySimAll(tfs[i], pick(f, table, aspect))); return { ref, transforms: tfs, // Rigid landmarks in IMAGE space: the head-pose signal. Frame removal is // decided from head motion, not from the mouth, so this has to survive the // fit rather than being consumed by it. rigid, // Residual rises with out-of-plane rotation, which no 2D similarity can // remove. High values mean this section wants a different head plate. residual: tfs.map((tf, i) => fitResidual(tf, rigid[i], ref)), outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER, aspect))), inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER, aspect))), oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL, aspect))), eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER, aspect))), aperture: dense.map((f, i) => { const a = applySimAll(tfs[i], pick(f, APERTURE, aspect)); return Math.hypot(a[0].x - a[1].x, a[0].y - a[1].y); }), // Eyes. Lid rings are a feature and get traced like the mouth; corners and // lid centres are the measurement frame; the iris blocks are raw until // pairIrises decides which is which. lidR: map(EYE_R_RING), lidL: map(EYE_L_RING), cornersR: map(EYE_R_CORNERS), cornersL: map(EYE_L_CORNERS), lidsR: map(EYE_R_LIDS), lidsL: map(EYE_L_LIDS), irisA: hasIris ? map(IRIS_A) : null, irisB: hasIris ? map(IRIS_B) : null, browA: map(BROW_A_RING), browB: map(BROW_B_RING), }; } /* ---------- brows ---------- */ // Two correspondences resolved from geometry, for the same reason the iris // pairing is: a wrong guess here is survivable enough to escape notice. // // Which ring is which brow follows MediaPipe's left/right naming, which is the // naming that would have put the irises on the wrong eyes. Which END of a ring // is the OUTER one matters more: get it backwards and the tilt mirrors, so // inner-up "worried" renders as outer-up, which is a different expression // rather than a broken one. It would read as a directed performance choice and // never be questioned. // // Both are decided by voting across every frame against landmarks already known // to be rigid, so one bad detection cannot swing them. export function pairBrows(stab) { const N = stab.browA.length; const cen = (ring) => { let x = 0; for (const p of ring) x += p.x; return x / ring.length; }; let side = 0, ends = 0; for (let f = 0; f < N; f++) { const cR = mid(stab.cornersR[f][0], stab.cornersR[f][1]).x; const cL = mid(stab.cornersL[f][0], stab.cornersL[f][1]).x; side += Math.abs(cen(stab.browA[f]) - cR) < Math.abs(cen(stab.browA[f]) - cL) ? 1 : -1; // EYE_R_CORNERS is [outer, inner], so this asks whether slot 0 of the ring // sits nearer the eye's outer corner than its inner one. const ring = side > 0 ? stab.browA[f] : stab.browB[f]; const co = side > 0 ? stab.cornersR[f] : stab.cornersL[f]; const s0 = ring[BROW_END_0[0]]; ends += Math.abs(s0.x - co[0].x) < Math.abs(s0.x - co[1].x) ? 1 : -1; } return { right: side > 0 ? 'browA' : 'browB', left: side > 0 ? 'browB' : 'browA', outerAtSlot0: ends > 0, }; } // Brow height above its own eye, at each end, in eye widths. // // Measured against the eye's CORNER MIDPOINT, not the lid: the corners are // rigid, so a blink cannot read as a brow raise. That is the same trap the gaze // origin has and it is worth avoiding twice - brows and lids move together // constantly, and a brow that jumped on every blink would look like a tic. // // Two ends rather than one height, because raise and tilt are different // expressions built from the same measurement: both ends up is surprise, inner // up alone is worry, inner down is anger. One number could not tell them apart. export function browSignals(stab) { const N = stab.browA.length; const pairing = pairBrows(stab); const endOuter = pairing.outerAtSlot0 ? BROW_END_0 : BROW_END_1; const endInner = pairing.outerAtSlot0 ? BROW_END_1 : BROW_END_0; const out = { R: [], L: [], pairing }; for (let f = 0; f < N; f++) { for (const [side, corners] of [['R', stab.cornersR], ['L', stab.cornersL]]) { const ring = stab[pairing[side === 'R' ? 'right' : 'left']][f]; const c = mid(corners[f][0], corners[f][1]); const w = dist(corners[f][0], corners[f][1]); const at = (pair) => (ring[pair[0]].y + ring[pair[1]].y) / 2; // y grows downward, so a brow ABOVE the eye gives a positive raise. out[side].push({ x: (c.y - at(endOuter)) / w, y: (c.y - at(endInner)) / w }); } } return out; } /* ---------- eyes ---------- */ const mid = (a, b) => ({ x: (a.x + b.x) / 2, y: (a.y + b.y) / 2 }); const dist = (a, b) => Math.hypot(a.x - b.x, a.y - b.y); // Which iris block belongs to which eye is RESOLVED FROM THE DATA, not declared // in a table. // // The naming in MediaPipe's own material is viewer-relative in some places and // subject-relative in others, and the two blocks are otherwise // indistinguishable. Getting it backwards swaps the irises, which looks almost // right - each eye still has a disc in roughly the right place - so it survives // a casual eyeball and then reads as a subtly wall-eyed character for the rest // of the project. Proximity to the eye's corner midpoint settles it in one // comparison, is impossible to get wrong, and keeps working if the model is // ever renumbered. // // Voted across every frame rather than read off frame zero: one bad detection // should not decide the whole shot. export function pairIrises(stab) { if (!stab.irisA) return null; let votes = 0; for (let f = 0; f < stab.irisA.length; f++) { const cR = mid(stab.cornersR[f][0], stab.cornersR[f][1]); votes += dist(stab.irisA[f][0], cR) < dist(stab.irisB[f][0], cR) ? 1 : -1; } return votes > 0 ? { right: 'irisA', left: 'irisB' } : { right: 'irisB', left: 'irisA' }; } // Per-frame eye measurements, in units of eye width. Measurement only - every // threshold and every stylisation is applied by the callers. // // Everything here stays in HEAD-LOCAL space, which is the same space the mouth // lives in and the same space the registered photo underlay is drawn in. An // earlier version pinned each eye into a fixed socket fitted to its corners' // mean over the shot. That does remove the wobble, but it removes too much: the // residual from out-of-plane rotation is real motion of the eye relative to the // head, it is still there in the footage, and pinning it away leaves the drawn // eyes hanging still over a photo whose eyes are moving. The eye has to track // the face exactly as the mouth does. // // The wobble the socket was aimed at is dealt with the way docs/design.md deals // with it everywhere else - the bounded contour average, the same knob and the // same radius the mouth uses - and by placing the iris in the frame of the // ALREADY-SMOOTHED lid ring, so the iris cannot jitter independently of the eye // it sits in. See buildEyes in app.js. export function eyeSignals(stab) { const N = stab.transforms.length; const pairing = pairIrises(stab); const openR = [], openL = [], gazeRaw = [], gazeR = [], gazeL = []; for (let f = 0; f < N; f++) { const cR = mid(stab.cornersR[f][0], stab.cornersR[f][1]); const cL = mid(stab.cornersL[f][0], stab.cornersL[f][1]); const wR = dist(stab.cornersR[f][0], stab.cornersR[f][1]); const wL = dist(stab.cornersL[f][0], stab.cornersL[f][1]); // Openness is the lid gap over the CORNER distance. Normalising by the // corners rather than by anything derived from the lids keeps the // denominator rigid, so the ratio measures the lid and nothing else, and // one threshold carries across takes, faces and framings. openR.push(dist(stab.lidsR[f][0], stab.lidsR[f][1]) / wR); openL.push(dist(stab.lidsL[f][0], stab.lidsL[f][1]) / wL); if (!pairing) { gazeRaw.push({ x: 0, y: 0 }); gazeR.push({ x: 0, y: 0 }); gazeL.push({ x: 0, y: 0 }); continue; } const iR = stab[pairing.right][f][0], iL = stab[pairing.left][f][0]; // Gaze is the iris centre relative to the CORNER MIDPOINT, in eye widths - // a pure offset WITHIN the eye, with the eye's own position divided out, so // that quantising it quantises the glance and not the head motion carrying // it. // // Measuring against the lid ring's centroid instead would track the lid: // every blink pulls that centroid down and would fake a glance at the // floor, on precisely the frames where the eye is most conspicuous. The // corners are in RIGID, so this origin and this denominator are both immune // to the performance they are measuring. const gR = { x: (iR.x - cR.x) / wR, y: (iR.y - cR.y) / wR }; const gL = { x: (iL.x - cL.x) / wL, y: (iL.y - cL.y) / wL }; // ONE gaze for both eyes, and deliberately so. At 320x200 an iris is a // handful of pixels and its centre comes from five landmarks on an eye // twenty pixels wide, so the difference between the two measurements is // noise, not vergence - and independent per-eye noise reads as wall-eyed // immediately, which is the most expensive artefact on a face. Openness // stays per-eye, because a wink is real performance and should survive. gazeRaw.push({ x: (gR.x + gL.x) / 2, y: (gR.y + gL.y) / 2 }); // Kept separately purely as a diagnostic. The two eyes should agree; when // they disagree in a sustained way rather than frame to frame, that is not // noise but out-of-plane head rotation biasing the projected iris offset, // and no 2D measurement can undo it. gazeR.push(gR); gazeL.push(gL); } return { openR, openL, gazeRaw, gazeR, gazeL, hasIris: !!pairing }; } // Where "not looking anywhere in particular" sits on THIS face. Everything the // character does is measured as a departure from it, so getting it wrong does // not bias the gaze slightly - it re-points the whole performance. // // `median` is the default and the safe one: the middle of the take, per axis. // docs/design.md already gives this rule for the anchor fit - the reference is // the MEAN configuration over the shot, not one frame - and gaze needs it for // the same reason. The median rather than the mean because a couple of frames // of hard glance should not drag the rest-point after them. // // `neutral` reads the origin off the take's neutral frame instead, which is // only correct when there genuinely is a held neutral to read. That frame is // chosen by MINIMUM MOUTH APERTURE, and a closed mouth says nothing whatever // about where the eyes are pointed - so on footage with no deliberate neutral // at the top it is an arbitrary frame, and whichever way the performer happened // to glance on it becomes "straight ahead" for the entire shot. It is kept // because it is right when the take was shot for this tool, and because being // able to switch is how you find out that it was not. export function gazeOrigin(gazeRaw, mode = 'median', neutral = 0, radius = 2) { if (mode === 'neutral') { let sx = 0, sy = 0, n = 0; // A window, not a single frame: one frame of a five-landmark iris centre is // worth about a pixel of noise, and that pixel would become a permanent // squint in the output. for (let f = neutral - radius; f <= neutral + radius; f++) { const k = Math.min(gazeRaw.length - 1, Math.max(0, f)); sx += gazeRaw[k].x; sy += gazeRaw[k].y; n++; } return { x: sx / n, y: sy / n }; } const mid1 = (vals) => { const v = vals.slice().sort((a, b) => a - b); return v.length % 2 ? v[(v.length - 1) / 2] : (v[v.length / 2 - 1] + v[v.length / 2]) / 2; }; return { x: mid1(gazeRaw.map((g) => g.x)), y: mid1(gazeRaw.map((g) => g.y)) }; } // Snap a two-channel track onto a grid, then require a new cell to hold before // it takes. Gaze uses it for (x, y); brows use it for (outer raise, inner raise), // where sharing the dwell is the point - a brow whose inner end arrived a frame // before its outer end would crawl instead of snapping. // // This is the "Primitive - quantised" row of the part table in docs/design.md, // and it is not a stylisation imposed on the truth: real eyes move in saccades, // holding a fixation and then jumping. The smooth drift left in the measurement // is tracker noise plus head-compensation error, so snapping to a grid and // requiring a dwell removes the noise and recovers the saccade in the same // operation - the rare case where the aesthetic rule and the physiology agree. // // The dwell is what stops a gaze parked on a cell boundary from chattering // between two cells forever. It is meaningless without a grid, because // continuous values never repeat, so step 0 short-circuits both. export function quantizeSnap(track, step, dwell) { if (!(step > 0)) return track.map((g) => ({ x: g.x, y: g.y })); const q = track.map((g) => ({ x: Math.round(g.x / step) * step, y: Math.round(g.y / step) * step, })); if (dwell <= 0 || !q.length) return q; const out = []; let live = q[0], pend = q[0], run = 0; for (const g of q) { if (g.x === pend.x && g.y === pend.y) run++; else { pend = g; run = 1; } if (run > dwell && (pend.x !== live.x || pend.y !== live.y)) live = pend; out.push(live); } return out; } // Resolve openness into a shut/open decision per frame. // // `dwell` is the same guard the teeth get: a lid hovering at the threshold must // commit before the state changes, so it cannot flicker. // // `hold` is the one that is NOT like the teeth, and it is the whole reason // blinks are worth special-casing. A blink is 100-150ms, which at 12fps is one // frame and at 24fps is two or three - and a single frame of closed eye reads // as a dropped frame, not as a blink. Animators draw a blink over two or three // drawings for exactly that reason. So once the eye shuts it stays shut for // `hold` frames, which turns an unreadable flicker into a beat. // // The hysteresis runs the other way from the teeth: shutting needs a clear // signal, and once shut the eye is given the benefit of the doubt on reopening, // because the lid landmarks are least reliable mid-blink. export function resolveBlink(open, { cut, dwell, hold }) { const N = open.length; const shut = new Array(N).fill(false); let live = false; // current state let run = 0; // frames the opposing reading has persisted let held = 0; // frames spent in the current state for (let f = 0; f < N; f++) { const reading = live ? open[f] < cut * 1.35 : open[f] < cut; if (reading === live) run = 0; else { run++; // Leaving a blink additionally requires the blink to have been on screen // long enough to be legible; entering one never waits. if (run > dwell && (!live || held >= hold)) { live = reading; held = 0; run = 0; } } held++; shut[f] = live; } return shut; } // Stage 4: fixed-index subsample of a stabilised ring, then map from normalised // face space into character raster space. export function toRasterRing(stabRing, ringTable, n, xform) { return subsampleSlots(ringTable.length, n).map((s) => xform(stabRing[s])); } // Stage 6: key selection. // // Keys go on velocity MINIMA, not on distance thresholds. A threshold fires at // the frame it was crossed - partway through a transition - so every pose lands // mushy and late. A minimum is where the shape is momentarily parked, which is // the pose a viewer actually reads. // // Minima alone are not enough: during a long hold the velocity wobbles near zero // and produces a key per wobble. So a candidate minimum is only accepted if the // shape has actually moved since the last accepted key (distThresh) and the // minimum hold has elapsed (minHold). export function selectKeys(shapes, opts) { const { minHold, distThresh, velSmooth, exposure } = opts; const N = shapes.length; if (N === 0) return { keys: [], velocity: [], candidates: [] }; const vel = new Array(N).fill(0); for (let t = 1; t < N; t++) { let acc = 0; for (let i = 0; i < shapes[t].length; i++) { acc += Math.hypot(shapes[t][i].x - shapes[t - 1][i].x, shapes[t][i].y - shapes[t - 1][i].y); } vel[t] = acc / shapes[t].length; } const sv = movingAverage(vel, velSmooth); const candidates = []; for (let t = 1; t < N - 1; t++) { if (sv[t] <= sv[t - 1] && sv[t] <= sv[t + 1]) candidates.push(t); } const shapeDist = (a, b) => { let acc = 0; for (let i = 0; i < a.length; i++) acc += Math.hypot(a[i].x - b[i].x, a[i].y - b[i].y); return acc / a.length; }; const accepted = [0]; for (const t of candidates) { const last = accepted[accepted.length - 1]; if (t - last < minHold) continue; if (shapeDist(shapes[t], shapes[last]) < distThresh) continue; accepted.push(t); } // Snap onto the exposure grid. f is what renders; src is provenance. const keys = []; for (const src of accepted) { const f = Math.round(src / exposure) * exposure; const prev = keys[keys.length - 1]; if (prev && prev.f === f) { // Two extremes collapsed onto one grid slot: keep the stronger one. if (sv[src] < sv[prev.src]) { prev.src = src; prev.frame = src; } continue; } keys.push({ f, src, frame: src }); } return { keys, velocity: sv, candidates }; } // Resolve which key is live on a given output frame under interp=hold. // "Most recent key at or before f" - lookup, not policy. export function activeKey(keys, f) { let hit = keys[0]; for (const k of keys) { if (k.f <= f) hit = k; else break; } return hit; } // Temporal smoothing of a contour, per vertex, across time. // // docs/design.md says to smooth the transform and never the contour. That // was correct while keys were sparse: sampling at velocity minima rejected // per-frame detector noise for free. With a key on every frame the noise is // visible as a shimmer along the lip edge, so a bounded exception applies - // the window must stay SHORTER than the shortest articulation worth keeping. // At 12fps, mouth movement spans 3-6 frames and detector noise is per-frame, so // a radius of 1 separates them and a radius of 3 would start eating speech. // // `radius` in frames either side: 0 off, 1 = 3-frame average, 2 = 5-frame. export function smoothContours(rings, radius) { if (radius <= 0) return rings; const half = Math.floor(radius), N = rings.length, V = rings[0].length; const out = []; for (let t = 0; t < N; t++) { const frame = []; for (let v = 0; v < V; v++) { let sx = 0, sy = 0, c = 0; for (let j = t - half; j <= t + half; j++) { const k = Math.min(N - 1, Math.max(0, j)); sx += rings[k][v].x; sy += rings[k][v].y; c++; } frame.push({ x: sx / c, y: sy / c }); } out.push(frame); } return out; } // Which frames need their own PLATE drawing. // // This is frame removal, not keyframe extraction: every frame is a candidate and // the question is which can be dropped. Walk forward holding the current drawing // until the head has moved further than `tol` from it, then a new drawing is // required. The cost being managed is an artist drawing a head, which is why the // signal is head pose and not the mouth - the mouth is traced and free. export function suggestPlateFrames(rigid, tol) { const dist = (a, b) => { let m = 0; for (let i = 0; i < a.length; i++) m = Math.max(m, Math.hypot(a[i].x - b[i].x, a[i].y - b[i].y)); return m; }; const keep = [0]; let anchor = 0; for (let f = 1; f < rigid.length; f++) { if (dist(rigid[f], rigid[anchor]) > tol) { keep.push(f); anchor = f; } } return keep; } // Nearest kept frame at or before f - the plate that is on screen. export function heldFrame(kept, f) { let hit = kept[0]; for (const k of kept) { if (k <= f) hit = k; else break; } return hit; } // Hold every output frame back onto an exposure grid: 1 = on 1s, 2 = on 2s, and // so on. Frame 5 at exposure 2 reads the pose from frame 4. // // This is where "aesthetic sparseness" belongs. docs/design.md used to put it at // the extraction rate - pick 12fps and the timing is already chosen - but that // makes the timing a property of a directory of PNGs, so auditioning 12 against // 24 means re-ripping the clip and re-running detection over all of it. Rip // dense once and quantise here instead: the dense track stays at the camera's // rate, the decision stays reversible, and the audio clock is untouched, so // sync cannot drift while you try timings. // // Floor, never round. Rounding would let an output frame read a pose from the // FUTURE, which is a lead - a separate control, applied after this one, for a // separate reason. export function exposeIndex(f, exposure) { return exposure > 1 ? Math.floor(f / exposure) * exposure : f; } // Shift a performance track against the clock, clamped at the ends. // // Pure and exported so the shift can actually be asserted: "the slider feels // like it does nothing" is otherwise indistinguishable from "the slider does // nothing", and at 24fps a lead of 1 is 42ms, which is small enough to doubt. export function shiftIndex(f, lead, n) { return Math.min(n - 1, Math.max(0, f + lead)); }