Fix horizontal stretch from MediaPipe's anisotropic normalised space

MediaPipe normalises x by image WIDTH and y by image HEIGHT, so for a 1080x1920
clip one unit of x is 1080px and one unit of y is 1920px. makeXform applied a
single scale to both, stretching everything horizontally by H/W - 1.78x on this
footage. The photo underlay looked equally squashed because frameAffine divided
x by imgW, matching the equally wrong vector shapes rather than disagreeing
with them.

Fixed at ingest: landmarks convert to an isotropic space whose unit is one image
height (x *= W/H), so equal numbers mean equal pixels everywhere downstream.
Pixel mapping follows - both axes divide by imgH.

This also silently fixes head roll. fitSimilarity was fitting a rotation in a
sheared space, so the "similarity" it recovered was not one, and stabilisation
of rolled heads was subtly wrong.

selftest: a shape circular in pixel space must stay circular in raster space,
checked at 1080x1920, 1920x1080 and 640x640. Fails at ratio 1.78 without the
conversion.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Your Name 2026-09-24 15:03:12 -04:00
parent de90bfd907
commit 29b0bd6690
4 changed files with 64 additions and 14 deletions

View file

@ -5,13 +5,22 @@
import { RIGID, LIPS_OUTER, LIPS_INNER, APERTURE, FACE_OVAL, EYE_INNER, subsampleSlots } from './landmarks.js';
import { fitSimilarity, applySimAll, applySim, fitResidual, procrustesMean, smoothTransforms, movingAverage } from './mathutil.js';
const pick = (lm, idx) => idx.map((i) => ({ x: lm[i].x, y: lm[i].y }));
// MediaPipe normalises x by image WIDTH and y by image HEIGHT, so its normalised
// space is anisotropic: for a 1080x1920 frame, one unit of x is 1080px and one
// unit of y is 1920px. Treating those as comparable stretches everything
// horizontally by H/W, and worse, makes fitSimilarity fit a "rotation" in a
// sheared space, so head roll comes out subtly wrong as well.
//
// Multiplying x by aspect = W/H converts to an ISOTROPIC space whose unit is one
// image height, so equal numbers mean equal pixels. Everything downstream -
// Procrustes, the similarity fit, the raster transform - depends on that.
const pick = (lm, idx, aspect) => idx.map((i) => ({ x: lm[i].x * aspect, y: lm[i].y }));
// Stage 1-3: fit the rigid transform per frame, smooth its parameters, then map
// every contour through it into the reference frame. The result is head-local:
// translation, roll and depth-scale of the head are gone.
export function stabilize(dense, smoothRadius) {
const rigid = dense.map((f) => pick(f, RIGID));
export function stabilize(dense, smoothRadius, aspect = 1) {
const rigid = dense.map((f) => pick(f, RIGID, aspect));
const ref = procrustesMean(rigid);
const raw = rigid.map((r) => fitSimilarity(r, ref));
const tfs = smoothTransforms(raw, smoothRadius);
@ -26,12 +35,12 @@ export function stabilize(dense, smoothRadius) {
// Residual rises with out-of-plane rotation, which no 2D similarity can
// remove. High values mean this section wants a different head plate.
residual: tfs.map((tf, i) => fitResidual(tf, rigid[i], ref)),
outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER))),
inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER))),
oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL))),
eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER))),
outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER, aspect))),
inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER, aspect))),
oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL, aspect))),
eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER, aspect))),
aperture: dense.map((f, i) => {
const a = applySimAll(tfs[i], pick(f, APERTURE));
const a = applySimAll(tfs[i], pick(f, APERTURE, aspect));
return Math.hypot(a[0].x - a[1].x, a[0].y - a[1].y);
}),
};