diff --git a/js/app.js b/js/app.js index d4f2528..6e6fc61 100644 --- a/js/app.js +++ b/js/app.js @@ -23,6 +23,7 @@ const state = { keep: new Set(), // frames that get their own plate drawing frame: 0, playing: false, faceBox: null, fps: 12, audio: null, // fps comes from manifest.json, never guessed + aspect: 1, // imgW/imgH; converts MediaPipe's anisotropic space }; const el = (id) => document.getElementById(id); @@ -134,7 +135,7 @@ function rebuild(resetKeep) { const o = opts(); const N = state.dense.length; - state.stab = stabilize(state.dense, o.smoothWin); + state.stab = stabilize(state.dense, o.smoothWin, state.aspect); const ap = state.stab.aperture; const head = Math.max(1, Math.floor(N / 4)); @@ -440,11 +441,13 @@ async function runFrames() { } const { dense, missing } = await detectAll(images); state.images = images; state.dense = dense; + state.aspect = images[0].naturalWidth / images[0].naturalHeight; el('scrub').max = dense.length - 1; state.frame = 0; rebuild(true); const dur = (dense.length / state.fps).toFixed(2); - status(`${images.length} frames · ${state.fps}fps · ${dur}s` + + status(`${images.length} frames · ${images[0].naturalWidth}x${images[0].naturalHeight} · ` + + `${state.fps}fps · ${dur}s` + (state.audio ? ' · audio loaded' : ' · no audio') + (missing.length ? ` · no face on ${missing.length} (held previous)` : ''), missing.length ? 'warn' : 'ok'); @@ -455,6 +458,8 @@ function runSynthetic() { state.images = []; attachAudio(null); state.fps = 12; + state.aspect = 1; // synthetic landmarks are generated square + state.dense = synthDense(72); el('scrub').max = 71; state.frame = 0; diff --git a/js/pipeline.js b/js/pipeline.js index 90ae12f..d75ea99 100644 --- a/js/pipeline.js +++ b/js/pipeline.js @@ -5,13 +5,22 @@ import { RIGID, LIPS_OUTER, LIPS_INNER, APERTURE, FACE_OVAL, EYE_INNER, subsampleSlots } from './landmarks.js'; import { fitSimilarity, applySimAll, applySim, fitResidual, procrustesMean, smoothTransforms, movingAverage } from './mathutil.js'; -const pick = (lm, idx) => idx.map((i) => ({ x: lm[i].x, y: lm[i].y })); +// MediaPipe normalises x by image WIDTH and y by image HEIGHT, so its normalised +// space is anisotropic: for a 1080x1920 frame, one unit of x is 1080px and one +// unit of y is 1920px. Treating those as comparable stretches everything +// horizontally by H/W, and worse, makes fitSimilarity fit a "rotation" in a +// sheared space, so head roll comes out subtly wrong as well. +// +// Multiplying x by aspect = W/H converts to an ISOTROPIC space whose unit is one +// image height, so equal numbers mean equal pixels. Everything downstream - +// Procrustes, the similarity fit, the raster transform - depends on that. +const pick = (lm, idx, aspect) => idx.map((i) => ({ x: lm[i].x * aspect, y: lm[i].y })); // Stage 1-3: fit the rigid transform per frame, smooth its parameters, then map // every contour through it into the reference frame. The result is head-local: // translation, roll and depth-scale of the head are gone. -export function stabilize(dense, smoothRadius) { - const rigid = dense.map((f) => pick(f, RIGID)); +export function stabilize(dense, smoothRadius, aspect = 1) { + const rigid = dense.map((f) => pick(f, RIGID, aspect)); const ref = procrustesMean(rigid); const raw = rigid.map((r) => fitSimilarity(r, ref)); const tfs = smoothTransforms(raw, smoothRadius); @@ -26,12 +35,12 @@ export function stabilize(dense, smoothRadius) { // Residual rises with out-of-plane rotation, which no 2D similarity can // remove. High values mean this section wants a different head plate. residual: tfs.map((tf, i) => fitResidual(tf, rigid[i], ref)), - outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER))), - inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER))), - oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL))), - eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER))), + outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER, aspect))), + inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER, aspect))), + oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL, aspect))), + eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER, aspect))), aperture: dense.map((f, i) => { - const a = applySimAll(tfs[i], pick(f, APERTURE)); + const a = applySimAll(tfs[i], pick(f, APERTURE, aspect)); return Math.hypot(a[0].x - a[1].x, a[0].y - a[1].y); }), }; diff --git a/js/selftest.js b/js/selftest.js index cbf616e..0e41def 100644 --- a/js/selftest.js +++ b/js/selftest.js @@ -113,6 +113,40 @@ export function run() { const apRange = Math.max(...stab.aperture) - Math.min(...stab.aperture); ok('stabilisation preserves mouth motion', apRange > 0.05, `aperture range ${apRange.toFixed(4)}`); + // ASPECT: a shape that is circular in PIXEL space must stay circular in raster + // space. MediaPipe normalises x by width and y by height, so for a portrait + // frame equal normalised numbers are unequal pixel distances; feeding those + // straight through stretches everything horizontally by H/W. This asserts the + // isotropic conversion, and fails at ~1.78 for a 1080x1920 clip without it. + for (const [W, H] of [[1080, 1920], [1920, 1080], [640, 640]]) { + const aspect = W / H; + const N = 24, cx = 0.5, cy = 0.5, rPx = 200; + // a true circle of radius rPx, expressed in MediaPipe normalised coords + const circleFrames = []; + for (let t = 0; t < 4; t++) { + const pts = new Array(478).fill(null).map(() => ({ x: 0.5, y: 0.5, z: 0 })); + RIGID.forEach((id, k) => { + const a = (k / RIGID.length) * Math.PI * 2; + pts[id] = { x: cx + (120 * Math.cos(a)) / W, y: cy + (120 * Math.sin(a)) / H, z: 0 }; + }); + LIPS_OUTER.forEach((id, k) => { + const a = -(k / LIPS_OUTER.length) * Math.PI * 2; + pts[id] = { x: cx + (rPx * Math.cos(a)) / W, y: cy + (rPx * Math.sin(a)) / H, z: 0 }; + }); + FACE_OVAL.forEach((id, k) => { + const a = -(k / FACE_OVAL.length) * Math.PI * 2; + pts[id] = { x: cx + (420 * Math.cos(a)) / W, y: cy + (420 * Math.sin(a)) / H, z: 0 }; + }); + circleFrames.push(pts); + } + const st2 = stabilize(circleFrames, 0, aspect); + const ring = toRasterRing(st2.outer[0], LIPS_OUTER, 16, (p) => p); + const xs = ring.map((p) => p.x), ys = ring.map((p) => p.y); + const ratio = (Math.max(...xs) - Math.min(...xs)) / (Math.max(...ys) - Math.min(...ys)); + ok(`circle stays circular at ${W}x${H}`, Math.abs(ratio - 1) < 0.02, + `w/h ratio ${ratio.toFixed(4)}`); + } + // key selection const xf = (p) => ({ x: p.x * 320, y: p.y * 200 }); const shapes = stab.outer.map((r) => toRasterRing(r, LIPS_OUTER, 8, xf)); diff --git a/js/underlay.js b/js/underlay.js index 81ba004..14f8478 100644 --- a/js/underlay.js +++ b/js/underlay.js @@ -10,11 +10,13 @@ import { applySim } from './mathutil.js'; // Compose pixel-space -> raster-space into one affine. // -// MediaPipe normalises x by width and y by height, so normalised space is a -// stretched pixel space and the composition is a general affine rather than a -// similarity. Three mapped points determine it exactly. +// Landmarks are converted to an isotropic space (unit = one image height) before +// fitting, so pixels map in the same way: BOTH axes divide by imgH, not by their +// own dimension. Dividing x by imgW here instead is what stretched the underlay +// horizontally by H/W and made it disagree with nothing - it matched the equally +// wrong vector shapes. export function frameAffine(tf, xform, imgW, imgH) { - const map = (px, py) => xform(applySim(tf, { x: px / imgW, y: py / imgH })); + const map = (px, py) => xform(applySim(tf, { x: px / imgH, y: py / imgH })); const P0 = map(0, 0), P1 = map(imgW, 0), P2 = map(0, imgH); return { a: (P1.x - P0.x) / imgW, b: (P1.y - P0.y) / imgW,