Fix horizontal stretch from MediaPipe's anisotropic normalised space
MediaPipe normalises x by image WIDTH and y by image HEIGHT, so for a 1080x1920 clip one unit of x is 1080px and one unit of y is 1920px. makeXform applied a single scale to both, stretching everything horizontally by H/W - 1.78x on this footage. The photo underlay looked equally squashed because frameAffine divided x by imgW, matching the equally wrong vector shapes rather than disagreeing with them. Fixed at ingest: landmarks convert to an isotropic space whose unit is one image height (x *= W/H), so equal numbers mean equal pixels everywhere downstream. Pixel mapping follows - both axes divide by imgH. This also silently fixes head roll. fitSimilarity was fitting a rotation in a sheared space, so the "similarity" it recovered was not one, and stabilisation of rolled heads was subtly wrong. selftest: a shape circular in pixel space must stay circular in raster space, checked at 1080x1920, 1920x1080 and 640x640. Fails at ratio 1.78 without the conversion. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
de90bfd907
commit
29b0bd6690
4 changed files with 64 additions and 14 deletions
|
|
@ -23,6 +23,7 @@ const state = {
|
||||||
keep: new Set(), // frames that get their own plate drawing
|
keep: new Set(), // frames that get their own plate drawing
|
||||||
frame: 0, playing: false, faceBox: null,
|
frame: 0, playing: false, faceBox: null,
|
||||||
fps: 12, audio: null, // fps comes from manifest.json, never guessed
|
fps: 12, audio: null, // fps comes from manifest.json, never guessed
|
||||||
|
aspect: 1, // imgW/imgH; converts MediaPipe's anisotropic space
|
||||||
};
|
};
|
||||||
|
|
||||||
const el = (id) => document.getElementById(id);
|
const el = (id) => document.getElementById(id);
|
||||||
|
|
@ -134,7 +135,7 @@ function rebuild(resetKeep) {
|
||||||
const o = opts();
|
const o = opts();
|
||||||
const N = state.dense.length;
|
const N = state.dense.length;
|
||||||
|
|
||||||
state.stab = stabilize(state.dense, o.smoothWin);
|
state.stab = stabilize(state.dense, o.smoothWin, state.aspect);
|
||||||
|
|
||||||
const ap = state.stab.aperture;
|
const ap = state.stab.aperture;
|
||||||
const head = Math.max(1, Math.floor(N / 4));
|
const head = Math.max(1, Math.floor(N / 4));
|
||||||
|
|
@ -440,11 +441,13 @@ async function runFrames() {
|
||||||
}
|
}
|
||||||
const { dense, missing } = await detectAll(images);
|
const { dense, missing } = await detectAll(images);
|
||||||
state.images = images; state.dense = dense;
|
state.images = images; state.dense = dense;
|
||||||
|
state.aspect = images[0].naturalWidth / images[0].naturalHeight;
|
||||||
el('scrub').max = dense.length - 1;
|
el('scrub').max = dense.length - 1;
|
||||||
state.frame = 0;
|
state.frame = 0;
|
||||||
rebuild(true);
|
rebuild(true);
|
||||||
const dur = (dense.length / state.fps).toFixed(2);
|
const dur = (dense.length / state.fps).toFixed(2);
|
||||||
status(`${images.length} frames · ${state.fps}fps · ${dur}s` +
|
status(`${images.length} frames · ${images[0].naturalWidth}x${images[0].naturalHeight} · ` +
|
||||||
|
`${state.fps}fps · ${dur}s` +
|
||||||
(state.audio ? ' · audio loaded' : ' · no audio') +
|
(state.audio ? ' · audio loaded' : ' · no audio') +
|
||||||
(missing.length ? ` · no face on ${missing.length} (held previous)` : ''),
|
(missing.length ? ` · no face on ${missing.length} (held previous)` : ''),
|
||||||
missing.length ? 'warn' : 'ok');
|
missing.length ? 'warn' : 'ok');
|
||||||
|
|
@ -455,6 +458,8 @@ function runSynthetic() {
|
||||||
state.images = [];
|
state.images = [];
|
||||||
attachAudio(null);
|
attachAudio(null);
|
||||||
state.fps = 12;
|
state.fps = 12;
|
||||||
|
state.aspect = 1; // synthetic landmarks are generated square
|
||||||
|
|
||||||
state.dense = synthDense(72);
|
state.dense = synthDense(72);
|
||||||
el('scrub').max = 71;
|
el('scrub').max = 71;
|
||||||
state.frame = 0;
|
state.frame = 0;
|
||||||
|
|
|
||||||
|
|
@ -5,13 +5,22 @@
|
||||||
import { RIGID, LIPS_OUTER, LIPS_INNER, APERTURE, FACE_OVAL, EYE_INNER, subsampleSlots } from './landmarks.js';
|
import { RIGID, LIPS_OUTER, LIPS_INNER, APERTURE, FACE_OVAL, EYE_INNER, subsampleSlots } from './landmarks.js';
|
||||||
import { fitSimilarity, applySimAll, applySim, fitResidual, procrustesMean, smoothTransforms, movingAverage } from './mathutil.js';
|
import { fitSimilarity, applySimAll, applySim, fitResidual, procrustesMean, smoothTransforms, movingAverage } from './mathutil.js';
|
||||||
|
|
||||||
const pick = (lm, idx) => idx.map((i) => ({ x: lm[i].x, y: lm[i].y }));
|
// MediaPipe normalises x by image WIDTH and y by image HEIGHT, so its normalised
|
||||||
|
// space is anisotropic: for a 1080x1920 frame, one unit of x is 1080px and one
|
||||||
|
// unit of y is 1920px. Treating those as comparable stretches everything
|
||||||
|
// horizontally by H/W, and worse, makes fitSimilarity fit a "rotation" in a
|
||||||
|
// sheared space, so head roll comes out subtly wrong as well.
|
||||||
|
//
|
||||||
|
// Multiplying x by aspect = W/H converts to an ISOTROPIC space whose unit is one
|
||||||
|
// image height, so equal numbers mean equal pixels. Everything downstream -
|
||||||
|
// Procrustes, the similarity fit, the raster transform - depends on that.
|
||||||
|
const pick = (lm, idx, aspect) => idx.map((i) => ({ x: lm[i].x * aspect, y: lm[i].y }));
|
||||||
|
|
||||||
// Stage 1-3: fit the rigid transform per frame, smooth its parameters, then map
|
// Stage 1-3: fit the rigid transform per frame, smooth its parameters, then map
|
||||||
// every contour through it into the reference frame. The result is head-local:
|
// every contour through it into the reference frame. The result is head-local:
|
||||||
// translation, roll and depth-scale of the head are gone.
|
// translation, roll and depth-scale of the head are gone.
|
||||||
export function stabilize(dense, smoothRadius) {
|
export function stabilize(dense, smoothRadius, aspect = 1) {
|
||||||
const rigid = dense.map((f) => pick(f, RIGID));
|
const rigid = dense.map((f) => pick(f, RIGID, aspect));
|
||||||
const ref = procrustesMean(rigid);
|
const ref = procrustesMean(rigid);
|
||||||
const raw = rigid.map((r) => fitSimilarity(r, ref));
|
const raw = rigid.map((r) => fitSimilarity(r, ref));
|
||||||
const tfs = smoothTransforms(raw, smoothRadius);
|
const tfs = smoothTransforms(raw, smoothRadius);
|
||||||
|
|
@ -26,12 +35,12 @@ export function stabilize(dense, smoothRadius) {
|
||||||
// Residual rises with out-of-plane rotation, which no 2D similarity can
|
// Residual rises with out-of-plane rotation, which no 2D similarity can
|
||||||
// remove. High values mean this section wants a different head plate.
|
// remove. High values mean this section wants a different head plate.
|
||||||
residual: tfs.map((tf, i) => fitResidual(tf, rigid[i], ref)),
|
residual: tfs.map((tf, i) => fitResidual(tf, rigid[i], ref)),
|
||||||
outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER))),
|
outer: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_OUTER, aspect))),
|
||||||
inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER))),
|
inner: dense.map((f, i) => applySimAll(tfs[i], pick(f, LIPS_INNER, aspect))),
|
||||||
oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL))),
|
oval: dense.map((f, i) => applySimAll(tfs[i], pick(f, FACE_OVAL, aspect))),
|
||||||
eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER))),
|
eyes: dense.map((f, i) => applySimAll(tfs[i], pick(f, EYE_INNER, aspect))),
|
||||||
aperture: dense.map((f, i) => {
|
aperture: dense.map((f, i) => {
|
||||||
const a = applySimAll(tfs[i], pick(f, APERTURE));
|
const a = applySimAll(tfs[i], pick(f, APERTURE, aspect));
|
||||||
return Math.hypot(a[0].x - a[1].x, a[0].y - a[1].y);
|
return Math.hypot(a[0].x - a[1].x, a[0].y - a[1].y);
|
||||||
}),
|
}),
|
||||||
};
|
};
|
||||||
|
|
|
||||||
|
|
@ -113,6 +113,40 @@ export function run() {
|
||||||
const apRange = Math.max(...stab.aperture) - Math.min(...stab.aperture);
|
const apRange = Math.max(...stab.aperture) - Math.min(...stab.aperture);
|
||||||
ok('stabilisation preserves mouth motion', apRange > 0.05, `aperture range ${apRange.toFixed(4)}`);
|
ok('stabilisation preserves mouth motion', apRange > 0.05, `aperture range ${apRange.toFixed(4)}`);
|
||||||
|
|
||||||
|
// ASPECT: a shape that is circular in PIXEL space must stay circular in raster
|
||||||
|
// space. MediaPipe normalises x by width and y by height, so for a portrait
|
||||||
|
// frame equal normalised numbers are unequal pixel distances; feeding those
|
||||||
|
// straight through stretches everything horizontally by H/W. This asserts the
|
||||||
|
// isotropic conversion, and fails at ~1.78 for a 1080x1920 clip without it.
|
||||||
|
for (const [W, H] of [[1080, 1920], [1920, 1080], [640, 640]]) {
|
||||||
|
const aspect = W / H;
|
||||||
|
const N = 24, cx = 0.5, cy = 0.5, rPx = 200;
|
||||||
|
// a true circle of radius rPx, expressed in MediaPipe normalised coords
|
||||||
|
const circleFrames = [];
|
||||||
|
for (let t = 0; t < 4; t++) {
|
||||||
|
const pts = new Array(478).fill(null).map(() => ({ x: 0.5, y: 0.5, z: 0 }));
|
||||||
|
RIGID.forEach((id, k) => {
|
||||||
|
const a = (k / RIGID.length) * Math.PI * 2;
|
||||||
|
pts[id] = { x: cx + (120 * Math.cos(a)) / W, y: cy + (120 * Math.sin(a)) / H, z: 0 };
|
||||||
|
});
|
||||||
|
LIPS_OUTER.forEach((id, k) => {
|
||||||
|
const a = -(k / LIPS_OUTER.length) * Math.PI * 2;
|
||||||
|
pts[id] = { x: cx + (rPx * Math.cos(a)) / W, y: cy + (rPx * Math.sin(a)) / H, z: 0 };
|
||||||
|
});
|
||||||
|
FACE_OVAL.forEach((id, k) => {
|
||||||
|
const a = -(k / FACE_OVAL.length) * Math.PI * 2;
|
||||||
|
pts[id] = { x: cx + (420 * Math.cos(a)) / W, y: cy + (420 * Math.sin(a)) / H, z: 0 };
|
||||||
|
});
|
||||||
|
circleFrames.push(pts);
|
||||||
|
}
|
||||||
|
const st2 = stabilize(circleFrames, 0, aspect);
|
||||||
|
const ring = toRasterRing(st2.outer[0], LIPS_OUTER, 16, (p) => p);
|
||||||
|
const xs = ring.map((p) => p.x), ys = ring.map((p) => p.y);
|
||||||
|
const ratio = (Math.max(...xs) - Math.min(...xs)) / (Math.max(...ys) - Math.min(...ys));
|
||||||
|
ok(`circle stays circular at ${W}x${H}`, Math.abs(ratio - 1) < 0.02,
|
||||||
|
`w/h ratio ${ratio.toFixed(4)}`);
|
||||||
|
}
|
||||||
|
|
||||||
// key selection
|
// key selection
|
||||||
const xf = (p) => ({ x: p.x * 320, y: p.y * 200 });
|
const xf = (p) => ({ x: p.x * 320, y: p.y * 200 });
|
||||||
const shapes = stab.outer.map((r) => toRasterRing(r, LIPS_OUTER, 8, xf));
|
const shapes = stab.outer.map((r) => toRasterRing(r, LIPS_OUTER, 8, xf));
|
||||||
|
|
|
||||||
|
|
@ -10,11 +10,13 @@ import { applySim } from './mathutil.js';
|
||||||
|
|
||||||
// Compose pixel-space -> raster-space into one affine.
|
// Compose pixel-space -> raster-space into one affine.
|
||||||
//
|
//
|
||||||
// MediaPipe normalises x by width and y by height, so normalised space is a
|
// Landmarks are converted to an isotropic space (unit = one image height) before
|
||||||
// stretched pixel space and the composition is a general affine rather than a
|
// fitting, so pixels map in the same way: BOTH axes divide by imgH, not by their
|
||||||
// similarity. Three mapped points determine it exactly.
|
// own dimension. Dividing x by imgW here instead is what stretched the underlay
|
||||||
|
// horizontally by H/W and made it disagree with nothing - it matched the equally
|
||||||
|
// wrong vector shapes.
|
||||||
export function frameAffine(tf, xform, imgW, imgH) {
|
export function frameAffine(tf, xform, imgW, imgH) {
|
||||||
const map = (px, py) => xform(applySim(tf, { x: px / imgW, y: py / imgH }));
|
const map = (px, py) => xform(applySim(tf, { x: px / imgH, y: py / imgH }));
|
||||||
const P0 = map(0, 0), P1 = map(imgW, 0), P2 = map(0, imgH);
|
const P0 = map(0, 0), P1 = map(imgW, 0), P2 = map(0, imgH);
|
||||||
return {
|
return {
|
||||||
a: (P1.x - P0.x) / imgW, b: (P1.y - P0.y) / imgW,
|
a: (P1.x - P0.x) / imgW, b: (P1.y - P0.y) / imgW,
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue