arthur/js/underlay.js
Your Name 29b0bd6690 Fix horizontal stretch from MediaPipe's anisotropic normalised space
MediaPipe normalises x by image WIDTH and y by image HEIGHT, so for a 1080x1920
clip one unit of x is 1080px and one unit of y is 1920px. makeXform applied a
single scale to both, stretching everything horizontally by H/W - 1.78x on this
footage. The photo underlay looked equally squashed because frameAffine divided
x by imgW, matching the equally wrong vector shapes rather than disagreeing
with them.

Fixed at ingest: landmarks convert to an isotropic space whose unit is one image
height (x *= W/H), so equal numbers mean equal pixels everywhere downstream.
Pixel mapping follows - both axes divide by imgH.

This also silently fixes head roll. fitSimilarity was fitting a rotation in a
sheared space, so the "similarity" it recovered was not one, and stabilisation
of rolled heads was subtly wrong.

selftest: a shape circular in pixel space must stay circular in raster space,
checked at 1080x1920, 1920x1080 and 640x640. Fails at ratio 1.78 without the
conversion.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-24 15:03:12 -04:00

65 lines
2.8 KiB
JavaScript

// Registered photo underlay: the source frame mapped into raster space through
// the same transform chain the vector shapes go through, so a drawing made over
// it lands on the shapes.
//
// Without registration an underlay is decorative. With it, the photo is
// stabilised exactly as the contours are - the head sits still - and tracing over
// it produces plate art already aligned to the mouth.
import { applySim } from './mathutil.js';
// Compose pixel-space -> raster-space into one affine.
//
// Landmarks are converted to an isotropic space (unit = one image height) before
// fitting, so pixels map in the same way: BOTH axes divide by imgH, not by their
// own dimension. Dividing x by imgW here instead is what stretched the underlay
// horizontally by H/W and made it disagree with nothing - it matched the equally
// wrong vector shapes.
export function frameAffine(tf, xform, imgW, imgH) {
const map = (px, py) => xform(applySim(tf, { x: px / imgH, y: py / imgH }));
const P0 = map(0, 0), P1 = map(imgW, 0), P2 = map(0, imgH);
return {
a: (P1.x - P0.x) / imgW, b: (P1.y - P0.y) / imgW,
c: (P2.x - P0.x) / imgH, d: (P2.y - P0.y) / imgH,
e: P0.x, f: P0.y,
};
}
// Draw the registered source frame into a raster-sized 2D context.
export function drawRegistered(ctx, img, tf, xform, zoom, alpha = 1) {
const m = frameAffine(tf, xform, img.naturalWidth, img.naturalHeight);
ctx.save();
ctx.globalAlpha = alpha;
ctx.setTransform(m.a * zoom, m.b * zoom, m.c * zoom, m.d * zoom, m.e * zoom, m.f * zoom);
ctx.imageSmoothingEnabled = true;
ctx.drawImage(img, 0, 0);
ctx.restore();
ctx.setTransform(1, 0, 0, 1, 0, 0);
}
// Quantise a registered frame straight into palette indices.
//
// Doubles as a look test: it shows what the footage becomes in the chosen ramp,
// with no dithering and no antialiasing, which is the question "will these tones
// read" asked directly of the source rather than of a drawing.
export function posterizeInto(raster, img, tf, xform, paletteHex) {
const tmp = document.createElement('canvas');
tmp.width = raster.w; tmp.height = raster.h;
const g = tmp.getContext('2d', { willReadFrequently: true });
g.fillStyle = '#000'; g.fillRect(0, 0, raster.w, raster.h);
drawRegistered(g, img, tf, xform, 1, 1);
const px = g.getImageData(0, 0, raster.w, raster.h).data;
const pal = paletteHex.map((h) => {
const s = h.replace('#', '');
return [parseInt(s.slice(0, 2), 16), parseInt(s.slice(2, 4), 16), parseInt(s.slice(4, 6), 16)];
});
for (let i = 0, n = raster.w * raster.h; i < n; i++) {
const r = px[i * 4], g2 = px[i * 4 + 1], b = px[i * 4 + 2];
let best = 0, bd = Infinity;
for (let k = 0; k < pal.length; k++) {
const d = (pal[k][0] - r) ** 2 + (pal[k][1] - g2) ** 2 + (pal[k][2] - b) ** 2;
if (d < bd) { bd = d; best = k; }
}
raster.buf[i] = best;
}
}