arthur/js/mathutil.js

135 lines
5 KiB
JavaScript
Raw Permalink Normal View History

// 2D similarity transforms and temporal smoothing.
// Least-squares similarity (translation + rotation + uniform scale, 4 DOF)
// mapping P onto Q. Closed form; no iteration.
//
// Deliberately NOT affine or homography: the extra degrees of freedom absorb
// out-of-plane head rotation as shear/perspective and smear it into the mouth.
// Four DOF removes exactly translation, roll and depth-scale, and leaves yaw and
// pitch as a measurable residual.
export function fitSimilarity(P, Q) {
const n = P.length;
let pcx = 0, pcy = 0, qcx = 0, qcy = 0;
for (let i = 0; i < n; i++) {
pcx += P[i].x; pcy += P[i].y;
qcx += Q[i].x; qcy += Q[i].y;
}
pcx /= n; pcy /= n; qcx /= n; qcy /= n;
let a = 0, b = 0, norm = 0;
for (let i = 0; i < n; i++) {
const px = P[i].x - pcx, py = P[i].y - pcy;
const qx = Q[i].x - qcx, qy = Q[i].y - qcy;
a += px * qx + py * qy; // dot
b += px * qy - py * qx; // cross
norm += px * px + py * py;
}
const theta = Math.atan2(b, a);
const s = norm > 1e-12 ? Math.hypot(a, b) / norm : 1;
const c = Math.cos(theta), sn = Math.sin(theta);
return {
s, theta,
tx: qcx - s * (c * pcx - sn * pcy),
ty: qcy - s * (sn * pcx + c * pcy),
};
}
export function applySim(tf, p) {
const c = Math.cos(tf.theta), sn = Math.sin(tf.theta);
return {
x: tf.s * (c * p.x - sn * p.y) + tf.tx,
y: tf.s * (sn * p.x + c * p.y) + tf.ty,
};
}
export function applySimAll(tf, pts) {
return pts.map((p) => applySim(tf, p));
}
// Residual RMS after the fit, in the units of Q. Rises with out-of-plane
// rotation, so it is the signal for "this section is not stabilisable".
export function fitResidual(tf, P, Q) {
let acc = 0;
for (let i = 0; i < P.length; i++) {
const m = applySim(tf, P[i]);
acc += (m.x - Q[i].x) ** 2 + (m.y - Q[i].y) ** 2;
}
return Math.sqrt(acc / P.length);
}
// Generalised Procrustes: the reference is the MEAN rigid configuration over the
// shot, not frame zero, so no single frame's idiosyncrasies get baked into every
// other frame. Three passes is plenty.
export function procrustesMean(framesRigid, iters = 3) {
let ref = framesRigid[0].map((p) => ({ x: p.x, y: p.y }));
for (let it = 0; it < iters; it++) {
const acc = ref.map(() => ({ x: 0, y: 0 }));
for (const rig of framesRigid) {
const tf = fitSimilarity(rig, ref);
const moved = applySimAll(tf, rig);
for (let i = 0; i < acc.length; i++) { acc[i].x += moved[i].x; acc[i].y += moved[i].y; }
}
ref = acc.map((p) => ({ x: p.x / framesRigid.length, y: p.y / framesRigid.length }));
}
return ref;
}
// `radius` is in frames either side: 0 is off, 1 averages over 3 frames, 2 over
// 5. Expressed as a radius rather than a window so that "off" is 0 and every
// value is symmetric - an even window would be lopsided in time.
function movingAverage(vals, radius) {
if (radius <= 0) return vals.slice();
const half = Math.floor(radius), out = new Array(vals.length);
for (let i = 0; i < vals.length; i++) {
let acc = 0, cnt = 0;
for (let j = i - half; j <= i + half; j++) {
const k = Math.min(vals.length - 1, Math.max(0, j));
acc += vals[k]; cnt++;
}
out[i] = acc / cnt;
}
return out;
}
export { movingAverage };
// Smooth the four transform parameters, NEVER the contour. Landmark jitter of a
// pixel is smeared into the mouth by the inverse transform, so the transform is
// where the low-pass belongs; smoothing the contour would destroy the
// performance, which is the entire asset.
// Angles are smoothed as (cos, sin) so wrapping cannot produce a spike.
export function smoothTransforms(tfs, radius) {
const c = movingAverage(tfs.map((t) => Math.cos(t.theta)), radius);
const sn = movingAverage(tfs.map((t) => Math.sin(t.theta)), radius);
const s = movingAverage(tfs.map((t) => t.s), radius);
const tx = movingAverage(tfs.map((t) => t.tx), radius);
const ty = movingAverage(tfs.map((t) => t.ty), radius);
return tfs.map((_, i) => ({
theta: Math.atan2(sn[i], c[i]), s: s[i], tx: tx[i], ty: ty[i],
}));
}
Eyes: lids, blinking, line of sight Three parts per eye, stacked the way the mouth is - dark lash ring, sclera inside it, iris inside that, square pupil in the iris. A blink then costs nothing: when the lid shuts the traced ring goes flat and the lash line collapses to a lens, which is a closed eye, drawn correctly, for free. Lids are a FEATURE, rotoscoped like the mouth: head-local, a key on every frame, the same contour avg knob. The iris is a PRIMITIVE - a disc at a quantised position - and that is where the stylisation lives. Line of sight. Gaze is the iris centre relative to the midpoint of the eye's two corners, in units of corner distance. Both corners are in RIGID, so the origin and the scale are immune to the performance being measured; against the lid ring's centroid instead, every blink would drag the origin down and fake a glance at the floor on exactly the frames where the eye is most visible. Both eyes share one gaze - at this size the difference between the two measurements is noise, not vergence, and independent per-eye noise reads as wall-eyed immediately. Openness stays per-eye so a wink survives. Gaze is then quantised to a pixel grid with a dwell, which is not a stylisation imposed on the truth: real eyes move in saccades, and the smooth drift left in the measurement is tracker noise plus head-compensation error. Snapping to a grid removes the noise and recovers the saccade in one operation. The iris is placed in the frame of the already-smoothed, already-subsampled lid ring - slots 0 and 8 of a 16-slot ring are the corners, and subsampling to any even budget keeps them at 0 and n/2 - so it cannot drift relative to its own eye. Size is authored from the take mean, never remeasured per frame: a radius that breathes by a fraction of a pixel flickers a pixel on and off around the whole silhouette. iris anchor toggles steady/free/locked, because how much the eye wanders turns out to be an aesthetic choice and not only a correctness one. Blinking gets hysteresis and a dwell like the teeth, plus one knob they do not have: blink hold. A blink is one frame at 12fps and a single frame of closed eye reads as a dropped frame, so once the eye shuts it stays shut long enough to be legible. Detection accuracy is not the problem; legibility is. The pupil is a square because at three pixels a circle is a plus sign with the corners gnawed off, and it changes shape as it moves. Drawn from a rounded centre shared with the iris so it is exactly its nominal size on every frame. Iris/pupil clip by colour key against the indexed buffer, the way Animator Pro would: the lid crops the iris at extreme gaze for free, so nothing has to clamp the gaze, which would flatten the performance at the extremes that carry it. Which iris block belongs to which eye is RESOLVED from geometry, not declared. A swap looks almost right - each eye still has a disc roughly where it belongs - so it survives an eyeball and then reads as a subtly wall-eyed character forever. Voted across every frame; the test feeds a deliberately swapped track. Also: exposure. Aesthetic sparseness was set by the extraction rate, which made the timing a property of a directory of PNGs - auditioning 12 against 24 meant re-ripping and re-detecting the whole clip. It is now a render-time grid, on 1s/2s/3s/4s, so the dense track keeps everything and the audio clock is untouched. The take format already carried an exposure field; it was never driven. Everything rides the same grid, because a head cutting on the odd frames while the mouth cuts on the even ones reads as two performances laid over each other. 41 -> 91 assertions. The load-bearing new ones: the iris pairing follows a swapped track, a blink does not fake a change of gaze, a stencilled disc cannot spill past its clip, a 3px pupil is 3x3 at every sub-pixel centre, and exposure never reads a pose from the future. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-24 18:06:04 -04:00
// Push a ring outward from its centroid by a FIXED distance, not by a scale
// factor.
//
// Scaling collapses with the shape: a shut eyelid scaled by 1.1 is still a shut
// eyelid, so the lash line - the only thing left to draw when the eye is closed
// - would vanish exactly on the frames where it is the whole drawing. A fixed
// radial offset gives a band of roughly constant thickness that survives the
// ring going degenerate, and it keeps a star-shaped ring simple, which
// docs/design.md requires of every cut part.
export function offsetRing(pts, d) {
if (!d) return pts;
let cx = 0, cy = 0;
for (const p of pts) { cx += p.x; cy += p.y; }
cx /= pts.length; cy /= pts.length;
return pts.map((p) => {
const dx = p.x - cx, dy = p.y - cy;
const m = Math.hypot(dx, dy);
// A vertex sitting exactly on the centroid has no outward direction. Leave
// it where it is rather than emitting NaN and poisoning the whole ring.
return m < 1e-9 ? { x: p.x, y: p.y }
: { x: p.x + (dx / m) * d, y: p.y + (dy / m) * d };
});
}