Three parts per eye, stacked the way the mouth is - dark lash ring, sclera inside it, iris inside that, square pupil in the iris. A blink then costs nothing: when the lid shuts the traced ring goes flat and the lash line collapses to a lens, which is a closed eye, drawn correctly, for free. Lids are a FEATURE, rotoscoped like the mouth: head-local, a key on every frame, the same contour avg knob. The iris is a PRIMITIVE - a disc at a quantised position - and that is where the stylisation lives. Line of sight. Gaze is the iris centre relative to the midpoint of the eye's two corners, in units of corner distance. Both corners are in RIGID, so the origin and the scale are immune to the performance being measured; against the lid ring's centroid instead, every blink would drag the origin down and fake a glance at the floor on exactly the frames where the eye is most visible. Both eyes share one gaze - at this size the difference between the two measurements is noise, not vergence, and independent per-eye noise reads as wall-eyed immediately. Openness stays per-eye so a wink survives. Gaze is then quantised to a pixel grid with a dwell, which is not a stylisation imposed on the truth: real eyes move in saccades, and the smooth drift left in the measurement is tracker noise plus head-compensation error. Snapping to a grid removes the noise and recovers the saccade in one operation. The iris is placed in the frame of the already-smoothed, already-subsampled lid ring - slots 0 and 8 of a 16-slot ring are the corners, and subsampling to any even budget keeps them at 0 and n/2 - so it cannot drift relative to its own eye. Size is authored from the take mean, never remeasured per frame: a radius that breathes by a fraction of a pixel flickers a pixel on and off around the whole silhouette. iris anchor toggles steady/free/locked, because how much the eye wanders turns out to be an aesthetic choice and not only a correctness one. Blinking gets hysteresis and a dwell like the teeth, plus one knob they do not have: blink hold. A blink is one frame at 12fps and a single frame of closed eye reads as a dropped frame, so once the eye shuts it stays shut long enough to be legible. Detection accuracy is not the problem; legibility is. The pupil is a square because at three pixels a circle is a plus sign with the corners gnawed off, and it changes shape as it moves. Drawn from a rounded centre shared with the iris so it is exactly its nominal size on every frame. Iris/pupil clip by colour key against the indexed buffer, the way Animator Pro would: the lid crops the iris at extreme gaze for free, so nothing has to clamp the gaze, which would flatten the performance at the extremes that carry it. Which iris block belongs to which eye is RESOLVED from geometry, not declared. A swap looks almost right - each eye still has a disc roughly where it belongs - so it survives an eyeball and then reads as a subtly wall-eyed character forever. Voted across every frame; the test feeds a deliberately swapped track. Also: exposure. Aesthetic sparseness was set by the extraction rate, which made the timing a property of a directory of PNGs - auditioning 12 against 24 meant re-ripping and re-detecting the whole clip. It is now a render-time grid, on 1s/2s/3s/4s, so the dense track keeps everything and the audio clock is untouched. The take format already carried an exposure field; it was never driven. Everything rides the same grid, because a head cutting on the odd frames while the mouth cuts on the even ones reads as two performances laid over each other. 41 -> 91 assertions. The load-bearing new ones: the iris pairing follows a swapped track, a blink does not fake a change of gaze, a stencilled disc cannot spill past its clip, a 3px pupil is 3x3 at every sub-pixel centre, and exposure never reads a pose from the future. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
112 lines
5.4 KiB
JavaScript
112 lines
5.4 KiB
JavaScript
// Synthetic landmark frames, shaped exactly like FaceLandmarker output.
|
|
//
|
|
// Exists so the whole chain downstream of detection - Procrustes, smoothing,
|
|
// stabilisation, key selection, rasterising, take writing - can be exercised and
|
|
// verified without a video file. A synthetic face is also the only way to test
|
|
// stabilisation against a KNOWN head motion, since real footage gives no ground
|
|
// truth to compare against.
|
|
import { LIPS_OUTER, LIPS_INNER, FACE_OVAL, RIGID,
|
|
EYE_R_RING, EYE_L_RING, IRIS_A, IRIS_B } from './landmarks.js';
|
|
|
|
const NUM = 478;
|
|
|
|
// `swapIris` places the two iris blocks on the opposite eyes. It exists so the
|
|
// pairing resolver can be tested against a track it actually disagrees with:
|
|
// a resolver checked only against the convention it was written for is checking
|
|
// nothing at all.
|
|
export function synthDense(nFrames = 72, { swapIris = false } = {}) {
|
|
const frames = [];
|
|
for (let t = 0; t < nFrames; t++) {
|
|
const pts = new Array(NUM);
|
|
for (let i = 0; i < NUM; i++) pts[i] = { x: 0.5, y: 0.5, z: 0 };
|
|
|
|
// Known head motion: drift, sway, roll and a slow scale change, plus a
|
|
// little per-frame jitter so transform smoothing has something to remove.
|
|
const ph = t / nFrames;
|
|
const hx = 0.5 + 0.045 * Math.sin(ph * Math.PI * 2) + (Math.random() - 0.5) * 0.002;
|
|
const hy = 0.5 + 0.02 * Math.cos(ph * Math.PI * 3) + (Math.random() - 0.5) * 0.002;
|
|
const roll = 0.18 * Math.sin(ph * Math.PI * 2.5);
|
|
const scale = 1 + 0.06 * Math.sin(ph * Math.PI * 1.5);
|
|
const cr = Math.cos(roll), sr = Math.sin(roll);
|
|
const place = (i, lx, ly) => {
|
|
const sx = lx * scale, sy = ly * scale;
|
|
pts[i] = { x: hx + cr * sx - sr * sy, y: hy + sr * sx + cr * sy, z: 0 };
|
|
};
|
|
|
|
// Mouth opens in four sustained beats with holds between, so key selection
|
|
// has genuine extremes and genuine plateaux to find.
|
|
const beat = Math.floor(t / 9) % 4;
|
|
const target = [0.004, 0.05, 0.022, 0.0];
|
|
const openAmt = target[beat];
|
|
const wide = 0.10 + (beat === 1 ? 0.012 : beat === 3 ? -0.008 : 0);
|
|
|
|
place(RIGID[4], 0.000, -0.050); place(RIGID[5], 0.000, -0.020);
|
|
place(RIGID[6], 0.000, 0.012);
|
|
|
|
// Eyes. The corners (RIGID[0..3]) are placed BY the lid rings rather than
|
|
// separately, because they are slots 0 and 8 of those rings: writing them
|
|
// twice is how the mouth grew a bowtie, and a corner that disagrees with
|
|
// its own ring would make the eye self-intersect at some vertex budgets
|
|
// and not others.
|
|
//
|
|
// A blink is ONE frame, which is the honest hard case: at 12fps that is
|
|
// what a real blink costs, and it is exactly the length that reads as a
|
|
// dropped frame rather than as a blink unless `hold` extends it.
|
|
const blink = t > 5 && t % 19 === 0;
|
|
const openness = blink ? 0.05 : 1;
|
|
|
|
// Gaze holds and then jumps, the way gaze actually behaves, with a little
|
|
// jitter on top so quantisation has noise to remove and the dwell has
|
|
// something to suppress.
|
|
const LOOK = [[0, 0], [0.16, 0.0], [-0.16, 0.05], [0.0, -0.09]];
|
|
const [gx, gy] = LOOK[Math.floor(t / 11) % LOOK.length];
|
|
const jit = () => (Math.random() - 0.5) * 0.012;
|
|
|
|
// Half the corner separation, and the lid half-height at full open.
|
|
const EYE_RX = 0.0235, EYE_RY = 0.011, EYE_Y = -0.044;
|
|
const eye = (ring, cx, dir, iris) => {
|
|
const n = ring.length;
|
|
for (let k = 0; k < n; k++) {
|
|
// dir flips the traversal so each ring runs the direction its real
|
|
// table does: slot 0 outer corner, 4 upper lid, 8 inner, 12 lower.
|
|
const a = dir > 0 ? Math.PI + (k / n) * Math.PI * 2 : -(k / n) * Math.PI * 2;
|
|
place(ring[k], cx + EYE_RX * Math.cos(a),
|
|
EYE_Y + EYE_RY * openness * Math.sin(a));
|
|
}
|
|
// Iris: centre first, then four ring points, as the refined mesh emits.
|
|
const ix = cx + (gx + jit()) * EYE_RX * 2, iy = EYE_Y + (gy + jit()) * EYE_RX * 2;
|
|
place(iris[0], ix, iy);
|
|
for (let k = 1; k < iris.length; k++) {
|
|
const a = ((k - 1) / (iris.length - 1)) * Math.PI * 2;
|
|
place(iris[k], ix + 0.008 * Math.cos(a), iy + 0.008 * Math.sin(a));
|
|
}
|
|
};
|
|
eye(EYE_R_RING, -0.0515, 1, swapIris ? IRIS_B : IRIS_A);
|
|
eye(EYE_L_RING, 0.0515, -1, swapIris ? IRIS_A : IRIS_B);
|
|
|
|
// Lip rings as ellipse arcs, traversed so ring ORDER matches the tables:
|
|
// slot 0 = right corner, 5 = top centre, 10 = left corner, 15 = bottom
|
|
// centre, with y growing downward. Getting this convention wrong swaps two
|
|
// opposite vertices and the ring self-intersects into a bowtie - see the
|
|
// ring-simplicity assertion in selftest.
|
|
const ring = (table, rx, ry, cy) => {
|
|
const n = table.length;
|
|
for (let k = 0; k < n; k++) {
|
|
const a = -(k / n) * Math.PI * 2;
|
|
place(table[k], rx * Math.cos(a), cy + ry * Math.sin(a));
|
|
}
|
|
};
|
|
ring(LIPS_OUTER, wide / 2, 0.012 + openAmt * 0.6, 0.075);
|
|
// APERTURE (13, 14) are slots 5 and 15 of the inner ring, so the ring itself
|
|
// places them at the vertical extremes. Writing them again afterwards is what
|
|
// produced the bowtie; the aperture is simply the inner ring's height.
|
|
ring(LIPS_INNER, wide / 2.6, 0.001 + openAmt, 0.075);
|
|
|
|
for (let k = 0; k < FACE_OVAL.length; k++) {
|
|
const a = -Math.PI / 2 + (k / FACE_OVAL.length) * Math.PI * 2;
|
|
place(FACE_OVAL[k], 0.105 * Math.cos(a), 0.145 * Math.sin(a) + 0.01);
|
|
}
|
|
frames.push(pts);
|
|
}
|
|
return frames;
|
|
}
|