From de90bfd907b19bda49823b0b265048a6f2a010b4 Mon Sep 17 00:00:00 2001 From: Your Name Date: Thu, 24 Sep 2026 14:57:59 -0400 Subject: [PATCH] Registered photo underlay as the plate reference The generated face oval was never going to be good enough to draw from: MediaPipe's face oval is the FACE boundary, cut at the hairline and excluding hair, ears, jaw and neck, so it is an egg by construction. Segmentation would give a real head outline but costs a 16MB model and per-frame inference for a shape that gets replaced by a drawing anyway. So the plate layer becomes switchable, and the useful modes are photographic: the source frame mapped into raster space through the same transform chain the contours go through. Registration is the whole point - the head sits still and a drawing traced from the underlay is already aligned to the mouth. An unregistered underlay would be decoration. - underlay.js: pixel->raster affine (a general affine, since MediaPipe normalises x by width and y by height), registered draw, palette posterise - plate modes: photo / photo dim / posterized / oval / oval+photo / none, B cycles - worksheet cells are registered composites rather than raw crops - Save frame 4x writes a 1280x800 PNG to draw on - selftest: FACE_OVAL simplicity, which was never asserted; a wrong ordering there reads as a lumpy plate rather than an obvious bowtie Co-Authored-By: Claude Opus 5 --- .gitignore | 2 ++ README.md | 25 ++++++++++++++ index.html | 17 +++++++-- js/app.js | 93 +++++++++++++++++++++++++++++++++++++++++--------- js/selftest.js | 11 ++++++ js/underlay.js | 63 ++++++++++++++++++++++++++++++++++ 6 files changed, 191 insertions(+), 20 deletions(-) create mode 100644 js/underlay.js diff --git a/.gitignore b/.gitignore index bedda0e..caf400b 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,5 @@ frames/ *.task *.take + +*.tflite diff --git a/README.md b/README.md index cbd26fc..85b29bc 100644 --- a/README.md +++ b/README.md @@ -62,6 +62,31 @@ is hand-drawn head plates, which this tool does not yet do. | closed-mouth cut | Aperture below which the mouth interior is emitted as `hidden`. | | suggest tolerance | Max head movement before a new plate drawing is required. Affects **Suggest** only. | +## The plate is reference, not art + +The plate layer has several representations because its job changes. Cycle with +B: + +| Mode | For | +| --- | --- | +| photo dim / photo | **Drawing over.** The source frame, stabilised. | +| posterized | The footage quantised into the ramp — a look test asked of the source rather than of a drawing. | +| oval | A flat stand-in, to judge the mouth against something. | +| oval + photo | Checking the stand-in against the real head. | +| none | Mouth alone. | + +Photo modes are **registered**: the frame is mapped into raster space through the +same transform chain the contours go through, so the head sits still and a +drawing traced from it is already aligned to the mouth. An unregistered underlay +would be decorative. + +`MediaPipe`'s face oval is the *face* boundary — it cuts at the hairline and +excludes hair, ears, jaw underside and neck — so as a head silhouette it is an +egg by construction, and no landmark precision fixes that. Hence the photo. + +**Save frame 4x** writes the current registered composite as a 1280×800 PNG to +draw on. + ## Two kinds of sparseness Sparseness has two unrelated causes, and conflating them was the original design diff --git a/index.html b/index.html index b74d5ed..37475e3 100644 --- a/index.html +++ b/index.html @@ -68,6 +68,15 @@ + + @@ -88,7 +97,9 @@

flat render — 320×200 indexed

-
audio drives the clock — dropped frames, never drift
+
audio drives the clock — dropped frames, never drift
+ plate representation: B cycles · photo modes are registered + into raster space, so tracing them lands on the mouth
@@ -97,7 +108,7 @@
- ← → step · X delete · K keep · + ← → step · X delete · K keep · B background · click to select · double-click or shift-click to toggle · the mouth keeps every frame regardless
@@ -131,7 +142,7 @@
-

drawings needed — one per kept frame, with the range it holds

+

drawings needed — registered reference per kept frame, with the range it holds

diff --git a/js/app.js b/js/app.js index 5fe666f..d4f2528 100644 --- a/js/app.js +++ b/js/app.js @@ -2,6 +2,7 @@ import { FaceLandmarker, FilesetResolver } from 'https://cdn.jsdelivr.net/npm/@m import { LIPS_OUTER, LIPS_INNER, FACE_OVAL } from './landmarks.js'; import { stabilize, toRasterRing, smoothContours, suggestPlateFrames, heldFrame } from './pipeline.js'; import { IndexedRaster } from './raster.js'; +import { drawRegistered, posterizeInto } from './underlay.js'; import { writeTake } from './take.js'; import { synthDense } from './synth.js'; @@ -180,16 +181,66 @@ function faceBoxes() { /* ---------- render ---------- */ -function renderFrame(f) { +// The plate layer has several representations because its job changes: a flat +// shape to judge the mouth against, or a registered photograph to draw over. +// Only the latter is any use as reference art, and the generated oval is only a +// stand-in until a drawing exists. +function renderFrame(f, mode = plateMode()) { const r = new IndexedRaster(RW, RH); - r.clear(IDX.bg); - const pf = heldFrame(keptSorted(), f); // the plate that is on screen - r.fillPoly(state.plates[pf], IDX.base); - r.fillPoly(state.outer[f], IDX.dark); // mouth at full rate + const pf = heldFrame(keptSorted(), f); // the plate frame on screen + + if (mode === 'posterize' && state.images[pf]) { + posterizeInto(r, state.images[pf], state.stab.transforms[pf], state.xform, + PALETTE.map((p) => p.hex)); + } else { + r.clear(IDX.bg); + if (mode === 'oval' || mode === 'oval+photo') r.fillPoly(state.plates[pf], IDX.base); + } + + r.fillPoly(state.outer[f], IDX.dark); // mouth keeps every frame if (!state.hidden[f]) r.fillPoly(state.inner[f], IDX.mouth); return r; } +const plateMode = () => el('plateMode').value; + +// Photo modes composite under the indexed layer, so the flat shapes stay exactly +// as they render while the reference sits behind them. +function compositeRender(canvas, f, zoom) { + const mode = plateMode(); + const pf = heldFrame(keptSorted(), f); + const img = state.images[pf]; + const showPhoto = img && (mode === 'photo' || mode === 'photo-dim' || mode === 'oval+photo'); + + canvas.width = RW * zoom; canvas.height = RH * zoom; + const g = canvas.getContext('2d'); + g.fillStyle = PALETTE[IDX.bg].hex; + g.fillRect(0, 0, canvas.width, canvas.height); + + if (showPhoto) { + drawRegistered(g, img, state.stab.transforms[pf], state.xform, zoom, + mode === 'photo-dim' ? 0.34 : 1); + } + + const r = renderFrame(f, mode === 'oval+photo' ? 'oval' : (showPhoto ? 'off' : mode)); + const img2 = r.toImageData(PALETTE.map((p) => p.hex), zoom); + if (showPhoto) { + // Keep the photo visible wherever the indexed layer is background. + const bg = PALETTE[IDX.bg].hex.replace('#', ''); + const br = parseInt(bg.slice(0, 2), 16), bgn = parseInt(bg.slice(2, 4), 16), bb = parseInt(bg.slice(4, 6), 16); + const d = img2.data; + for (let i = 0; i < d.length; i += 4) { + if (d[i] === br && d[i + 1] === bgn && d[i + 2] === bb) d[i + 3] = 0; + } + const tmp = document.createElement('canvas'); + tmp.width = img2.width; tmp.height = img2.height; + tmp.getContext('2d').putImageData(img2, 0, 0); + g.drawImage(tmp, 0, 0); + } else { + g.putImageData(img2, 0, 0); + } +} + const keptSorted = () => [...state.keep].sort((a, b) => a - b); function blit(canvas, raster, zoom) { @@ -254,7 +305,7 @@ function drawPanes() { strokePts(g2, z(state.outer[f]), '#4ade80'); if (!state.hidden[f]) strokePts(g2, z(state.inner[f]), '#f87171'); - blit(el('cv-render'), renderFrame(f), ZOOM); + compositeRender(el('cv-render'), f, ZOOM); } function strokePts(g, pts, color, lw = 1) { @@ -322,18 +373,11 @@ function drawWorksheet() { const until = (i + 1 < kept.length ? kept[i + 1] : state.dense.length) - 1; const cell = document.createElement('div'); cell.className = 'cell'; + // Registered, not raw-cropped: the worksheet frame is in raster space, so a + // drawing traced from it is already aligned to the mouth. const cv = document.createElement('canvas'); - const im = state.images[f]; - if (im) { - const b = state.faceBox[f]; - const sx = b.x0 * im.naturalWidth, sy = b.y0 * im.naturalHeight; - const sw = (b.x1 - b.x0) * im.naturalWidth, sh = (b.y1 - b.y0) * im.naturalHeight; - cv.width = 150; cv.height = Math.round(150 * (sh / sw)); - cv.getContext('2d').drawImage(im, sx, sy, sw, sh, 0, 0, cv.width, cv.height); - } else { - blit(cv, renderFrame(f), 1); - cv.style.width = '150px'; - } + compositeRender(cv, f, 1); + cv.style.width = '176px'; const cap = document.createElement('span'); cap.textContent = `f${f}` + (until > f ? ` → ${until}` : '') + ` (${until - f + 1}f)`; cell.append(cv, cap); @@ -438,6 +482,17 @@ el('scrub').addEventListener('input', (e) => { seekTo(+e.target.value); drawAll( el('btn-frames').onclick = runFrames; el('btn-synth').onclick = runSynthetic; el('btn-export').onclick = exportTake; +el('plateMode').addEventListener('change', () => { if (state.dense) drawAll(); }); +el('btn-saveframe').onclick = () => { + if (!state.dense) return; + const cv = document.createElement('canvas'); + compositeRender(cv, state.frame, 4); // 1280x800: enough to draw on + const a = document.createElement('a'); + a.href = cv.toDataURL('image/png'); + a.download = `${el('takename').value || 'take'}_f${String(state.frame).padStart(4, '0')}.png`; + a.click(); + status(`saved registered frame f${state.frame} at 4x`, 'ok'); +}; el('btn-keepall').onclick = () => { if (state.dense) { rebuild(true); } }; el('btn-suggest').onclick = () => { if (!state.dense) return; @@ -478,6 +533,10 @@ window.addEventListener('keydown', (e) => { else if (e.key === 'ArrowLeft') { seekTo(Math.max(0, state.frame - 1)); drawAll(); } else if (e.key === 'Backspace' || e.key === 'Delete' || e.key === 'x') { state.keep.delete(state.frame === 0 ? -1 : state.frame); drawAll(); + } else if (e.key === 'b') { + const sel = el('plateMode'); + sel.selectedIndex = (sel.selectedIndex + 1) % sel.options.length; + drawAll(); } else if (e.key === 'k' || e.key === ' ') { if (state.frame !== 0) state.keep.add(state.frame); drawAll(); diff --git a/js/selftest.js b/js/selftest.js index af45d75..cbf616e 100644 --- a/js/selftest.js +++ b/js/selftest.js @@ -80,6 +80,17 @@ export function run() { ok(`${label} ring is simple at every vertex budget`, !worst, worst || ''); } + // FACE_OVAL traversal: never checked before, and a wrong ordering here shows up + // as a lumpy plate rather than an obvious bowtie, so it needs asserting. + { + let bad = null; + for (let f = 0; f < dense.length && !bad; f++) { + const h = ringSelfIntersections(FACE_OVAL.map((i) => dense[f][i])); + if (h.length) bad = `frame ${f} edges ${JSON.stringify(h[0])}`; + } + ok('FACE_OVAL is a simple ring on every frame', !bad, bad || ''); + } + // similarity fit recovers a known transform const src = [{ x: 0, y: 0 }, { x: 1, y: 0 }, { x: 0, y: 1 }, { x: 2, y: 3 }]; const truth = { s: 1.7, theta: 0.6, tx: 4, ty: -2 }; diff --git a/js/underlay.js b/js/underlay.js new file mode 100644 index 0000000..81ba004 --- /dev/null +++ b/js/underlay.js @@ -0,0 +1,63 @@ +// Registered photo underlay: the source frame mapped into raster space through +// the same transform chain the vector shapes go through, so a drawing made over +// it lands on the shapes. +// +// Without registration an underlay is decorative. With it, the photo is +// stabilised exactly as the contours are - the head sits still - and tracing over +// it produces plate art already aligned to the mouth. + +import { applySim } from './mathutil.js'; + +// Compose pixel-space -> raster-space into one affine. +// +// MediaPipe normalises x by width and y by height, so normalised space is a +// stretched pixel space and the composition is a general affine rather than a +// similarity. Three mapped points determine it exactly. +export function frameAffine(tf, xform, imgW, imgH) { + const map = (px, py) => xform(applySim(tf, { x: px / imgW, y: py / imgH })); + const P0 = map(0, 0), P1 = map(imgW, 0), P2 = map(0, imgH); + return { + a: (P1.x - P0.x) / imgW, b: (P1.y - P0.y) / imgW, + c: (P2.x - P0.x) / imgH, d: (P2.y - P0.y) / imgH, + e: P0.x, f: P0.y, + }; +} + +// Draw the registered source frame into a raster-sized 2D context. +export function drawRegistered(ctx, img, tf, xform, zoom, alpha = 1) { + const m = frameAffine(tf, xform, img.naturalWidth, img.naturalHeight); + ctx.save(); + ctx.globalAlpha = alpha; + ctx.setTransform(m.a * zoom, m.b * zoom, m.c * zoom, m.d * zoom, m.e * zoom, m.f * zoom); + ctx.imageSmoothingEnabled = true; + ctx.drawImage(img, 0, 0); + ctx.restore(); + ctx.setTransform(1, 0, 0, 1, 0, 0); +} + +// Quantise a registered frame straight into palette indices. +// +// Doubles as a look test: it shows what the footage becomes in the chosen ramp, +// with no dithering and no antialiasing, which is the question "will these tones +// read" asked directly of the source rather than of a drawing. +export function posterizeInto(raster, img, tf, xform, paletteHex) { + const tmp = document.createElement('canvas'); + tmp.width = raster.w; tmp.height = raster.h; + const g = tmp.getContext('2d', { willReadFrequently: true }); + g.fillStyle = '#000'; g.fillRect(0, 0, raster.w, raster.h); + drawRegistered(g, img, tf, xform, 1, 1); + const px = g.getImageData(0, 0, raster.w, raster.h).data; + const pal = paletteHex.map((h) => { + const s = h.replace('#', ''); + return [parseInt(s.slice(0, 2), 16), parseInt(s.slice(2, 4), 16), parseInt(s.slice(4, 6), 16)]; + }); + for (let i = 0, n = raster.w * raster.h; i < n; i++) { + const r = px[i * 4], g2 = px[i * 4 + 1], b = px[i * 4 + 2]; + let best = 0, bd = Infinity; + for (let k = 0; k < pal.length; k++) { + const d = (pal[k][0] - r) ** 2 + (pal[k][1] - g2) ** 2 + (pal[k][2] - b) ** 2; + if (d < bd) { bd = d; best = k; } + } + raster.buf[i] = best; + } +}