Three causes, all of them mine: Otsu always returns a split, including on a homogeneous region - given a dark cavity with no teeth it invents a threshold and calls half the pixels bright. The gate is now the separation between the two class means, which is the only thing that says whether the split means anything. Coverage was the wrong signal: it is high both when the mouth is full of teeth and when the region is uniformly dark and Otsu has split noise. The row scan tracked the last qualifying row anywhere rather than where the run from the top stops, so one bright row near the bottom - a lit lower lip inside the ring - pushed the line to full height. It now breaks at the first failing row once the run has started. MediaPipe's inner lip landmarks sit slightly outside the real opening, so the sampled region included lip pixels, which are bright and sit exactly at the boundary where they do most damage. The ring is now eroded toward its centroid before sampling, with the amount exposed as a knob. Adds a diagnostic panel showing the sampled crop, pixels above threshold, and the resolved line, because tuning this from numbers alone does not tell you whether the region being measured is even the right region. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
176 lines
7 KiB
JavaScript
176 lines
7 KiB
JavaScript
// Mouth interior from image content.
|
|
//
|
|
// MediaPipe has no landmarks inside the lips - the inner ring bounds the cavity
|
|
// and everything within it is just pixels. So teeth have to come from the
|
|
// picture, and the question is how to do that without reintroducing the boil
|
|
// that per-frame detection causes.
|
|
//
|
|
// The answer is to extract a SCALAR, not a shape. Tracing the bright blob would
|
|
// give a new contour every frame with no vertex correspondence - exactly the
|
|
// failure docs/roto-puppet.md warns about. Instead the teeth polygon is the
|
|
// inner lip ring clipped to a horizontal line, and only that line's height is
|
|
// measured. The silhouette is therefore always the mouth's own shape (stable by
|
|
// construction) and the only thing that varies per frame is one number, which
|
|
// smooths trivially. It is also how the shape is drawn by hand: a band bounded
|
|
// by the lip.
|
|
|
|
// Otsu's threshold over a luminance histogram. Self-tuning, so exposure changes
|
|
// between frames do not shift what counts as "bright".
|
|
export function otsuForTest(h, t) { return otsu(h, t); }
|
|
|
|
function otsu(hist, total) {
|
|
let sum = 0;
|
|
for (let i = 0; i < 256; i++) sum += i * hist[i];
|
|
let sumB = 0, wB = 0, best = 0, bestVar = -1, bestMB = 0, bestMF = 0;
|
|
for (let t = 0; t < 256; t++) {
|
|
wB += hist[t];
|
|
if (!wB) continue;
|
|
const wF = total - wB;
|
|
if (!wF) break;
|
|
sumB += t * hist[t];
|
|
const mB = sumB / wB, mF = (sum - sumB) / wF; // mB = dark class, mF = bright
|
|
const between = wB * wF * (mB - mF) * (mB - mF);
|
|
if (between > bestVar) { bestVar = between; best = t; bestMB = mB; bestMF = mF; }
|
|
}
|
|
return { thr: best, mDark: bestMB, mBright: bestMF };
|
|
}
|
|
|
|
const pointInPoly = (pts, x, y) => {
|
|
let inside = false;
|
|
for (let i = 0, j = pts.length - 1; i < pts.length; j = i++) {
|
|
if ((pts[i].y > y) !== (pts[j].y > y) &&
|
|
x < ((pts[j].x - pts[i].x) * (y - pts[i].y)) / (pts[j].y - pts[i].y) + pts[i].x) inside = !inside;
|
|
}
|
|
return inside;
|
|
};
|
|
|
|
// Shrink a ring toward its centroid.
|
|
//
|
|
// MediaPipe's inner lip landmarks sit slightly OUTSIDE the actual opening, so
|
|
// sampling the ring as given includes lip pixels - which are bright, and sit
|
|
// right at the cavity boundary where they do the most damage.
|
|
function erode(pts, k) {
|
|
let cx = 0, cy = 0;
|
|
for (const p of pts) { cx += p.x; cy += p.y; }
|
|
cx /= pts.length; cy /= pts.length;
|
|
return pts.map((p) => ({ x: cx + (p.x - cx) * (1 - k), y: cy + (p.y - cy) * (1 - k) }));
|
|
}
|
|
|
|
// Measure one frame: how far down the cavity the bright region reaches, as a
|
|
// fraction of cavity height, plus the contrast that justified calling it bright.
|
|
//
|
|
// `wantDebug` returns the sampled crop with the classification drawn on it.
|
|
// Tuning this blind is miserable; the numbers alone do not say whether the
|
|
// region being measured is even the right region.
|
|
export function measureInterior(img, innerNorm, ctx, opts = {}, wantDebug = false) {
|
|
const minContrast = opts.minContrast ?? 0.14;
|
|
const rowFrac = opts.rowFrac ?? 0.4;
|
|
const inner = erode(innerNorm, opts.erode ?? 0.18);
|
|
|
|
const none = { teethT: 0, contrast: 0, coverage: 0, debug: null };
|
|
let x0 = 1, y0 = 1, x1 = 0, y1 = 0;
|
|
for (const p of inner) {
|
|
x0 = Math.min(x0, p.x); y0 = Math.min(y0, p.y);
|
|
x1 = Math.max(x1, p.x); y1 = Math.max(y1, p.y);
|
|
}
|
|
const W = img.naturalWidth, H = img.naturalHeight;
|
|
const px0 = Math.max(0, Math.floor(x0 * W)), py0 = Math.max(0, Math.floor(y0 * H));
|
|
const pw = Math.min(W - px0, Math.ceil((x1 - x0) * W)), ph = Math.min(H - py0, Math.ceil((y1 - y0) * H));
|
|
if (pw < 4 || ph < 4) return none;
|
|
|
|
ctx.canvas.width = pw; ctx.canvas.height = ph;
|
|
ctx.drawImage(img, px0, py0, pw, ph, 0, 0, pw, ph);
|
|
const img0 = ctx.getImageData(0, 0, pw, ph);
|
|
const data = img0.data;
|
|
|
|
const poly = inner.map((p) => ({ x: p.x * W - px0, y: p.y * H - py0 }));
|
|
const hist = new Uint32Array(256);
|
|
const lum = new Float32Array(pw * ph);
|
|
const mask = new Uint8Array(pw * ph);
|
|
let n = 0;
|
|
for (let y = 0; y < ph; y++) {
|
|
for (let x = 0; x < pw; x++) {
|
|
if (!pointInPoly(poly, x + 0.5, y + 0.5)) continue;
|
|
const o = (y * pw + x) * 4;
|
|
const l = (0.299 * data[o] + 0.587 * data[o + 1] + 0.114 * data[o + 2]) | 0;
|
|
const i = y * pw + x;
|
|
lum[i] = l; mask[i] = 1; hist[l]++; n++;
|
|
}
|
|
}
|
|
if (n < 16) return none;
|
|
|
|
const { thr, mDark, mBright } = otsu(hist, n);
|
|
|
|
// Otsu ALWAYS returns a split, including on a homogeneous region: given a dark
|
|
// cavity with no teeth it invents a threshold and calls half the pixels
|
|
// bright. The separation between the two class means is what says whether the
|
|
// split means anything, so it is the actual gate.
|
|
const contrast = (mBright - mDark) / 255;
|
|
if (contrast < minContrast) {
|
|
return { teethT: 0, contrast, coverage: 0, debug: wantDebug ? debugCanvas(img0, mask, lum, thr, pw, ph, -1) : null };
|
|
}
|
|
|
|
// Scan from the top and STOP at the first row that fails. Teeth hang from the
|
|
// upper lip, so what matters is the contiguous run, not whether some row near
|
|
// the bottom happens to qualify - tracking the latter is what made the band
|
|
// fill the whole mouth.
|
|
let lastRow = -1, started = false, bright = 0;
|
|
for (let y = 0; y < ph; y++) {
|
|
let rowIn = 0, rowBright = 0;
|
|
for (let x = 0; x < pw; x++) {
|
|
const i = y * pw + x;
|
|
if (!mask[i]) continue;
|
|
rowIn++;
|
|
if (lum[i] > thr) { rowBright++; bright++; }
|
|
}
|
|
if (rowIn < 2) continue;
|
|
const ok = rowBright / rowIn > rowFrac;
|
|
if (ok) { started = true; lastRow = y; }
|
|
else if (started) break;
|
|
}
|
|
|
|
return {
|
|
teethT: lastRow < 0 ? 0 : (lastRow + 1) / ph,
|
|
contrast,
|
|
coverage: bright / n,
|
|
debug: wantDebug ? debugCanvas(img0, mask, lum, thr, pw, ph, lastRow) : null,
|
|
};
|
|
}
|
|
|
|
// The sampled crop with the classification painted on: sampled region tinted,
|
|
// pixels above threshold in green, the resolved teeth line in amber.
|
|
function debugCanvas(img0, mask, lum, thr, pw, ph, lastRow) {
|
|
const c = document.createElement('canvas');
|
|
c.width = pw; c.height = ph;
|
|
const g = c.getContext('2d');
|
|
const out = new ImageData(pw, ph);
|
|
for (let i = 0; i < pw * ph; i++) {
|
|
const o = i * 4;
|
|
const [r, gr, b] = [img0.data[o], img0.data[o + 1], img0.data[o + 2]];
|
|
if (!mask[i]) { out.data[o] = r * 0.3; out.data[o + 1] = gr * 0.3; out.data[o + 2] = b * 0.3; }
|
|
else if (lum[i] > thr) { out.data[o] = 60; out.data[o + 1] = 230; out.data[o + 2] = 120; }
|
|
else { out.data[o] = r; out.data[o + 1] = gr; out.data[o + 2] = b; }
|
|
out.data[o + 3] = 255;
|
|
}
|
|
g.putImageData(out, 0, 0);
|
|
if (lastRow >= 0) {
|
|
g.fillStyle = '#fbbf24';
|
|
g.fillRect(0, lastRow, pw, 1);
|
|
}
|
|
return c;
|
|
}
|
|
|
|
// Sutherland-Hodgman against the half-plane y <= limit.
|
|
export function clipPolyAbove(pts, limit) {
|
|
const out = [];
|
|
for (let i = 0; i < pts.length; i++) {
|
|
const a = pts[i], b = pts[(i + 1) % pts.length];
|
|
const ain = a.y <= limit, bin = b.y <= limit;
|
|
if (ain) out.push(a);
|
|
if (ain !== bin) {
|
|
const t = (limit - a.y) / (b.y - a.y);
|
|
out.push({ x: a.x + t * (b.x - a.x), y: limit });
|
|
}
|
|
}
|
|
return out;
|
|
}
|