diff --git a/README.md b/README.md index 107a7c2..cbd26fc 100644 --- a/README.md +++ b/README.md @@ -24,7 +24,7 @@ python3 -m http.server 8777 # from this directory For real footage: ```sh -./extract.sh /path/to/clip.mp4 24 # -> frames/0001.png … +./extract.sh /path/to/clip.mov 12 # -> frames/*.png, audio.wav, manifest.json ``` then **Load frames**. MediaPipe's wasm is fetched from jsdelivr on first use; @@ -34,6 +34,14 @@ Frames are pre-extracted rather than decoded in the page because browser video seeking is approximate and `requestVideoFrameCallback` only delivers frames at playback speed — neither gives a deterministic per-frame pass. +`manifest.json` records the true extraction rate. The page reads it rather than +assuming, because a guessed fps desynchronises audio from picture — and sync is +the one thing this view exists to show. + +**Audio is the playback clock**: `frame = floor(audio.currentTime * fps)`. A slow +render loop therefore drops frames instead of drifting, and ½x / ¼x work by +setting `playbackRate` with the picture following for free. + ## Shooting for it Near-frontal, good light, consistent scale, head reasonably still. Hold a neutral @@ -49,18 +57,33 @@ is hand-drawn head plates, which this tool does not yet do. | Knob | What it does | | --- | --- | | vertices | Lip vertex budget. The reduction past what the footage supports *is* the style. | -| min hold | Minimum frames between keys. | -| change gate | Mean vertex movement required before a new key is accepted. | -| vel smoothing | Window on the velocity signal used to find extremes. | -| anchor smoothing | Window on the four similarity parameters. Smooths the *transform*, never the contour. | -| exposure | Grid that key frames snap onto. | +| contour avg ±f | Radius in frames. 0 off, 1 = ±1. Removes per-frame landmark jitter. | +| anchor avg ±f | Radius on the four similarity parameters. Smooths the *transform*. | | closed-mouth cut | Aperture below which the mouth interior is emitted as `hidden`. | +| suggest tolerance | Max head movement before a new plate drawing is required. Affects **Suggest** only. | -Keys go on **velocity minima**, not distance thresholds: a threshold fires at the -frame it was crossed — partway through a transition — so poses land mushy and -late. The timeline shows the velocity curve, candidate minima, accepted keys -(`f`) and their pre-snap extremes (`src`); a large `f`/`src` gap means min-hold -and exposure are fighting. +## Two kinds of sparseness + +Sparseness has two unrelated causes, and conflating them was the original design +error here. **Aesthetic** sparseness is set by the extraction rate — pick 12fps and +you have already chosen your timing. **Labour** sparseness is a human drawing +each one, and it binds only on the plate. + +So the mouth keeps **every** frame: it is traced, and therefore free. In limited +animation lip sync is routinely the densest element, on 1s, while heads hold on +2s and 3s. + +The frame strip is the editing surface for the other half: which frames need +their own plate drawing. Everything starts kept; delete what you don't want. +**Suggest** runs error-tolerance decimation over head pose as a starting point, +then you hand-correct. + +`contour avg` is a deliberate, bounded exception to "never smooth the contour" in +`../docs/roto-puppet.md`. That rule held while keys were sparse, because sampling +at velocity minima rejected detector noise for free. With a key on every frame it +does not, so a radius shorter than the shortest articulation worth keeping is +justified — at 12fps, articulation spans 3–6 frames and detector noise is +per-frame, so ±1 separates them and ±3 starts eating speech. ## Tests @@ -79,8 +102,9 @@ eyeball. ## Not done yet -Eyes and irises; hand-drawn head plates and per-plate mouth slots; real +Eyes and irises; hand-drawn head plates and per-plate mouth slots (the strip +decides *which frames need one*, but you cannot yet supply the drawing); real performer→character calibration (currently identity, fitting the face oval to the -canvas); the override layer; anything on the Animator Pro side. The placeholder -plate is a frozen face-oval polygon — it exists so the mouth has a face to read +canvas); the override layer; anything on the Animator Pro side. The plate is a +face-oval polygon per kept frame — it exists so the mouth has a face to read against, not to look good. diff --git a/audio.wav b/audio.wav new file mode 100644 index 0000000..9c39a31 Binary files /dev/null and b/audio.wav differ diff --git a/extract.sh b/extract.sh index 50a4bbb..cdbc9c6 100755 --- a/extract.sh +++ b/extract.sh @@ -1,17 +1,38 @@ #!/usr/bin/env bash -# Extract a clip to a PNG sequence for the take builder. +# Extract a clip to a PNG sequence + audio + manifest for the take builder. # # Frames are pre-extracted rather than decoded in the page on purpose: browser # video seeking by currentTime is approximate and requestVideoFrameCallback only # delivers frames at playback speed, so neither gives a deterministic per-frame # pass. A PNG sequence is exact, instantly seekable, and reproducible. +# +# Audio comes out alongside because the page uses it as the PLAYBACK CLOCK - +# frame = floor(audio.currentTime * fps) - so picture and sound cannot drift +# apart no matter how long the shot is or how slow the render loop runs. set -euo pipefail src="${1:?usage: ./extract.sh CLIP [FPS] [OUTDIR]}" -fps="${2:-24}" +fps="${2:-12}" out="${3:-frames}" -rm -rf "$out" -mkdir -p "$out" +rm -rf "$out"; mkdir -p "$out" ffmpeg -hide_banner -loglevel warning -i "$src" -vf "fps=$fps" "$out/%04d.png" -echo "$(ls -1 "$out" | wc -l) frames at ${fps}fps -> $out/" +count=$(ls -1 "$out" | wc -l) + +# Mono is enough for judging sync and halves the file. Absent audio is not fatal. +if ffprobe -v error -select_streams a:0 -show_entries stream=codec_type \ + -of csv=p=0 "$src" 2>/dev/null | grep -q audio; then + ffmpeg -hide_banner -loglevel warning -y -i "$src" -vn -ac 1 -ar 44100 audio.wav + audio='"audio.wav"' + echo "audio -> audio.wav" +else + audio='null' + echo "no audio stream" +fi + +# The page must know the true extraction rate: if it guessed, audio and picture +# would drift. Source of truth lives here, next to the frames it describes. +printf '{"fps":%s,"frames":%s,"dir":"%s","audio":%s,"source":"%s"}\n' \ + "$fps" "$count" "$out" "$audio" "$(basename "$src")" > manifest.json + +echo "$count frames at ${fps}fps -> $out/ (manifest.json written)" diff --git a/index.html b/index.html index dfad8d9..b74d5ed 100644 --- a/index.html +++ b/index.html @@ -63,7 +63,8 @@ - + + @@ -87,6 +88,7 @@

flat render — 320×200 indexed

+
audio drives the clock — dropped frames, never drift
diff --git a/js/app.js b/js/app.js index 10c6b03..5fe666f 100644 --- a/js/app.js +++ b/js/app.js @@ -21,6 +21,7 @@ const state = { outer: null, inner: null, plates: null, hidden: null, keep: new Set(), // frames that get their own plate drawing frame: 0, playing: false, faceBox: null, + fps: 12, audio: null, // fps comes from manifest.json, never guessed }; const el = (id) => document.getElementById(id); @@ -44,6 +45,24 @@ const loadImage = (src) => new Promise((r) => { im.onload = () => r(im); im.onerror = () => r(null); im.src = src; }); +// The extraction rate is read, not assumed. Guessing it would desynchronise +// audio from picture, which is the one thing this view exists to show. +async function loadManifest() { + try { + const r = await fetch('./manifest.json', { cache: 'no-store' }); + if (!r.ok) return null; + return await r.json(); + } catch { return null; } +} + +function attachAudio(name) { + const a = el('audio'); + if (!name) { a.removeAttribute('src'); a.hidden = true; state.audio = null; return; } + a.src = './' + name; + a.hidden = false; + state.audio = a; +} + async function loadFrameSequence() { const dir = el('framedir').value.replace(/\/$/, ''); const imgs = []; @@ -199,7 +218,7 @@ function drawPanes() { const f = state.frame, kept = keptSorted(); const pf = heldFrame(kept, f); el('framelabel').textContent = - `f ${f} / ${state.dense.length - 1} · plate from f${pf}` + + `f ${f} / ${state.dense.length - 1} · ${(f / state.fps).toFixed(2)}s · plate f${pf}` + (state.keep.has(f) ? ' · KEPT' : ' · held'); const c1 = el('cv-source'), g1 = c1.getContext('2d'); @@ -279,7 +298,7 @@ function drawStrip() { tag.textContent = f; cell.append(cv, tag); cell.onclick = (ev) => { - state.frame = f; + seekTo(f); if (ev.shiftKey) toggle(f); drawAll(); }; @@ -318,7 +337,7 @@ function drawWorksheet() { const cap = document.createElement('span'); cap.textContent = `f${f}` + (until > f ? ` → ${until}` : '') + ` (${until - f + 1}f)`; cell.append(cv, cap); - cell.onclick = () => { state.frame = f; drawAll(); }; + cell.onclick = () => { seekTo(f); drawAll(); }; host.append(cell); }); } @@ -330,7 +349,7 @@ function exportTake() { const N = state.dense.length; const take = { name: el('takename').value || 'line_01', - frames: N, width: RW, height: RH, exposure: 1, + frames: N, width: RW, height: RH, exposure: 1, fps: state.fps, palette: PALETTE, slot: { x: RW / 2, y: RH / 2 }, parts: [ @@ -361,6 +380,15 @@ function exportTake() { async function runFrames() { try { + const man = await loadManifest(); + if (man) { + state.fps = man.fps; + el('framedir').value = man.dir || 'frames'; + attachAudio(man.audio); + } else { + attachAudio(null); + status('no manifest.json — assuming 12fps, no audio. Re-run extract.sh', 'warn'); + } const images = await loadFrameSequence(); if (!images.length) { status(`no frames in ${el('framedir').value}/ — run extract.sh first`, 'err'); @@ -371,14 +399,18 @@ async function runFrames() { el('scrub').max = dense.length - 1; state.frame = 0; rebuild(true); - status(missing.length - ? `${images.length} frames · no face on ${missing.length} (held previous)` - : `${images.length} frames detected`, missing.length ? 'warn' : 'ok'); + const dur = (dense.length / state.fps).toFixed(2); + status(`${images.length} frames · ${state.fps}fps · ${dur}s` + + (state.audio ? ' · audio loaded' : ' · no audio') + + (missing.length ? ` · no face on ${missing.length} (held previous)` : ''), + missing.length ? 'warn' : 'ok'); } catch (e) { status(e.message, 'err'); console.error(e); } } function runSynthetic() { state.images = []; + attachAudio(null); + state.fps = 12; state.dense = synthDense(72); el('scrub').max = 71; state.frame = 0; @@ -396,7 +428,13 @@ for (const id of ['verts', 'smoothWin', 'contourSmooth', 'apertureThresh', 'tol' el(id + 'v').textContent = el(id).value; } -el('scrub').addEventListener('input', (e) => { state.frame = +e.target.value; drawAll(); }); +function seekTo(f) { + state.frame = f; + if (state.audio) state.audio.currentTime = f / state.fps; + el('scrub').value = f; +} + +el('scrub').addEventListener('input', (e) => { seekTo(+e.target.value); drawAll(); }); el('btn-frames').onclick = runFrames; el('btn-synth').onclick = runSynthetic; el('btn-export').onclick = exportTake; @@ -409,34 +447,68 @@ el('btn-suggest').onclick = () => { status(`suggested ${kept.length} drawings at tolerance ${opts().tol.toFixed(3)} — now hand-correct`, 'ok'); }; el('btn-play').onclick = () => { + if (!state.dense) return; state.playing = !state.playing; el('btn-play').textContent = state.playing ? 'Stop' : 'Play'; - if (state.playing) tick(); + if (state.playing) { + if (state.audio) { + state.audio.playbackRate = +el('speed').value; + // Restart from the top if we are sitting at the end. + if (state.frame >= state.dense.length - 1) state.frame = 0; + state.audio.currentTime = state.frame / state.fps; + state.audio.play().catch((e) => status('audio blocked: ' + e.message, 'warn')); + } + last = 0; + tick(); + } else if (state.audio) { + state.audio.pause(); + } }; +el('speed').addEventListener('change', () => { + // playbackRate retimes the clock, and the picture follows it for free. + if (state.audio) state.audio.playbackRate = +el('speed').value; +}); + // Keyboard is the point: stepping and deleting 37 frames by mouse is miserable. window.addEventListener('keydown', (e) => { if (!state.dense || e.target.tagName === 'INPUT') return; const N = state.dense.length; - if (e.key === 'ArrowRight') { state.frame = Math.min(N - 1, state.frame + 1); drawAll(); } - else if (e.key === 'ArrowLeft') { state.frame = Math.max(0, state.frame - 1); drawAll(); } + if (e.key === 'ArrowRight') { seekTo(Math.min(N - 1, state.frame + 1)); drawAll(); } + else if (e.key === 'ArrowLeft') { seekTo(Math.max(0, state.frame - 1)); drawAll(); } else if (e.key === 'Backspace' || e.key === 'Delete' || e.key === 'x') { state.keep.delete(state.frame === 0 ? -1 : state.frame); drawAll(); } else if (e.key === 'k' || e.key === ' ') { if (state.frame !== 0) state.keep.add(state.frame); drawAll(); } else return; - el('scrub').value = state.frame; e.preventDefault(); }); let last = 0; function tick(ts = 0) { if (!state.playing) return; - if (ts - last > 1000 / +el('playfps').value) { + const N = state.dense.length; + let f; + if (state.audio) { + // Audio is the clock. Deriving the frame from currentTime rather than + // counting means a slow render loop drops frames instead of drifting out + // of sync, which is the behaviour you want when judging lip sync. + f = Math.floor(state.audio.currentTime * state.fps); + if (f >= N || state.audio.ended) { + state.audio.currentTime = 0; + state.audio.play().catch(() => {}); + f = 0; + } + } else { + const rate = state.fps * +el('speed').value; + if (ts - last < 1000 / rate) { requestAnimationFrame(tick); return; } last = ts; - state.frame = (state.frame + 1) % state.dense.length; - el('scrub').value = state.frame; + f = (state.frame + 1) % N; + } + if (f !== state.frame) { + state.frame = f; + el('scrub').value = f; drawPanes(); drawStrip(); } requestAnimationFrame(tick); diff --git a/js/take.js b/js/take.js index 1aa6217..5225626 100644 --- a/js/take.js +++ b/js/take.js @@ -7,7 +7,8 @@ const r = (v) => Math.round(v); export function writeTake(take) { const L = []; - L.push(`take name=${take.name} frames=${take.frames} width=${take.width} height=${take.height} exposure=${take.exposure}`); + L.push(`take name=${take.name} frames=${take.frames} width=${take.width} height=${take.height}` + + ` exposure=${take.exposure}${take.fps ? ` fps=${take.fps}` : ''}`); L.push(''); take.palette.forEach((p, i) => L.push(`pal ${i} ${p.name.padEnd(11)} ${i}`)); L.push(''); diff --git a/manifest.json b/manifest.json new file mode 100644 index 0000000..3efb3c1 --- /dev/null +++ b/manifest.json @@ -0,0 +1 @@ +{"fps":12,"frames":37,"dir":"frames","audio":"audio.wav","source":"IMG_8486.MOV"}