diff --git a/README.md b/README.md
index 107a7c2..cbd26fc 100644
--- a/README.md
+++ b/README.md
@@ -24,7 +24,7 @@ python3 -m http.server 8777 # from this directory
For real footage:
```sh
-./extract.sh /path/to/clip.mp4 24 # -> frames/0001.png …
+./extract.sh /path/to/clip.mov 12 # -> frames/*.png, audio.wav, manifest.json
```
then **Load frames**. MediaPipe's wasm is fetched from jsdelivr on first use;
@@ -34,6 +34,14 @@ Frames are pre-extracted rather than decoded in the page because browser video
seeking is approximate and `requestVideoFrameCallback` only delivers frames at
playback speed — neither gives a deterministic per-frame pass.
+`manifest.json` records the true extraction rate. The page reads it rather than
+assuming, because a guessed fps desynchronises audio from picture — and sync is
+the one thing this view exists to show.
+
+**Audio is the playback clock**: `frame = floor(audio.currentTime * fps)`. A slow
+render loop therefore drops frames instead of drifting, and ½x / ¼x work by
+setting `playbackRate` with the picture following for free.
+
## Shooting for it
Near-frontal, good light, consistent scale, head reasonably still. Hold a neutral
@@ -49,18 +57,33 @@ is hand-drawn head plates, which this tool does not yet do.
| Knob | What it does |
| --- | --- |
| vertices | Lip vertex budget. The reduction past what the footage supports *is* the style. |
-| min hold | Minimum frames between keys. |
-| change gate | Mean vertex movement required before a new key is accepted. |
-| vel smoothing | Window on the velocity signal used to find extremes. |
-| anchor smoothing | Window on the four similarity parameters. Smooths the *transform*, never the contour. |
-| exposure | Grid that key frames snap onto. |
+| contour avg ±f | Radius in frames. 0 off, 1 = ±1. Removes per-frame landmark jitter. |
+| anchor avg ±f | Radius on the four similarity parameters. Smooths the *transform*. |
| closed-mouth cut | Aperture below which the mouth interior is emitted as `hidden`. |
+| suggest tolerance | Max head movement before a new plate drawing is required. Affects **Suggest** only. |
-Keys go on **velocity minima**, not distance thresholds: a threshold fires at the
-frame it was crossed — partway through a transition — so poses land mushy and
-late. The timeline shows the velocity curve, candidate minima, accepted keys
-(`f`) and their pre-snap extremes (`src`); a large `f`/`src` gap means min-hold
-and exposure are fighting.
+## Two kinds of sparseness
+
+Sparseness has two unrelated causes, and conflating them was the original design
+error here. **Aesthetic** sparseness is set by the extraction rate — pick 12fps and
+you have already chosen your timing. **Labour** sparseness is a human drawing
+each one, and it binds only on the plate.
+
+So the mouth keeps **every** frame: it is traced, and therefore free. In limited
+animation lip sync is routinely the densest element, on 1s, while heads hold on
+2s and 3s.
+
+The frame strip is the editing surface for the other half: which frames need
+their own plate drawing. Everything starts kept; delete what you don't want.
+**Suggest** runs error-tolerance decimation over head pose as a starting point,
+then you hand-correct.
+
+`contour avg` is a deliberate, bounded exception to "never smooth the contour" in
+`../docs/roto-puppet.md`. That rule held while keys were sparse, because sampling
+at velocity minima rejected detector noise for free. With a key on every frame it
+does not, so a radius shorter than the shortest articulation worth keeping is
+justified — at 12fps, articulation spans 3–6 frames and detector noise is
+per-frame, so ±1 separates them and ±3 starts eating speech.
## Tests
@@ -79,8 +102,9 @@ eyeball.
## Not done yet
-Eyes and irises; hand-drawn head plates and per-plate mouth slots; real
+Eyes and irises; hand-drawn head plates and per-plate mouth slots (the strip
+decides *which frames need one*, but you cannot yet supply the drawing); real
performer→character calibration (currently identity, fitting the face oval to the
-canvas); the override layer; anything on the Animator Pro side. The placeholder
-plate is a frozen face-oval polygon — it exists so the mouth has a face to read
+canvas); the override layer; anything on the Animator Pro side. The plate is a
+face-oval polygon per kept frame — it exists so the mouth has a face to read
against, not to look good.
diff --git a/audio.wav b/audio.wav
new file mode 100644
index 0000000..9c39a31
Binary files /dev/null and b/audio.wav differ
diff --git a/extract.sh b/extract.sh
index 50a4bbb..cdbc9c6 100755
--- a/extract.sh
+++ b/extract.sh
@@ -1,17 +1,38 @@
#!/usr/bin/env bash
-# Extract a clip to a PNG sequence for the take builder.
+# Extract a clip to a PNG sequence + audio + manifest for the take builder.
#
# Frames are pre-extracted rather than decoded in the page on purpose: browser
# video seeking by currentTime is approximate and requestVideoFrameCallback only
# delivers frames at playback speed, so neither gives a deterministic per-frame
# pass. A PNG sequence is exact, instantly seekable, and reproducible.
+#
+# Audio comes out alongside because the page uses it as the PLAYBACK CLOCK -
+# frame = floor(audio.currentTime * fps) - so picture and sound cannot drift
+# apart no matter how long the shot is or how slow the render loop runs.
set -euo pipefail
src="${1:?usage: ./extract.sh CLIP [FPS] [OUTDIR]}"
-fps="${2:-24}"
+fps="${2:-12}"
out="${3:-frames}"
-rm -rf "$out"
-mkdir -p "$out"
+rm -rf "$out"; mkdir -p "$out"
ffmpeg -hide_banner -loglevel warning -i "$src" -vf "fps=$fps" "$out/%04d.png"
-echo "$(ls -1 "$out" | wc -l) frames at ${fps}fps -> $out/"
+count=$(ls -1 "$out" | wc -l)
+
+# Mono is enough for judging sync and halves the file. Absent audio is not fatal.
+if ffprobe -v error -select_streams a:0 -show_entries stream=codec_type \
+ -of csv=p=0 "$src" 2>/dev/null | grep -q audio; then
+ ffmpeg -hide_banner -loglevel warning -y -i "$src" -vn -ac 1 -ar 44100 audio.wav
+ audio='"audio.wav"'
+ echo "audio -> audio.wav"
+else
+ audio='null'
+ echo "no audio stream"
+fi
+
+# The page must know the true extraction rate: if it guessed, audio and picture
+# would drift. Source of truth lives here, next to the frames it describes.
+printf '{"fps":%s,"frames":%s,"dir":"%s","audio":%s,"source":"%s"}\n' \
+ "$fps" "$count" "$out" "$audio" "$(basename "$src")" > manifest.json
+
+echo "$count frames at ${fps}fps -> $out/ (manifest.json written)"
diff --git a/index.html b/index.html
index dfad8d9..b74d5ed 100644
--- a/index.html
+++ b/index.html
@@ -63,7 +63,8 @@
-
+
+
@@ -87,6 +88,7 @@
flat render — 320×200 indexed
+
audio drives the clock — dropped frames, never drift
diff --git a/js/app.js b/js/app.js
index 10c6b03..5fe666f 100644
--- a/js/app.js
+++ b/js/app.js
@@ -21,6 +21,7 @@ const state = {
outer: null, inner: null, plates: null, hidden: null,
keep: new Set(), // frames that get their own plate drawing
frame: 0, playing: false, faceBox: null,
+ fps: 12, audio: null, // fps comes from manifest.json, never guessed
};
const el = (id) => document.getElementById(id);
@@ -44,6 +45,24 @@ const loadImage = (src) => new Promise((r) => {
im.onload = () => r(im); im.onerror = () => r(null); im.src = src;
});
+// The extraction rate is read, not assumed. Guessing it would desynchronise
+// audio from picture, which is the one thing this view exists to show.
+async function loadManifest() {
+ try {
+ const r = await fetch('./manifest.json', { cache: 'no-store' });
+ if (!r.ok) return null;
+ return await r.json();
+ } catch { return null; }
+}
+
+function attachAudio(name) {
+ const a = el('audio');
+ if (!name) { a.removeAttribute('src'); a.hidden = true; state.audio = null; return; }
+ a.src = './' + name;
+ a.hidden = false;
+ state.audio = a;
+}
+
async function loadFrameSequence() {
const dir = el('framedir').value.replace(/\/$/, '');
const imgs = [];
@@ -199,7 +218,7 @@ function drawPanes() {
const f = state.frame, kept = keptSorted();
const pf = heldFrame(kept, f);
el('framelabel').textContent =
- `f ${f} / ${state.dense.length - 1} · plate from f${pf}` +
+ `f ${f} / ${state.dense.length - 1} · ${(f / state.fps).toFixed(2)}s · plate f${pf}` +
(state.keep.has(f) ? ' · KEPT' : ' · held');
const c1 = el('cv-source'), g1 = c1.getContext('2d');
@@ -279,7 +298,7 @@ function drawStrip() {
tag.textContent = f;
cell.append(cv, tag);
cell.onclick = (ev) => {
- state.frame = f;
+ seekTo(f);
if (ev.shiftKey) toggle(f);
drawAll();
};
@@ -318,7 +337,7 @@ function drawWorksheet() {
const cap = document.createElement('span');
cap.textContent = `f${f}` + (until > f ? ` → ${until}` : '') + ` (${until - f + 1}f)`;
cell.append(cv, cap);
- cell.onclick = () => { state.frame = f; drawAll(); };
+ cell.onclick = () => { seekTo(f); drawAll(); };
host.append(cell);
});
}
@@ -330,7 +349,7 @@ function exportTake() {
const N = state.dense.length;
const take = {
name: el('takename').value || 'line_01',
- frames: N, width: RW, height: RH, exposure: 1,
+ frames: N, width: RW, height: RH, exposure: 1, fps: state.fps,
palette: PALETTE,
slot: { x: RW / 2, y: RH / 2 },
parts: [
@@ -361,6 +380,15 @@ function exportTake() {
async function runFrames() {
try {
+ const man = await loadManifest();
+ if (man) {
+ state.fps = man.fps;
+ el('framedir').value = man.dir || 'frames';
+ attachAudio(man.audio);
+ } else {
+ attachAudio(null);
+ status('no manifest.json — assuming 12fps, no audio. Re-run extract.sh', 'warn');
+ }
const images = await loadFrameSequence();
if (!images.length) {
status(`no frames in ${el('framedir').value}/ — run extract.sh first`, 'err');
@@ -371,14 +399,18 @@ async function runFrames() {
el('scrub').max = dense.length - 1;
state.frame = 0;
rebuild(true);
- status(missing.length
- ? `${images.length} frames · no face on ${missing.length} (held previous)`
- : `${images.length} frames detected`, missing.length ? 'warn' : 'ok');
+ const dur = (dense.length / state.fps).toFixed(2);
+ status(`${images.length} frames · ${state.fps}fps · ${dur}s` +
+ (state.audio ? ' · audio loaded' : ' · no audio') +
+ (missing.length ? ` · no face on ${missing.length} (held previous)` : ''),
+ missing.length ? 'warn' : 'ok');
} catch (e) { status(e.message, 'err'); console.error(e); }
}
function runSynthetic() {
state.images = [];
+ attachAudio(null);
+ state.fps = 12;
state.dense = synthDense(72);
el('scrub').max = 71;
state.frame = 0;
@@ -396,7 +428,13 @@ for (const id of ['verts', 'smoothWin', 'contourSmooth', 'apertureThresh', 'tol'
el(id + 'v').textContent = el(id).value;
}
-el('scrub').addEventListener('input', (e) => { state.frame = +e.target.value; drawAll(); });
+function seekTo(f) {
+ state.frame = f;
+ if (state.audio) state.audio.currentTime = f / state.fps;
+ el('scrub').value = f;
+}
+
+el('scrub').addEventListener('input', (e) => { seekTo(+e.target.value); drawAll(); });
el('btn-frames').onclick = runFrames;
el('btn-synth').onclick = runSynthetic;
el('btn-export').onclick = exportTake;
@@ -409,34 +447,68 @@ el('btn-suggest').onclick = () => {
status(`suggested ${kept.length} drawings at tolerance ${opts().tol.toFixed(3)} — now hand-correct`, 'ok');
};
el('btn-play').onclick = () => {
+ if (!state.dense) return;
state.playing = !state.playing;
el('btn-play').textContent = state.playing ? 'Stop' : 'Play';
- if (state.playing) tick();
+ if (state.playing) {
+ if (state.audio) {
+ state.audio.playbackRate = +el('speed').value;
+ // Restart from the top if we are sitting at the end.
+ if (state.frame >= state.dense.length - 1) state.frame = 0;
+ state.audio.currentTime = state.frame / state.fps;
+ state.audio.play().catch((e) => status('audio blocked: ' + e.message, 'warn'));
+ }
+ last = 0;
+ tick();
+ } else if (state.audio) {
+ state.audio.pause();
+ }
};
+el('speed').addEventListener('change', () => {
+ // playbackRate retimes the clock, and the picture follows it for free.
+ if (state.audio) state.audio.playbackRate = +el('speed').value;
+});
+
// Keyboard is the point: stepping and deleting 37 frames by mouse is miserable.
window.addEventListener('keydown', (e) => {
if (!state.dense || e.target.tagName === 'INPUT') return;
const N = state.dense.length;
- if (e.key === 'ArrowRight') { state.frame = Math.min(N - 1, state.frame + 1); drawAll(); }
- else if (e.key === 'ArrowLeft') { state.frame = Math.max(0, state.frame - 1); drawAll(); }
+ if (e.key === 'ArrowRight') { seekTo(Math.min(N - 1, state.frame + 1)); drawAll(); }
+ else if (e.key === 'ArrowLeft') { seekTo(Math.max(0, state.frame - 1)); drawAll(); }
else if (e.key === 'Backspace' || e.key === 'Delete' || e.key === 'x') {
state.keep.delete(state.frame === 0 ? -1 : state.frame); drawAll();
} else if (e.key === 'k' || e.key === ' ') {
if (state.frame !== 0) state.keep.add(state.frame);
drawAll();
} else return;
- el('scrub').value = state.frame;
e.preventDefault();
});
let last = 0;
function tick(ts = 0) {
if (!state.playing) return;
- if (ts - last > 1000 / +el('playfps').value) {
+ const N = state.dense.length;
+ let f;
+ if (state.audio) {
+ // Audio is the clock. Deriving the frame from currentTime rather than
+ // counting means a slow render loop drops frames instead of drifting out
+ // of sync, which is the behaviour you want when judging lip sync.
+ f = Math.floor(state.audio.currentTime * state.fps);
+ if (f >= N || state.audio.ended) {
+ state.audio.currentTime = 0;
+ state.audio.play().catch(() => {});
+ f = 0;
+ }
+ } else {
+ const rate = state.fps * +el('speed').value;
+ if (ts - last < 1000 / rate) { requestAnimationFrame(tick); return; }
last = ts;
- state.frame = (state.frame + 1) % state.dense.length;
- el('scrub').value = state.frame;
+ f = (state.frame + 1) % N;
+ }
+ if (f !== state.frame) {
+ state.frame = f;
+ el('scrub').value = f;
drawPanes(); drawStrip();
}
requestAnimationFrame(tick);
diff --git a/js/take.js b/js/take.js
index 1aa6217..5225626 100644
--- a/js/take.js
+++ b/js/take.js
@@ -7,7 +7,8 @@ const r = (v) => Math.round(v);
export function writeTake(take) {
const L = [];
- L.push(`take name=${take.name} frames=${take.frames} width=${take.width} height=${take.height} exposure=${take.exposure}`);
+ L.push(`take name=${take.name} frames=${take.frames} width=${take.width} height=${take.height}` +
+ ` exposure=${take.exposure}${take.fps ? ` fps=${take.fps}` : ''}`);
L.push('');
take.palette.forEach((p, i) => L.push(`pal ${i} ${p.name.padEnd(11)} ${i}`));
L.push('');
diff --git a/manifest.json b/manifest.json
new file mode 100644
index 0000000..3efb3c1
--- /dev/null
+++ b/manifest.json
@@ -0,0 +1 @@
+{"fps":12,"frames":37,"dir":"frames","audio":"audio.wav","source":"IMG_8486.MOV"}