Keep footage audio with generated face symbols

This commit is contained in:
Your Name 2026-10-01 23:38:16 -04:00
parent 0cb9d0ebf6
commit f5a39aee39
3 changed files with 65 additions and 15 deletions

View file

@ -64,9 +64,16 @@
`frozen` is what `flow/freeze/clip` makes: a `:main` that places one symbol per
tracked face. `:main` becomes the named symbol — it is what holds the faces in
stage pixels, so it is the thing worth placing — and it gets the take's SOUND
as an audio node of its own, source frames `range` of footage `footage-id`, so
wherever the symbol is placed it is heard.
stage pixels, so it is the thing worth placing. Each generated face symbol gets
an automatic association with the take's SOUND, source frames `range` of
footage `footage-id`. Thus the take is heard through the faces it places, and
a face subsequently placed by itself still brings its sound. This is the same
shape a future manual symbol/sound association can write; dropping a face is
not a special operation.
A multi-face take consequently reaches the same recording through several
symbols. `nest/audio-tracks` collapses simultaneous copies carrying the same
`:media-link`; placing those faces at different times still schedules each one.
The tracking identities, and the analysis they were measured by, come along
only when `clip` has no analysis of its own and no face had to be renamed. A
@ -76,20 +83,27 @@
[clip frozen label footage-id range]
(let [{c :clip ids :ids} (symbols clip frozen [:main] {:main (symbol-id label)})
sid (ids :main)
faces (mapv ids (sort-by str (keys (:subjects frozen))))
;; Older/plain imported clips have no subject table. They are still a
;; take, so their wrapper remains the only honest owner of the sound.
sound-hosts (if (seq faces) faces [sid])
sound {:id :sound :name "sound" :kind :audio :parent nil
:z "z-sound" :source {:footage footage-id}
;; All automatic copies name one recording. The audio walk
;; uses this identity only to avoid mixing that recording once
;; per detected face when the complete take is played.
:media-link [:footage footage-id range]
;; Source frame `start` plays on the symbol's 0.
:span range
:time {:mode :map :at (- (first range)) :rate 1}}
tracked? (and (nil? (:analysis clip))
(every? #(= % (ids %)) (keys (:subjects frozen))))]
{:sid sid
:tracked? tracked?
:clip (cond-> (-> c
(assoc-in [:symbols sid :name] (str label))
(assoc-in [:symbols sid :nodes :sound]
{:id :sound :name "sound" :kind :audio :parent nil
:z "z-sound" :source {:footage footage-id}
;; Source frame `start` plays on the symbol's 0.
:span range
:time {:mode :map
:at (- (first range))
:rate 1}}))
:clip (cond-> (-> (reduce (fn [document face]
(assoc-in document [:symbols face :nodes :sound] sound))
c sound-hosts)
(assoc-in [:symbols sid :name] (str label)))
tracked? (-> (assoc :analysis (:analysis frozen))
(update :subjects merge (:subjects frozen))
(update :features merge (:features frozen))

View file

@ -134,7 +134,12 @@
(defn audio-tracks
"Flatten audible source intervals through cel and parent clocks.
A held visual source is silent. Every returned track carries a source offset,
an output interval, and automation mapped into the open symbol's time."
an output interval, and automation mapped into the open symbol's time.
Automatically associated face audio can be reached more than once in a
multi-face take. Equal `:media-link`s at the same source and output interval
are one recording, not a louder mix; at different placements they remain
separate scheduled clips."
[clip sid]
(letfn [(to-local [m f] (* (:rate m) (- f (:at m))))
(to-outer [m f] (+ (:at m) (/ f (:rate m))))
@ -200,7 +205,17 @@
periods))))
nil))))
(sort-by (comp str key) nodes))))]
(vec (walk sid (clip/grid-time clip sid) [0 (clip/output-frames clip sid)] [] #{}))))
(let [tracks (walk sid (clip/grid-time clip sid)
[0 (clip/output-frames clip sid)] [] #{})]
(second
(reduce (fn [[seen out] track]
(let [link (:media-link track)
k (when link [link (:source track) (:span track)
(select-keys (:time track) [:at :rate :offset])])]
(if (and k (contains? seen k))
[seen out]
[(cond-> seen k (conj k)) (conj out track)])))
[#{} []] tracks)))))
(defn- retime
"Node `n` with its own time map replaced by `m`, and nothing else touched: its

View file

@ -89,6 +89,27 @@
[track] (nest/audio-tracks slow :main)]
(is (= 0.5 (* (get-in track [:time :rate]) (/ (:fps slow) (:fps track)))))))))
(deftest generated-faces-own-their-footage-sound
(let [face (fn [id] {:id id :fps 30 :frames 20 :nodes {}})
frozen {:fps 30 :subjects {:face-1 {} :face-2 {}}
:symbols {:main {:id :main :fps 30 :frames 20
:nodes {:face-1 {:id :face-1 :kind :instance :z "a1"
:source {:symbol :face-1}}
:face-2 {:id :face-2 :kind :instance :z "a2"
:source {:symbol :face-2}}}}
:face-1 (face :face-1)
:face-2 (face :face-2)}}
{doc :clip take :sid} (bring/take (clip/blank) frozen "take" "video" [3 13])
alone (clip/place-symbol doc nil :main :face-1 4 :placed-face nil)]
(is (nil? (get-in doc [:symbols take :nodes :sound]))
"the wrapper does not own a sound the face would lose")
(is (= {:footage "video"}
(get-in doc [:symbols :face-1 :nodes :sound :source])))
(is (= 1 (count (nest/audio-tracks doc take)))
"the same take recording is not mixed once per detected face")
(is (= 1 (count (nest/audio-tracks alone :main)))
"placing a generated face by itself brings its associated sound")))
;; ---------------------------------------------------------------------------
;; the preserve-snap
;;