From c70acb8f649da76ff9618408ea2109ebde81acdf Mon Sep 17 00:00:00 2001 From: EtienneLescot Date: Fri, 25 Sep 2026 23:02:47 +0200 Subject: [PATCH 1/2] feat(audio): a soft limiter and -16 LUFS loudness at export The export hard-clipped at full scale after a +12 dB trim, and nothing levelled the voice. Replace the clamp with a look-ahead peak limiter (-1.5 dBFS ceiling, 5 ms ramp, 80 ms release) and bring every voice file (the recording's audio and each voiceover take) to -16 LUFS integrated, measured per BS.1770-4 with EBU R128 gating over the whole file, with at most +12 dB of boost. The gain is a property of the file, not of the edit, so the preview asks the addon for the same number (loudnessGainDb) and plays the voice at the exported level. The limiter runs at export only. --- crates/compositor-view-napi/src/lib.rs | 29 + crates/compositor/src/audio.rs | 561 ++++++++++++++++-- crates/compositor/src/audio_jobs.rs | 14 +- crates/compositor/src/scene.rs | 14 + electron/electron-env.d.ts | 5 + electron/ipc/handlers.ts | 28 + .../services/compositorViewService.ts | 11 + electron/native/compositor-view/addon.d.ts | 8 + electron/preload.ts | 4 + .../ai-edition/VirtualPreview.audio.test.ts | 24 +- src/components/ai-edition/VirtualPreview.tsx | 96 ++- src/i18n/locales/ar/settings.json | 2 +- src/i18n/locales/cs/settings.json | 2 +- src/i18n/locales/de/settings.json | 2 +- src/i18n/locales/en/settings.json | 2 +- src/i18n/locales/es/settings.json | 2 +- src/i18n/locales/fr/settings.json | 2 +- src/i18n/locales/it/settings.json | 2 +- src/i18n/locales/ja-JP/settings.json | 2 +- src/i18n/locales/ko-KR/settings.json | 2 +- src/i18n/locales/pt-BR/settings.json | 2 +- src/i18n/locales/ru/settings.json | 2 +- src/i18n/locales/tr/settings.json | 2 +- src/i18n/locales/vi/settings.json | 2 +- src/i18n/locales/zh-CN/settings.json | 2 +- src/i18n/locales/zh-TW/settings.json | 2 +- src/native/sceneDescription.test.ts | 18 + src/native/sceneDescription.ts | 21 +- 28 files changed, 770 insertions(+), 93 deletions(-) diff --git a/crates/compositor-view-napi/src/lib.rs b/crates/compositor-view-napi/src/lib.rs index b2a80ac61..7a9f87638 100644 --- a/crates/compositor-view-napi/src/lib.rs +++ b/crates/compositor-view-napi/src/lib.rs @@ -824,3 +824,32 @@ pub fn remux_seekable(input_path: String, output_path: String) -> AsyncTask Result { + Ok(f64::from(openscreen_compositor::audio::loudness_gain_db(&self.path))) + } + + fn resolve(&mut self, _env: Env, out: Self::Output) -> Result { + Ok(out) + } +} + +/// Le gain de normalisation de loudness (dB) que l'export applique à ce fichier voix — le +/// même nombre, par la même fonction, pour que la preview de l'éditeur joue la voix au +/// niveau où l'export l'écrira. Voir `openscreen_compositor::audio::loudness_gain_db`. +/// +/// `AsyncTask` : la mesure décode tout l'audio du fichier, ce qui se compte en secondes sur +/// un long enregistrement. Le résultat est mis en cache dans le processus, donc l'export qui +/// suit ne le refait pas. +#[napi] +pub fn loudness_gain_db(path: String) -> AsyncTask { + AsyncTask::new(LoudnessGainTask { path }) +} diff --git a/crates/compositor/src/audio.rs b/crates/compositor/src/audio.rs index ee90c1374..35851fd71 100644 --- a/crates/compositor/src/audio.rs +++ b/crates/compositor/src/audio.rs @@ -5,11 +5,14 @@ use crate::ffi::*; use crate::regions::SpeedSegment; -use crate::scene::{SceneAudio, SceneAudioTrack}; +use crate::scene::{SceneAudio, SceneAudioTrack, SceneAudioTrackKind}; use anyhow::{bail, Result}; +use std::collections::{HashMap, VecDeque}; use std::f32::consts::PI; use std::ffi::CString; use std::ptr; +use std::sync::{Arc, Mutex, OnceLock}; +use std::time::SystemTime; pub const AUDIO_OUTPUT_SAMPLE_RATE: i32 = 48_000; pub const AUDIO_OUTPUT_CHANNELS: usize = 2; @@ -28,25 +31,25 @@ const PASSTHROUGH_EPSILON: f64 = 1e-3; pub type PlanarPcm = Vec>; -/// Apply the editor's output trim to the assembled timeline. +/// Apply the editor's output trim to the assembled timeline, then keep it under the ceiling. /// -/// One stage, and keeping it that way takes some resisting. This runs on the assembled -/// timeline — trimmed, speed-adjusted, concatenated — while the editor preview plays the -/// untouched SOURCE file, seeked. A linear gain is the only operation that means the same -/// thing on both, which is what lets the editor claim that what you hear is what you export. +/// This runs on the assembled timeline — trimmed, speed-adjusted, concatenated — while the +/// editor preview plays the untouched SOURCE file, seeked. A linear gain is the one operation +/// that means the same thing on both, which is why the level work that has to match the +/// preview happens elsewhere, as static gains: the voice's loudness normalisation is a gain +/// per FILE (`loudness_gain_db`), applied to each clip's audio before assembly and asked for +/// by the preview too. /// -/// Three things that look like they belong here and do not: -/// - a filter or a compressor, which carries state across cuts here and not in the preview; -/// - a loudness normaliser, whose makeup is a single scalar measured over the whole assembled -/// programme — the preview never holds that programme, and the value moves with every trim; -/// - a sync offset, which shipped here once. It is expressed in TIMELINE seconds at this -/// point in the pipeline, but the preview would apply it in SOURCE seconds, so a 2x speed -/// region halved it; and because the shift is uniform over the assembled programme, near a -/// cut the export pulls audio across the junction while the preview only has the active -/// asset loaded. +/// The limiter is the one stage here that the preview does not run. It is stateful, so it +/// would not see the same signal on both sides, and it only acts on peaks above +/// `LIMITER_CEILING`: below that the output is the trimmed input, sample for sample. What +/// it replaced was a hard clamp at full scale, which squared off every peak the trim or the +/// normalisation pushed over. /// -/// Any of them means either rendering the export's audio assembly preview-side, or accepting -/// and documenting a divergence — not a quiet extra stage in this function. +/// A sync offset shipped here once and was removed: it is expressed in TIMELINE seconds at +/// this point, but the preview would apply it in SOURCE seconds, so a 2x speed region halved +/// it, and near a cut the export pulled audio across the junction while the preview only has +/// the active asset loaded. /// /// The bound mirrors `AUDIO_GAIN_DB_LIMIT` in editorSettings.ts. The result stays the same /// length so video and following clips cannot drift. @@ -61,11 +64,258 @@ pub fn finish_audio(mut pcm: PlanarPcm, settings: SceneAudio) -> PlanarPcm { let trim = 10.0f32.powf(settings.gain_db.clamp(-12.0, 12.0) / 20.0); for sample in pcm.iter_mut().flatten() { - *sample = (*sample * trim).clamp(-1.0, 1.0); + *sample *= trim; } + limit_peaks(&mut pcm); pcm } +/// Highest sample the export writes: −1.5 dBFS. The limiter reads samples, not the waveform +/// between them, and the AAC encoder overshoots a little on top: measured, a −1.0 dBFS ceiling +/// came out of the encoder at −0.9 dBTP. The extra half decibel keeps the file under the usual +/// −1 dBTP delivery limit. +pub const LIMITER_CEILING: f32 = 0.841_395_1; +/// How early the gain starts to come down before a peak, as a straight ramp: 5 ms. +const LIMITER_LOOKAHEAD: usize = 240; +/// Time constant of the recovery once a peak has passed. +const LIMITER_RELEASE_SEC: f32 = 0.08; + +/// Gain that brings sample `index` of every channel under the ceiling, 1 when it already is. +fn gain_to_ceiling(pcm: &[Vec], index: usize) -> f32 { + let peak = pcm.iter().fold(0.0f32, |peak, channel| peak.max(channel[index].abs())); + if peak > LIMITER_CEILING { + LIMITER_CEILING / peak + } else { + 1.0 + } +} + +/// Look-ahead peak limiter, linked across channels so the stereo image does not move. +/// +/// The gain each sample needs (`gain_to_ceiling`) is held at its minimum over the next +/// `LIMITER_LOOKAHEAD` samples, then averaged over the last `LIMITER_LOOKAHEAD`. Every value +/// in that average comes from a window that contains the current sample, so the average never +/// exceeds what the sample needs: the ceiling holds exactly, and the gain still reaches each +/// peak as a straight 5 ms ramp instead of a step. The recovery after it is exponential, and +/// a signal that never crosses the ceiling passes through untouched. +fn limit_peaks(pcm: &mut PlanarPcm) { + let len = pcm.first().map(Vec::len).unwrap_or(0); + let window = LIMITER_LOOKAHEAD as isize; + let release = 1.0 - (-1.0 / (LIMITER_RELEASE_SEC * AUDIO_OUTPUT_SAMPLE_RATE as f32)).exp(); + // Rising minimum of the needed gain over the look-ahead window, as (index, gain). + let mut held: VecDeque<(usize, f32)> = VecDeque::new(); + let mut ring = vec![1.0f32; LIMITER_LOOKAHEAD]; + let mut ring_sum = LIMITER_LOOKAHEAD as f64; + let mut next = 0usize; + let mut gain = 1.0f32; + // Starting a window early fills the average with minima that already cover sample 0. + for index in (1 - window)..len as isize { + let horizon = ((index + window) as usize).min(len); + while next < horizon { + let needed = gain_to_ceiling(pcm, next); + while held.back().is_some_and(|&(_, g)| g >= needed) { + held.pop_back(); + } + held.push_back((next, needed)); + next += 1; + } + let first = index.max(0) as usize; + while held.front().is_some_and(|&(i, _)| i < first) { + held.pop_front(); + } + let minimum = held.front().map_or(1.0, |&(_, g)| g); + let slot = (index + window) as usize % LIMITER_LOOKAHEAD; + ring_sum += f64::from(minimum - ring[slot]); + ring[slot] = minimum; + if index < 0 { + continue; + } + let sample = index as usize; + // The last `min` only absorbs the running sum's rounding: the average already obeys it. + gain = ((ring_sum / LIMITER_LOOKAHEAD as f64) as f32) + .min(gain + (1.0 - gain) * release) + .min(gain_to_ceiling(pcm, sample)); + if gain < 1.0 { + for channel in pcm.iter_mut() { + channel[sample] *= gain; + } + } + } +} + +/// Integrated loudness every voice source is brought to: −16 LUFS, the usual delivery target +/// for spoken content online (it is what Apple Podcasts asks for). +pub const LOUDNESS_TARGET_LUFS: f64 = -16.0; +/// Most the normalisation will raise a file. Without a denoiser, a recording that is mostly +/// room noise (a microphone left open over a silent demo) would otherwise come out as loud +/// hiss; a voice quieter than −28 LUFS lands below the target instead. +pub const LOUDNESS_MAX_BOOST_DB: f64 = 12.0; + +/// ITU-R BS.1770-4 K-weighting at 48 kHz — the pre-filter's high shelf, then the RLB +/// high-pass — as `[b0, b1, b2, a1, a2]`. +const K_WEIGHTING: [[f64; 5]; 2] = [ + [ + 1.535_124_859_586_97, + -2.691_696_189_406_38, + 1.198_392_810_852_85, + -1.690_659_293_182_41, + 0.732_480_774_215_85, + ], + [1.0, -2.0, 1.0, -1.990_047_454_833_98, 0.990_072_250_366_21], +]; +/// 100 ms: four of these make a 400 ms gating block, and one is the 75 % overlap hop. +const LOUDNESS_STEP: usize = AUDIO_OUTPUT_SAMPLE_RATE as usize / 10; + +/// Integrated loudness per ITU-R BS.1770-4, gated as EBU R128 specifies (−70 LUFS absolute, +/// −10 LU relative). Fed in order, one window at a time, so a long recording never sits in +/// memory whole: only one number per 100 ms is kept. +struct LoudnessMeter { + /// Transposed direct form II state, per channel and per filter stage. + state: [[[f64; 2]; 2]; AUDIO_OUTPUT_CHANNELS], + /// Channel-summed, K-weighted mean square of each complete 100 ms step. + steps: Vec, + partial: f64, + partial_len: usize, +} + +impl LoudnessMeter { + fn new() -> Self { + Self { + state: [[[0.0; 2]; 2]; AUDIO_OUTPUT_CHANNELS], + steps: Vec::new(), + partial: 0.0, + partial_len: 0, + } + } + + fn push(&mut self, pcm: &[Vec]) { + let len = pcm.first().map(Vec::len).unwrap_or(0); + for index in 0..len { + let mut energy = 0.0f64; + for (channel, state) in self.state.iter_mut().enumerate() { + let sample = pcm.get(channel).and_then(|plane| plane.get(index)); + let mut x = f64::from(sample.copied().unwrap_or(0.0)); + for (c, z) in K_WEIGHTING.iter().zip(state.iter_mut()) { + let y = c[0] * x + z[0]; + z[0] = c[1] * x - c[3] * y + z[1]; + z[1] = c[2] * x - c[4] * y; + x = y; + } + energy += x * x; + } + self.partial += energy; + self.partial_len += 1; + if self.partial_len == LOUDNESS_STEP { + self.steps.push(self.partial / LOUDNESS_STEP as f64); + self.partial = 0.0; + self.partial_len = 0; + } + } + } + + /// LUFS, or `None` when no 400 ms block clears the absolute gate: silence has no loudness + /// to correct. + fn integrated(&self) -> Option { + let lufs = |power: f64| -0.691 + 10.0 * power.log10(); + let mean = |powers: &[f64]| powers.iter().sum::() / powers.len() as f64; + let audible: Vec = self + .steps + .windows(4) + .map(|block| block.iter().sum::() / 4.0) + .filter(|&power| lufs(power) > -70.0) + .collect(); + if audible.is_empty() { + return None; + } + let relative_gate = lufs(mean(&audible)) - 10.0; + let gated: Vec = audible + .into_iter() + .filter(|&power| lufs(power) > relative_gate) + .collect(); + Some(lufs(mean(&gated))) + } +} + +/// One minute at a time: 23 MB of PCM, whatever the recording's length. +const LOUDNESS_WINDOW_SEC: f64 = 60.0; +/// A day of windows. Only a file whose timestamps never end would get here. +const LOUDNESS_MAX_WINDOWS: usize = 24 * 60; + +/// Integrated loudness of every audio track in `path`, mixed exactly as the export mixes +/// them (`decode_clip_audio`), over the whole file. `Ok(None)`: no audio, or only silence. +pub fn measure_file_loudness(path: &str) -> Result> { + let mut meter = LoudnessMeter::new(); + for window in 0..LOUDNESS_MAX_WINDOWS { + let start = window as f64 * LOUDNESS_WINDOW_SEC; + let Some((pcm, more)) = decode_audio_window(path, start, start + LOUDNESS_WINDOW_SEC)? + else { + return Ok(None); + }; + // The last window is zero-padded past the end of the file; silence never clears + // the absolute gate, so the padding does not move the result. + meter.push(&pcm); + if !more { + break; + } + } + Ok(meter.integrated()) +} + +/// The gain, in dB, that takes audio measured at `loudness` to the target: negative for a +/// hot recording, positive up to `LOUDNESS_MAX_BOOST_DB` for a quiet one, 0 for silence. +fn loudness_gain_for(loudness: Option) -> f32 { + loudness.map_or(0.0, |lufs| { + (LOUDNESS_TARGET_LUFS - lufs).min(LOUDNESS_MAX_BOOST_DB) as f32 + }) +} + +/// Loudness-normalisation gain for one voice source, in dB. +/// +/// Measured over the WHOLE file, not over the part the timeline keeps. That makes it a +/// property of the recording rather than of the edit: a trim does not move the level of what +/// is left, a timeline cut from several takes brings each take to the target on its own, and +/// the editor preview — which plays the source file, not the assembled programme — can apply +/// the very same number (it asks for it through the addon's `loudnessGainDb`). +/// +/// Cached per process by path, size and modification time, so the preview's request and +/// every export job of every clip cut from the same file share one measurement. A file that +/// cannot be read measures as 0 dB, the same degradation a clip without audio gets. +pub fn loudness_gain_db(path: &str) -> f32 { + type Key = (String, u64, Option); + static CACHE: OnceLock>>>> = OnceLock::new(); + let Ok(metadata) = std::fs::metadata(path) else { + return 0.0; + }; + let key = (path.to_owned(), metadata.len(), metadata.modified().ok()); + let cell = CACHE + .get_or_init(Default::default) + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .entry(key) + .or_default() + .clone(); + // Measured outside the map's lock: another file never waits on this one, and a second + // request for the same file waits for the first measurement instead of repeating it. + *cell.get_or_init(|| match measure_file_loudness(path) { + Ok(loudness) => loudness_gain_for(loudness), + Err(error) => { + eprintln!("[audio] loudness of {path} not measured ({error:#}); left as recorded"); + 0.0 + } + }) +} + +/// Multiply every sample by the gain `gain_db` stands for. A no-op at 0 dB. +pub fn apply_gain_db(pcm: &mut PlanarPcm, gain_db: f32) { + if gain_db == 0.0 { + return; + } + let gain = 10.0f32.powf(gain_db / 20.0); + for sample in pcm.iter_mut().flatten() { + *sample *= gain; + } +} + extern "C" { fn sn_fmt_stream(s: *mut AVFormatContext, i: i32) -> *mut AVStream; // bindgen rend `AVFormatContext` opaque (atteinte seulement par pointeur), d'où l'accesseur @@ -233,6 +483,16 @@ impl Drop for AudioTrackDecoder { /// défaut dans l'exporteur navigateur ; l'app n'emprunte plus ce chemin, d'où la même correction /// ici, dans le chemin natif. pub fn decode_clip_audio(path: &str, source_start_sec: f64, source_end_sec: f64) -> Result> { + Ok(decode_audio_window(path, source_start_sec, source_end_sec)?.map(|(pcm, _)| pcm)) +} + +/// `decode_clip_audio`, plus whether the file still has audio past `source_end_sec`: what a +/// caller walking a whole file one window at a time needs to know where to stop. +fn decode_audio_window( + path: &str, + source_start_sec: f64, + source_end_sec: f64, +) -> Result> { unsafe { decode_clip_audio_inner(path, source_start_sec, source_end_sec) } } @@ -240,7 +500,7 @@ unsafe fn decode_clip_audio_inner( path: &str, source_start_sec: f64, source_end_sec: f64, -) -> Result> { +) -> Result> { let mut fmt: *mut AVFormatContext = ptr::null_mut(); let cpath = CString::new(path)?; averr( @@ -402,14 +662,15 @@ unsafe fn decode_clip_audio_inner( let target_samples = (((source_end_sec - source_start_sec).max(0.0) * AUDIO_OUTPUT_SAMPLE_RATE as f64) .round()) as usize; + // A track stops either on a frame at or past the window's end, or at the end of the file. + let more = tracks.iter().any(|track| track.reached_end); let aligned: Vec<(f64, &PlanarPcm)> = tracks .iter() .map(|track| (track.origin_sec.unwrap_or(source_start_sec), &track.decoded)) .collect(); - Ok(Some(mix_aligned_tracks( - &aligned, - source_start_sec, - target_samples, + Ok(Some(( + mix_aligned_tracks(&aligned, source_start_sec, target_samples), + more, ))) } @@ -444,16 +705,10 @@ fn mix_aligned_tracks( } } } - // Sommer plusieurs pistes peut dépasser la pleine échelle : on écrête. Sur une source - // mono-piste, sommer dans un buffer nul est l'identité et on n'écrête pas — le comportement - // d'avant ce mixage est conservé tel quel. - if tracks.len() > 1 { - for plane in mixed.iter_mut() { - for sample in plane.iter_mut() { - *sample = sample.clamp(-1.0, 1.0); - } - } - } + // Sommer plusieurs pistes peut dépasser la pleine échelle, et on ne l'écrête PLUS ici : le + // flottant garde ces crêtes intactes jusqu'au gain de loudness (qui baisse souvent un mix + // chaud) puis au limiteur de `finish_audio`, seul endroit où le signal touche un plafond. + // Écrêter ici cassait des crêtes que la suite aurait ramenées sous la pleine échelle. mixed } @@ -1438,7 +1693,13 @@ pub fn mix_external_tracks(mut programme: PlanarPcm, tracks: &[SceneAudioTrack]) }; // The app's own range is -60..+12 dB (the inspector slider); clamping at // -12 here floored every quiet bed at a tenth of the attenuation asked for. - let gain = 10.0f32.powf(track.gain_db.clamp(-60.0, 12.0) / 20.0); + // A voiceover is voice, levelled like the recording (see `loudness_gain_db`); its + // own gain then trims from there, exactly as the preview applies it. + let normalisation = match track.kind { + SceneAudioTrackKind::Voiceover => loudness_gain_db(&track.path), + SceneAudioTrackKind::Music => 0.0, + }; + let gain = 10.0f32.powf((track.gain_db.clamp(-60.0, 12.0) + normalisation) / 20.0); overlay_track_pcm( &mut programme, &decoded, @@ -1907,8 +2168,7 @@ mod tests { gain_db: 0.0, trim_start_sec: 2.0, trim_end_sec: Some(1.0), // end <= start: empty window, never decoded - fade_in_sec: 0.0, - fade_out_sec: 0.0, + ..Default::default() }]; // The empty window is skipped before any decode, so the programme is // untouched even though the path does not exist. @@ -1928,8 +2188,7 @@ mod tests { gain_db: 0.0, trim_start_sec: 0.0, trim_end_sec: Some(3600.0), - fade_in_sec: 0.0, - fade_out_sec: 0.0, + ..Default::default() }]; let out = mix_external_tracks(programme, &tracks); assert_eq!(out[0], vec![0.4, 0.4]); @@ -1938,7 +2197,7 @@ mod tests { #[test] fn single_track_is_not_clamped() { // Promesse de non-régression : une source mono-piste ressort telle quelle, y compris - // hors pleine échelle. Seul le mixage multipiste écrête. + // hors pleine échelle. let track = planar(&[1.5, -1.5]); let mixed = mix_aligned_tracks(&[(0.0, &track)], 0.0, 2); assert_eq!(mixed[0], vec![1.5, -1.5]); @@ -1965,11 +2224,15 @@ mod tests { } #[test] - fn multi_track_sum_is_clamped_to_full_scale() { + fn a_multi_track_sum_past_full_scale_is_not_clipped_at_decode() { + // The loudness gain comes next and often brings a hot mix down; squaring the peaks + // off here would bake in a clip the rest of the chain could have avoided. The + // limiter in `finish_audio` is the one place the signal meets a ceiling. let a = planar(&[0.8, -0.8]); let b = planar(&[0.8, -0.8]); let mixed = mix_aligned_tracks(&[(0.0, &a), (0.0, &b)], 0.0, 2); - assert_eq!(mixed[0], vec![1.0, -1.0]); + assert!((mixed[0][0] - 1.6).abs() < 1e-6); + assert!((mixed[0][1] + 1.6).abs() < 1e-6); } #[test] @@ -1999,12 +2262,13 @@ mod tests { /// else stands between what the editor plays and what this writes. #[test] fn output_trim_is_the_same_scalar_the_preview_applies() { + // 0.2 × +12 dB is 0.796: under the limiter's ceiling, so nothing but the trim acts. for gain_db in [-12.0f32, -6.0206, 0.0, 6.0206, 12.0] { let result = finish_audio( - planar(&[0.25, -0.25]), + planar(&[0.2, -0.2]), SceneAudio { gain_db }, ); - let expected = (0.25 * 10.0f32.powf(gain_db / 20.0)).clamp(-1.0, 1.0); + let expected = 0.2 * 10.0f32.powf(gain_db / 20.0); assert!( (result[0][0] - expected).abs() < 1e-6, "gain {gain_db} dB: got {}, want {expected}", @@ -2112,14 +2376,209 @@ mod tests { assert!(programme[0][0] < 10.0f32.powf(-12.0 / 20.0)); } + /// `secs` of a 1 kHz tone per segment, at the segment's level in dBFS (sine peak, as + /// EBU Tech 3341 states its test signals), on both channels, phase continuous. + fn tone_segments(segments: &[(f32, f64)]) -> PlanarPcm { + let rate = AUDIO_OUTPUT_SAMPLE_RATE as f64; + let mut plane = Vec::new(); + for &(dbfs, secs) in segments { + let amplitude = 10.0f32.powf(dbfs / 20.0); + for _ in 0..(secs * rate).round() as usize { + let t = plane.len() as f64 / rate; + plane.push(amplitude * (2.0 * std::f64::consts::PI * 1000.0 * t).sin() as f32); + } + } + vec![plane.clone(), plane] + } + + fn loudness_of(pcm: &PlanarPcm) -> Option { + let mut meter = LoudnessMeter::new(); + meter.push(pcm); + meter.integrated() + } + + fn assert_lufs(measured: Option, expected: f64) { + let measured = measured.expect("a signal above the absolute gate has a loudness"); + assert!( + (measured - expected).abs() <= 0.1, + "measured {measured:.3} LUFS, EBU Tech 3341 wants {expected} ±0.1" + ); + } + + // The EBU Tech 3341 minimum-requirement cases that exercise what this meter computes: + // calibration, the relative gate and the absolute gate. + #[test] + fn a_stereo_1khz_tone_reads_its_own_level_in_lufs() { + assert_lufs(loudness_of(&tone_segments(&[(-23.0, 20.0)])), -23.0); + assert_lufs(loudness_of(&tone_segments(&[(-33.0, 20.0)])), -33.0); + } + + #[test] + fn the_relative_gate_ignores_the_quiet_passages() { + assert_lufs( + loudness_of(&tone_segments(&[(-36.0, 10.0), (-23.0, 60.0), (-36.0, 10.0)])), + -23.0, + ); + assert_lufs( + loudness_of(&tone_segments(&[(-26.0, 20.0), (-20.0, 20.1), (-26.0, 20.0)])), + -23.0, + ); + } + + #[test] + fn the_absolute_gate_ignores_near_silence() { + assert_lufs( + loudness_of(&tone_segments(&[ + (-72.0, 10.0), + (-36.0, 10.0), + (-23.0, 60.0), + (-36.0, 10.0), + (-72.0, 10.0), + ])), + -23.0, + ); + // Nothing but silence and a tone under the gate: no loudness to correct. + assert_eq!(loudness_of(&tone_segments(&[(-80.0, 5.0)])), None); + assert_eq!(loudness_of(&vec![vec![0.0; 48_000]; 2]), None); + } + + #[test] + fn feeding_the_meter_in_windows_changes_nothing() { + // The file walk feeds one minute at a time; the filter state and the 100 ms steps + // must carry across the seams, uneven ones included. + let pcm = tone_segments(&[(-30.0, 3.0), (-18.0, 4.0), (-40.0, 3.0)]); + let mut meter = LoudnessMeter::new(); + let mut start = 0; + for length in [1usize, 4799, 4801, 100_000, 7] { + let end = start + length; + meter.push(&pcm.iter().map(|plane| plane[start..end].to_vec()).collect::>()); + start = end; + } + meter.push(&pcm.iter().map(|plane| plane[start..].to_vec()).collect::>()); + let whole = loudness_of(&pcm).unwrap(); + assert!((meter.integrated().unwrap() - whole).abs() < 1e-9); + } + + #[test] + fn the_gain_brings_a_file_to_the_target_and_caps_the_boost() { + assert_eq!(loudness_gain_for(Some(-23.0)), 7.0); + assert_eq!(loudness_gain_for(Some(-10.0)), -6.0); + // A room-noise recording at −50 LUFS is raised 12 dB, not 34. + assert_eq!(loudness_gain_for(Some(-50.0)), LOUDNESS_MAX_BOOST_DB as f32); + assert_eq!(loudness_gain_for(None), 0.0); + } + + /// A 16-bit stereo 48 kHz WAV of `pcm`: ffmpeg reads it, so the test walks the same + /// decode as a real recording without a media fixture. + fn write_wav(path: &std::path::Path, pcm: &PlanarPcm) { + let frames = pcm[0].len(); + let data_len = (frames * 4) as u32; + let mut bytes = Vec::with_capacity(44 + frames * 4); + bytes.extend_from_slice(b"RIFF"); + bytes.extend_from_slice(&(36 + data_len).to_le_bytes()); + bytes.extend_from_slice(b"WAVEfmt "); + bytes.extend_from_slice(&16u32.to_le_bytes()); + bytes.extend_from_slice(&1u16.to_le_bytes()); // PCM + bytes.extend_from_slice(&2u16.to_le_bytes()); + bytes.extend_from_slice(&48_000u32.to_le_bytes()); + bytes.extend_from_slice(&(48_000u32 * 4).to_le_bytes()); + bytes.extend_from_slice(&4u16.to_le_bytes()); + bytes.extend_from_slice(&16u16.to_le_bytes()); + bytes.extend_from_slice(b"data"); + bytes.extend_from_slice(&data_len.to_le_bytes()); + for index in 0..frames { + for plane in pcm { + let value = (plane[index].clamp(-1.0, 1.0) * 32767.0).round() as i16; + bytes.extend_from_slice(&value.to_le_bytes()); + } + } + std::fs::write(path, bytes).unwrap(); + } + + #[test] + fn a_file_is_measured_whole_across_its_windows() { + // 125 s crosses two window seams and ends inside a third, zero-padded window: the + // walk must neither stop early nor count the padding. + let path = std::env::temp_dir().join(format!("openscreen-loudness-{}.wav", std::process::id())); + write_wav(&path, &tone_segments(&[(-23.0, 5.0), (-29.0, 115.0), (-23.0, 5.0)])); + let file = path.to_str().unwrap(); + let (_, more) = decode_audio_window(file, 60.0, 120.0).unwrap().unwrap(); + assert!(more, "audio continues past 120 s"); + let (_, more) = decode_audio_window(file, 120.0, 180.0).unwrap().unwrap(); + assert!(!more, "the file ends at 125 s"); + let measured = measure_file_loudness(file).unwrap(); + let expected = loudness_of(&tone_segments(&[(-23.0, 5.0), (-29.0, 115.0), (-23.0, 5.0)])); + let _ = std::fs::remove_file(&path); + assert!( + (measured.unwrap() - expected.unwrap()).abs() < 0.05, + "file walk {measured:?} vs whole signal {expected:?}" + ); + } + + #[test] + fn an_unreadable_file_is_left_as_recorded() { + assert_eq!(loudness_gain_db("/no/such/recording.mp4"), 0.0); + } + + #[test] + fn a_signal_under_the_ceiling_leaves_finishing_untouched() { + let pcm = sine(1.0); + let result = finish_audio(pcm.clone(), SceneAudio { gain_db: 0.0 }); + assert_eq!(result, pcm, "the limiter must be transparent below its ceiling"); + } + + #[test] + fn the_limiter_holds_its_ceiling_and_the_length() { + // A tone at +6 dBFS with a +12 dB trim on top, and a burst 20 dB hotter still: the + // old clamp squared every one of these off at full scale. + let mut pcm = sine(1.0); + for plane in pcm.iter_mut() { + for (index, sample) in plane.iter_mut().enumerate() { + *sample *= if (24_000..24_480).contains(&index) { 40.0 } else { 4.0 }; + } + } + let len = pcm[0].len(); + let result = finish_audio(pcm, SceneAudio { gain_db: 12.0 }); + assert_eq!(result[0].len(), len); + let peak = result.iter().flatten().fold(0.0f32, |peak, s| peak.max(s.abs())); + assert!(peak <= LIMITER_CEILING * (1.0 + 1e-6), "peak {peak} over the ceiling"); + assert!(peak > LIMITER_CEILING * 0.99, "the ceiling is used, not undershot: {peak}"); + } + + #[test] + fn the_gain_ramps_down_before_a_peak_and_recovers_after_it() { + // A steady 0.5 with one sample at 2.0 halfway: the level must glide down over the + // look-ahead to exactly what the peak needs, and come back up afterwards, instead of + // stepping at the peak. + let spike = 24_000; + let mut plane = vec![0.5f32; 48_000]; + plane[spike] = 2.0; + let mut pcm = vec![plane.clone(), plane]; + limit_peaks(&mut pcm); + let out = &pcm[0]; + assert_eq!(out[spike - LIMITER_LOOKAHEAD - 1], 0.5, "untouched before the ramp"); + assert!((out[spike] - LIMITER_CEILING).abs() < 1e-6, "the peak lands on the ceiling"); + let floor = 0.5 * LIMITER_CEILING / 2.0; + let step = (0.5 - floor) / LIMITER_LOOKAHEAD as f32; + for index in spike - LIMITER_LOOKAHEAD..spike - 1 { + let fall = out[index] - out[index + 1]; + assert!( + fall >= 0.0 && fall <= step * 1.01, + "sample {index}: a {fall} drop is not a straight {step} ramp" + ); + } + // Five release time constants later the gain is back within 1 %. + let later = spike + (5.0 * LIMITER_RELEASE_SEC * 48_000.0) as usize; + assert!(out[later] > 0.495 && out[later] <= 0.5, "not recovered: {}", out[later]); + } + #[test] - fn output_is_clipped_to_full_scale_and_keeps_its_length() { - // The trim can push a hot signal past full scale; the timeline must come back the - // same length either way, or video and the following clips drift against it. - let result = finish_audio(planar(&[0.9, -0.9, 0.1]), SceneAudio { gain_db: 12.0 }); - assert_eq!(result[0].len(), 3); - assert_eq!(result[0][0], 1.0); - assert_eq!(result[0][1], -1.0); - assert!((result[0][2] - 0.1 * 10.0f32.powf(12.0 / 20.0)).abs() < 1e-6); + fn the_limiter_moves_both_channels_together() { + // A peak on the left alone must lower the right by the same gain, or the stereo + // image would lurch towards the right on every limited peak. + let mut pcm = vec![vec![0.5f32; 4_800], vec![0.5f32; 4_800]]; + pcm[0][2_400] = 2.0; + limit_peaks(&mut pcm); + assert!((pcm[1][2_400] - 0.5 * LIMITER_CEILING / 2.0).abs() < 1e-6); } } diff --git a/crates/compositor/src/audio_jobs.rs b/crates/compositor/src/audio_jobs.rs index 8d97be40c..5f690ac07 100644 --- a/crates/compositor/src/audio_jobs.rs +++ b/crates/compositor/src/audio_jobs.rs @@ -21,7 +21,9 @@ //! n'a pas d'état partagé entre contextes, et le décodeur vidéo du parcours en a un autre //! sur le même chemin, en lecture seule lui aussi. -use crate::audio::{decode_clip_audio, stretch_clip_pcm_by_speed, PlanarPcm}; +use crate::audio::{ + apply_gain_db, decode_clip_audio, loudness_gain_db, stretch_clip_pcm_by_speed, PlanarPcm, +}; use crate::regions::SpeedSegment; use std::collections::VecDeque; use std::thread::JoinHandle; @@ -34,7 +36,9 @@ use std::thread::JoinHandle; /// est tout ce qu'on cherche ici. const MAX_INFLIGHT_AUDIO_JOBS: usize = 4; -/// Le corps d'un job : décode la fenêtre gardée du clip et l'étire sur ses spans de vitesse. +/// Le corps d'un job : décode la fenêtre gardée du clip, l'étire sur ses spans de vitesse et +/// l'amène au niveau de loudness cible avec le gain mesuré sur le fichier entier +/// (`loudness_gain_db`, celui que la preview applique aussi). /// /// Rend `None` quand le clip se déclare audio mais n'a pas de flux décodable, ou quand le /// décodage échoue — dans les deux cas l'export continue et le clip sort muet, comme avant @@ -49,7 +53,11 @@ pub fn decode_and_stretch_clip_audio( out_fps: f64, ) -> Option { match decode_clip_audio(screen_path, source_start_sec, source_end_sec) { - Ok(Some(pcm)) => Some(stretch_clip_pcm_by_speed(&pcm, speed_segments, out_fps)), + Ok(Some(pcm)) => { + let mut pcm = stretch_clip_pcm_by_speed(&pcm, speed_segments, out_fps); + apply_gain_db(&mut pcm, loudness_gain_db(screen_path)); + Some(pcm) + } Ok(None) => { eprintln!( "[pipeline] warning: clip #{clip_index} déclaré audio mais sans flux décodable; silence conservé" diff --git a/crates/compositor/src/scene.rs b/crates/compositor/src/scene.rs index 0787ac638..40713319a 100644 --- a/crates/compositor/src/scene.rs +++ b/crates/compositor/src/scene.rs @@ -598,6 +598,20 @@ pub struct SceneAudioTrack { pub fade_in_sec: f64, #[serde(default)] pub fade_out_sec: f64, + /// A voiceover is voice: it is loudness-normalised like the recording's own audio. + /// A music bed is not. `#[serde(default)]` reads an older payload as music, which is + /// what every imported file was before voiceovers were recorded in the app. + #[serde(default)] + pub kind: SceneAudioTrackKind, +} + +/// `AxcutAudioTrack["kind"]` on the app side. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum SceneAudioTrackKind { + #[default] + Music, + Voiceover, } #[derive(Debug, Clone, Copy, Deserialize)] diff --git a/electron/electron-env.d.ts b/electron/electron-env.d.ts index be65ea224..c5a8e945f 100644 --- a/electron/electron-env.d.ts +++ b/electron/electron-env.d.ts @@ -373,6 +373,11 @@ interface Window { message?: string; error?: string; }>; + getLoudnessGain: (filePath: string) => Promise<{ + success: boolean; + gainDb: number; + message?: string; + }>; clearCurrentVideoPath: () => Promise<{ success: boolean }>; saveProjectFile: ( projectData: unknown, diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index fe60a1534..0e75615ce 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -92,6 +92,7 @@ import { macSystemPickerEnabled, markMacSystemPickerUnavailable, } from "../native-bridge/screen/macPickerSession"; +import { CompositorViewService } from "../native-bridge/services/compositorViewService"; import { getMacPermissions, showPermissionsWindow } from "../permissions"; import { scoreDeviceNameMatch } from "../recording/deviceNameMatching"; import { @@ -4240,6 +4241,33 @@ export function registerIpcHandlers( }, ); + // The loudness-normalisation gain the export applies to a voice file, measured by the + // compositor over the whole file and cached there, so the preview plays the voice at the + // level the export writes it. `gainDb: 0` whenever there is nothing to apply — no addon, a + // file with no audio, a failed read — which is the preview as it played before. + const loudnessService = new CompositorViewService(); + ipcMain.handle( + "get-loudness-gain", + async ( + _, + filePath: string, + ): Promise<{ success: boolean; gainDb: number; message?: string }> => { + try { + // Same approval gate as every other read of a renderer-supplied path. + const normalizedPath = readableApprovedPath(filePath); + if (!normalizedPath) { + return { success: false, gainDb: 0, message: "File path is not approved" }; + } + return { + success: true, + gainDb: (await loudnessService.loudnessGainDb(normalizedPath)) ?? 0, + }; + } catch (error) { + return { success: false, gainDb: 0, message: String(error) }; + } + }, + ); + // Cap renderer-requested chunk sizes so a buggy or compromised renderer // cannot make the main process allocate an arbitrarily large buffer. const MAX_IPC_CHUNK_BYTES = 64 * 1024 * 1024; diff --git a/electron/native-bridge/services/compositorViewService.ts b/electron/native-bridge/services/compositorViewService.ts index 39e64827d..b08a1a06d 100644 --- a/electron/native-bridge/services/compositorViewService.ts +++ b/electron/native-bridge/services/compositorViewService.ts @@ -737,4 +737,15 @@ export class CompositorViewService { } return addon.remuxSeekable(inputPath, outputPath); } + + /** The loudness-normalisation gain (dB) the export applies to this voice file, so the + * preview can play it at the same level. Null when the addon is absent or predates it: + * the preview then plays the file as recorded, the export still normalises. */ + async loudnessGainDb(filePath: string): Promise { + const addon = this.ensureAddon(); + if (!addon?.loudnessGainDb) { + return null; + } + return addon.loudnessGainDb(filePath); + } } diff --git a/electron/native/compositor-view/addon.d.ts b/electron/native/compositor-view/addon.d.ts index 1a32e554d..697c0ec1c 100644 --- a/electron/native/compositor-view/addon.d.ts +++ b/electron/native/compositor-view/addon.d.ts @@ -205,6 +205,14 @@ export interface CompositorViewAddon { * (dev trees keep a stale binary until the next `build-linux-compositor-addon.mjs`), * and the caller degrades to "leave the file alone" rather than failing the save. */ remuxSeekable?(inputPath: string, outputPath: string): Promise; + + /** Loudness-normalisation gain in dB that the export applies to this voice file (the + * recording's own audio, or a voiceover take), measured over the whole file. The + * preview applies the same number so it plays the voice at the exported level. + * 0 for a file with no audio, only silence, or that cannot be read. + * + * Optional for the same reason as `remuxSeekable`: a stale `.node` predates it. */ + loudnessGainDb?(path: string): Promise; } /** diff --git a/electron/preload.ts b/electron/preload.ts index cba14229c..732cb010a 100644 --- a/electron/preload.ts +++ b/electron/preload.ts @@ -342,6 +342,10 @@ contextBridge.exposeInMainWorld("electronAPI", { preparePreviewAudioTrack: (filePath: string) => { return ipcRenderer.invoke("prepare-preview-audio-track", filePath); }, + /** Loudness-normalisation gain the export applies to a voice file. See the handler. */ + getLoudnessGain: (filePath: string) => { + return ipcRenderer.invoke("get-loudness-gain", filePath); + }, clearCurrentVideoPath: () => { return ipcRenderer.invoke("clear-current-video-path"); }, diff --git a/src/components/ai-edition/VirtualPreview.audio.test.ts b/src/components/ai-edition/VirtualPreview.audio.test.ts index 624272848..f2187e0a5 100644 --- a/src/components/ai-edition/VirtualPreview.audio.test.ts +++ b/src/components/ai-edition/VirtualPreview.audio.test.ts @@ -9,11 +9,12 @@ import { timelineAudioFadeAt, } from "./VirtualPreview"; -/** Minimal stand-in: the function only ever touches `gain.gain.value`. */ +/** Minimal stand-in: the function only ever touches the two nodes' `gain.value`. */ function fakeGraph(): PreviewAudioGraph { return { context: {} as AudioContext, gain: { gain: { value: Number.NaN } } as GainNode, + voice: { gain: { value: Number.NaN } } as GainNode, }; } @@ -83,6 +84,27 @@ describe("applyPreviewAudioSettings", () => { expect(element.volume).toBe(0.25); expect(graph.gain.gain.value).toBeCloseTo(0.5, 4); }); + + it("levels the recording with the export's loudness gain, under the output trim", () => { + // The export multiplies each clip by `loudness_gain_db` of its file, then the whole + // mix by the trim. The preview has to play the same product, or the voice is heard + // at one level while editing and another in the file. + const graph = fakeGraph(); + applyPreviewAudioSettings(graph, [], -6.0206, 9.5424); + expect(graph.voice.gain.value).toBeCloseTo(3, 3); + expect(graph.gain.gain.value).toBeCloseTo(0.5, 4); + // No measurement yet (or nothing to correct): unity, the file as recorded. + applyPreviewAudioSettings(graph, [], 0); + expect(graph.voice.gain.value).toBe(1); + }); + + it("folds the loudness gain into the element-volume fallback, still capped at unity", () => { + const element = { volume: Number.NaN } as HTMLAudioElement; + applyPreviewAudioSettings(null, [element], -12.0412, 6.0206); + expect(element.volume).toBeCloseTo(0.5, 4); + applyPreviewAudioSettings(null, [element], 0, 6.0206); + expect(element.volume).toBe(1); + }); }); describe("resolveTimelineAudioPlayback", () => { diff --git a/src/components/ai-edition/VirtualPreview.tsx b/src/components/ai-edition/VirtualPreview.tsx index 3d73f92ed..1741debd6 100644 --- a/src/components/ai-edition/VirtualPreview.tsx +++ b/src/components/ai-edition/VirtualPreview.tsx @@ -158,33 +158,43 @@ export function timelineAudioFadeAt( export interface PreviewAudioGraph { context: AudioContext; + /** Output trim: everything the preview plays goes through it. */ gain: GainNode; + /** The recording's own audio (primary + supplemental elements), on its way to `gain`. */ + voice: GainNode; } /** - * The preview's ONLY audio processing is the output trim, and that is deliberate: it is - * the same `10 ** (dB / 20)` scalar `finish_audio` applies natively, so what the editor - * plays is what the export writes. + * The preview's audio processing is static gains only, and that is deliberate: each is the + * same `10 ** (dB / 20)` scalar the export applies natively, so what the editor plays is + * what the export writes. + * + * - `gainDb` is the output trim, `finish_audio`'s gain. + * - `voiceGainDb` is the loudness normalisation of the recording being played. The + * compositor measures it over the whole file and applies it to every clip cut from that + * file at export (`loudness_gain_db`); the preview asks for the same number. * * Nothing with state belongs here. The export runs on the assembled timeline (trimmed, - * speed-adjusted, concatenated); the preview runs on the untouched source file, seeked. - * A filter or a compressor would see a different signal on each side and drift — and an - * offline stage measured over the whole programme (a loudness normaliser) cannot exist - * here at all, because the preview never holds that programme. + * speed-adjusted, concatenated); the preview runs on the untouched source file, seeked. A + * filter or a compressor would see a different signal on each side and drift. That is why + * the export's peak limiter, which only acts above −1.5 dBFS, has no counterpart here. */ export function applyPreviewAudioSettings( graph: PreviewAudioGraph | null, elements: Array, gainDb: number, + voiceGainDb = 0, ): void { const outputGain = audioGainScalar(gainDb); + const voiceGain = audioGainScalar(voiceGainDb); if (!graph) { for (const element of elements) { - if (element) element.volume = Math.min(1, outputGain); + if (element) element.volume = Math.min(1, outputGain * voiceGain); } return; } graph.gain.gain.value = outputGain; + graph.voice.gain.value = voiceGain; } /** First clip (by timeline order) starting strictly after `afterTimelineStartSec` — @@ -375,6 +385,47 @@ export function VirtualPreview({ }; }, [activeSource?.filePath]); + // Loudness normalisation of every voice file the preview plays: the recording mounted now + // and each voiceover take. The export levels them to −16 LUFS with a gain the compositor + // measures over the whole file (`loudness_gain_db`), so asking it for that gain is what + // makes the preview play the voice at the exported level. Keyed by path: the gain belongs + // to the file, and the compositor caches it for the export that follows. Until it arrives + // the file plays as recorded — the first second or two after a recording is opened. + const [loudnessGainDbByPath, setLoudnessGainDbByPath] = useState>( + () => new Map(), + ); + const loudnessGainDbByPathRef = useRef(loudnessGainDbByPath); + loudnessGainDbByPathRef.current = loudnessGainDbByPath; + const requestedLoudnessRef = useRef(new Set()); + const voicePathsKey = [ + activeSource?.filePath, + ...audioTracks + .filter((track) => track.kind === "voiceover") + .map((track) => audioSources.find((source) => source.id === track.assetId)?.filePath), + ] + .filter((path): path is string => Boolean(path)) + .join("\n"); + useEffect(() => { + const getLoudnessGain = window.electronAPI?.getLoudnessGain; + if (!getLoudnessGain) return; + for (const path of voicePathsKey.split("\n")) { + if (!path || requestedLoudnessRef.current.has(path)) continue; + requestedLoudnessRef.current.add(path); + void getLoudnessGain(path).then( + (result) => + setLoudnessGainDbByPath((previous) => + new Map(previous).set(path, result.success ? result.gainDb : 0), + ), + () => undefined, + ); + } + }, [voicePathsKey]); + const voiceGainDb = activeSource?.filePath + ? (loudnessGainDbByPath.get(activeSource.filePath) ?? 0) + : 0; + const voiceGainDbRef = useRef(voiceGainDb); + voiceGainDbRef.current = voiceGainDb; + // Which imported-track elements are actually mounted (a track is rendered only once its // asset URL resolves — see the JSX). Re-routing the graph is keyed on this set, NOT on the // tracks' gains: a level change is applied live on the existing node by the rAF, so it must @@ -406,7 +457,9 @@ export function VirtualPreview({ } const gain = context.createGain(); gain.connect(context.destination); - return { context, gain }; + const voice = context.createGain(); + voice.connect(gain); + return { context, gain, voice }; } catch { return null; } @@ -414,7 +467,7 @@ export function VirtualPreview({ if (!graph) { // WebAudio can be unavailable in unit tests or under a denied audio policy. No source // node was created, so `volume` still reaches the output — capped at 0 dB. - applyPreviewAudioSettings(null, elements, audioGainDbRef.current); + applyPreviewAudioSettings(null, elements, audioGainDbRef.current, voiceGainDbRef.current); return; } @@ -427,7 +480,7 @@ export function VirtualPreview({ audioSourceNodesRef.current.set(element, source); } source.disconnect(); - source.connect(graph.gain); + source.connect(graph.voice); connectedSources.push(source); } catch { // Routing THIS element failed; leave the others alone. Once @@ -463,12 +516,13 @@ export function VirtualPreview({ } } audioGraphRef.current = graph; - applyPreviewAudioSettings(graph, elements, audioGainDbRef.current); + applyPreviewAudioSettings(graph, elements, audioGainDbRef.current, voiceGainDbRef.current); return () => { audioGraphRef.current = null; for (const source of connectedSources) source.disconnect(); for (const trackGain of trackGainNodes) trackGain.disconnect(); audioTrackGainNodesRef.current = new Map(); + graph.voice.disconnect(); graph.gain.disconnect(); }; }, [ @@ -511,8 +565,9 @@ export function VirtualPreview({ audioGraphRef.current, [primaryAudioRef.current, supplementalAudioRef.current], settings.audioGainDb, + voiceGainDb, ); - }, [settings.audioGainDb]); + }, [settings.audioGainDb, voiceGainDb]); const setPrimaryAudioElement = useCallback((element: HTMLAudioElement | null) => { primaryAudioRef.current = element; @@ -610,6 +665,8 @@ export function VirtualPreview({ //